@tangle-network/agent-runtime 0.143.0 → 0.153.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. package/README.md +32 -0
  2. package/dist/{activation-mi8TgSLM.js → activation-DKxQsHiO.js} +3 -3
  3. package/dist/{activation-mi8TgSLM.js.map → activation-DKxQsHiO.js.map} +1 -1
  4. package/dist/agent.d.ts +1 -1
  5. package/dist/agent.js +3 -3
  6. package/dist/{analyst-loop-CvkmfhUB.js → analyst-loop-B2kTHKNh.js} +2 -2
  7. package/dist/{analyst-loop-CvkmfhUB.js.map → analyst-loop-B2kTHKNh.js.map} +1 -1
  8. package/dist/analyst-loop.js +1 -1
  9. package/dist/{authoring-DuyShLdw.js → authoring-Tk2xFys0.js} +2 -2
  10. package/dist/{authoring-DuyShLdw.js.map → authoring-Tk2xFys0.js.map} +1 -1
  11. package/dist/candidate-execution/index.js +4 -4
  12. package/dist/{candidate-execution-CD5PoNmC.js → candidate-execution-CB9RJbbJ.js} +4 -4
  13. package/dist/{candidate-execution-CD5PoNmC.js.map → candidate-execution-CB9RJbbJ.js.map} +1 -1
  14. package/dist/{conversation-1hNtDeMM.js → conversation-D64WyWrK.js} +4 -4
  15. package/dist/{conversation-1hNtDeMM.js.map → conversation-D64WyWrK.js.map} +1 -1
  16. package/dist/conversation.d.ts +1 -1
  17. package/dist/conversation.js +1 -1
  18. package/dist/durable.d.ts +102 -4
  19. package/dist/durable.js +259 -10
  20. package/dist/durable.js.map +1 -1
  21. package/dist/{environment-provider-B8eZ85Ga.d.ts → environment-provider-DJG0yIYl.d.ts} +2 -2
  22. package/dist/{environment-provider-DCBJ13VP.js → environment-provider-D_MXw3jK.js} +88 -9
  23. package/dist/environment-provider-D_MXw3jK.js.map +1 -0
  24. package/dist/environment-provider.d.ts +1 -1
  25. package/dist/environment-provider.js +1 -1
  26. package/dist/executable-spec-BUcUcEPk.js +73 -0
  27. package/dist/executable-spec-BUcUcEPk.js.map +1 -0
  28. package/dist/{graph-De_XEsfH.js → graph-M0Qr4863.js} +4 -4
  29. package/dist/{graph-De_XEsfH.js.map → graph-M0Qr4863.js.map} +1 -1
  30. package/dist/{improvement-cycle-Bi-Tf4QF.js → improvement-cycle-CDxlGQGE.js} +7 -7
  31. package/dist/{improvement-cycle-Bi-Tf4QF.js.map → improvement-cycle-CDxlGQGE.js.map} +1 -1
  32. package/dist/{index-B8rKtH3U.d.ts → index-BihHgUAI.d.ts} +4 -4
  33. package/dist/{index-C2caGomG.d.ts → index-Dlz7ejRx.d.ts} +339 -503
  34. package/dist/{index-D4sLkOPY.d.ts → index-LQfgFJoX.d.ts} +2 -2
  35. package/dist/index.d.ts +7 -7
  36. package/dist/index.js +13 -13
  37. package/dist/intelligence.d.ts +4 -4
  38. package/dist/intelligence.js +6 -6
  39. package/dist/kernel.d.ts +6 -6
  40. package/dist/kernel.js +13 -13
  41. package/dist/{knowledge-CVE6Wk1i.js → knowledge-DM-i40b9.js} +10 -7
  42. package/dist/{knowledge-CVE6Wk1i.js.map → knowledge-DM-i40b9.js.map} +1 -1
  43. package/dist/knowledge.d.ts +1 -1
  44. package/dist/knowledge.js +1 -1
  45. package/dist/{loop-runner-bin-BoHdQqLm.d.ts → loop-runner-bin-D-BW51yr.d.ts} +3 -3
  46. package/dist/{loop-runner-bin-DdaI1nLL.js → loop-runner-bin-pD5WJuqJ.js} +4 -4
  47. package/dist/{loop-runner-bin-DdaI1nLL.js.map → loop-runner-bin-pD5WJuqJ.js.map} +1 -1
  48. package/dist/loop-runner-bin.d.ts +1 -1
  49. package/dist/loop-runner-bin.js +1 -1
  50. package/dist/{materialization-P86EciBw.js → materialization-Bo6j3Hg9.js} +257 -61
  51. package/dist/materialization-Bo6j3Hg9.js.map +1 -0
  52. package/dist/mcp/bin.js +3 -3
  53. package/dist/mcp/index.d.ts +3 -3
  54. package/dist/mcp/index.js +6 -6
  55. package/dist/{model-policy-DgRJIJse.js → model-policy-Sw4ywhtL.js} +68 -6
  56. package/dist/model-policy-Sw4ywhtL.js.map +1 -0
  57. package/dist/{openai-tools-BiAWKjh6.js → openai-tools-CWBywy-e.js} +2 -2
  58. package/dist/{openai-tools-BiAWKjh6.js.map → openai-tools-CWBywy-e.js.map} +1 -1
  59. package/dist/{prepare-D3n7cA39.js → prepare-C3eFd3UA.js} +2 -2
  60. package/dist/{prepare-D3n7cA39.js.map → prepare-C3eFd3UA.js.map} +1 -1
  61. package/dist/primeintellect/index.d.ts +1 -1
  62. package/dist/profiles.js +1 -1
  63. package/dist/{protected-model-port-bq6bD9YX.js → protected-model-port-Dn2b6peS.js} +2 -2
  64. package/dist/{protected-model-port-bq6bD9YX.js.map → protected-model-port-Dn2b6peS.js.map} +1 -1
  65. package/dist/{redact-DexTuAiX.d.ts → redact-NAptmqKw.d.ts} +3 -3
  66. package/dist/{researcher-CiSY6Wzb.js → researcher-BMFFPwKb.js} +2 -2
  67. package/dist/{researcher-CiSY6Wzb.js.map → researcher-BMFFPwKb.js.map} +1 -1
  68. package/dist/{run-layout-Be_Z0dF8.js → run-layout-C2jgwsCq.js} +111 -6
  69. package/dist/run-layout-C2jgwsCq.js.map +1 -0
  70. package/dist/{runtime-J6Toky8K.d.ts → runtime-B-5dp5Iz.d.ts} +214 -14
  71. package/dist/{runtime-BnqjbOGX.js → runtime-RY_rvv7t.js} +85 -82
  72. package/dist/runtime-RY_rvv7t.js.map +1 -0
  73. package/dist/{sandbox-events-COwCWUVB.js → sandbox-events-DMijDq2p.js} +49 -2
  74. package/dist/sandbox-events-DMijDq2p.js.map +1 -0
  75. package/dist/{spawn-journal-D2HKUImC.js → spawn-journal-CLguaA2A.js} +8 -43
  76. package/dist/spawn-journal-CLguaA2A.js.map +1 -0
  77. package/dist/{stream-agent-turn-C1Vs_3B9.d.ts → stream-agent-turn-BGyszuPH.d.ts} +2 -2
  78. package/dist/{stream-agent-turn-CXRugSGO.js → stream-agent-turn-D6HoaRPV.js} +90 -14
  79. package/dist/stream-agent-turn-D6HoaRPV.js.map +1 -0
  80. package/dist/{structural-rollout-CJgj9YXH.js → structural-rollout-P1YxFRfl.js} +11 -8
  81. package/dist/structural-rollout-P1YxFRfl.js.map +1 -0
  82. package/dist/{supervise-hCgG6ROJ.js → supervise-BezbzKiJ.js} +1550 -1186
  83. package/dist/supervise-BezbzKiJ.js.map +1 -0
  84. package/dist/{supervisor-s8SfZYbW.js → supervisor-DBVlPONC.js} +597 -194
  85. package/dist/supervisor-DBVlPONC.js.map +1 -0
  86. package/dist/testing.d.ts +2 -2
  87. package/dist/testing.js +12 -12
  88. package/dist/{top-app-BMOSKK06.js → top-app-tKo9OrON.js} +189 -86
  89. package/dist/top-app-tKo9OrON.js.map +1 -0
  90. package/dist/tui/bin.js +1 -1
  91. package/dist/tui/index.d.ts +32 -4
  92. package/dist/tui/index.js +1 -1
  93. package/dist/{types-Df9vulGc.d.ts → types-BPQhQRrC.d.ts} +507 -10
  94. package/dist/{util-C1bv6yog.js → util-Bncw0o_F.js} +41 -2
  95. package/dist/util-Bncw0o_F.js.map +1 -0
  96. package/dist/{workspace-archive-C52LjkeN.js → workspace-archive-Dm73lTfB.js} +2 -2
  97. package/dist/{workspace-archive-C52LjkeN.js.map → workspace-archive-Dm73lTfB.js.map} +1 -1
  98. package/package.json +4 -4
  99. package/dist/environment-provider-DCBJ13VP.js.map +0 -1
  100. package/dist/materialization-P86EciBw.js.map +0 -1
  101. package/dist/model-policy-DgRJIJse.js.map +0 -1
  102. package/dist/retained-run-binding-Dp2e8xne.js +0 -268
  103. package/dist/retained-run-binding-Dp2e8xne.js.map +0 -1
  104. package/dist/run-layout-Be_Z0dF8.js.map +0 -1
  105. package/dist/runtime-BnqjbOGX.js.map +0 -1
  106. package/dist/sandbox-events-COwCWUVB.js.map +0 -1
  107. package/dist/spawn-journal-D2HKUImC.js.map +0 -1
  108. package/dist/stream-agent-turn-CXRugSGO.js.map +0 -1
  109. package/dist/structural-rollout-CJgj9YXH.js.map +0 -1
  110. package/dist/supervise-hCgG6ROJ.js.map +0 -1
  111. package/dist/supervisor-s8SfZYbW.js.map +0 -1
  112. package/dist/top-app-BMOSKK06.js.map +0 -1
  113. package/dist/util-C1bv6yog.js.map +0 -1
@@ -1,1127 +1,492 @@
1
+ import { S as runtimeOwnedScopeOwnerRuntime, _ as runtimeOwnedDriveHarnessProviderEvidence, b as runtimeOwnedExecutorProviderEvidence, d as recordRuntimeOwnedDriveHarnessProviderEvidence, i as attestRuntimeOwnedScopeOwner, s as inheritRuntimeOwnedExecutorAttestation, v as runtimeOwnedExecutorExecutionBinding, x as runtimeOwnedPendingExecutorMaterialization, y as runtimeOwnedExecutorMaterialization } from "./materialization-Bo6j3Hg9.js";
1
2
  import { f as RuntimeRunStateError, i as ConfigError, m as ValidationError, o as NotFoundError, r as BackendTransportError, t as AgentEvalError$1 } from "./errors-CDZ8XsVj.js";
2
3
  import { t as detachedSnapshot } from "./snapshot-cK9K420n.js";
3
- import { S as contentAddress, b as parseWorkerToolTraceArtifact, d as prepareJsonlAppend, f as writeAllBytes, i as InMemorySpawnJournal, n as FileSpawnJournal, r as InMemoryResultBlobStore, t as FileResultBlobStore, u as parseCommittedJsonLines, x as workerTraceAnalysisStore } from "./spawn-journal-D2HKUImC.js";
4
- import { n as chargedTokens } from "./util-C1bv6yog.js";
5
- import { S as runtimeOwnedScopeOwnerRuntime, _ as runtimeOwnedDriveHarnessProviderEvidence, b as runtimeOwnedExecutorProviderEvidence, d as recordRuntimeOwnedDriveHarnessProviderEvidence, i as attestRuntimeOwnedScopeOwner, s as inheritRuntimeOwnedExecutorAttestation, v as runtimeOwnedExecutorExecutionBinding, x as runtimeOwnedPendingExecutorMaterialization, y as runtimeOwnedExecutorMaterialization } from "./materialization-P86EciBw.js";
6
- import { a as concreteProfileModel, d as harnessRunsAgent, n as assertModelAllowed, r as assertProfileModelsAllowed, s as profileModelExecutionSettings, t as assertExecutableAgentProfile, u as agentHarness } from "./model-policy-DgRJIJse.js";
7
- import { $ as createInbox, B as WORKER_TRACE_PROPAGATION, D as DEFAULT_SUCCESSFUL_SHUTDOWN_MS, Et as buildLoopSpanNodes, I as createExecutor, It as toOtelAttributes, L as createExecutorRegistry, M as bindReusableExecutorExecutionId, Mt as generateSpanId, N as bridgeStopSignalKey, O as teardownExecutor, P as captureReusableExecutorConfig, R as snapshotExecutorConfig, _t as runBrainLoop, a as pickBestDelivered, an as promptControlProfileMaterialization, c as driverChild, d as deriveNodeExecutionIdentity, en as assertProfileMaterialization, f as meterRuntimeOwnedAccounting, fn as worktreeCliProfileMaterialization, gt as routerBrain, h as scopeOwnerExecutorNodeContext, in as profileMaterializationAxes$1, it as createPeerMailbox, j as spendFromUsageEvents, k as assertValidBudget, kt as createOtelExporter, l as withDriverExecutor, m as recordScopeOwnerMaterialization, n as createSupervisor, nn as defineProfileMaterializationContract, o as runFinalizer, on as promptModelProfileMaterialization, p as meterRuntimeOwnedProviderAttempt, r as bestDelivered, rn as fullProfileMaterialization, s as runTree, tn as controlProfileMaterialization, w as freeSlots } from "./supervisor-s8SfZYbW.js";
4
+ import { S as contentAddress, b as parseWorkerToolTraceArtifact, d as prepareJsonlAppend, f as writeAllBytes, i as InMemorySpawnJournal, n as FileSpawnJournal, r as InMemoryResultBlobStore, t as FileResultBlobStore, u as parseCommittedJsonLines, x as workerTraceAnalysisStore } from "./spawn-journal-CLguaA2A.js";
5
+ import { r as chargedTokens } from "./util-Bncw0o_F.js";
6
+ import { a as concreteProfileModel, c as profileModelExecutionSettings, d as agentHarness, f as harnessRunsAgent, n as assertModelAllowed, o as enforceTokenLimits, r as assertProfileModelsAllowed, s as profileBridgeWireModel, t as assertExecutableAgentProfile } from "./model-policy-Sw4ywhtL.js";
7
+ import { B as createExecutorRegistry, D as DEFAULT_SUCCESSFUL_SHUTDOWN_MS, F as bridgeRuntimeAttachmentsKey, Ft as generateSpanId, I as bridgeStopSignalKey, L as captureReusableExecutorConfig, M as bindReusableExecutorExecutionId, Mt as createOtelExporter, N as bridgeAdmissionRead, O as teardownExecutor, P as bridgeModelRouteRefusal, U as WORKER_TRACE_PROPAGATION, V as snapshotExecutorConfig, _n as worktreeCliProfileMaterialization, a as pickBestDelivered, an as defineProfileMaterializationContract, bt as runBrainLoop, c as driverChild, cn as promptControlProfileMaterialization, d as deriveNodeExecutionIdentity, f as meterRuntimeOwnedAccounting, h as scopeOwnerExecutorNodeContext, hn as unsupportedProfileDimensions, in as controlProfileMaterialization, j as spendFromUsageEvents, k as assertValidBudget, kt as buildLoopSpanNodes, l as withDriverExecutor, ln as promptModelProfileMaterialization, m as recordScopeOwnerMaterialization, n as createSupervisor, nt as createInbox, o as runFinalizer, on as fullProfileMaterialization, p as meterRuntimeOwnedProviderAttempt, pn as renderUnsupported, r as bestDelivered, rn as assertProfileMaterialization, s as runTree, sn as profileMaterializationAxes$1, st as createPeerMailbox, t as createRootHandle, w as freeSlots, yt as routerBrain, z as createExecutor, zt as toOtelAttributes } from "./supervisor-DBVlPONC.js";
8
8
  import { t as composeRuntimeHooks } from "./runtime-hooks-tXpAarhW.js";
9
- import { _ as writeWorkerCancellation, a as readWorkerCancellation, i as readWorkerCancelRequests } from "./run-layout-Be_Z0dF8.js";
9
+ import { C as writeWorkerCancellation, S as writeRunCancellation, a as readRunCancelRequest, c as readWorkerCancellation, o as readRunCancellation, s as readWorkerCancelRequests } from "./run-layout-C2jgwsCq.js";
10
10
  import { t as createStdioToolServer } from "./tool-server-DEmLr9YY.js";
11
11
  import { n as UI_LENSES } from "./substrate-B0TYNrXn.js";
12
12
  import { agentProfileSchema, canonicalAgentProfileDigest, canonicalCandidateDigest, validateAgentProfileSecurity } from "@tangle-network/agent-interface";
13
13
  import { argHash, errorStreakDetector, observeAll, repeatedActionDetector } from "@tangle-network/agent-eval";
14
14
  import { randomUUID } from "node:crypto";
15
15
  import path, { dirname, join, resolve } from "node:path";
16
+ import { isMaterializerHarness } from "@tangle-network/agent-profile-materialize";
16
17
  import { mkdir, readFile, rename, writeFile } from "node:fs/promises";
17
18
  import { readFileSync } from "node:fs";
18
19
  import { createServer } from "node:http";
19
20
  import { Readable, Writable } from "node:stream";
20
- //#region src/runtime/supervise/completion-gate.ts
21
+ //#region src/runtime/supervise/detector-monitor.ts
21
22
  /**
22
23
  *
23
- * The completion-oracle: **settled DELIVERED.**
24
- *
25
- * Foreman's one hard lesson (0/18 self-improvement deliverables) "done" must mean a check
26
- * PASSED, not the agent's say-so. `gateOnDeliverable` wraps an `Executor` so its settlement
27
- * is `valid` ONLY when the deliverable check passes. The child still RUNS and settles (its
28
- * spend is conserved into the pool either way), but a child that ran WITHOUT delivering
29
- * settles `valid:false` — so a keep-best driver never counts it as done, and a gate never
30
- * inflates with self-judged wins.
31
- *
32
- * Dual-purpose by construction:
33
- * - product: the agent fleet only advances on real, checked deliverables.
34
- * - proof: the gate's `valid` is the honest settle — equal-k comparisons can't be gamed by an
35
- * arm that "ran" without producing the artifact.
36
- *
37
- * The check is a DEPLOYABLE oracle — a test command, a state verifier, the commit0 judge —
38
- * read off the child's output, never the model judging itself. A throwing check is
39
- * fail-closed (not delivered), never a crash.
24
+ * The ONLINE analyst: watch a `TraceSource` and fold each tool span through agent-eval's published
25
+ * streaming detector kernel (`repeatedActionDetector`/`errorStreakDetector` — the SAME kernel the
26
+ * control loop folds), firing `onSignal` the moment a worker loops or error-storms. Substrate-
27
+ * agnostic: it consumes spans from any source (owned router/bridge loop OR a sandbox box session),
28
+ * never the raw tool seam. Detection logic + the failure taxonomy live in agent-eval; not reimplemented.
40
29
  *
41
30
  * @experimental
42
31
  */
43
- /**
44
- * Wrap an `Executor` so its settlement `valid` reflects the deliverable check, not the
45
- * inner verdict. Handles both `execute` shapes (one-shot `Promise<ExecutorResult>` and
46
- * streaming `AsyncIterable<UsageEvent>` + `resultArtifact()`); the check runs once the inner
47
- * executor has produced its output. The inner `score` is preserved; only `valid` is gated.
48
- */
49
- function gateOnDeliverable(inner, deliverable) {
50
- let gated;
51
- const check = async (out, baseScore) => {
52
- let delivered;
32
+ /** The default online panel for a tool-call pipe: a worker repeating the same call, or hammering
33
+ * consecutive errors. (No-progress needs a domain progress-probe, so it is opt-in, not default.)
34
+ *
35
+ * Coverage note: `repeated-action` works for EVERY harness (it needs only tool name + args, which
36
+ * every adapter provides). `error-streak` needs per-call status opencode carries it inline
37
+ * (`state.status`, VALIDATED live), but claude-code/codex tool-call parts do NOT (their errors live
38
+ * in separate result blocks not yet decoded), so error-streak is silent for those until result-block
39
+ * decoding is added + live-validated. It is in the panel because it is correct where status exists. */
40
+ function defaultToolDetectors() {
41
+ return [repeatedActionDetector({ maxRepeated: 3 }), errorStreakDetector({ maxErrors: 3 })];
42
+ }
43
+ /** Subscribe to a `TraceSource` and run the streaming detectors over its live spans. Returns an
44
+ * unsubscribe. A defensive `argHash` failure (circular args) never throws out of the side-channel. */
45
+ function watchTrace(source, opts = {}) {
46
+ const detectors = opts.detectors ?? defaultToolDetectors();
47
+ return source.onSpan((span) => {
48
+ let fingerprint;
53
49
  try {
54
- delivered = await deliverable.check(out) === true;
50
+ fingerprint = `${span.toolName}|${argHash(span.args)}`;
55
51
  } catch {
56
- delivered = false;
52
+ fingerprint = `${span.toolName}|<unhashable>`;
57
53
  }
58
- return {
59
- valid: delivered,
60
- score: baseScore ?? (delivered ? 1 : 0)
61
- };
62
- };
63
- /**
64
- * Ask the delivery question once, from whatever the inner executor managed to produce.
65
- *
66
- * Fail-closed on the artifact being unavailable: an executor that never produced one delivered
67
- * nothing, and leaving `gated` unset keeps the existing invalid-by-default reading.
68
- */
69
- const settleVerdict = async () => {
70
- let art;
71
- try {
72
- art = inner.resultArtifact();
73
- } catch {
74
- return;
54
+ const signals = observeAll(detectors, {
55
+ actionFingerprint: fingerprint,
56
+ ...span.status ? { status: span.status } : {},
57
+ label: span.toolName
58
+ });
59
+ for (const s of signals) opts.onSignal?.(s, span);
60
+ });
61
+ }
62
+ //#endregion
63
+ //#region src/runtime/supervise/event-bus.ts
64
+ /** Create the child→parent coordination bus: one typed pipe for settled outputs, questions, and analyst findings, with a priority-ordered pull queue and a pass-through subscribe lane.
65
+ * @experimental In-process queue; durability is a transport swap that does not exist yet. */
66
+ function createEventBus(now = Date.now) {
67
+ const queue = [];
68
+ const log = [];
69
+ const subscribers = [];
70
+ const byKind = {};
71
+ const staged = /* @__PURE__ */ new WeakMap();
72
+ let seq = 0;
73
+ let published = 0;
74
+ let pulled = 0;
75
+ const matches = (r, kinds) => !kinds || kinds.includes(r.event.type);
76
+ const bestIndex = (kinds) => {
77
+ let best = -1;
78
+ let bestPriority = Number.NEGATIVE_INFINITY;
79
+ for (let i = 0; i < queue.length; i++) {
80
+ const r = queue[i];
81
+ if (!r || !matches(r, kinds)) continue;
82
+ if (r.priority > bestPriority) {
83
+ best = i;
84
+ bestPriority = r.priority;
85
+ }
75
86
  }
76
- gated = await check(art.out, art.verdict?.score);
87
+ return best;
77
88
  };
78
- return inheritRuntimeOwnedExecutorAttestation(inner, {
79
- runtime: inner.runtime,
80
- ...inner.budgetExempt !== void 0 ? { budgetExempt: inner.budgetExempt } : {},
81
- ...inner.deliver ? { deliver: (m) => inner.deliver?.(m) } : {},
82
- ...inner.progress ? { progress: () => inner.progress?.() } : {},
83
- ...inner.traceSource ? { traceSource: () => inner.traceSource?.() } : {},
84
- ...inner.metered ? { metered: () => inner.metered?.() } : {},
85
- execute(task, signal) {
86
- const r = inner.execute(task, signal);
87
- if (isAsyncIterable$1(r)) return (async function* () {
88
- try {
89
- for await (const ev of r) yield ev;
90
- } finally {
91
- await settleVerdict();
92
- }
93
- })();
94
- return (async () => {
95
- let res;
96
- try {
97
- res = await r;
98
- } catch (error) {
99
- await settleVerdict();
100
- throw error;
101
- }
102
- gated = await check(res.out, res.verdict?.score);
103
- return {
104
- ...res,
105
- verdict: gated
106
- };
107
- })();
89
+ return {
90
+ async publish(event, opts) {
91
+ const record = staged.get(event) ?? {
92
+ seq: seq++,
93
+ at: now(),
94
+ priority: opts?.priority ?? 0,
95
+ event
96
+ };
97
+ staged.set(event, record);
98
+ for (const handler of subscribers) await handler(record);
99
+ staged.delete(event);
100
+ if (opts?.queue !== false) queue.push(record);
101
+ log.push(record);
102
+ published += 1;
103
+ byKind[event.type] = (byKind[event.type] ?? 0) + 1;
104
+ return record;
108
105
  },
109
- teardown: (grace) => inner.teardown(grace),
110
- resultArtifact() {
111
- const art = inner.resultArtifact();
106
+ pull(kinds) {
107
+ const i = bestIndex(kinds);
108
+ if (i < 0) return void 0;
109
+ pulled++;
110
+ return queue.splice(i, 1)[0]?.event;
111
+ },
112
+ subscribe(handler) {
113
+ subscribers.push(handler);
114
+ return () => {
115
+ const i = subscribers.indexOf(handler);
116
+ if (i >= 0) subscribers.splice(i, 1);
117
+ };
118
+ },
119
+ pending(kinds) {
120
+ return kinds ? queue.filter((r) => matches(r, kinds)).length : queue.length;
121
+ },
122
+ history() {
123
+ return log;
124
+ },
125
+ stats() {
112
126
  return {
113
- ...art,
114
- verdict: gated ?? art.verdict
127
+ published,
128
+ pulled,
129
+ byKind: { ...byKind }
115
130
  };
116
131
  }
117
- });
118
- }
119
- /**
120
- * Transform a Runtime executor's terminal artifact without losing its private
121
- * profile-materialization attestation or altering its measured spend. This is
122
- * the composition point for deterministic post-processing and grading; callers
123
- * must not rebuild an Executor around a model transport merely to change `out`.
124
- */
125
- function mapExecutorResult(inner, map) {
126
- let mapped;
127
- const settle = async (result, task) => {
128
- const transformed = await map(result, task);
129
- mapped = {
130
- outRef: transformed.outRef,
131
- out: transformed.out,
132
- ...transformed.verdict ? { verdict: transformed.verdict } : {},
133
- spent: result.spent
134
- };
135
- return mapped;
136
132
  };
137
- return inheritRuntimeOwnedExecutorAttestation(inner, {
138
- runtime: inner.runtime,
139
- ...inner.budgetExempt !== void 0 ? { budgetExempt: inner.budgetExempt } : {},
140
- ...inner.deliver ? { deliver: (message) => inner.deliver?.(message) } : {},
141
- ...inner.progress ? { progress: () => inner.progress?.() } : {},
142
- ...inner.traceSource ? { traceSource: () => inner.traceSource?.() } : {},
143
- ...inner.accounting ? { accounting: () => inner.accounting?.() } : {},
144
- ...inner.metered ? { metered: () => inner.metered?.() } : {},
145
- execute(task, signal) {
146
- const execution = inner.execute(task, signal);
147
- if (isAsyncIterable$1(execution)) return (async function* () {
148
- for await (const event of execution) yield event;
149
- await settle(inner.resultArtifact(), task);
150
- })();
151
- return (async () => settle(await execution, task))();
152
- },
153
- teardown: (grace) => inner.teardown(grace),
154
- resultArtifact() {
155
- if (!mapped) throw new Error("mapExecutorResult: resultArtifact() read before execute()");
156
- return mapped;
157
- }
158
- });
159
- }
160
- function isAsyncIterable$1(v) {
161
- return v != null && typeof v[Symbol.asyncIterator] === "function";
162
133
  }
163
134
  //#endregion
164
- //#region src/runtime/supervise/otel-spans.ts
135
+ //#region src/mcp/tools/coordination.ts
165
136
  /**
166
- * Supervisor tree → OTLP spans. OPT-IN, off by default.
167
- *
168
- * WHY. A supervised tree is legible today only by parsing this package's own spawn journal, so
169
- * every other multi-agent shape on the machine (a coding-CLI's subagents, a pi fanout, ad-hoc tool
170
- * parallelism) needs its own bespoke reader. A span carrying `parent_span_id` IS a tree, and any
171
- * system can emit one — so emitting spans makes the supervisor readable by the same viewer as
172
- * everything else, with no per-system reader.
173
- *
174
- * WHAT IT IS NOT. This is telemetry, never the record of truth. The spawn journal remains the sole
175
- * durable ledger for replay/resume and for cost; nothing here is read back, and a run whose export
176
- * fails is unaffected in every observable way. The two data models are deliberately separate.
177
- *
178
- * HOW IT ATTACHES. This is a pure `RuntimeHooks` observer over the lifecycle events `Scope` ALREADY
179
- * emits — `agent.spawn` (a node opened), `agent.child` (a node settled), `agent.turn` (a driver
180
- * inference turn was metered). It adds no event, mutates no journal, and changes no result. Because
181
- * `Scope` re-seeds the same hooks into every nested scope (`makeNestedScopeSeam`), one observer sees
182
- * the WHOLE recursion at arbitrary depth.
183
137
  *
184
- * SPAN SHAPE. One span per supervised node, opened at spawn and closed at settle, parented to its
185
- * parent node's span; the run root is the trace. Driver inference rides as an `LLM` child span under
186
- * the node that metered it. Attributes reuse the vocabulary the rest of the stack already reads
187
- * (`openinference.span.kind`, `agent.name`, `llm.token_count.*`, `llm.cost_usd`, `tangle.cost.usd`)
188
- * — see `@tangle-network/agent-eval`'s `src/trace/attribute-vocabulary.ts`, which is the consumer.
138
+ * MCP binding for a live `Scope`. A sandbox driver gets the same small verbs
139
+ * the in-process driver has: spawn, observe, await, steer, ask/answer, analyze,
140
+ * and stop. Settled outputs remain Scope artifacts; product code can project
141
+ * them into any UI/report envelope it needs.
189
142
  *
190
- * UNKNOWN IS NEVER ZERO. `Spend.tokensKnown === false` / `usdKnown === false` mark work that
191
- * HAPPENED with an unreported count. Those spans OMIT the token/cost attribute entirely and set
192
- * `tangle.supervise.tokens_known` / `tangle.supervise.cost_known` to `false`, so a reader can never
193
- * mistake an unmeasured turn for a free one.
194
- */
195
- /** OTEL status codes (`UNSET` / `OK` / `ERROR`) — the numeric wire values `OtelSpan.status` carries. */
196
- const STATUS_UNSET = 0;
197
- const STATUS_OK = 1;
198
- const STATUS_ERROR = 2;
199
- /** Longest string attribute value written from free-form detail, so an oversized turn payload
200
- * cannot inflate a span. Identity/label attributes we control are never truncated. */
201
- const MAX_DETAIL_CHARS = 256;
202
- /**
203
- * Build the span recorder for one supervised run, or `undefined` when no exporter resolves — the
204
- * off-by-default path. A run that passes no `exporter` and no `exportConfig` never reaches this
205
- * function at all; one that passes an `exportConfig` with no endpoint (and no env endpoint) gets
206
- * `undefined` here, so "configured but unreachable" also costs nothing.
143
+ * @experimental
207
144
  */
208
- function createSupervisorSpanRecorder(opts) {
209
- const exporter = opts.exporter ?? createOtelExporter(opts.exportConfig);
210
- if (!exporter) return void 0;
211
- const ownsExporter = opts.exporter === void 0;
212
- const now = opts.now ?? Date.now;
213
- const traceId = normalizeTraceId(opts.traceId, opts.runId);
214
- const rootSpanId = generateSpanId();
215
- const rootStartMs = now();
216
- const base = {
217
- "tangle.run.id": opts.runId,
218
- "tangle.sessionId": opts.runId,
219
- ...opts.attributes ?? {}
220
- };
221
- /** Node id → its open span. Seeded with the run id ⇒ the root span, because a depth-0 spawn's
222
- * `parentId` is the run id itself and every deeper spawn's is a real node id. */
223
- const open = /* @__PURE__ */ new Map();
224
- const spanIdOf = /* @__PURE__ */ new Map([[opts.runId, rootSpanId]]);
225
- let finished = false;
226
- /** Every export is best-effort: a throwing exporter must never reach the run. */
227
- const emit = (span) => {
228
- try {
229
- exporter.exportSpan(span);
230
- } catch {}
231
- };
232
- const span = (spanId, parentSpanId, name, startMs, endMs, attrs, status, message) => ({
233
- traceId,
234
- spanId,
235
- ...parentSpanId ? { parentSpanId } : {},
236
- name,
237
- kind: 1,
238
- startTimeUnixNano: msToNano(startMs),
239
- endTimeUnixNano: msToNano(Math.max(startMs, endMs)),
240
- attributes: toOtelAttributes(attrs),
241
- status: {
242
- code: status,
243
- ...message ? { message } : {}
244
- }
245
- });
246
- function onSpawn(event) {
247
- const p = record(event.payload);
248
- const childId = str(p.childId);
249
- if (!childId) return;
250
- const label = str(p.label) ?? "node";
251
- const runtime = str(p.runtime);
252
- const isWait = runtime === "wait";
253
- const attrs = {
254
- ...base,
255
- "openinference.span.kind": isWait ? "CHAIN" : "AGENT",
256
- "agent.name": label,
257
- "tangle.supervise.node.id": childId,
258
- "tangle.supervise.node.label": label,
259
- "tangle.supervise.node.kind": isWait ? "wait" : "agent",
260
- "tangle.supervise.tree.root": event.runId
261
- };
262
- if (event.parentId) attrs["tangle.supervise.node.parent_id"] = event.parentId;
263
- if (runtime) attrs["tangle.supervise.node.runtime"] = runtime;
264
- if (typeof p.depth === "number") attrs["tangle.supervise.node.depth"] = p.depth;
265
- if (typeof event.stepIndex === "number") attrs["tangle.supervise.node.ordinal"] = event.stepIndex;
266
- if (p.resumed === true) attrs["tangle.supervise.node.resumed"] = true;
267
- assignBudget(attrs, p.budget);
268
- const spanId = generateSpanId();
269
- spanIdOf.set(childId, spanId);
270
- open.set(childId, {
271
- spanId,
272
- parentSpanId: event.parentId && spanIdOf.get(event.parentId) || rootSpanId,
273
- name: label,
274
- startMs: event.timestamp,
275
- attrs
276
- });
145
+ /** Producer-side cleanliness for the `finding` event. The findings payload is arbitrary analyst
146
+ * output, the digest a subscriber computes (RFC 8785) throws on ANY `undefined` value — nested
147
+ * included and a throwing subscriber leaves the event invisible to EVERY subscriber. The
148
+ * producer, not the digest, owns keeping the event canonical: an `undefined` payload is stripped
149
+ * to key-absence, everything else is JSON round-tripped (nested `undefined` object values drop,
150
+ * `undefined` array slots become `null`), and a payload JSON cannot represent at all (cycle,
151
+ * BigInt, bare function) becomes a record OF that fact — degraded findings beat a vanished
152
+ * event. */
153
+ function canonicalFindingEvent(finding) {
154
+ if (finding.findings === void 0) {
155
+ const { findings: _absent, ...present } = finding;
156
+ return present;
277
157
  }
278
- function onSettled(event) {
279
- const p = record(event.payload);
280
- const childId = str(p.childId);
281
- if (!childId) return;
282
- const node = open.get(childId);
283
- if (!node) return;
284
- open.delete(childId);
285
- const status = str(p.status);
286
- const down = status === "down";
287
- const attrs = {
288
- ...node.attrs,
289
- "tangle.supervise.node.status": status ?? "done"
158
+ try {
159
+ return {
160
+ ...finding,
161
+ findings: JSON.parse(JSON.stringify(finding.findings))
290
162
  };
291
- if (typeof event.stepIndex === "number") attrs["tangle.supervise.node.seq"] = event.stepIndex;
292
- if (typeof p.outRef === "string") attrs["tangle.supervise.node.out_ref"] = p.outRef;
293
- if (typeof p.valid === "boolean") attrs["tangle.supervise.verdict.valid"] = p.valid;
294
- if (typeof p.score === "number") attrs["tangle.supervise.verdict.score"] = p.score;
295
- if (down) {
296
- attrs["error.type"] = p.infra === true ? "infra" : "child-down";
297
- const reason = str(p.reason);
298
- if (reason) attrs["error.message"] = truncate(reason);
299
- if (typeof p.infra === "boolean") attrs["tangle.supervise.node.infra"] = p.infra;
300
- }
301
- const wokeBy = str(record(p.wait).settled);
302
- if (wokeBy) attrs["tangle.supervise.wait.settled"] = wokeBy;
303
- assignSpend(attrs, p.spent);
304
- emit(span(node.spanId, node.parentSpanId, node.name, node.startMs, event.timestamp, attrs, down ? STATUS_ERROR : STATUS_OK, down ? str(p.reason) ?? "child down" : void 0));
305
- }
306
- function onTurn(event) {
307
- const p = record(event.payload);
308
- const parentId = event.parentId;
309
- const parentSpanId = parentId && spanIdOf.get(parentId) || rootSpanId;
310
- const attrs = {
311
- ...base,
312
- "openinference.span.kind": "LLM",
313
- "inference.observation_kind": "LLM",
314
- "tangle.supervise.node.kind": "inference"
163
+ } catch (error) {
164
+ return {
165
+ ...finding,
166
+ findings: { nonCanonicalFindings: error instanceof Error ? error.message : String(error) }
315
167
  };
316
- if (parentId) attrs["tangle.supervise.node.id"] = parentId;
317
- for (const [key, value] of Object.entries(p)) {
318
- if (key === "spend") continue;
319
- if (key === "driver" && typeof value === "string") {
320
- attrs["agent.name"] = value;
321
- attrs["inference.agent_name"] = value;
322
- continue;
323
- }
324
- if (key === "model" && typeof value === "string") {
325
- attrs["llm.model_name"] = value;
326
- continue;
327
- }
328
- if (Array.isArray(value)) {
329
- const scalars = value.filter((v) => typeof v === "string" || typeof v === "number");
330
- if (scalars.length > 0) attrs[`tangle.supervise.turn.${key}`] = truncate(scalars.join(","));
331
- continue;
332
- }
333
- if (typeof value === "string") attrs[`tangle.supervise.turn.${key}`] = truncate(value);
334
- else if (typeof value === "number" || typeof value === "boolean") attrs[`tangle.supervise.turn.${key}`] = value;
335
- }
336
- const spend = assignSpend(attrs, p.spend);
337
- const endMs = event.timestamp + (spend?.ms ?? 0);
338
- emit(span(generateSpanId(), parentSpanId, "gen_ai.client.inference", event.timestamp, endMs, attrs, STATUS_OK));
339
- }
340
- return {
341
- hooks: { onEvent(event) {
342
- if (finished) return;
343
- try {
344
- if (event.phase !== "after") return;
345
- if (event.target === "agent.spawn") onSpawn(event);
346
- else if (event.target === "agent.child") onSettled(event);
347
- else if (event.target === "agent.turn") onTurn(event);
348
- } catch {}
349
- } },
350
- traceId,
351
- rootSpanId,
352
- workerTrace(spawningNodeId) {
353
- return {
354
- traceId,
355
- parentSpanId: spanIdOf.get(spawningNodeId) ?? rootSpanId
356
- };
357
- },
358
- async finish(outcome) {
359
- if (finished) return;
360
- finished = true;
361
- const endMs = now();
362
- try {
363
- for (const [nodeId, node] of open) emit(span(node.spanId, node.parentSpanId, node.name, node.startMs, endMs, {
364
- ...node.attrs,
365
- "tangle.supervise.node.status": "unsettled",
366
- "tangle.supervise.node.settled": false,
367
- "tangle.supervise.node.id": nodeId
368
- }, STATUS_UNSET));
369
- open.clear();
370
- emit(span(rootSpanId, opts.parentSpanId, "supervisor.run", rootStartMs, endMs, rootAttrs(base, opts.agentName ?? "supervisor", outcome), rootStatus(outcome), rootMessage(outcome)));
371
- await exporter.flush();
372
- if (ownsExporter) await exporter.shutdown();
373
- } catch {}
374
- }
375
- };
376
- }
377
- function rootAttrs(base, agentName, outcome) {
378
- const attrs = {
379
- ...base,
380
- "openinference.span.kind": "AGENT",
381
- "inference.observation_kind": "AGENT",
382
- "agent.name": agentName,
383
- "inference.agent_name": agentName,
384
- "tangle.supervise.node.kind": "root"
385
- };
386
- const result = outcome?.result;
387
- if (result) {
388
- attrs["tangle.supervise.result"] = result.kind;
389
- if (result.kind === "no-winner") {
390
- attrs["tangle.supervise.reason"] = result.reason;
391
- attrs["tangle.supervise.down_count"] = result.downCount;
392
- if (result.reason === "driver-failed") {
393
- attrs["error.type"] = result.error.name;
394
- attrs["error.message"] = truncate(result.error.message);
395
- }
396
- }
397
- assignSpend(attrs, result.spentTotal);
398
- }
399
- if (outcome?.error !== void 0) {
400
- attrs["tangle.supervise.result"] = "error";
401
- const err = outcome.error;
402
- attrs["error.type"] = err instanceof Error ? err.name : typeof err;
403
- attrs["error.message"] = truncate(err instanceof Error ? err.message : String(err));
404
168
  }
405
- return attrs;
406
169
  }
407
- function rootStatus(outcome) {
408
- if (outcome?.error !== void 0) return STATUS_ERROR;
409
- if (!outcome?.result) return STATUS_UNSET;
410
- return outcome.result.kind === "winner" ? STATUS_OK : STATUS_ERROR;
170
+ /** Normalize the two spellings of an analyst-on-settle entry to the route form. */
171
+ function normalizeAnalyzeOnSettle(entry) {
172
+ return typeof entry === "string" ? { kind: entry } : entry;
411
173
  }
412
- function rootMessage(outcome) {
413
- if (outcome?.error !== void 0) return truncate(outcome.error instanceof Error ? outcome.error.message : String(outcome.error));
414
- const result = outcome?.result;
415
- return result && result.kind === "no-winner" ? result.reason : void 0;
416
- }
417
- /**
418
- * Write a `Spend` onto a span, honouring the "a missing measurement is never zero" invariant: a
419
- * channel marked NOT known contributes no number at all and instead flags itself, so nothing
420
- * downstream can sum an unmeasured turn as free. Returns the spend it read (for the duration).
421
- */
422
- function assignSpend(attrs, value) {
423
- if (!isRecord(value)) return void 0;
424
- const spend = value;
425
- const tokens = isRecord(spend.tokens) ? spend.tokens : void 0;
426
- if (spend.tokensKnown === false) attrs["tangle.supervise.tokens_known"] = false;
427
- else if (tokens) {
428
- if (typeof tokens.input === "number") attrs["llm.token_count.prompt"] = tokens.input;
429
- if (typeof tokens.output === "number") attrs["llm.token_count.completion"] = tokens.output;
430
- }
431
- if (spend.usdKnown === false) attrs["tangle.supervise.cost_known"] = false;
432
- else if (typeof spend.usd === "number" && Number.isFinite(spend.usd)) {
433
- attrs["llm.cost_usd"] = spend.usd;
434
- attrs["tangle.cost.usd"] = spend.usd;
435
- }
436
- if (typeof spend.iterations === "number") attrs["tangle.supervise.iterations"] = spend.iterations;
437
- if (typeof spend.ms === "number" && spend.ms > 0) attrs["tangle.supervise.duration_ms"] = spend.ms;
438
- return spend;
439
- }
440
- function assignBudget(attrs, value) {
441
- if (!isRecord(value)) return;
442
- const budget = value;
443
- if (typeof budget.maxTokens === "number") attrs["tangle.supervise.budget.max_tokens"] = budget.maxTokens;
444
- if (typeof budget.maxIterations === "number") attrs["tangle.supervise.budget.max_iterations"] = budget.maxIterations;
445
- if (typeof budget.maxUsd === "number") attrs["tangle.supervise.budget.max_usd"] = budget.maxUsd;
174
+ /** Every cause at zero — a pre-flight publishes its whole ledger from the first read. */
175
+ function emptyPreflightCounts() {
176
+ return {
177
+ "model-route": 0,
178
+ "bridge-full": 0,
179
+ "unmountable-tool": 0
180
+ };
446
181
  }
182
+ /** Default ceiling for a single `await_event` block (ms). Chosen well under any reasonable remote
183
+ * MCP client request timeout so the call returns a `pending` liveness snapshot instead of erroring;
184
+ * the supervisor re-polls until the worker settles. */
185
+ const DEFAULT_AWAIT_EVENT_TIMEOUT_MS = 15e3;
186
+ /** The reserved coordination verb names — the complete set `createCoordinationTools` can emit
187
+ * (the analyst pair is conditional but still reserved). A driver's extra WORK tools must not
188
+ * collide with any of these, or it could no longer coordinate; callers validate eagerly against
189
+ * this set so the conflict fails loud at construction, not buried in a swallowed `act()` throw. */
190
+ const coordinationVerbNames = [
191
+ "spawn_agent",
192
+ "observe_agent",
193
+ "steer_agent",
194
+ "await_event",
195
+ "list_questions",
196
+ "answer_question",
197
+ "ask_parent",
198
+ "submit_result",
199
+ "stop",
200
+ "list_analysts",
201
+ "run_analyst"
202
+ ];
203
+ const idArg = {
204
+ type: "string",
205
+ description: "The workerId returned by spawn_agent."
206
+ };
447
207
  /**
448
- * A trace id must be 32 hex characters. A caller-supplied one is used verbatim when it already is;
449
- * anything else (and the default) is DERIVED from the run id by content address — deterministic, so
450
- * a resumed run rejoins the trace its first process opened rather than forking a new one.
208
+ * Strip zod's object-KEY CODEC artifact from a derived JSON Schema, at every depth.
209
+ *
210
+ * `z.record(z.string(), …)` converts to `propertyNames: { type: 'string', pattern:
211
+ * '^u(?:[0-9a-f]{4})*$' }` — a marker of the lossy key round-trip, not a constraint anything
212
+ * enforces: `agentProfileSchema.safeParse({ name: 'r', tools: { bash: true, Read: false } })`
213
+ * succeeds with those plain keys. Published verbatim it reads to a model as "every key must be a
214
+ * run of hex quads", and the model either emits hex-encoded garbage keys or drops the field —
215
+ * on `tools`, `permissions`, `metadata`, `mcp`, `mcp.*.env`, and `model.metadata`, which is most
216
+ * of what a parent actually configures.
217
+ *
218
+ * Every `propertyNames` in the canonical profile schema is this artifact (25 of 25 at Interface
219
+ * 0.40), and the canonical schema constrains no key by pattern, so dropping the keyword outright
220
+ * loses nothing real and cannot be defeated by zod changing the encoding's exact regex.
221
+ *
222
+ * Rebuilds rather than mutates: the canonical conversion output must stay untouched for callers
223
+ * that compare against it.
451
224
  */
452
- function normalizeTraceId(traceId, runId) {
453
- if (traceId && /^[0-9a-f]{32}$/i.test(traceId)) return traceId.toLowerCase();
454
- return contentAddress(traceId ?? runId).slice(7, 39);
455
- }
456
- function msToNano(ms) {
457
- return (BigInt(Number.isFinite(ms) ? Math.floor(ms) : 0) * 1000000n).toString();
458
- }
459
- function isRecord(value) {
460
- return typeof value === "object" && value !== null && !Array.isArray(value);
461
- }
462
- function record(value) {
463
- return isRecord(value) ? value : {};
464
- }
465
- function str(value) {
466
- return typeof value === "string" && value.length > 0 ? value : void 0;
467
- }
468
- function truncate(value) {
469
- return value.length > MAX_DETAIL_CHARS ? `${value.slice(0, MAX_DETAIL_CHARS)}…` : value;
470
- }
471
- //#endregion
472
- //#region src/runtime/supervise/coordination-log.ts
225
+ const stripKeyCodecArtifacts = (node) => {
226
+ if (Array.isArray(node)) return node.map(stripKeyCodecArtifacts);
227
+ if (!node || typeof node !== "object") return node;
228
+ return Object.fromEntries(Object.entries(node).filter(([key]) => key !== "propertyNames").map(([key, value]) => [key, stripKeyCodecArtifacts(value)]));
229
+ };
230
+ /** The canonical profile fields a spawning parent actually sets on a child, in the order a reader
231
+ * needs them, each with the description published alongside it. Everything else stays legal to
232
+ * pass see {@link deriveSpawnProfileArg}.
233
+ *
234
+ * Why these eleven, with the measured numbers (serialized JSON bytes, Interface 0.40). The
235
+ * canonical `properties` map is 10629 bytes, against 2424 bytes for every argument of every other
236
+ * coordination tool COMBINED publishing it whole makes one parameter four times the rest of the
237
+ * surface a driver re-reads each turn. Of the seven omitted fields (tags, connections, subagents,
238
+ * hooks, modes, confidential, extensions) none is something a parent hands a worker; `permissions`
239
+ * (321 bytes) IS, so it is published a child that must not touch the network or the filesystem
240
+ * is fenced there and nowhere else. `mcp` (2663 bytes) and `resources` (3122 bytes) are the two a
241
+ * parent is least likely to author inline and were together 85% of the published cost, so they
242
+ * carry a brief shape plus a description naming the full form instead of the canonical sub-tree. */
243
+ const spawnProfileFields = [
244
+ {
245
+ name: "name",
246
+ description: "Short identifier for this child, e.g. \"researcher\" or \"patch-writer\". A child normally sets it to its role; it shows up in traces and worker labels."
247
+ },
248
+ {
249
+ name: "description",
250
+ description: "One line saying what this child is for. Read by humans and by a parent listing its workers."
251
+ },
252
+ {
253
+ name: "version",
254
+ description: "Optional version string for this profile, so two revisions of the same role are distinguishable in evidence. A child usually omits it."
255
+ },
256
+ {
257
+ name: "harness",
258
+ description: "Which coding backend runs the child. Omit to inherit the run default; set it only when this child needs a specific backend (e.g. a long refactor on \"claude-code\")."
259
+ },
260
+ {
261
+ name: "model",
262
+ description: "Model routing: `default` model id, optional `small` for cheap sub-calls, `provider`, and `reasoningEffort`. A child typically sets `default` and `reasoningEffort` — raise effort for a hard reasoning task, lower it for bulk mechanical work."
263
+ },
264
+ {
265
+ name: "prompt",
266
+ description: "The child's standing instructions: `systemPrompt` (who it is and how it works) and `instructions` (durable rules). This is the role; the separate `task` argument carries what to do right now. A child with no systemPrompt has no role — always set one."
267
+ },
268
+ {
269
+ name: "tools",
270
+ description: "Per-tool on/off map keyed by tool name, e.g. `{ \"bash\": true, \"webfetch\": false }`. Omit to inherit the backend's default tool set; set it to narrow a child to the tools its task needs. Keys are plain tool names."
271
+ },
272
+ {
273
+ name: "permissions",
274
+ description: "Per-tool permission decisions: each tool name maps to \"allow\", \"deny\", or \"ask\", or to a nested map of finer-grained rules. This is where a parent fences a child it does not fully trust — a read-only child denies write and network tools here, not in `tools`."
275
+ },
276
+ {
277
+ name: "mcp",
278
+ description: "MCP tool servers to mount for the child, keyed by server name. Brief form: each value is either `{ transport: \"stdio\", command, args?, env?, cwd? }` or `{ transport: \"sse\" | \"http\", url, headers? }`. A child normally sets this only when it needs a tool server the task requires; the canonical AgentProfile schema carries the full form (secret-ref values for env/headers, per-server metadata) and governs validation.",
279
+ brief: {
280
+ type: "object",
281
+ additionalProperties: { type: "object" }
282
+ }
283
+ },
284
+ {
285
+ name: "resources",
286
+ description: "Files, tools, skills, and agents materialized into the child workspace before it starts. Brief form: `{ files?, tools?, skills?, agents? }`, each an array whose entries are `{ kind: \"inline\", name, content }` or `{ kind: \"github\", repository?, path, ref? }` (`files` entries wrap that as `{ path, resource, executable? }`). A child typically gets `files` for seed inputs and `skills` for a procedure it must follow; the canonical AgentProfile schema carries the full form and governs validation.",
287
+ brief: {
288
+ type: "object",
289
+ properties: {
290
+ files: {
291
+ type: "array",
292
+ items: { type: "object" }
293
+ },
294
+ tools: {
295
+ type: "array",
296
+ items: { type: "object" }
297
+ },
298
+ skills: {
299
+ type: "array",
300
+ items: { type: "object" }
301
+ },
302
+ agents: {
303
+ type: "array",
304
+ items: { type: "object" }
305
+ }
306
+ },
307
+ additionalProperties: true
308
+ }
309
+ },
310
+ {
311
+ name: "metadata",
312
+ description: "Free-form key/value bag carried with the profile (plain string keys). Use it for run bookkeeping a reader will want later; it does not change how the child executes."
313
+ }
314
+ ];
473
315
  /**
474
- * Durable side-log for coordination evidence the spawn journal does not own: questions, analyst
475
- * findings, answer decisions, authorized continuation receipts, delivery-attempt markers, and
476
- * delivery outcomes. A durable run
477
- * (`supervise({ runDir })`) appends them as they publish and loads them on resume, so a restarted
478
- * coordinator retains the exact evidence produced by prior processes.
316
+ * Build the published shape of `spawn_agent`'s `profile` argument from the canonical
317
+ * `agentProfileSchema` conversion's `properties` map, so it cannot drift from the profile the
318
+ * runtime materializes.
479
319
  *
480
- * Answer down-events also fold status on load: a question answered before the crash reloads as
481
- * `answered`, not as a re-blocking `open`. Settled events are skipped (the spawn journal is their
482
- * ledger). A receipt followed by an attempt but no outcome proves the process died in the delivery
483
- * window; that outcome remains unknown and no prior instruction is auto-delivered.
320
+ * DEGRADES, never throws. A canonical field that is absent renamed or removed upstream — is
321
+ * simply omitted from the published shape, and a canonical schema that is no longer an object
322
+ * publishes no properties at all. This function is reached from a statically-imported module, so a
323
+ * throw here bricks `import '@tangle-network/agent-runtime/kernel'` for every consumer over an
324
+ * upstream rename that costs them, at worst, one advisory field. The drift itself is still caught
325
+ * loudly — as a test assertion in `tests/kernel/coordination.test.ts`, in CI, where it is our
326
+ * problem rather than at a consumer's import, where it is theirs.
484
327
  *
485
- * JSONL, one fsynced record per event, keyed by `runId` several runs may share one log file
486
- * exactly as they share one spawn-journal file.
328
+ * Permissive on purpose: `additionalProperties: true` with no `required` list, so every canonical
329
+ * field this shape omits stays legal to pass. This tool layer performs no profile validation.
487
330
  *
488
- * @experimental
331
+ * @internal exported for the drift and degradation tests; not part of the package's public API.
489
332
  */
490
- /** Persist prior context plus exact continuation authorization, attempt, and result evidence.
491
- * Settlements have their own journal. */
492
- function persisted(event) {
493
- return event.type !== "settled";
494
- }
495
- /** FS-backed `CoordinationLog`: append-only JSONL, fsynced per record. */
496
- var FileCoordinationLog = class {
497
- path;
498
- appendTail = Promise.resolve();
499
- constructor(path) {
500
- this.path = path;
501
- }
502
- async append(runId, record, ownerId) {
503
- if (!persisted(record.event)) return;
504
- const append = this.appendTail.then(() => this.appendRecord(runId, record, ownerId));
505
- this.appendTail = append.catch(() => void 0);
506
- return append;
507
- }
508
- async appendRecord(runId, busRecord, ownerId) {
509
- const fs = await import("node:fs/promises");
510
- const path = await import("node:path");
511
- await fs.mkdir(path.dirname(this.path), { recursive: true });
512
- const record = {
513
- runId,
514
- ...ownerId !== void 0 ? { ownerId } : {},
515
- ...busRecord
516
- };
517
- const needsSeparator = await prepareJsonlAppend(this.path);
518
- const fh = await fs.open(this.path, "a");
519
- try {
520
- await writeAllBytes(fh, `${needsSeparator ? "\n" : ""}${JSON.stringify(record)}\n`);
521
- await fh.sync();
522
- } finally {
523
- await fh.close();
524
- }
333
+ function deriveSpawnProfileArg(canonicalProperties) {
334
+ const published = [];
335
+ for (const field of spawnProfileFields) {
336
+ const canonical = canonicalProperties?.[field.name];
337
+ if (canonical === void 0) continue;
338
+ const shape = field.brief ?? stripKeyCodecArtifacts(canonical);
339
+ published.push([field.name, {
340
+ ...shape,
341
+ description: field.description
342
+ }]);
525
343
  }
526
- async load(runId, ownerId) {
527
- const fs = await import("node:fs/promises");
528
- let text;
529
- try {
530
- text = await fs.readFile(this.path, "utf8");
531
- } catch (err) {
532
- if (isNoEntError(err)) return emptyPriorCoordination(ownerId);
533
- throw err;
534
- }
535
- const byId = /* @__PURE__ */ new Map();
536
- const findings = [];
537
- const continuations = [];
538
- const deliveryEvidence = [];
539
- const mail = [];
540
- const records = [];
541
- let legacySeq = 0;
542
- for (const stored of parseCommittedJsonLines(text, this.path)) {
543
- if (stored.runId !== runId) continue;
544
- if (ownerId !== void 0 && stored.ownerId !== ownerId) continue;
545
- const record = "seq" in stored ? {
546
- seq: stored.seq,
547
- at: stored.at,
548
- priority: stored.priority,
549
- event: stored.event
550
- } : {
551
- seq: legacySeq++,
552
- at: Date.parse(stored.at),
553
- priority: 0,
554
- event: stored.event
555
- };
556
- records.push(record);
557
- const ev = record.event;
558
- if (ev.type === "delivery-attempt" || ev.type === "steer" || ev.type === "answer") deliveryEvidence.push(ev);
559
- if (ev.type === "question") byId.set(ev.question.id, ev.question);
560
- else if (ev.type === "finding") findings.push(ev.finding);
561
- else if (ev.type === "answer") {
562
- const prior = byId.get(ev.questionId);
563
- if (prior && ev.down.delivered) byId.set(ev.questionId, {
564
- ...prior,
565
- status: "answered",
566
- decision: {
567
- kind: "answer",
568
- answer: ev.down.instruction,
569
- by: "prior-run"
570
- }
571
- });
572
- } else if (ev.type === "instruction") continuations.push(ev.instruction);
573
- else if (ev.type === "mail") mail.push(ev.mail);
574
- }
575
- return {
576
- ...ownerId !== void 0 ? { ownerId } : {},
577
- questions: [...byId.values()],
578
- findings,
579
- continuations,
580
- deliveryEvidence,
581
- mail,
582
- records
583
- };
584
- }
585
- };
586
- function emptyPriorCoordination(ownerId) {
587
344
  return {
588
- ...ownerId !== void 0 ? { ownerId } : {},
589
- questions: [],
590
- findings: [],
591
- continuations: [],
592
- deliveryEvidence: [],
593
- mail: [],
594
- records: []
345
+ type: "object",
346
+ description: "The child agent profile to run — this is the shape the DEFAULT worker seam accepts; a run wired with a custom makeWorkerAgent may accept a different one. The properties below are derived from the canonical AgentProfile schema and reduced to what a spawning parent sets (`mcp` and `resources` in a brief form); every other canonical field — tags, connections, subagents, hooks, modes, confidential, extensions — may still be passed. This tool does not validate the profile: the canonical AgentProfile schema governs validation downstream.",
347
+ properties: Object.fromEntries(published),
348
+ additionalProperties: true
595
349
  };
596
350
  }
597
- function isNoEntError(err) {
598
- return typeof err === "object" && err !== null && "code" in err && err.code === "ENOENT";
599
- }
600
- //#endregion
601
- //#region src/runtime/supervise/run-context.ts
602
- /**
603
- *
604
- * `createInMemoryRunContext` the one-call bundle of the in-memory stores a
605
- * `createSupervisor().run(root, task, opts)` needs: a fresh `InMemorySpawnJournal`
606
- * (the event-sourced spawn log), a fresh `InMemoryResultBlobStore` (the
607
- * content-addressed `outRef` payload store the driver's `observe`/`finalize` reads
608
- * settled outputs through), and a fresh `createExecutorRegistry()` (the open
609
- * `AgentSpec → Executor` resolver).
610
- *
611
- * It exists to kill the boilerplate every offline/local supervised run repeats by
612
- * hand — three constructors threaded into `SupervisorOpts` — and to single-source the
613
- * ONE wiring invariant that is easy to get wrong: when the root is the recursive
614
- * `driverAgent` LLM-driver brain AND it may spawn DRIVER children (agents
615
- * driving agents), the registry MUST be wrapped with `withDriverExecutor` so a
616
- * `role: 'driver'` child resolves to the nested-scope executor — and that SAME blob
617
- * store MUST be the one passed to `driverAgent({ blobs })`, or the driver
618
- * reads from a different store than the scope writes to. Pass `{ withDriver: true }`
619
- * and reuse the returned `blobs` for both.
620
- *
621
- * The spread shape matches `SupervisorOpts` exactly, so the call site reads:
622
- * const run = createInMemoryRunContext()
623
- * await createSupervisor().run(root, task, { budget, runId, ...run })
351
+ spawnProfileFields.map((f) => f.name);
352
+ let spawnProfileArgCache;
353
+ /** The published `profile` shape, computed on FIRST tool-definition access and memoized — not at
354
+ * module load. The conversion walks the whole canonical profile tree, and the strip walk rebuilds
355
+ * it: 3.5ms on the first `createCoordinationTools`, 0.017ms on every later one (measured, Interface
356
+ * 0.40, node 24). `src/runtime/index.ts` imports this module statically, so paying that at import
357
+ * taxes every consumer of the kernel entrypoint — including the ones that never build a
358
+ * coordination toolbox. The memo keeps it at once per process for the ones that do.
624
359
  *
625
- * @experimental
626
- */
627
- /**
628
- * Build a fresh in-memory run context. Every call returns NEW stores (no shared global
629
- * state between runs), so two runs never cross-contaminate their journals/blobs.
630
- */
631
- function createInMemoryRunContext(opts = {}) {
632
- const base = createExecutorRegistry();
633
- return {
634
- journal: new InMemorySpawnJournal(),
635
- blobs: new InMemoryResultBlobStore(),
636
- executors: opts.withDriver ? withDriverExecutor(base) : base
637
- };
360
+ * Conversion choices. `io: 'input'` is the CALLER's view — pre-default, pre-transform — which is
361
+ * what a spawning parent may pass, not what the runtime ends up holding. `unrepresentable: 'any'`
362
+ * keeps the conversion total: the canonical schema contains transforms with no JSON Schema form,
363
+ * and zod's default is to throw on them, which would leave the tool with no published shape. */
364
+ function spawnProfileArg() {
365
+ if (!spawnProfileArgCache) spawnProfileArgCache = deepFreeze(deriveSpawnProfileArg(agentProfileSchema.toJSONSchema({
366
+ io: "input",
367
+ target: "draft-07",
368
+ unrepresentable: "any"
369
+ }).properties));
370
+ return spawnProfileArgCache;
638
371
  }
639
- /**
640
- * Build a DURABLE run context: the spawn journal and the result blobs are file-backed (fsynced
641
- * per append/write) under `dir`, and the context carries `resume: true` so spreading it into
642
- * `SupervisorOpts` makes the supervisor `loadTree`-first. A run that dies mid-flight therefore
643
- * resumes when it is re-run with the SAME `runId` and the SAME `dir`: the committed children come
644
- * back on `Scope.resume` (rehydrated by `replaySpawnTree`) instead of being re-executed.
645
- *
646
- * Layout: `${dir}/spawn-journal.jsonl` (one JSONL record per event), `${dir}/blobs/` (one
647
- * content-addressed JSON file per settled result), and `${dir}/coordination-log.jsonl`
648
- * (questions, findings, answer decisions, and authorized continuation receipts retained as
649
- * evidence). The directory is created on first write.
650
- *
651
- * Opt-in by construction — `createInMemoryRunContext()` is unchanged and stays the default, so no
652
- * existing consumer writes to disk or resumes unless it asks for this.
653
- */
654
- function createFileRunContext(dir, opts = {}) {
655
- const base = createExecutorRegistry();
656
- return {
657
- journal: new FileSpawnJournal(`${dir}/spawn-journal.jsonl`),
658
- blobs: new FileResultBlobStore(`${dir}/blobs`),
659
- executors: opts.withDriver ? withDriverExecutor(base) : base,
660
- resume: true,
661
- coordinationLog: new FileCoordinationLog(`${dir}/coordination-log.jsonl`)
372
+ /** Build the driver's MCP tools over a live scope. */
373
+ function createCoordinationTools(opts) {
374
+ const deliverable = opts.deliverable;
375
+ let stopped = false;
376
+ let reason;
377
+ let stopNotified = false;
378
+ let submitted;
379
+ let questionSeq = 0;
380
+ const ledger = [];
381
+ const questions = [...opts.priorQuestions ?? []];
382
+ const questionPolicy = opts.questionPolicy ?? "auto";
383
+ const notifyStop = () => {
384
+ if (stopNotified) return;
385
+ stopNotified = true;
386
+ opts.onStop?.(reason);
662
387
  };
663
- }
664
- //#endregion
665
- //#region src/runtime/supervise/detector-monitor.ts
666
- /**
667
- *
668
- * The ONLINE analyst: watch a `TraceSource` and fold each tool span through agent-eval's published
669
- * streaming detector kernel (`repeatedActionDetector`/`errorStreakDetector` the SAME kernel the
670
- * control loop folds), firing `onSignal` the moment a worker loops or error-storms. Substrate-
671
- * agnostic: it consumes spans from any source (owned router/bridge loop OR a sandbox box session),
672
- * never the raw tool seam. Detection logic + the failure taxonomy live in agent-eval; not reimplemented.
673
- *
674
- * @experimental
675
- */
676
- /** The default online panel for a tool-call pipe: a worker repeating the same call, or hammering
677
- * consecutive errors. (No-progress needs a domain progress-probe, so it is opt-in, not default.)
678
- *
679
- * Coverage note: `repeated-action` works for EVERY harness (it needs only tool name + args, which
680
- * every adapter provides). `error-streak` needs per-call status — opencode carries it inline
681
- * (`state.status`, VALIDATED live), but claude-code/codex tool-call parts do NOT (their errors live
682
- * in separate result blocks not yet decoded), so error-streak is silent for those until result-block
683
- * decoding is added + live-validated. It is in the panel because it is correct where status exists. */
684
- function defaultToolDetectors() {
685
- return [repeatedActionDetector({ maxRepeated: 3 }), errorStreakDetector({ maxErrors: 3 })];
686
- }
687
- /** Subscribe to a `TraceSource` and run the streaming detectors over its live spans. Returns an
688
- * unsubscribe. A defensive `argHash` failure (circular args) never throws out of the side-channel. */
689
- function watchTrace(source, opts = {}) {
690
- const detectors = opts.detectors ?? defaultToolDetectors();
691
- return source.onSpan((span) => {
692
- let fingerprint;
693
- try {
694
- fingerprint = `${span.toolName}|${argHash(span.args)}`;
695
- } catch {
696
- fingerprint = `${span.toolName}|<unhashable>`;
697
- }
698
- const signals = observeAll(detectors, {
699
- actionFingerprint: fingerprint,
700
- ...span.status ? { status: span.status } : {},
701
- label: span.toolName
388
+ const completedKeys = /* @__PURE__ */ new Set();
389
+ const keyByWorker = /* @__PURE__ */ new Map();
390
+ const profileNameByWorker = /* @__PURE__ */ new Map();
391
+ const liveHandles = /* @__PURE__ */ new Map();
392
+ let unkeyedAssignmentOrdinal = nextUnkeyedAssignmentOrdinal(opts.scope);
393
+ const preflightCounts = emptyPreflightCounts();
394
+ for (const [key, prior] of opts.scope.resume?.keys ?? []) if (prior.state === "completed") completedKeys.add(key);
395
+ const nodeForWorker = (id) => opts.scope.view.nodes.find((node) => node.id === id) ?? opts.scope.resume?.view.nodes.find((node) => node.id === id);
396
+ const projectSettled = (settled, resumed = false) => {
397
+ const node = nodeForWorker(settled.handle.id);
398
+ const assignmentId = settled.handle.assignmentId ?? node?.assignmentId;
399
+ const identity = settled.handle.identity ?? node?.identity;
400
+ const materialization = settled.handle.materialization ?? node?.materialization;
401
+ const executionBindings = settled.handle.executionBindings ?? node?.executionBindings;
402
+ const settledAt = settled.settledAt ?? node?.settledAt;
403
+ const trace = settled.trace ?? node?.trace ?? {
404
+ status: "unavailable",
405
+ reason: "legacy-settlement-without-trace-evidence"
406
+ };
407
+ const common = {
408
+ id: settled.handle.id,
409
+ ...assignmentId === void 0 ? {} : { assignmentId },
410
+ ...identity === void 0 ? {} : { identity },
411
+ ...materialization === void 0 ? {} : { materialization },
412
+ ...executionBindings === void 0 ? {} : { executionBindings },
413
+ ...settledAt === void 0 ? {} : { settledAt },
414
+ trace,
415
+ ...resumed ? { resumed: true } : {}
416
+ };
417
+ return deepFreezeDetached(settled.kind === "done" ? {
418
+ ...common,
419
+ status: "done",
420
+ spent: settled.spent,
421
+ ...settled.verdict?.score === void 0 ? {} : { score: settled.verdict.score },
422
+ ...settled.verdict?.valid === void 0 ? {} : { valid: settled.verdict.valid },
423
+ outRef: settled.outRef
424
+ } : {
425
+ ...common,
426
+ status: "down",
427
+ ...node?.spent === void 0 ? {} : { spent: node.spent },
428
+ reason: settled.reason
702
429
  });
703
- for (const s of signals) opts.onSignal?.(s, span);
704
- });
705
- }
706
- //#endregion
707
- //#region src/runtime/supervise/event-bus.ts
708
- /** Create the child→parent coordination bus: one typed pipe for settled outputs, questions, and analyst findings, with a priority-ordered pull queue and a pass-through subscribe lane.
709
- * @experimental In-process queue; durability is a transport swap that does not exist yet. */
710
- function createEventBus(now = Date.now) {
711
- const queue = [];
712
- const log = [];
713
- const subscribers = [];
714
- const byKind = {};
715
- const staged = /* @__PURE__ */ new WeakMap();
716
- let seq = 0;
717
- let published = 0;
718
- let pulled = 0;
719
- const matches = (r, kinds) => !kinds || kinds.includes(r.event.type);
720
- const bestIndex = (kinds) => {
721
- let best = -1;
722
- let bestPriority = Number.NEGATIVE_INFINITY;
723
- for (let i = 0; i < queue.length; i++) {
724
- const r = queue[i];
725
- if (!r || !matches(r, kinds)) continue;
726
- if (r.priority > bestPriority) {
727
- best = i;
728
- bestPriority = r.priority;
430
+ };
431
+ const resumedWorkers = [];
432
+ for (const s of opts.scope.resume?.settled ?? []) {
433
+ const worker = projectSettled(s, true);
434
+ resumedWorkers.push(worker);
435
+ ledger.push(worker);
436
+ }
437
+ const bus = createEventBus();
438
+ if (opts.onEvent) {
439
+ const cb = opts.onEvent;
440
+ bus.subscribe((rec) => cb(rec.event, rec));
441
+ }
442
+ const resumeEvents = opts.replaySettlements ? resumedWorkers.map((worker) => deepFreezeDetached({
443
+ type: "settled",
444
+ worker
445
+ })) : [];
446
+ let resumeEventIndex = 0;
447
+ let readyInFlight;
448
+ const ready = () => {
449
+ if (resumeEventIndex >= resumeEvents.length) return Promise.resolve();
450
+ if (readyInFlight) return readyInFlight;
451
+ readyInFlight = (async () => {
452
+ while (resumeEventIndex < resumeEvents.length) {
453
+ const event = resumeEvents[resumeEventIndex];
454
+ if (!event) break;
455
+ await bus.publish(event);
456
+ resumeEventIndex += 1;
729
457
  }
730
- }
731
- return best;
458
+ })().finally(() => {
459
+ readyInFlight = void 0;
460
+ });
461
+ return readyInFlight;
732
462
  };
733
- return {
734
- async publish(event, opts) {
735
- const record = staged.get(event) ?? {
736
- seq: seq++,
737
- at: now(),
738
- priority: opts?.priority ?? 0,
739
- event
740
- };
741
- staged.set(event, record);
742
- for (const handler of subscribers) await handler(record);
743
- staged.delete(event);
744
- if (opts?.queue !== false) queue.push(record);
745
- log.push(record);
746
- published += 1;
747
- byKind[event.type] = (byKind[event.type] ?? 0) + 1;
748
- return record;
749
- },
750
- pull(kinds) {
751
- const i = bestIndex(kinds);
752
- if (i < 0) return void 0;
753
- pulled++;
754
- return queue.splice(i, 1)[0]?.event;
755
- },
756
- subscribe(handler) {
757
- subscribers.push(handler);
758
- return () => {
759
- const i = subscribers.indexOf(handler);
760
- if (i >= 0) subscribers.splice(i, 1);
761
- };
762
- },
763
- pending(kinds) {
764
- return kinds ? queue.filter((r) => matches(r, kinds)).length : queue.length;
765
- },
766
- history() {
767
- return log;
768
- },
769
- stats() {
770
- return {
771
- published,
772
- pulled,
773
- byKind: { ...byKind }
774
- };
775
- }
463
+ const urgencyPriority = (u) => u === "blocks-run" ? 20 : u === "blocks-step" ? 10 : 0;
464
+ const str = (v, field) => {
465
+ if (typeof v !== "string" || v.length === 0) throw new Error(`coordination tools: "${field}" must be a non-empty string`);
466
+ return v;
776
467
  };
777
- }
778
- //#endregion
779
- //#region src/mcp/tools/coordination.ts
780
- /**
781
- *
782
- * MCP binding for a live `Scope`. A sandbox driver gets the same small verbs
783
- * the in-process driver has: spawn, observe, await, steer, ask/answer, analyze,
784
- * and stop. Settled outputs remain Scope artifacts; product code can project
785
- * them into any UI/report envelope it needs.
786
- *
787
- * @experimental
788
- */
789
- /** Producer-side cleanliness for the `finding` event. The findings payload is arbitrary analyst
790
- * output, the digest a subscriber computes (RFC 8785) throws on ANY `undefined` value — nested
791
- * included — and a throwing subscriber leaves the event invisible to EVERY subscriber. The
792
- * producer, not the digest, owns keeping the event canonical: an `undefined` payload is stripped
793
- * to key-absence, everything else is JSON round-tripped (nested `undefined` object values drop,
794
- * `undefined` array slots become `null`), and a payload JSON cannot represent at all (cycle,
795
- * BigInt, bare function) becomes a record OF that fact — degraded findings beat a vanished
796
- * event. */
797
- function canonicalFindingEvent(finding) {
798
- if (finding.findings === void 0) {
799
- const { findings: _absent, ...present } = finding;
800
- return present;
801
- }
802
- try {
803
- return {
804
- ...finding,
805
- findings: JSON.parse(JSON.stringify(finding.findings))
468
+ const obj = (raw) => {
469
+ if (!raw || typeof raw !== "object") throw new Error("coordination tools: arguments must be an object");
470
+ return raw;
471
+ };
472
+ const mergeBudget = (base, raw) => {
473
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) throw new Error("coordination tools: \"budget\" must be an object");
474
+ const o = raw;
475
+ const field = (name) => {
476
+ const v = o[name];
477
+ if (v === void 0) return void 0;
478
+ if (typeof v !== "number" || !Number.isFinite(v)) throw new Error(`coordination tools: "budget.${name}" must be a finite number`);
479
+ return v;
806
480
  };
807
- } catch (error) {
808
- return {
809
- ...finding,
810
- findings: { nonCanonicalFindings: error instanceof Error ? error.message : String(error) }
811
- };
812
- }
813
- }
814
- /** Normalize the two spellings of an analyst-on-settle entry to the route form. */
815
- function normalizeAnalyzeOnSettle(entry) {
816
- return typeof entry === "string" ? { kind: entry } : entry;
817
- }
818
- /** Default ceiling for a single `await_event` block (ms). Chosen well under any reasonable remote
819
- * MCP client request timeout so the call returns a `pending` liveness snapshot instead of erroring;
820
- * the supervisor re-polls until the worker settles. */
821
- const DEFAULT_AWAIT_EVENT_TIMEOUT_MS = 15e3;
822
- /** The reserved coordination verb names — the complete set `createCoordinationTools` can emit
823
- * (the analyst pair is conditional but still reserved). A driver's extra WORK tools must not
824
- * collide with any of these, or it could no longer coordinate; callers validate eagerly against
825
- * this set so the conflict fails loud at construction, not buried in a swallowed `act()` throw. */
826
- const coordinationVerbNames = [
827
- "spawn_agent",
828
- "observe_agent",
829
- "steer_agent",
830
- "await_event",
831
- "list_questions",
832
- "answer_question",
833
- "ask_parent",
834
- "submit_result",
835
- "stop",
836
- "list_analysts",
837
- "run_analyst"
838
- ];
839
- const idArg = {
840
- type: "string",
841
- description: "The workerId returned by spawn_agent."
842
- };
843
- /**
844
- * Strip zod's object-KEY CODEC artifact from a derived JSON Schema, at every depth.
845
- *
846
- * `z.record(z.string(), …)` converts to `propertyNames: { type: 'string', pattern:
847
- * '^u(?:[0-9a-f]{4})*$' }` — a marker of the lossy key round-trip, not a constraint anything
848
- * enforces: `agentProfileSchema.safeParse({ name: 'r', tools: { bash: true, Read: false } })`
849
- * succeeds with those plain keys. Published verbatim it reads to a model as "every key must be a
850
- * run of hex quads", and the model either emits hex-encoded garbage keys or drops the field —
851
- * on `tools`, `permissions`, `metadata`, `mcp`, `mcp.*.env`, and `model.metadata`, which is most
852
- * of what a parent actually configures.
853
- *
854
- * Every `propertyNames` in the canonical profile schema is this artifact (25 of 25 at Interface
855
- * 0.40), and the canonical schema constrains no key by pattern, so dropping the keyword outright
856
- * loses nothing real and cannot be defeated by zod changing the encoding's exact regex.
857
- *
858
- * Rebuilds rather than mutates: the canonical conversion output must stay untouched for callers
859
- * that compare against it.
860
- */
861
- const stripKeyCodecArtifacts = (node) => {
862
- if (Array.isArray(node)) return node.map(stripKeyCodecArtifacts);
863
- if (!node || typeof node !== "object") return node;
864
- return Object.fromEntries(Object.entries(node).filter(([key]) => key !== "propertyNames").map(([key, value]) => [key, stripKeyCodecArtifacts(value)]));
865
- };
866
- /** The canonical profile fields a spawning parent actually sets on a child, in the order a reader
867
- * needs them, each with the description published alongside it. Everything else stays legal to
868
- * pass — see {@link deriveSpawnProfileArg}.
869
- *
870
- * Why these eleven, with the measured numbers (serialized JSON bytes, Interface 0.40). The
871
- * canonical `properties` map is 10629 bytes, against 2424 bytes for every argument of every other
872
- * coordination tool COMBINED — publishing it whole makes one parameter four times the rest of the
873
- * surface a driver re-reads each turn. Of the seven omitted fields (tags, connections, subagents,
874
- * hooks, modes, confidential, extensions) none is something a parent hands a worker; `permissions`
875
- * (321 bytes) IS, so it is published — a child that must not touch the network or the filesystem
876
- * is fenced there and nowhere else. `mcp` (2663 bytes) and `resources` (3122 bytes) are the two a
877
- * parent is least likely to author inline and were together 85% of the published cost, so they
878
- * carry a brief shape plus a description naming the full form instead of the canonical sub-tree. */
879
- const spawnProfileFields = [
880
- {
881
- name: "name",
882
- description: "Short identifier for this child, e.g. \"researcher\" or \"patch-writer\". A child normally sets it to its role; it shows up in traces and worker labels."
883
- },
884
- {
885
- name: "description",
886
- description: "One line saying what this child is for. Read by humans and by a parent listing its workers."
887
- },
888
- {
889
- name: "version",
890
- description: "Optional version string for this profile, so two revisions of the same role are distinguishable in evidence. A child usually omits it."
891
- },
892
- {
893
- name: "harness",
894
- description: "Which coding backend runs the child. Omit to inherit the run default; set it only when this child needs a specific backend (e.g. a long refactor on \"claude-code\")."
895
- },
896
- {
897
- name: "model",
898
- description: "Model routing: `default` model id, optional `small` for cheap sub-calls, `provider`, and `reasoningEffort`. A child typically sets `default` and `reasoningEffort` — raise effort for a hard reasoning task, lower it for bulk mechanical work."
899
- },
900
- {
901
- name: "prompt",
902
- description: "The child's standing instructions: `systemPrompt` (who it is and how it works) and `instructions` (durable rules). This is the role; the separate `task` argument carries what to do right now. A child with no systemPrompt has no role — always set one."
903
- },
904
- {
905
- name: "tools",
906
- description: "Per-tool on/off map keyed by tool name, e.g. `{ \"bash\": true, \"webfetch\": false }`. Omit to inherit the backend's default tool set; set it to narrow a child to the tools its task needs. Keys are plain tool names."
907
- },
908
- {
909
- name: "permissions",
910
- description: "Per-tool permission decisions: each tool name maps to \"allow\", \"deny\", or \"ask\", or to a nested map of finer-grained rules. This is where a parent fences a child it does not fully trust — a read-only child denies write and network tools here, not in `tools`."
911
- },
912
- {
913
- name: "mcp",
914
- description: "MCP tool servers to mount for the child, keyed by server name. Brief form: each value is either `{ transport: \"stdio\", command, args?, env?, cwd? }` or `{ transport: \"sse\" | \"http\", url, headers? }`. A child normally sets this only when it needs a tool server the task requires; the canonical AgentProfile schema carries the full form (secret-ref values for env/headers, per-server metadata) and governs validation.",
915
- brief: {
916
- type: "object",
917
- additionalProperties: { type: "object" }
918
- }
919
- },
920
- {
921
- name: "resources",
922
- description: "Files, tools, skills, and agents materialized into the child workspace before it starts. Brief form: `{ files?, tools?, skills?, agents? }`, each an array whose entries are `{ kind: \"inline\", name, content }` or `{ kind: \"github\", repository?, path, ref? }` (`files` entries wrap that as `{ path, resource, executable? }`). A child typically gets `files` for seed inputs and `skills` for a procedure it must follow; the canonical AgentProfile schema carries the full form and governs validation.",
923
- brief: {
924
- type: "object",
925
- properties: {
926
- files: {
927
- type: "array",
928
- items: { type: "object" }
929
- },
930
- tools: {
931
- type: "array",
932
- items: { type: "object" }
933
- },
934
- skills: {
935
- type: "array",
936
- items: { type: "object" }
937
- },
938
- agents: {
939
- type: "array",
940
- items: { type: "object" }
941
- }
942
- },
943
- additionalProperties: true
944
- }
945
- },
946
- {
947
- name: "metadata",
948
- description: "Free-form key/value bag carried with the profile (plain string keys). Use it for run bookkeeping a reader will want later; it does not change how the child executes."
949
- }
950
- ];
951
- /**
952
- * Build the published shape of `spawn_agent`'s `profile` argument from the canonical
953
- * `agentProfileSchema` conversion's `properties` map, so it cannot drift from the profile the
954
- * runtime materializes.
955
- *
956
- * DEGRADES, never throws. A canonical field that is absent — renamed or removed upstream — is
957
- * simply omitted from the published shape, and a canonical schema that is no longer an object
958
- * publishes no properties at all. This function is reached from a statically-imported module, so a
959
- * throw here bricks `import '@tangle-network/agent-runtime/kernel'` for every consumer over an
960
- * upstream rename that costs them, at worst, one advisory field. The drift itself is still caught
961
- * loudly — as a test assertion in `tests/kernel/coordination.test.ts`, in CI, where it is our
962
- * problem rather than at a consumer's import, where it is theirs.
963
- *
964
- * Permissive on purpose: `additionalProperties: true` with no `required` list, so every canonical
965
- * field this shape omits stays legal to pass. This tool layer performs no profile validation.
966
- *
967
- * @internal exported for the drift and degradation tests; not part of the package's public API.
968
- */
969
- function deriveSpawnProfileArg(canonicalProperties) {
970
- const published = [];
971
- for (const field of spawnProfileFields) {
972
- const canonical = canonicalProperties?.[field.name];
973
- if (canonical === void 0) continue;
974
- const shape = field.brief ?? stripKeyCodecArtifacts(canonical);
975
- published.push([field.name, {
976
- ...shape,
977
- description: field.description
978
- }]);
979
- }
980
- return {
981
- type: "object",
982
- description: "The child agent profile to run — this is the shape the DEFAULT worker seam accepts; a run wired with a custom makeWorkerAgent may accept a different one. The properties below are derived from the canonical AgentProfile schema and reduced to what a spawning parent sets (`mcp` and `resources` in a brief form); every other canonical field — tags, connections, subagents, hooks, modes, confidential, extensions — may still be passed. This tool does not validate the profile: the canonical AgentProfile schema governs validation downstream.",
983
- properties: Object.fromEntries(published),
984
- additionalProperties: true
985
- };
986
- }
987
- spawnProfileFields.map((f) => f.name);
988
- let spawnProfileArgCache;
989
- /** The published `profile` shape, computed on FIRST tool-definition access and memoized — not at
990
- * module load. The conversion walks the whole canonical profile tree, and the strip walk rebuilds
991
- * it: 3.5ms on the first `createCoordinationTools`, 0.017ms on every later one (measured, Interface
992
- * 0.40, node 24). `src/runtime/index.ts` imports this module statically, so paying that at import
993
- * taxes every consumer of the kernel entrypoint — including the ones that never build a
994
- * coordination toolbox. The memo keeps it at once per process for the ones that do.
995
- *
996
- * Conversion choices. `io: 'input'` is the CALLER's view — pre-default, pre-transform — which is
997
- * what a spawning parent may pass, not what the runtime ends up holding. `unrepresentable: 'any'`
998
- * keeps the conversion total: the canonical schema contains transforms with no JSON Schema form,
999
- * and zod's default is to throw on them, which would leave the tool with no published shape. */
1000
- function spawnProfileArg() {
1001
- if (!spawnProfileArgCache) spawnProfileArgCache = deepFreeze(deriveSpawnProfileArg(agentProfileSchema.toJSONSchema({
1002
- io: "input",
1003
- target: "draft-07",
1004
- unrepresentable: "any"
1005
- }).properties));
1006
- return spawnProfileArgCache;
1007
- }
1008
- /** Build the driver's MCP tools over a live scope. */
1009
- function createCoordinationTools(opts) {
1010
- const deliverable = opts.deliverable;
1011
- let stopped = false;
1012
- let reason;
1013
- let stopNotified = false;
1014
- let submitted;
1015
- let questionSeq = 0;
1016
- const ledger = [];
1017
- const questions = [...opts.priorQuestions ?? []];
1018
- const questionPolicy = opts.questionPolicy ?? "auto";
1019
- const notifyStop = () => {
1020
- if (stopNotified) return;
1021
- stopNotified = true;
1022
- opts.onStop?.(reason);
1023
- };
1024
- const completedKeys = /* @__PURE__ */ new Set();
1025
- const keyByWorker = /* @__PURE__ */ new Map();
1026
- const profileNameByWorker = /* @__PURE__ */ new Map();
1027
- const liveHandles = /* @__PURE__ */ new Map();
1028
- let unkeyedAssignmentOrdinal = nextUnkeyedAssignmentOrdinal(opts.scope);
1029
- for (const [key, prior] of opts.scope.resume?.keys ?? []) if (prior.state === "completed") completedKeys.add(key);
1030
- const nodeForWorker = (id) => opts.scope.view.nodes.find((node) => node.id === id) ?? opts.scope.resume?.view.nodes.find((node) => node.id === id);
1031
- const projectSettled = (settled, resumed = false) => {
1032
- const node = nodeForWorker(settled.handle.id);
1033
- const assignmentId = settled.handle.assignmentId ?? node?.assignmentId;
1034
- const identity = settled.handle.identity ?? node?.identity;
1035
- const materialization = settled.handle.materialization ?? node?.materialization;
1036
- const executionBindings = settled.handle.executionBindings ?? node?.executionBindings;
1037
- const settledAt = settled.settledAt ?? node?.settledAt;
1038
- const trace = settled.trace ?? node?.trace ?? {
1039
- status: "unavailable",
1040
- reason: "legacy-settlement-without-trace-evidence"
1041
- };
1042
- const common = {
1043
- id: settled.handle.id,
1044
- ...assignmentId === void 0 ? {} : { assignmentId },
1045
- ...identity === void 0 ? {} : { identity },
1046
- ...materialization === void 0 ? {} : { materialization },
1047
- ...executionBindings === void 0 ? {} : { executionBindings },
1048
- ...settledAt === void 0 ? {} : { settledAt },
1049
- trace,
1050
- ...resumed ? { resumed: true } : {}
1051
- };
1052
- return deepFreezeDetached(settled.kind === "done" ? {
1053
- ...common,
1054
- status: "done",
1055
- spent: settled.spent,
1056
- ...settled.verdict?.score === void 0 ? {} : { score: settled.verdict.score },
1057
- ...settled.verdict?.valid === void 0 ? {} : { valid: settled.verdict.valid },
1058
- outRef: settled.outRef
1059
- } : {
1060
- ...common,
1061
- status: "down",
1062
- ...node?.spent === void 0 ? {} : { spent: node.spent },
1063
- reason: settled.reason
1064
- });
1065
- };
1066
- const resumedWorkers = [];
1067
- for (const s of opts.scope.resume?.settled ?? []) {
1068
- const worker = projectSettled(s, true);
1069
- resumedWorkers.push(worker);
1070
- ledger.push(worker);
1071
- }
1072
- const bus = createEventBus();
1073
- if (opts.onEvent) {
1074
- const cb = opts.onEvent;
1075
- bus.subscribe((rec) => cb(rec.event, rec));
1076
- }
1077
- const resumeEvents = opts.replaySettlements ? resumedWorkers.map((worker) => deepFreezeDetached({
1078
- type: "settled",
1079
- worker
1080
- })) : [];
1081
- let resumeEventIndex = 0;
1082
- let readyInFlight;
1083
- const ready = () => {
1084
- if (resumeEventIndex >= resumeEvents.length) return Promise.resolve();
1085
- if (readyInFlight) return readyInFlight;
1086
- readyInFlight = (async () => {
1087
- while (resumeEventIndex < resumeEvents.length) {
1088
- const event = resumeEvents[resumeEventIndex];
1089
- if (!event) break;
1090
- await bus.publish(event);
1091
- resumeEventIndex += 1;
1092
- }
1093
- })().finally(() => {
1094
- readyInFlight = void 0;
1095
- });
1096
- return readyInFlight;
1097
- };
1098
- const urgencyPriority = (u) => u === "blocks-run" ? 20 : u === "blocks-step" ? 10 : 0;
1099
- const str = (v, field) => {
1100
- if (typeof v !== "string" || v.length === 0) throw new Error(`coordination tools: "${field}" must be a non-empty string`);
1101
- return v;
1102
- };
1103
- const obj = (raw) => {
1104
- if (!raw || typeof raw !== "object") throw new Error("coordination tools: arguments must be an object");
1105
- return raw;
1106
- };
1107
- const mergeBudget = (base, raw) => {
1108
- if (!raw || typeof raw !== "object" || Array.isArray(raw)) throw new Error("coordination tools: \"budget\" must be an object");
1109
- const o = raw;
1110
- const field = (name) => {
1111
- const v = o[name];
1112
- if (v === void 0) return void 0;
1113
- if (typeof v !== "number" || !Number.isFinite(v)) throw new Error(`coordination tools: "budget.${name}" must be a finite number`);
1114
- return v;
1115
- };
1116
- const maxIterations = field("maxIterations");
1117
- const maxTokens = field("maxTokens");
1118
- const maxUsd = field("maxUsd");
1119
- const deadlineMs = field("deadlineMs");
1120
- const merged = {
1121
- maxIterations: maxIterations ?? base.maxIterations,
1122
- maxTokens: maxTokens ?? base.maxTokens,
1123
- ...(maxUsd ?? base.maxUsd) === void 0 ? {} : { maxUsd: maxUsd ?? base.maxUsd },
1124
- ...(deadlineMs ?? base.deadlineMs) === void 0 ? {} : { deadlineMs: deadlineMs ?? base.deadlineMs }
481
+ const maxIterations = field("maxIterations");
482
+ const maxTokens = field("maxTokens");
483
+ const maxUsd = field("maxUsd");
484
+ const deadlineMs = field("deadlineMs");
485
+ const merged = {
486
+ maxIterations: maxIterations ?? base.maxIterations,
487
+ maxTokens: maxTokens ?? base.maxTokens,
488
+ ...(maxUsd ?? base.maxUsd) === void 0 ? {} : { maxUsd: maxUsd ?? base.maxUsd },
489
+ ...(deadlineMs ?? base.deadlineMs) === void 0 ? {} : { deadlineMs: deadlineMs ?? base.deadlineMs }
1125
490
  };
1126
491
  assertValidBudget(merged, "coordination tools: budget");
1127
492
  return merged;
@@ -1809,7 +1174,7 @@ function createCoordinationTools(opts) {
1809
1174
  },
1810
1175
  required: ["profile", "task"]
1811
1176
  },
1812
- handler: (raw) => {
1177
+ handler: async (raw) => {
1813
1178
  const a = obj(raw);
1814
1179
  const key = a.key === void 0 ? void 0 : str(a.key, "key");
1815
1180
  if (!(key !== void 0 && completedKeys.has(key)) && !usesTreeWideLimit() && maxLiveWorkers !== void 0 && maxLiveWorkers > 0 && liveWorkerCount() >= maxLiveWorkers) return Promise.resolve({
@@ -1835,6 +1200,23 @@ function createCoordinationTools(opts) {
1835
1200
  });
1836
1201
  const task = deepFreezeDetached(a.task);
1837
1202
  const label = typeof a.label === "string" ? a.label : "worker";
1203
+ if (opts.preflightSpawn) {
1204
+ const refusal = await opts.preflightSpawn(profile, {
1205
+ label,
1206
+ ...key !== void 0 ? { key } : {},
1207
+ task
1208
+ });
1209
+ if (refusal) {
1210
+ preflightCounts[refusal.cause] += 1;
1211
+ return {
1212
+ error: "preflight-refused",
1213
+ cause: refusal.cause,
1214
+ detail: refusal.detail,
1215
+ live: liveWorkerCount(),
1216
+ freeSlots: freeWorkerSlots()
1217
+ };
1218
+ }
1219
+ }
1838
1220
  const budget = Object.freeze(a.budget === void 0 ? opts.perWorker : mergeBudget(opts.perWorker, a.budget));
1839
1221
  const assignmentId = key !== void 0 ? `key:${key}` : `ordinal:${unkeyedAssignmentOrdinal++}`;
1840
1222
  const peerMailUrl = peerMail?.mintCapability(assignmentId);
@@ -2262,63 +1644,711 @@ function createCoordinationTools(opts) {
2262
1644
  if (handle === void 0) return void 0;
2263
1645
  handle.abort(reason);
2264
1646
  return {
2265
- id: target.id,
2266
- label: target.label
1647
+ id: target.id,
1648
+ label: target.label
1649
+ };
1650
+ };
1651
+ return {
1652
+ tools,
1653
+ ready,
1654
+ history: () => bus.history(),
1655
+ raiseFinding: (finding) => bus.publish({
1656
+ type: "finding",
1657
+ finding: canonicalFindingEvent(finding)
1658
+ }).then(() => void 0),
1659
+ stats: () => opts.preflightSpawn === void 0 ? bus.stats() : {
1660
+ ...bus.stats(),
1661
+ preflight: { ...preflightCounts }
1662
+ },
1663
+ isStopped: () => stopped,
1664
+ stopReason: () => reason,
1665
+ submittedResult: () => submitted,
1666
+ settled: () => ledger,
1667
+ questions: () => questions,
1668
+ drainResolved,
1669
+ abortWorker,
1670
+ ...peerMail ? { peerMail } : {}
1671
+ };
1672
+ }
1673
+ function nextUnkeyedAssignmentOrdinal(scope) {
1674
+ let next = 0;
1675
+ const views = [scope.resume?.view, scope.view];
1676
+ for (const view of views) {
1677
+ if (view === void 0) continue;
1678
+ for (const node of view.nodes) {
1679
+ const match = /^ordinal:(\d+)$/.exec(node.assignmentId ?? "");
1680
+ if (match === null) continue;
1681
+ const ordinal = Number(match[1]);
1682
+ if (!Number.isSafeInteger(ordinal)) throw new Error(`coordination: durable assignment id '${node.assignmentId}' exceeds the safe ordinal range`);
1683
+ next = Math.max(next, ordinal + 1);
1684
+ }
1685
+ }
1686
+ if (!Number.isSafeInteger(next)) throw new Error("coordination: durable assignment ordinal space is exhausted");
1687
+ return next;
1688
+ }
1689
+ function deepFreezeDetached(value) {
1690
+ return deepFreeze(structuredClone(value));
1691
+ }
1692
+ /** Stringify a findings payload for a routed delivery; never throws (a cyclic payload degrades to
1693
+ * its String form rather than killing the settle path). */
1694
+ function safeJsonText(value) {
1695
+ if (typeof value === "string") return value;
1696
+ try {
1697
+ return JSON.stringify(value) ?? String(value);
1698
+ } catch {
1699
+ return String(value);
1700
+ }
1701
+ }
1702
+ function deepFreeze(value, seen = /* @__PURE__ */ new Set()) {
1703
+ if (value === null || typeof value !== "object" || seen.has(value)) return value;
1704
+ seen.add(value);
1705
+ for (const child of Object.values(value)) deepFreeze(child, seen);
1706
+ return Object.freeze(value);
1707
+ }
1708
+ //#endregion
1709
+ //#region src/runtime/supervise/completion-gate.ts
1710
+ /**
1711
+ *
1712
+ * The completion-oracle: **settled ⟺ DELIVERED.**
1713
+ *
1714
+ * Foreman's one hard lesson (0/18 self-improvement deliverables) — "done" must mean a check
1715
+ * PASSED, not the agent's say-so. `gateOnDeliverable` wraps an `Executor` so its settlement
1716
+ * is `valid` ONLY when the deliverable check passes. The child still RUNS and settles (its
1717
+ * spend is conserved into the pool either way), but a child that ran WITHOUT delivering
1718
+ * settles `valid:false` — so a keep-best driver never counts it as done, and a gate never
1719
+ * inflates with self-judged wins.
1720
+ *
1721
+ * Dual-purpose by construction:
1722
+ * - product: the agent fleet only advances on real, checked deliverables.
1723
+ * - proof: the gate's `valid` is the honest settle — equal-k comparisons can't be gamed by an
1724
+ * arm that "ran" without producing the artifact.
1725
+ *
1726
+ * The check is a DEPLOYABLE oracle — a test command, a state verifier, the commit0 judge —
1727
+ * read off the child's output, never the model judging itself. A throwing check is
1728
+ * fail-closed (not delivered), never a crash.
1729
+ *
1730
+ * @experimental
1731
+ */
1732
+ /**
1733
+ * Wrap an `Executor` so its settlement `valid` reflects the deliverable check, not the
1734
+ * inner verdict. Handles both `execute` shapes (one-shot `Promise<ExecutorResult>` and
1735
+ * streaming `AsyncIterable<UsageEvent>` + `resultArtifact()`); the check runs once the inner
1736
+ * executor has produced its output. The inner `score` is preserved; only `valid` is gated.
1737
+ */
1738
+ function gateOnDeliverable(inner, deliverable) {
1739
+ let gated;
1740
+ const check = async (out, baseScore) => {
1741
+ let delivered;
1742
+ try {
1743
+ delivered = await deliverable.check(out) === true;
1744
+ } catch {
1745
+ delivered = false;
1746
+ }
1747
+ return {
1748
+ valid: delivered,
1749
+ score: baseScore ?? (delivered ? 1 : 0)
1750
+ };
1751
+ };
1752
+ /**
1753
+ * Ask the delivery question once, from whatever the inner executor managed to produce.
1754
+ *
1755
+ * Fail-closed on the artifact being unavailable: an executor that never produced one delivered
1756
+ * nothing, and leaving `gated` unset keeps the existing invalid-by-default reading.
1757
+ */
1758
+ const settleVerdict = async () => {
1759
+ let art;
1760
+ try {
1761
+ art = inner.resultArtifact();
1762
+ } catch {
1763
+ return;
1764
+ }
1765
+ gated = await check(art.out, art.verdict?.score);
1766
+ };
1767
+ return inheritRuntimeOwnedExecutorAttestation(inner, {
1768
+ runtime: inner.runtime,
1769
+ ...inner.budgetExempt !== void 0 ? { budgetExempt: inner.budgetExempt } : {},
1770
+ ...inner.deliver ? { deliver: (m) => inner.deliver?.(m) } : {},
1771
+ ...inner.progress ? { progress: () => inner.progress?.() } : {},
1772
+ ...inner.traceSource ? { traceSource: () => inner.traceSource?.() } : {},
1773
+ ...inner.metered ? { metered: () => inner.metered?.() } : {},
1774
+ execute(task, signal) {
1775
+ const r = inner.execute(task, signal);
1776
+ if (isAsyncIterable$1(r)) return (async function* () {
1777
+ try {
1778
+ for await (const ev of r) yield ev;
1779
+ } finally {
1780
+ await settleVerdict();
1781
+ }
1782
+ })();
1783
+ return (async () => {
1784
+ let res;
1785
+ try {
1786
+ res = await r;
1787
+ } catch (error) {
1788
+ await settleVerdict();
1789
+ throw error;
1790
+ }
1791
+ gated = await check(res.out, res.verdict?.score);
1792
+ return {
1793
+ ...res,
1794
+ verdict: gated
1795
+ };
1796
+ })();
1797
+ },
1798
+ teardown: (grace) => inner.teardown(grace),
1799
+ resultArtifact() {
1800
+ const art = inner.resultArtifact();
1801
+ return {
1802
+ ...art,
1803
+ verdict: gated ?? art.verdict
1804
+ };
1805
+ }
1806
+ });
1807
+ }
1808
+ /**
1809
+ * Transform a Runtime executor's terminal artifact without losing its private
1810
+ * profile-materialization attestation or altering its measured spend. This is
1811
+ * the composition point for deterministic post-processing and grading; callers
1812
+ * must not rebuild an Executor around a model transport merely to change `out`.
1813
+ */
1814
+ function mapExecutorResult(inner, map) {
1815
+ let mapped;
1816
+ const settle = async (result, task) => {
1817
+ const transformed = await map(result, task);
1818
+ mapped = {
1819
+ outRef: transformed.outRef,
1820
+ out: transformed.out,
1821
+ ...transformed.verdict ? { verdict: transformed.verdict } : {},
1822
+ spent: result.spent
1823
+ };
1824
+ return mapped;
1825
+ };
1826
+ return inheritRuntimeOwnedExecutorAttestation(inner, {
1827
+ runtime: inner.runtime,
1828
+ ...inner.budgetExempt !== void 0 ? { budgetExempt: inner.budgetExempt } : {},
1829
+ ...inner.deliver ? { deliver: (message) => inner.deliver?.(message) } : {},
1830
+ ...inner.progress ? { progress: () => inner.progress?.() } : {},
1831
+ ...inner.traceSource ? { traceSource: () => inner.traceSource?.() } : {},
1832
+ ...inner.accounting ? { accounting: () => inner.accounting?.() } : {},
1833
+ ...inner.metered ? { metered: () => inner.metered?.() } : {},
1834
+ execute(task, signal) {
1835
+ const execution = inner.execute(task, signal);
1836
+ if (isAsyncIterable$1(execution)) return (async function* () {
1837
+ for await (const event of execution) yield event;
1838
+ await settle(inner.resultArtifact(), task);
1839
+ })();
1840
+ return (async () => settle(await execution, task))();
1841
+ },
1842
+ teardown: (grace) => inner.teardown(grace),
1843
+ resultArtifact() {
1844
+ if (!mapped) throw new Error("mapExecutorResult: resultArtifact() read before execute()");
1845
+ return mapped;
1846
+ }
1847
+ });
1848
+ }
1849
+ function isAsyncIterable$1(v) {
1850
+ return v != null && typeof v[Symbol.asyncIterator] === "function";
1851
+ }
1852
+ //#endregion
1853
+ //#region src/runtime/supervise/otel-spans.ts
1854
+ /**
1855
+ * Supervisor tree → OTLP spans. OPT-IN, off by default.
1856
+ *
1857
+ * WHY. A supervised tree is legible today only by parsing this package's own spawn journal, so
1858
+ * every other multi-agent shape on the machine (a coding-CLI's subagents, a pi fanout, ad-hoc tool
1859
+ * parallelism) needs its own bespoke reader. A span carrying `parent_span_id` IS a tree, and any
1860
+ * system can emit one — so emitting spans makes the supervisor readable by the same viewer as
1861
+ * everything else, with no per-system reader.
1862
+ *
1863
+ * WHAT IT IS NOT. This is telemetry, never the record of truth. The spawn journal remains the sole
1864
+ * durable ledger for replay/resume and for cost; nothing here is read back, and a run whose export
1865
+ * fails is unaffected in every observable way. The two data models are deliberately separate.
1866
+ *
1867
+ * HOW IT ATTACHES. This is a pure `RuntimeHooks` observer over the lifecycle events `Scope` ALREADY
1868
+ * emits — `agent.spawn` (a node opened), `agent.child` (a node settled), `agent.turn` (a driver
1869
+ * inference turn was metered). It adds no event, mutates no journal, and changes no result. Because
1870
+ * `Scope` re-seeds the same hooks into every nested scope (`makeNestedScopeSeam`), one observer sees
1871
+ * the WHOLE recursion at arbitrary depth.
1872
+ *
1873
+ * SPAN SHAPE. One span per supervised node, opened at spawn and closed at settle, parented to its
1874
+ * parent node's span; the run root is the trace. Driver inference rides as an `LLM` child span under
1875
+ * the node that metered it. Attributes reuse the vocabulary the rest of the stack already reads
1876
+ * (`openinference.span.kind`, `agent.name`, `llm.token_count.*`, `llm.cost_usd`, `tangle.cost.usd`)
1877
+ * — see `@tangle-network/agent-eval`'s `src/trace/attribute-vocabulary.ts`, which is the consumer.
1878
+ *
1879
+ * UNKNOWN IS NEVER ZERO. `Spend.tokensKnown === false` / `usdKnown === false` mark work that
1880
+ * HAPPENED with an unreported count. Those spans OMIT the token/cost attribute entirely and set
1881
+ * `tangle.supervise.tokens_known` / `tangle.supervise.cost_known` to `false`, so a reader can never
1882
+ * mistake an unmeasured turn for a free one.
1883
+ */
1884
+ /** OTEL status codes (`UNSET` / `OK` / `ERROR`) — the numeric wire values `OtelSpan.status` carries. */
1885
+ const STATUS_UNSET = 0;
1886
+ const STATUS_OK = 1;
1887
+ const STATUS_ERROR = 2;
1888
+ /** Longest string attribute value written from free-form detail, so an oversized turn payload
1889
+ * cannot inflate a span. Identity/label attributes we control are never truncated. */
1890
+ const MAX_DETAIL_CHARS = 256;
1891
+ /**
1892
+ * Build the span recorder for one supervised run, or `undefined` when no exporter resolves — the
1893
+ * off-by-default path. A run that passes no `exporter` and no `exportConfig` never reaches this
1894
+ * function at all; one that passes an `exportConfig` with no endpoint (and no env endpoint) gets
1895
+ * `undefined` here, so "configured but unreachable" also costs nothing.
1896
+ */
1897
+ function createSupervisorSpanRecorder(opts) {
1898
+ const exporter = opts.exporter ?? createOtelExporter(opts.exportConfig);
1899
+ if (!exporter) return void 0;
1900
+ const ownsExporter = opts.exporter === void 0;
1901
+ const now = opts.now ?? Date.now;
1902
+ const traceId = normalizeTraceId(opts.traceId, opts.runId);
1903
+ const rootSpanId = generateSpanId();
1904
+ const rootStartMs = now();
1905
+ const base = {
1906
+ "tangle.run.id": opts.runId,
1907
+ "tangle.sessionId": opts.runId,
1908
+ ...opts.attributes ?? {}
1909
+ };
1910
+ /** Node id → its open span. Seeded with the run id ⇒ the root span, because a depth-0 spawn's
1911
+ * `parentId` is the run id itself and every deeper spawn's is a real node id. */
1912
+ const open = /* @__PURE__ */ new Map();
1913
+ const spanIdOf = /* @__PURE__ */ new Map([[opts.runId, rootSpanId]]);
1914
+ let finished = false;
1915
+ /** Every export is best-effort: a throwing exporter must never reach the run. */
1916
+ const emit = (span) => {
1917
+ try {
1918
+ exporter.exportSpan(span);
1919
+ } catch {}
1920
+ };
1921
+ const span = (spanId, parentSpanId, name, startMs, endMs, attrs, status, message) => ({
1922
+ traceId,
1923
+ spanId,
1924
+ ...parentSpanId ? { parentSpanId } : {},
1925
+ name,
1926
+ kind: 1,
1927
+ startTimeUnixNano: msToNano(startMs),
1928
+ endTimeUnixNano: msToNano(Math.max(startMs, endMs)),
1929
+ attributes: toOtelAttributes(attrs),
1930
+ status: {
1931
+ code: status,
1932
+ ...message ? { message } : {}
1933
+ }
1934
+ });
1935
+ function onSpawn(event) {
1936
+ const p = record(event.payload);
1937
+ const childId = str(p.childId);
1938
+ if (!childId) return;
1939
+ const label = str(p.label) ?? "node";
1940
+ const runtime = str(p.runtime);
1941
+ const isWait = runtime === "wait";
1942
+ const attrs = {
1943
+ ...base,
1944
+ "openinference.span.kind": isWait ? "CHAIN" : "AGENT",
1945
+ "agent.name": label,
1946
+ "tangle.supervise.node.id": childId,
1947
+ "tangle.supervise.node.label": label,
1948
+ "tangle.supervise.node.kind": isWait ? "wait" : "agent",
1949
+ "tangle.supervise.tree.root": event.runId
1950
+ };
1951
+ if (event.parentId) attrs["tangle.supervise.node.parent_id"] = event.parentId;
1952
+ if (runtime) attrs["tangle.supervise.node.runtime"] = runtime;
1953
+ if (typeof p.depth === "number") attrs["tangle.supervise.node.depth"] = p.depth;
1954
+ if (typeof event.stepIndex === "number") attrs["tangle.supervise.node.ordinal"] = event.stepIndex;
1955
+ if (p.resumed === true) attrs["tangle.supervise.node.resumed"] = true;
1956
+ assignBudget(attrs, p.budget);
1957
+ const spanId = generateSpanId();
1958
+ spanIdOf.set(childId, spanId);
1959
+ open.set(childId, {
1960
+ spanId,
1961
+ parentSpanId: event.parentId && spanIdOf.get(event.parentId) || rootSpanId,
1962
+ name: label,
1963
+ startMs: event.timestamp,
1964
+ attrs
1965
+ });
1966
+ }
1967
+ function onSettled(event) {
1968
+ const p = record(event.payload);
1969
+ const childId = str(p.childId);
1970
+ if (!childId) return;
1971
+ const node = open.get(childId);
1972
+ if (!node) return;
1973
+ open.delete(childId);
1974
+ const status = str(p.status);
1975
+ const down = status === "down";
1976
+ const attrs = {
1977
+ ...node.attrs,
1978
+ "tangle.supervise.node.status": status ?? "done"
1979
+ };
1980
+ if (typeof event.stepIndex === "number") attrs["tangle.supervise.node.seq"] = event.stepIndex;
1981
+ if (typeof p.outRef === "string") attrs["tangle.supervise.node.out_ref"] = p.outRef;
1982
+ if (typeof p.valid === "boolean") attrs["tangle.supervise.verdict.valid"] = p.valid;
1983
+ if (typeof p.score === "number") attrs["tangle.supervise.verdict.score"] = p.score;
1984
+ if (down) {
1985
+ attrs["error.type"] = p.infra === true ? "infra" : "child-down";
1986
+ const reason = str(p.reason);
1987
+ if (reason) attrs["error.message"] = truncate(reason);
1988
+ if (typeof p.infra === "boolean") attrs["tangle.supervise.node.infra"] = p.infra;
1989
+ }
1990
+ const wokeBy = str(record(p.wait).settled);
1991
+ if (wokeBy) attrs["tangle.supervise.wait.settled"] = wokeBy;
1992
+ assignSpend(attrs, p.spent);
1993
+ emit(span(node.spanId, node.parentSpanId, node.name, node.startMs, event.timestamp, attrs, down ? STATUS_ERROR : STATUS_OK, down ? str(p.reason) ?? "child down" : void 0));
1994
+ }
1995
+ function onTurn(event) {
1996
+ const p = record(event.payload);
1997
+ const parentId = event.parentId;
1998
+ const parentSpanId = parentId && spanIdOf.get(parentId) || rootSpanId;
1999
+ const attrs = {
2000
+ ...base,
2001
+ "openinference.span.kind": "LLM",
2002
+ "inference.observation_kind": "LLM",
2003
+ "tangle.supervise.node.kind": "inference"
2004
+ };
2005
+ if (parentId) attrs["tangle.supervise.node.id"] = parentId;
2006
+ for (const [key, value] of Object.entries(p)) {
2007
+ if (key === "spend") continue;
2008
+ if (key === "driver" && typeof value === "string") {
2009
+ attrs["agent.name"] = value;
2010
+ attrs["inference.agent_name"] = value;
2011
+ continue;
2012
+ }
2013
+ if (key === "model" && typeof value === "string") {
2014
+ attrs["llm.model_name"] = value;
2015
+ continue;
2016
+ }
2017
+ if (Array.isArray(value)) {
2018
+ const scalars = value.filter((v) => typeof v === "string" || typeof v === "number");
2019
+ if (scalars.length > 0) attrs[`tangle.supervise.turn.${key}`] = truncate(scalars.join(","));
2020
+ continue;
2021
+ }
2022
+ if (typeof value === "string") attrs[`tangle.supervise.turn.${key}`] = truncate(value);
2023
+ else if (typeof value === "number" || typeof value === "boolean") attrs[`tangle.supervise.turn.${key}`] = value;
2024
+ }
2025
+ const spend = assignSpend(attrs, p.spend);
2026
+ const endMs = event.timestamp + (spend?.ms ?? 0);
2027
+ emit(span(generateSpanId(), parentSpanId, "gen_ai.client.inference", event.timestamp, endMs, attrs, STATUS_OK));
2028
+ }
2029
+ return {
2030
+ hooks: { onEvent(event) {
2031
+ if (finished) return;
2032
+ try {
2033
+ if (event.phase !== "after") return;
2034
+ if (event.target === "agent.spawn") onSpawn(event);
2035
+ else if (event.target === "agent.child") onSettled(event);
2036
+ else if (event.target === "agent.turn") onTurn(event);
2037
+ } catch {}
2038
+ } },
2039
+ traceId,
2040
+ rootSpanId,
2041
+ workerTrace(spawningNodeId) {
2042
+ return {
2043
+ traceId,
2044
+ parentSpanId: spanIdOf.get(spawningNodeId) ?? rootSpanId
2045
+ };
2046
+ },
2047
+ async finish(outcome) {
2048
+ if (finished) return;
2049
+ finished = true;
2050
+ const endMs = now();
2051
+ try {
2052
+ for (const [nodeId, node] of open) emit(span(node.spanId, node.parentSpanId, node.name, node.startMs, endMs, {
2053
+ ...node.attrs,
2054
+ "tangle.supervise.node.status": "unsettled",
2055
+ "tangle.supervise.node.settled": false,
2056
+ "tangle.supervise.node.id": nodeId
2057
+ }, STATUS_UNSET));
2058
+ open.clear();
2059
+ emit(span(rootSpanId, opts.parentSpanId, "supervisor.run", rootStartMs, endMs, rootAttrs(base, opts.agentName ?? "supervisor", outcome), rootStatus(outcome), rootMessage(outcome)));
2060
+ await exporter.flush();
2061
+ if (ownsExporter) await exporter.shutdown();
2062
+ } catch {}
2063
+ }
2064
+ };
2065
+ }
2066
+ function rootAttrs(base, agentName, outcome) {
2067
+ const attrs = {
2068
+ ...base,
2069
+ "openinference.span.kind": "AGENT",
2070
+ "inference.observation_kind": "AGENT",
2071
+ "agent.name": agentName,
2072
+ "inference.agent_name": agentName,
2073
+ "tangle.supervise.node.kind": "root"
2074
+ };
2075
+ const result = outcome?.result;
2076
+ if (result) {
2077
+ attrs["tangle.supervise.result"] = result.kind;
2078
+ if (result.kind === "no-winner") {
2079
+ attrs["tangle.supervise.reason"] = result.reason;
2080
+ attrs["tangle.supervise.down_count"] = result.downCount;
2081
+ if (result.reason === "driver-failed") {
2082
+ attrs["error.type"] = result.error.name;
2083
+ attrs["error.message"] = truncate(result.error.message);
2084
+ }
2085
+ }
2086
+ assignSpend(attrs, result.spentTotal);
2087
+ }
2088
+ if (outcome?.error !== void 0) {
2089
+ attrs["tangle.supervise.result"] = "error";
2090
+ const err = outcome.error;
2091
+ attrs["error.type"] = err instanceof Error ? err.name : typeof err;
2092
+ attrs["error.message"] = truncate(err instanceof Error ? err.message : String(err));
2093
+ }
2094
+ return attrs;
2095
+ }
2096
+ function rootStatus(outcome) {
2097
+ if (outcome?.error !== void 0) return STATUS_ERROR;
2098
+ if (!outcome?.result) return STATUS_UNSET;
2099
+ return outcome.result.kind === "winner" ? STATUS_OK : STATUS_ERROR;
2100
+ }
2101
+ function rootMessage(outcome) {
2102
+ if (outcome?.error !== void 0) return truncate(outcome.error instanceof Error ? outcome.error.message : String(outcome.error));
2103
+ const result = outcome?.result;
2104
+ return result && result.kind === "no-winner" ? result.reason : void 0;
2105
+ }
2106
+ /**
2107
+ * Write a `Spend` onto a span, honouring the "a missing measurement is never zero" invariant: a
2108
+ * channel marked NOT known contributes no number at all and instead flags itself, so nothing
2109
+ * downstream can sum an unmeasured turn as free. Returns the spend it read (for the duration).
2110
+ */
2111
+ function assignSpend(attrs, value) {
2112
+ if (!isRecord(value)) return void 0;
2113
+ const spend = value;
2114
+ const tokens = isRecord(spend.tokens) ? spend.tokens : void 0;
2115
+ if (spend.tokensKnown === false) attrs["tangle.supervise.tokens_known"] = false;
2116
+ else if (tokens) {
2117
+ if (typeof tokens.input === "number") attrs["llm.token_count.prompt"] = tokens.input;
2118
+ if (typeof tokens.output === "number") attrs["llm.token_count.completion"] = tokens.output;
2119
+ }
2120
+ if (spend.usdKnown === false) attrs["tangle.supervise.cost_known"] = false;
2121
+ else if (typeof spend.usd === "number" && Number.isFinite(spend.usd)) {
2122
+ attrs["llm.cost_usd"] = spend.usd;
2123
+ attrs["tangle.cost.usd"] = spend.usd;
2124
+ }
2125
+ if (typeof spend.iterations === "number") attrs["tangle.supervise.iterations"] = spend.iterations;
2126
+ if (typeof spend.ms === "number" && spend.ms > 0) attrs["tangle.supervise.duration_ms"] = spend.ms;
2127
+ return spend;
2128
+ }
2129
+ function assignBudget(attrs, value) {
2130
+ if (!isRecord(value)) return;
2131
+ const budget = value;
2132
+ if (typeof budget.maxTokens === "number") attrs["tangle.supervise.budget.max_tokens"] = budget.maxTokens;
2133
+ if (typeof budget.maxIterations === "number") attrs["tangle.supervise.budget.max_iterations"] = budget.maxIterations;
2134
+ if (typeof budget.maxUsd === "number") attrs["tangle.supervise.budget.max_usd"] = budget.maxUsd;
2135
+ }
2136
+ /**
2137
+ * A trace id must be 32 hex characters. A caller-supplied one is used verbatim when it already is;
2138
+ * anything else (and the default) is DERIVED from the run id by content address — deterministic, so
2139
+ * a resumed run rejoins the trace its first process opened rather than forking a new one.
2140
+ */
2141
+ function normalizeTraceId(traceId, runId) {
2142
+ if (traceId && /^[0-9a-f]{32}$/i.test(traceId)) return traceId.toLowerCase();
2143
+ return contentAddress(traceId ?? runId).slice(7, 39);
2144
+ }
2145
+ function msToNano(ms) {
2146
+ return (BigInt(Number.isFinite(ms) ? Math.floor(ms) : 0) * 1000000n).toString();
2147
+ }
2148
+ function isRecord(value) {
2149
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2150
+ }
2151
+ function record(value) {
2152
+ return isRecord(value) ? value : {};
2153
+ }
2154
+ function str(value) {
2155
+ return typeof value === "string" && value.length > 0 ? value : void 0;
2156
+ }
2157
+ function truncate(value) {
2158
+ return value.length > MAX_DETAIL_CHARS ? `${value.slice(0, MAX_DETAIL_CHARS)}…` : value;
2159
+ }
2160
+ //#endregion
2161
+ //#region src/runtime/supervise/coordination-log.ts
2162
+ /**
2163
+ * Durable side-log for coordination evidence the spawn journal does not own: questions, analyst
2164
+ * findings, answer decisions, authorized continuation receipts, delivery-attempt markers, and
2165
+ * delivery outcomes. A durable run
2166
+ * (`supervise({ runDir })`) appends them as they publish and loads them on resume, so a restarted
2167
+ * coordinator retains the exact evidence produced by prior processes.
2168
+ *
2169
+ * Answer down-events also fold status on load: a question answered before the crash reloads as
2170
+ * `answered`, not as a re-blocking `open`. Settled events are skipped (the spawn journal is their
2171
+ * ledger). A receipt followed by an attempt but no outcome proves the process died in the delivery
2172
+ * window; that outcome remains unknown and no prior instruction is auto-delivered.
2173
+ *
2174
+ * JSONL, one fsynced record per event, keyed by `runId` — several runs may share one log file
2175
+ * exactly as they share one spawn-journal file.
2176
+ *
2177
+ * @experimental
2178
+ */
2179
+ /** Persist prior context plus exact continuation authorization, attempt, and result evidence.
2180
+ * Settlements have their own journal. */
2181
+ function persisted(event) {
2182
+ return event.type !== "settled";
2183
+ }
2184
+ /** FS-backed `CoordinationLog`: append-only JSONL, fsynced per record. */
2185
+ var FileCoordinationLog = class {
2186
+ path;
2187
+ appendTail = Promise.resolve();
2188
+ constructor(path) {
2189
+ this.path = path;
2190
+ }
2191
+ async append(runId, record, ownerId) {
2192
+ if (!persisted(record.event)) return;
2193
+ const append = this.appendTail.then(() => this.appendRecord(runId, record, ownerId));
2194
+ this.appendTail = append.catch(() => void 0);
2195
+ return append;
2196
+ }
2197
+ async appendRecord(runId, busRecord, ownerId) {
2198
+ const fs = await import("node:fs/promises");
2199
+ const path = await import("node:path");
2200
+ await fs.mkdir(path.dirname(this.path), { recursive: true });
2201
+ const record = {
2202
+ runId,
2203
+ ...ownerId !== void 0 ? { ownerId } : {},
2204
+ ...busRecord
2205
+ };
2206
+ const needsSeparator = await prepareJsonlAppend(this.path);
2207
+ const fh = await fs.open(this.path, "a");
2208
+ try {
2209
+ await writeAllBytes(fh, `${needsSeparator ? "\n" : ""}${JSON.stringify(record)}\n`);
2210
+ await fh.sync();
2211
+ } finally {
2212
+ await fh.close();
2213
+ }
2214
+ }
2215
+ async load(runId, ownerId) {
2216
+ const fs = await import("node:fs/promises");
2217
+ let text;
2218
+ try {
2219
+ text = await fs.readFile(this.path, "utf8");
2220
+ } catch (err) {
2221
+ if (isNoEntError(err)) return emptyPriorCoordination(ownerId);
2222
+ throw err;
2223
+ }
2224
+ const byId = /* @__PURE__ */ new Map();
2225
+ const findings = [];
2226
+ const continuations = [];
2227
+ const deliveryEvidence = [];
2228
+ const mail = [];
2229
+ const records = [];
2230
+ let legacySeq = 0;
2231
+ for (const stored of parseCommittedJsonLines(text, this.path)) {
2232
+ if (stored.runId !== runId) continue;
2233
+ if (ownerId !== void 0 && stored.ownerId !== ownerId) continue;
2234
+ const record = "seq" in stored ? {
2235
+ seq: stored.seq,
2236
+ at: stored.at,
2237
+ priority: stored.priority,
2238
+ event: stored.event
2239
+ } : {
2240
+ seq: legacySeq++,
2241
+ at: Date.parse(stored.at),
2242
+ priority: 0,
2243
+ event: stored.event
2244
+ };
2245
+ records.push(record);
2246
+ const ev = record.event;
2247
+ if (ev.type === "delivery-attempt" || ev.type === "steer" || ev.type === "answer") deliveryEvidence.push(ev);
2248
+ if (ev.type === "question") byId.set(ev.question.id, ev.question);
2249
+ else if (ev.type === "finding") findings.push(ev.finding);
2250
+ else if (ev.type === "answer") {
2251
+ const prior = byId.get(ev.questionId);
2252
+ if (prior && ev.down.delivered) byId.set(ev.questionId, {
2253
+ ...prior,
2254
+ status: "answered",
2255
+ decision: {
2256
+ kind: "answer",
2257
+ answer: ev.down.instruction,
2258
+ by: "prior-run"
2259
+ }
2260
+ });
2261
+ } else if (ev.type === "instruction") continuations.push(ev.instruction);
2262
+ else if (ev.type === "mail") mail.push(ev.mail);
2263
+ }
2264
+ return {
2265
+ ...ownerId !== void 0 ? { ownerId } : {},
2266
+ questions: [...byId.values()],
2267
+ findings,
2268
+ continuations,
2269
+ deliveryEvidence,
2270
+ mail,
2271
+ records
2267
2272
  };
2268
- };
2273
+ }
2274
+ };
2275
+ function emptyPriorCoordination(ownerId) {
2269
2276
  return {
2270
- tools,
2271
- ready,
2272
- history: () => bus.history(),
2273
- raiseFinding: (finding) => bus.publish({
2274
- type: "finding",
2275
- finding: canonicalFindingEvent(finding)
2276
- }).then(() => void 0),
2277
- stats: () => bus.stats(),
2278
- isStopped: () => stopped,
2279
- stopReason: () => reason,
2280
- submittedResult: () => submitted,
2281
- settled: () => ledger,
2282
- questions: () => questions,
2283
- drainResolved,
2284
- abortWorker,
2285
- ...peerMail ? { peerMail } : {}
2277
+ ...ownerId !== void 0 ? { ownerId } : {},
2278
+ questions: [],
2279
+ findings: [],
2280
+ continuations: [],
2281
+ deliveryEvidence: [],
2282
+ mail: [],
2283
+ records: []
2286
2284
  };
2287
2285
  }
2288
- function nextUnkeyedAssignmentOrdinal(scope) {
2289
- let next = 0;
2290
- const views = [scope.resume?.view, scope.view];
2291
- for (const view of views) {
2292
- if (view === void 0) continue;
2293
- for (const node of view.nodes) {
2294
- const match = /^ordinal:(\d+)$/.exec(node.assignmentId ?? "");
2295
- if (match === null) continue;
2296
- const ordinal = Number(match[1]);
2297
- if (!Number.isSafeInteger(ordinal)) throw new Error(`coordination: durable assignment id '${node.assignmentId}' exceeds the safe ordinal range`);
2298
- next = Math.max(next, ordinal + 1);
2299
- }
2300
- }
2301
- if (!Number.isSafeInteger(next)) throw new Error("coordination: durable assignment ordinal space is exhausted");
2302
- return next;
2303
- }
2304
- function deepFreezeDetached(value) {
2305
- return deepFreeze(structuredClone(value));
2286
+ function isNoEntError(err) {
2287
+ return typeof err === "object" && err !== null && "code" in err && err.code === "ENOENT";
2306
2288
  }
2307
- /** Stringify a findings payload for a routed delivery; never throws (a cyclic payload degrades to
2308
- * its String form rather than killing the settle path). */
2309
- function safeJsonText(value) {
2310
- if (typeof value === "string") return value;
2311
- try {
2312
- return JSON.stringify(value) ?? String(value);
2313
- } catch {
2314
- return String(value);
2315
- }
2289
+ //#endregion
2290
+ //#region src/runtime/supervise/run-context.ts
2291
+ /**
2292
+ *
2293
+ * `createInMemoryRunContext` — the one-call bundle of the in-memory stores a
2294
+ * `createSupervisor().run(root, task, opts)` needs: a fresh `InMemorySpawnJournal`
2295
+ * (the event-sourced spawn log), a fresh `InMemoryResultBlobStore` (the
2296
+ * content-addressed `outRef` payload store the driver's `observe`/`finalize` reads
2297
+ * settled outputs through), and a fresh `createExecutorRegistry()` (the open
2298
+ * `AgentSpec → Executor` resolver).
2299
+ *
2300
+ * It exists to kill the boilerplate every offline/local supervised run repeats by
2301
+ * hand — three constructors threaded into `SupervisorOpts` — and to single-source the
2302
+ * ONE wiring invariant that is easy to get wrong: when the root is the recursive
2303
+ * `driverAgent` LLM-driver brain AND it may spawn DRIVER children (agents
2304
+ * driving agents), the registry MUST be wrapped with `withDriverExecutor` so a
2305
+ * `role: 'driver'` child resolves to the nested-scope executor — and that SAME blob
2306
+ * store MUST be the one passed to `driverAgent({ blobs })`, or the driver
2307
+ * reads from a different store than the scope writes to. Pass `{ withDriver: true }`
2308
+ * and reuse the returned `blobs` for both.
2309
+ *
2310
+ * The spread shape matches `SupervisorOpts` exactly, so the call site reads:
2311
+ * const run = createInMemoryRunContext()
2312
+ * await createSupervisor().run(root, task, { budget, runId, ...run })
2313
+ *
2314
+ * @experimental
2315
+ */
2316
+ /**
2317
+ * Build a fresh in-memory run context. Every call returns NEW stores (no shared global
2318
+ * state between runs), so two runs never cross-contaminate their journals/blobs.
2319
+ */
2320
+ function createInMemoryRunContext(opts = {}) {
2321
+ const base = createExecutorRegistry();
2322
+ return {
2323
+ journal: new InMemorySpawnJournal(),
2324
+ blobs: new InMemoryResultBlobStore(),
2325
+ executors: opts.withDriver ? withDriverExecutor(base) : base
2326
+ };
2316
2327
  }
2317
- function deepFreeze(value, seen = /* @__PURE__ */ new Set()) {
2318
- if (value === null || typeof value !== "object" || seen.has(value)) return value;
2319
- seen.add(value);
2320
- for (const child of Object.values(value)) deepFreeze(child, seen);
2321
- return Object.freeze(value);
2328
+ /**
2329
+ * Build a DURABLE run context: the spawn journal and the result blobs are file-backed (fsynced
2330
+ * per append/write) under `dir`, and the context carries `resume: true` so spreading it into
2331
+ * `SupervisorOpts` makes the supervisor `loadTree`-first. A run that dies mid-flight therefore
2332
+ * resumes when it is re-run with the SAME `runId` and the SAME `dir`: the committed children come
2333
+ * back on `Scope.resume` (rehydrated by `replaySpawnTree`) instead of being re-executed.
2334
+ *
2335
+ * Layout: `${dir}/spawn-journal.jsonl` (one JSONL record per event), `${dir}/blobs/` (one
2336
+ * content-addressed JSON file per settled result), and `${dir}/coordination-log.jsonl`
2337
+ * (questions, findings, answer decisions, and authorized continuation receipts retained as
2338
+ * evidence). The directory is created on first write.
2339
+ *
2340
+ * Opt-in by construction — `createInMemoryRunContext()` is unchanged and stays the default, so no
2341
+ * existing consumer writes to disk or resumes unless it asks for this.
2342
+ */
2343
+ function createFileRunContext(dir, opts = {}) {
2344
+ const base = createExecutorRegistry();
2345
+ return {
2346
+ journal: new FileSpawnJournal(`${dir}/spawn-journal.jsonl`),
2347
+ blobs: new FileResultBlobStore(`${dir}/blobs`),
2348
+ executors: opts.withDriver ? withDriverExecutor(base) : base,
2349
+ resume: true,
2350
+ coordinationLog: new FileCoordinationLog(`${dir}/coordination-log.jsonl`)
2351
+ };
2322
2352
  }
2323
2353
  //#endregion
2324
2354
  //#region src/runtime/anytime.ts
@@ -2680,6 +2710,27 @@ function allOf(...rules) {
2680
2710
  };
2681
2711
  };
2682
2712
  }
2713
+ /**
2714
+ * Evaluate a rule against the run's settled work — the ONE evaluator both supervisor arms call.
2715
+ *
2716
+ * The router arm calls it before each driver inference turn; the harness arm calls it on each
2717
+ * worker settle. Ordering is the contract in both: the hard ceilings (`poolStarved`,
2718
+ * `deadlinePassed`, abort, the driver's own stop) are checked first and independently, so a stop
2719
+ * rule can only ever ADD a stop — it can never keep a run alive past a budget it has exhausted.
2720
+ *
2721
+ * Folding the whole roster each call is idempotent by worker id, so it costs O(settled) and never
2722
+ * double-counts. `settledAt` carries the instant the ledger recorded a settlement; `now()` is the
2723
+ * fallback resolution a per-turn guard has.
2724
+ */
2725
+ function progressStop(tracker, rule, ledger, scope, now, stallAfterMs) {
2726
+ for (const w of ledger.settled()) tracker.record({
2727
+ id: w.id,
2728
+ at: w.settledAt ?? now(),
2729
+ ...w.score !== void 0 ? { objective: w.score } : {},
2730
+ delivered: w.status === "done" && w.valid === true
2731
+ });
2732
+ return tracker.evaluate(rule, scope, stallAfterMs !== void 0 ? { stallAfterMs } : void 0);
2733
+ }
2683
2734
  //#endregion
2684
2735
  //#region src/runtime/supervise/coordination-driver.ts
2685
2736
  /**
@@ -2744,23 +2795,6 @@ function deadlinePassed(scope, now) {
2744
2795
  const b = scope.budget;
2745
2796
  return b.deadlineMs > 0 && now() >= b.deadlineMs;
2746
2797
  }
2747
- /**
2748
- * The PROGRESS-derived stop, evaluated strictly AFTER the hard ceilings above.
2749
- *
2750
- * Ordering is the contract, not a detail: `poolStarved` / `deadlinePassed` / abort / the driver's
2751
- * own stop are checked first and independently, so a stop rule can only ever ADD a stop — it can
2752
- * never keep a run alive past a budget it has exhausted. The rule reads the settled-work ledger
2753
- * (via the tracker) plus the live worker feed off the scope; it spends nothing to do so.
2754
- */
2755
- function progressStop(tracker, rule, coord, scope, now, stallAfterMs) {
2756
- for (const w of coord.settled()) tracker.record({
2757
- id: w.id,
2758
- at: w.settledAt ?? now(),
2759
- ...w.score !== void 0 ? { objective: w.score } : {},
2760
- delivered: w.status === "done" && w.valid === true
2761
- });
2762
- return tracker.evaluate(rule, scope, stallAfterMs !== void 0 ? { stallAfterMs } : void 0);
2763
- }
2764
2798
  function providerAttemptEvidence$1(model) {
2765
2799
  const attempts = Object.freeze([Object.freeze({ observations: Object.freeze(model === void 0 ? [] : [model]) })]);
2766
2800
  const models = Object.freeze(model === void 0 ? [] : [model]);
@@ -2814,26 +2848,46 @@ const SPAWN_JOURNAL_FILE = "spawn-journal.jsonl";
2814
2848
  /**
2815
2849
  * The worker-cancel ACKNOWLEDGER — the runtime-side half of `run-layout`'s `cancelWorker`
2816
2850
  * contract, run from the coordination driver's turn loop (one cancellation-inbox read per turn,
2817
- * no new process, no poller, no extra lifetime).
2851
+ * no new process, no poller, no extra lifetime). Every manager with a `controlDir` mounts one;
2852
+ * OWNERSHIP keeps them from colliding: a request naming a node id is owned by the manager whose
2853
+ * own id is that node's parent, and a label/profile-name reference is owned by the `'run'`-scoped
2854
+ * (root) manager only — so exactly one acknowledger can ever apply one operation.
2818
2855
  *
2819
2856
  * Two-phase, honestly reported: `cancel_requested` is written the moment a live worker's abort is
2820
2857
  * issued (through the per-child abort chain the scope already owns, so siblings are untouched);
2821
2858
  * `cancelled` is written only when that worker's settlement is DELIVERED on the settle path with
2822
2859
  * a terminal `down`, and then the record names every subtree node id proven terminated. A worker
2823
2860
  * that already settled — or that settles `done` despite the abort — records `not_live`; a
2824
- * reference matching nothing this manager knows stays pending (`cancelWorker` reports it
2861
+ * reference matching nothing this manager owns stays pending (`cancelWorker` reports it
2825
2862
  * `unknown`). No path reports success for a missing worker.
2826
2863
  *
2864
+ * Expiry is run end, not a clock: `finish()` (after the final post-drain pass) writes `not_live`
2865
+ * for every owned request never applied and `unknown` for an issued abort whose settle the run
2866
+ * ended too soon to observe. A pending request can therefore never outlive its run and abort a
2867
+ * future spawn that happens to reuse a label.
2868
+ *
2827
2869
  * Idempotency is a lookup, in-process and across processes: an operation with a durable
2828
2870
  * acknowledgement is returned as-is and never re-applied.
2829
2871
  */
2830
2872
  function createCancelAcknowledger(deps) {
2831
2873
  const tracked = /* @__PURE__ */ new Map();
2874
+ let runTracked;
2875
+ const abortIssuedAt = /* @__PURE__ */ new Map();
2832
2876
  const iso = () => new Date(deps.now()).toISOString();
2833
2877
  const write = (record) => {
2834
2878
  writeWorkerCancellation(deps.dir, record);
2835
2879
  tracked.set(record.operationId, record);
2836
2880
  };
2881
+ /** `ref` is exactly one of THIS manager's direct-child node ids (`${ownerId}:s<seq>`). */
2882
+ const directChildId = (ref) => ref.startsWith(`${deps.ownerId}:s`) && /^s\d+$/.test(ref.slice(deps.ownerId.length + 1));
2883
+ /** Whether this acknowledger owns `ref`. A node id deeper in this subtree belongs to the nested
2884
+ * manager that parents it; anything that is not a node id under this manager is a
2885
+ * label/profile-name reference, owned by the `'run'`-scoped manager alone. */
2886
+ const owned = (ref) => {
2887
+ if (directChildId(ref)) return true;
2888
+ if (deps.controlScope !== "run") return false;
2889
+ return !ref.startsWith(`${deps.ownerId}:`) && ref !== deps.ownerId;
2890
+ };
2837
2891
  const deliveredTerminal = (id) => {
2838
2892
  const row = deps.coord.settled().find((w) => w.id === id);
2839
2893
  if (row !== void 0) return row.status;
@@ -2852,6 +2906,7 @@ function createCancelAcknowledger(deps) {
2852
2906
  ...request.reason === void 0 ? {} : { reason: request.reason }
2853
2907
  };
2854
2908
  if (aborted !== void 0) {
2909
+ abortIssuedAt.set(request.operationId, base.observedAt);
2855
2910
  write({
2856
2911
  ...base,
2857
2912
  effect: "cancel_requested",
@@ -2870,6 +2925,17 @@ function createCancelAcknowledger(deps) {
2870
2925
  terminated: []
2871
2926
  });
2872
2927
  };
2928
+ /** The proven-terminated set for one record: the worker plus every subtree id with a terminal
2929
+ * journal record at/after the abort was issued. Union with what the record already names, so
2930
+ * the set only ever grows (a late teardown journal adds; nothing removes). */
2931
+ const provenTerminated = (record, workerId) => {
2932
+ const since = abortIssuedAt.get(record.operationId) ?? record.observedAt;
2933
+ return [.../* @__PURE__ */ new Set([
2934
+ ...record.terminated,
2935
+ workerId,
2936
+ ...terminatedDescendants(deps.dir, workerId, since)
2937
+ ])].sort();
2938
+ };
2873
2939
  const reconcile = (record) => {
2874
2940
  const workerId = record.workerId;
2875
2941
  if (workerId === void 0) return;
@@ -2880,7 +2946,7 @@ function createCancelAcknowledger(deps) {
2880
2946
  ...record,
2881
2947
  effect: "cancelled",
2882
2948
  observedAt: iso(),
2883
- terminated: [workerId, ...terminatedDescendants(deps.dir, workerId, record.requestedAt)],
2949
+ terminated: provenTerminated(record, workerId),
2884
2950
  detail: `worker '${workerId}' reached a terminal down state on the settle path`
2885
2951
  });
2886
2952
  return;
@@ -2893,8 +2959,56 @@ function createCancelAcknowledger(deps) {
2893
2959
  detail: `worker '${workerId}' settled done despite the abort request; nothing was terminated`
2894
2960
  });
2895
2961
  };
2896
- return { pass() {
2962
+ /** Re-scan a `cancelled` record while the manager still turns: a descendant whose teardown
2963
+ * journals after the lead's settle joins the set on a later pass instead of being lost. Only
2964
+ * a grown set is re-written; the window needs the in-process abort instant, so a record a
2965
+ * PRIOR process closed stays as that process proved it. */
2966
+ const regrow = (record) => {
2967
+ const workerId = record.workerId;
2968
+ if (workerId === void 0 || !abortIssuedAt.has(record.operationId)) return;
2969
+ const terminated = provenTerminated(record, workerId);
2970
+ if (terminated.length > record.terminated.length) write({
2971
+ ...record,
2972
+ observedAt: iso(),
2973
+ terminated
2974
+ });
2975
+ };
2976
+ /**
2977
+ * The RUN-scoped request: seen once, `cancel_requested` written the moment the run's cascading
2978
+ * abort is issued through the one controller the run already has. The `supervise()` settle path
2979
+ * records what the run then actually did — this manager cannot observe its own tree's terminal
2980
+ * state from inside `act`.
2981
+ *
2982
+ * Applied only at a TURN boundary, never on the final post-drain pass: by then the driver has
2983
+ * finished and drained, so a root abort could only void work that is already delivered. A
2984
+ * request that arrives that late expires in `finish()` instead — it terminated nothing.
2985
+ */
2986
+ const passRun = () => {
2987
+ if (deps.controlScope !== "run" || deps.abortRun === void 0) return;
2988
+ const request = readRunCancelRequest(deps.dir);
2989
+ if (request === void 0) return;
2990
+ if (runTracked !== void 0) return;
2991
+ const prior = readRunCancellation(deps.dir, request.operationId);
2992
+ if (prior !== void 0) {
2993
+ runTracked = prior;
2994
+ return;
2995
+ }
2996
+ const record = {
2997
+ operationId: request.operationId,
2998
+ effect: "cancel_requested",
2999
+ requestedAt: request.at,
3000
+ observedAt: iso(),
3001
+ ...request.reason === void 0 ? {} : { reason: request.reason },
3002
+ detail: "root abort issued to the whole run; termination not yet proven"
3003
+ };
3004
+ writeRunCancellation(deps.dir, record);
3005
+ runTracked = record;
3006
+ deps.abortRun(request.reason ?? "run cancel requested");
3007
+ };
3008
+ const pass = (phase) => {
3009
+ if (phase === "turn") passRun();
2897
3010
  for (const request of readWorkerCancelRequests(deps.dir)) {
3011
+ if (!owned(request.worker)) continue;
2898
3012
  let record = tracked.get(request.operationId);
2899
3013
  if (record === void 0) {
2900
3014
  record = readWorkerCancellation(deps.dir, request.operationId);
@@ -2905,16 +3019,58 @@ function createCancelAcknowledger(deps) {
2905
3019
  continue;
2906
3020
  }
2907
3021
  if (record.effect === "cancel_requested") reconcile(record);
3022
+ else if (record.effect === "cancelled") regrow(record);
2908
3023
  }
2909
- } };
3024
+ };
3025
+ return {
3026
+ pass,
3027
+ finish() {
3028
+ const runRequest = deps.controlScope === "run" && deps.abortRun !== void 0 ? readRunCancelRequest(deps.dir) : void 0;
3029
+ if (runRequest !== void 0 && readRunCancellation(deps.dir, runRequest.operationId) === void 0) writeRunCancellation(deps.dir, {
3030
+ operationId: runRequest.operationId,
3031
+ effect: "not_live",
3032
+ requestedAt: runRequest.at,
3033
+ observedAt: iso(),
3034
+ ...runRequest.reason === void 0 ? {} : { reason: runRequest.reason },
3035
+ detail: "run ended before the request was applied"
3036
+ });
3037
+ for (const request of readWorkerCancelRequests(deps.dir)) {
3038
+ if (!owned(request.worker)) continue;
3039
+ const record = tracked.get(request.operationId) ?? readWorkerCancellation(deps.dir, request.operationId);
3040
+ if (record === void 0) {
3041
+ write({
3042
+ operationId: request.operationId,
3043
+ worker: request.worker,
3044
+ effect: "not_live",
3045
+ requestedAt: request.at,
3046
+ observedAt: iso(),
3047
+ ...request.reason === void 0 ? {} : { reason: request.reason },
3048
+ detail: "run ended before the request was applied",
3049
+ terminated: []
3050
+ });
3051
+ continue;
3052
+ }
3053
+ if (record.effect === "cancel_requested") write({
3054
+ ...record,
3055
+ effect: "unknown",
3056
+ observedAt: iso(),
3057
+ detail: "abort issued; run ended before termination was observed"
3058
+ });
3059
+ }
3060
+ }
3061
+ };
2910
3062
  }
2911
3063
  /**
2912
3064
  * Subtree node ids with a terminal `down`/`cancelled` journal record at or after `sinceIso` —
2913
- * the descendants a cancelled lead's cascading abort took down, read from the durable spawn
2914
- * journal beside the run layout (the nested trees journal their terminal records before the lead
2915
- * itself settles). Ids are hierarchical (`parent:sN`), so `${nodeId}:` prefixes exactly the
2916
- * subtree. Tolerant of a missing or partially-written journal: evidence that cannot be read
2917
- * names fewer nodes, never wrong ones.
3065
+ * the abort-issue instant (the acknowledger's own `observedAt` on the `cancel_requested` record,
3066
+ * runtime clock), never the client's `requestedAt` read from the durable spawn journal beside
3067
+ * the run layout. The set is proven at acknowledgement time and is approximate about post-abort
3068
+ * causation: a descendant that died of its OWN cause after the abort was issued is
3069
+ * indistinguishable from the cascade and may be included; one whose teardown journals late joins
3070
+ * on a later acknowledger pass; a teardown journal still absent when the run ends is absent from
3071
+ * the set. Ids are hierarchical (`parent:sN`), so `${nodeId}:` prefixes exactly the subtree.
3072
+ * Tolerant of a missing or partially-written journal: evidence that cannot be read names fewer
3073
+ * nodes, never wrong ones.
2918
3074
  */
2919
3075
  function terminatedDescendants(dir, nodeId, sinceIso) {
2920
3076
  let raw;
@@ -2983,16 +3139,21 @@ function driverAgent(opts) {
2983
3139
  ...opts.watchWorkers ? { watchWorkers: opts.watchWorkers } : {},
2984
3140
  ...opts.stallAfterMs !== void 0 ? { stallAfterMs: opts.stallAfterMs } : {},
2985
3141
  ...opts.continuityByProfile ? { continuityByProfile: opts.continuityByProfile } : {},
3142
+ ...opts.preflightSpawn ? { preflightSpawn: opts.preflightSpawn } : {},
2986
3143
  ...opts.onEvent ? { onEvent: opts.onEvent } : {},
2987
3144
  ...opts.replaySettlements ? { replaySettlements: true } : {},
2988
3145
  ...opts.priorCoordination?.questions.length ? { priorQuestions: opts.priorCoordination.questions } : {}
2989
3146
  });
2990
3147
  await coord.ready();
3148
+ opts.onCoordinationTools?.(coord.tools);
2991
3149
  const acknowledger = opts.controlDir === void 0 ? void 0 : createCancelAcknowledger({
2992
3150
  dir: opts.controlDir,
2993
3151
  coord,
2994
3152
  scope,
2995
- now
3153
+ now,
3154
+ ownerId: scope.view.root,
3155
+ controlScope: opts.controlScope ?? "run",
3156
+ ...opts.abortRun ? { abortRun: opts.abortRun } : {}
2996
3157
  });
2997
3158
  for (const w of scope.resume?.waits ?? []) {
2998
3159
  const rearmed = scope.wait(w.spec, { label: w.label });
@@ -3162,7 +3323,7 @@ function driverAgent(opts) {
3162
3323
  maxTurns,
3163
3324
  hooks: {
3164
3325
  beforeTurn: (_turn, messages) => {
3165
- acknowledger?.pass();
3326
+ acknowledger?.pass("turn");
3166
3327
  const pending = inbox.drain();
3167
3328
  if (pending.length > 0) messages.push({
3168
3329
  role: "user",
@@ -3183,7 +3344,8 @@ function driverAgent(opts) {
3183
3344
  }
3184
3345
  });
3185
3346
  await coord.drainResolved();
3186
- acknowledger?.pass();
3347
+ acknowledger?.pass("final");
3348
+ acknowledger?.finish();
3187
3349
  const submitted = coord.submittedResult();
3188
3350
  if (submitted) return submitted.result;
3189
3351
  return runFinalizer(opts.finalizer ?? bestDelivered, {
@@ -5052,9 +5214,11 @@ async function serveCoordinationMcp(opts) {
5052
5214
  ...opts.replaySettlements ? { replaySettlements: true } : {},
5053
5215
  ...opts.questionPolicy ? { questionPolicy: opts.questionPolicy } : {},
5054
5216
  ...opts.priorQuestions?.length ? { priorQuestions: opts.priorQuestions } : {},
5217
+ ...opts.preflightSpawn ? { preflightSpawn: opts.preflightSpawn } : {},
5055
5218
  ...opts.peerMail ? { peerMail: typeof opts.peerMail === "object" && opts.peerMail.limits ? { limits: opts.peerMail.limits } : {} } : {}
5056
5219
  });
5057
5220
  await coord.ready();
5221
+ opts.onCoordinationTools?.(coord.tools);
5058
5222
  const mcp = createMcpServer({
5059
5223
  extraTools: [...coord.tools, ...opts.nodeTools ?? []],
5060
5224
  serverName: "coordination"
@@ -5635,6 +5799,29 @@ function assertCoordinationBinding(binding) {
5635
5799
  if (binding?.allowUnauthenticatedRemote === true) return;
5636
5800
  throw new ConfigError(`supervisorAgent: coordination.host=${JSON.stringify(host)} is not a loopback address and the coordination MCP has no authentication: any client that can reach the port could call spawn_agent/steer_agent and spend this run's budget. Bind a loopback host ("127.0.0.1", "localhost", "::1"), or set coordination.allowUnauthenticatedRemote: true to accept that exposure explicitly.`);
5637
5801
  }
5802
+ function createVerbSlot() {
5803
+ let bound;
5804
+ const verb = (name) => async (args) => {
5805
+ if (bound === void 0) throw new ValidationError(`supervisorAgent: coordination verb "${name}" was called before this manager's coordination tools were bound`);
5806
+ const tool = bound.find((descriptor) => descriptor.name === name);
5807
+ if (tool === void 0) throw new ValidationError(`supervisorAgent: coordination verb "${name}" is not mounted on this manager`);
5808
+ return tool.handler(args);
5809
+ };
5810
+ return {
5811
+ verbs: Object.freeze({
5812
+ spawnAgent: verb("spawn_agent"),
5813
+ awaitEvent: verb("await_event"),
5814
+ steerAgent: verb("steer_agent"),
5815
+ observeAgent: verb("observe_agent"),
5816
+ listQuestions: verb("list_questions"),
5817
+ answerQuestion: verb("answer_question"),
5818
+ runAnalyst: verb("run_analyst")
5819
+ }),
5820
+ bind(tools) {
5821
+ bound = tools;
5822
+ }
5823
+ };
5824
+ }
5638
5825
  const ROUTER_TRANSPORT_FIELDS = /* @__PURE__ */ new Set([
5639
5826
  "routerBaseUrl",
5640
5827
  "routerKey",
@@ -5682,12 +5869,13 @@ function buildSupervisorAgent(profile, deps, testBrain) {
5682
5869
  const coordination = deps.coordination ? { ...deps.coordination } : void 0;
5683
5870
  assertCoordinationBinding(coordination);
5684
5871
  if (harness === null && coordination !== void 0) throw new ConfigError("supervisorAgent: coordination binding is only meaningful for a harness-brained supervisor (profile.harness set). A router-brained supervisor calls the coordination verbs in process and serves no MCP, so this binding would be silently ignored.");
5872
+ if (harness === null && deps.peerMail) throw new ValidationError("supervisorAgent: peerMail is only served by a harness-brained supervisor (profile.harness set). A router-brained supervisor serves no coordination MCP listener, so there is no peer-mail post office to mint worker capabilities from.");
5685
5873
  if (harness !== null && deps.compaction) throw new ValidationError("supervisorAgent: compaction is only supported for router-brained supervisors (profile.harness omitted or cli-base)");
5686
5874
  if (harness === null) {
5687
5875
  assertRouterArmResourcePolicy(stableProfile);
5688
5876
  const brain = testBrain ?? routerBrainFromProfile(stableProfile, stableRouter);
5689
5877
  const inbox = createInbox();
5690
- const build = (priorCoordination, nodeTools, onEvent) => driverAgent({
5878
+ const build = (priorCoordination, nodeTools, onEvent, slot) => driverAgent({
5691
5879
  name,
5692
5880
  brain,
5693
5881
  ...testBrain === void 0 ? { expectedModel: resolveSupervisorModelId(stableProfile) } : {},
@@ -5707,6 +5895,7 @@ function buildSupervisorAgent(profile, deps, testBrain) {
5707
5895
  ...deps.watchWorkers ? { watchWorkers: deps.watchWorkers } : {},
5708
5896
  ...deps.stallAfterMs !== void 0 ? { stallAfterMs: deps.stallAfterMs } : {},
5709
5897
  ...deps.continuityByProfile ? { continuityByProfile: deps.continuityByProfile } : {},
5898
+ ...deps.preflightSpawn ? { preflightSpawn: deps.preflightSpawn } : {},
5710
5899
  ...deps.stopRule ? { stopRule: deps.stopRule } : {},
5711
5900
  ...deps.onProgressStop ? { onProgressStop: deps.onProgressStop } : {},
5712
5901
  ...deps.maxTurns !== void 0 ? { maxTurns: deps.maxTurns } : {},
@@ -5715,7 +5904,10 @@ function buildSupervisorAgent(profile, deps, testBrain) {
5715
5904
  ...deps.replaySettlements ? { replaySettlements: true } : {},
5716
5905
  ...priorCoordination ? { priorCoordination } : {},
5717
5906
  ...deps.finalizer ? { finalizer: deps.finalizer } : {},
5907
+ ...slot ? { onCoordinationTools: (tools) => slot.bind(tools) } : {},
5718
5908
  ...deps.controlDir === void 0 ? {} : { controlDir: deps.controlDir },
5909
+ ...deps.controlScope === void 0 ? {} : { controlScope: deps.controlScope },
5910
+ ...deps.abortRun ? { abortRun: deps.abortRun } : {},
5719
5911
  inbox
5720
5912
  });
5721
5913
  if (!deps.loadPriorCoordination && !resolveTools && !observeNodeEvent) return build(deps.priorCoordination, void 0, deps.onEvent);
@@ -5727,9 +5919,10 @@ function buildSupervisorAgent(profile, deps, testBrain) {
5727
5919
  async act(task, scope) {
5728
5920
  const context = nodeContextSeed ? supervisorNodeContext(nodeContextSeed, stableProfile, task, scope) : void 0;
5729
5921
  const priorCoordination = await deps.loadPriorCoordination?.();
5730
- const nodeTools = resolveTools && context ? await bindSupervisorTools(resolveTools, context, scope.signal) : void 0;
5922
+ const slot = createVerbSlot();
5923
+ const nodeTools = resolveTools && context ? await bindSupervisorTools(resolveTools, context, scope.signal, slot) : void 0;
5731
5924
  const onEvent = bindSupervisorNodeObserver(context, observeNodeEvent, deps.onEvent);
5732
- return build(priorCoordination, nodeTools, onEvent).act(task, scope);
5925
+ return build(priorCoordination, nodeTools, onEvent, slot).act(task, scope);
5733
5926
  }
5734
5927
  };
5735
5928
  }
@@ -5744,9 +5937,24 @@ function buildSupervisorAgent(profile, deps, testBrain) {
5744
5937
  async act(task, scope) {
5745
5938
  const context = nodeContextSeed ? supervisorNodeContext(nodeContextSeed, stableProfile, task, scope) : void 0;
5746
5939
  const priorCoordination = deps.loadPriorCoordination ? await deps.loadPriorCoordination() : deps.priorCoordination;
5747
- const nodeTools = resolveTools && context ? await bindSupervisorTools(resolveTools, context, scope.signal) : void 0;
5748
- const onEvent = bindSupervisorNodeObserver(context, observeNodeEvent, deps.onEvent);
5940
+ const slot = createVerbSlot();
5941
+ const nodeTools = resolveTools && context ? await bindSupervisorTools(resolveTools, context, scope.signal, slot) : void 0;
5942
+ const nodeObserver = bindSupervisorNodeObserver(context, observeNodeEvent, deps.onEvent);
5749
5943
  const stopController = new AbortController();
5944
+ const tracker = deps.stopRule ? createProgressTracker({ now: Date.now }) : void 0;
5945
+ let progressStopReason;
5946
+ let ledger;
5947
+ const rule = deps.stopRule;
5948
+ const onEvent = tracker && rule ? async (event, record) => {
5949
+ await nodeObserver?.(event, record);
5950
+ if (event.type !== "settled") return;
5951
+ if (ledger === void 0) throw new ValidationError("supervisorAgent: a worker settled before the coordination server was bound");
5952
+ const decision = progressStop(tracker, rule, ledger, scope, Date.now, deps.stallAfterMs);
5953
+ if (!decision.stop || progressStopReason !== void 0) return;
5954
+ progressStopReason = decision.reason;
5955
+ deps.onProgressStop?.(decision.reason);
5956
+ if (!stopController.signal.aborted) stopController.abort(decision.reason);
5957
+ } : nodeObserver;
5750
5958
  const mcp = await serveCoordinationMcp({
5751
5959
  scope,
5752
5960
  blobs: deps.blobs,
@@ -5768,9 +5976,13 @@ function buildSupervisorAgent(profile, deps, testBrain) {
5768
5976
  ...deps.continuityByProfile ? { continuityByProfile: deps.continuityByProfile } : {},
5769
5977
  ...onEvent ? { onEvent } : {},
5770
5978
  ...deps.replaySettlements ? { replaySettlements: true } : {},
5979
+ ...deps.preflightSpawn ? { preflightSpawn: deps.preflightSpawn } : {},
5980
+ ...deps.peerMail ? { peerMail: deps.peerMail } : {},
5771
5981
  ...priorCoordination?.questions.length ? { priorQuestions: priorCoordination.questions } : {},
5772
- ...nodeTools?.length ? { nodeTools } : {}
5982
+ ...nodeTools?.length ? { nodeTools } : {},
5983
+ onCoordinationTools: (tools) => slot.bind(tools)
5773
5984
  });
5985
+ ledger = mcp;
5774
5986
  try {
5775
5987
  const baseTokensLeft = scope.budget.tokensLeft;
5776
5988
  await runDriverWithRetry({
@@ -5828,12 +6040,13 @@ function supervisorNodeContext(seed, profile, task, scope) {
5828
6040
  task
5829
6041
  }, "supervisorAgent trusted node context");
5830
6042
  }
5831
- async function bindSupervisorTools(resolveTools, context, signal) {
6043
+ async function bindSupervisorTools(resolveTools, context, signal, slot) {
5832
6044
  const resolved = await resolveTools(context);
5833
6045
  if (!Array.isArray(resolved)) throw new ValidationError("supervisorAgent: resolveSupervisorTools must return an array");
5834
6046
  const invocationContext = Object.freeze({
5835
6047
  ...context,
5836
- signal
6048
+ signal,
6049
+ verbs: slot.verbs
5837
6050
  });
5838
6051
  const names = new Set(coordinationVerbNames);
5839
6052
  return Object.freeze(resolved.map((rawTool, index) => {
@@ -5876,7 +6089,7 @@ function routerBrainFromProfile(profile, router) {
5876
6089
  ...router.complete !== void 0 ? { complete: router.complete } : {},
5877
6090
  model: modelId,
5878
6091
  ...settings.retry !== void 0 ? { retry: settings.retry } : {},
5879
- ...settings.maxTokens !== void 0 ? { maxTokens: settings.maxTokens } : {},
6092
+ ...enforceTokenLimits(settings.tokenLimits, "router", "supervisorAgent").applied,
5880
6093
  ...settings.stream !== void 0 ? { stream: settings.stream } : {}
5881
6094
  }, {
5882
6095
  ...settings.temperature !== void 0 ? { temperature: settings.temperature } : {},
@@ -5932,6 +6145,7 @@ function workerFromBackend(backend, deliverable, seams) {
5932
6145
  if (!parsed.success) throw new ValidationError(`workerFromBackend: invalid AgentProfile: ${parsed.error.message}`);
5933
6146
  const profile = parsed.data;
5934
6147
  assertBackendProfileMaterialization(profile, capturedBackend, "workerFromBackend");
6148
+ assertBridgeProfileMaterializes(profile, capturedBackend, "workerFromBackend");
5935
6149
  const resumeSessionId = bridgeResumeSessionId(capturedBackend, spawnContext, bridgeSessionByWorker);
5936
6150
  const name = profile.name ?? "worker";
5937
6151
  const assignmentId = spawnContext?.assignmentId ?? `unscoped:${unscopedNamespace}:${unscopedOrdinal++}`;
@@ -6030,6 +6244,46 @@ function assertBackendProfileMaterialization(profile, backend, context) {
6030
6244
  assertProfileContract(profile, backendProfileMaterialization(backend), context);
6031
6245
  }
6032
6246
  /**
6247
+ * The dimensions cli-bridge lowers through its OWN native controls rather than the workspace plan,
6248
+ * per harness. The pre-spawn check must skip exactly these, or it refuses a profile the bridge
6249
+ * would have executed. Mirrors `provisionProfileWorkspace` / `provisionPiProfile` in cli-bridge
6250
+ * `src/backends/profile-support.ts`.
6251
+ */
6252
+ function bridgeMaterializationSkip(harness) {
6253
+ return harness === "pi" ? ["mcp", "extensions"] : ["mcp"];
6254
+ }
6255
+ /**
6256
+ * The two prompt intents, gated here because they are the ONLY profile dimensions cli-bridge
6257
+ * refuses independently of whether a harness materializes a workspace at all
6258
+ * (`assertProfilePromptIntentsSupported`, cli-bridge `src/backends/profile-support.ts`): a
6259
+ * backend either owns a control that reaches the harness's system-prompt position or it does not.
6260
+ * Every other dimension's verdict belongs to the workspace plan the executing backend builds, and
6261
+ * the run's own materialization receipt already refuses those after the fact.
6262
+ */
6263
+ const gatedPromptDimensions = ["systemPrompt", "appendSystemPrompt"];
6264
+ /**
6265
+ * Refuse a bridge-bound profile whose prompt intent the harness cannot execute, at the SYNCHRONOUS
6266
+ * spawn seam — before the reservation commits, the `spawned` event is journaled, or a token is
6267
+ * metered. The verdict is a pure function of the profile and the harness, but the bridge only
6268
+ * reaches it while assembling the prompt, by which point the child is spawned, metered, and
6269
+ * settled `down` to deliver an answer that was available before it started.
6270
+ *
6271
+ * Applies to the profiles the runtime SPAWNS — a worker and a nested driver child. A root manager
6272
+ * profile is the caller's own input and is not a spawn: the bridge answers it on the manager's
6273
+ * first turn without a child ever existing.
6274
+ *
6275
+ * Silent for every other backend: `cli-worktree` runs the full plan check on its own local plan
6276
+ * (`runWorktreeHarness`), and no other backend hands the profile to a harness materializer.
6277
+ */
6278
+ function assertBridgeProfileMaterializes(profile, backend, context) {
6279
+ if (backend.backend !== "bridge") return;
6280
+ const harness = agentHarness(profile.harness);
6281
+ if (harness === void 0 || !isMaterializerHarness(harness)) return;
6282
+ const unsupported = unsupportedProfileDimensions(profile, harness, gatedPromptDimensions, bridgeMaterializationSkip(harness));
6283
+ if (unsupported.length === 0) return;
6284
+ throw new ValidationError(`${context}: ${harness} cannot materialize the profile: ${renderUnsupported(unsupported)}`);
6285
+ }
6286
+ /**
6033
6287
  * The ROOT router-brained supervisor's materialization claim. The router arm consumes the
6034
6288
  * identity fields, the resolved system prompt (`prompt.systemPrompt` + `prompt.instructions` +
6035
6289
  * `resources.instructions`), and the resolved model id (`model.default`); the remaining model
@@ -6062,6 +6316,62 @@ const routerSupervisorProfileMaterialization = defineProfileMaterializationContr
6062
6316
  ]
6063
6317
  });
6064
6318
  const coordinationMcpAlias = "agent-runtime-coordination";
6319
+ /** How a harness sees a coordination verb once the MCP is mounted under its reserved alias. */
6320
+ const coordinationToolPrefix = `${coordinationMcpAlias.replaceAll("-", "_")}_`;
6321
+ /**
6322
+ * Tools a child REQUIRES that name the coordination MCP but no coordination verb.
6323
+ *
6324
+ * A profile can only receive a coordination tool this run actually serves, and the served set is
6325
+ * closed (`coordinationVerbNames`). A required name inside the reserved namespace that is not one
6326
+ * of them can never mount on any harness, for any backend, at any depth — the harness discovers it
6327
+ * only when it starts and exits (`pi exit 78: requested tool "…" is unavailable`), after the child
6328
+ * is spawned, journaled and metered.
6329
+ */
6330
+ function unmountedCoordinationTools(profile) {
6331
+ const served = new Set(coordinationVerbNames.map((verb) => `${coordinationToolPrefix}${verb}`));
6332
+ return Object.entries(profile.tools ?? {}).filter(([name, required]) => required === true && name.startsWith(coordinationToolPrefix)).map(([name]) => name).filter((name) => !served.has(name));
6333
+ }
6334
+ /**
6335
+ * The pre-flight `supervise` installs for a bridge backend. No new knob: the backend already says
6336
+ * where the bridge is, and these are the questions only the bridge can answer.
6337
+ *
6338
+ * Three causes, in cost order — the pure one first, so a deterministic refusal never pays for a
6339
+ * round trip:
6340
+ *
6341
+ * - `unmountable-tool` — pure; see {@link unmountedCoordinationTools}.
6342
+ * - `model-route` — `GET /v1/capabilities?model=<wire id>`. The bridge answers exactly this
6343
+ * question and 404s `no backend matches model "…"`. FAIL CLOSED: any answer that is not a route
6344
+ * refuses, including a transport error or an unexpected status, because a pre-flight that skips
6345
+ * itself on an error is the silent admission it exists to remove.
6346
+ * - `bridge-full` — `GET /health` → `admission.active >= admission.maxActive`. ADVISORY by
6347
+ * nature (admission can fill or drain a moment later), so only a POSITIVE reading of fullness
6348
+ * refuses: a `/health` that does not answer, or answers without an admission snapshot, is not
6349
+ * evidence that the bridge is full and admits the spawn.
6350
+ */
6351
+ function bridgeSpawnPreflight(seam) {
6352
+ return async (profile) => {
6353
+ const unmounted = unmountedCoordinationTools(profile);
6354
+ if (unmounted.length > 0) return {
6355
+ cause: "unmountable-tool",
6356
+ detail: `no coordination verb is named by ${unmounted.map((name) => JSON.stringify(name)).join(", ")}; this run serves ${coordinationVerbNames.join(", ")}`
6357
+ };
6358
+ const wireModel = profileBridgeWireModel(profile);
6359
+ if (wireModel === void 0) return {
6360
+ cause: "model-route",
6361
+ detail: "the child AgentProfile resolves no bridge wire model (harness + provider + model)"
6362
+ };
6363
+ const routeRefusal = await bridgeModelRouteRefusal(seam, wireModel);
6364
+ if (routeRefusal !== void 0) return {
6365
+ cause: "model-route",
6366
+ detail: routeRefusal
6367
+ };
6368
+ const admission = await bridgeAdmissionRead(seam);
6369
+ if (admission && admission.active >= admission.maxActive) return {
6370
+ cause: "bridge-full",
6371
+ detail: `bridge ${seam.bridgeUrl.replace(/\/$/, "")} admission is full: active ${admission.active} of maxActive ${admission.maxActive}`
6372
+ };
6373
+ };
6374
+ }
6065
6375
  const defaultAllowedMcpHosts = [];
6066
6376
  Object.freeze(defaultAllowedMcpHosts);
6067
6377
  /** Manager-authored profiles are untrusted until product policy says otherwise. Remote MCP and
@@ -6081,7 +6391,9 @@ function automaticDriverBackendSupported(backend) {
6081
6391
  /** Run a harness-brained manager through the same executor factory as its children. The manager's
6082
6392
  * full profile is preserved, the live coordination server is added under one reserved alias, and
6083
6393
  * every streamed turn is charged to the manager's scope before it may continue. */
6084
- function driveHarnessFromBackend(backend, executionId, now = Date.now) {
6394
+ function driveHarnessFromBackend(backend, executionId, now = Date.now, maxTurns) {
6395
+ if (maxTurns !== void 0 && maxTurns < 0) throw new ValidationError("driveHarnessFromBackend: maxTurns must be >= 0 (0 lifts the turn cap; bounds become the conserved pool + deadline + abort)");
6396
+ const turnCap = maxTurns ?? 0;
6085
6397
  const boundBackend = bindReusableExecutorExecutionId(captureReusableExecutorConfig(backend, "driveHarnessFromBackend"), executionId);
6086
6398
  const baseFactory = createExecutor(boundBackend);
6087
6399
  let activeExecutor;
@@ -6090,25 +6402,23 @@ function driveHarnessFromBackend(backend, executionId, now = Date.now) {
6090
6402
  if (!(scope.view.inFlight > 0 || scope.view.waiting > 0) && (initialBudget.tokensLeft <= 0 || initialBudget.iterationsLeft <= 0 || initialBudget.usdCapped && initialBudget.usdLeft <= 0 || initialBudget.deadlineMs > 0 && now() >= initialBudget.deadlineMs)) throw new ValidationError("driveHarnessFromBackend: supervisor budget exhausted");
6091
6403
  const canonicalDriverProfile = agentProfileSchema.parse(profile);
6092
6404
  if (canonicalDriverProfile.mcp?.[coordinationMcpAlias] !== void 0) throw new ValidationError(`driveHarnessFromBackend: profile MCP alias ${JSON.stringify(coordinationMcpAlias)} is reserved`);
6093
- const effectiveProfile = agentProfileSchema.parse({
6094
- ...canonicalDriverProfile,
6095
- mcp: {
6096
- ...canonicalDriverProfile.mcp,
6097
- [coordinationMcpAlias]: {
6098
- transport: "http",
6099
- url: coordinationMcpUrl
6100
- }
6101
- }
6102
- });
6103
6405
  const stableCoordinationTools = detachedSnapshot(coordinationTools, "driveHarnessFromBackend coordination tools");
6104
6406
  const spec = {
6105
- profile: effectiveProfile,
6106
- harness: boundBackend.backend === "sandbox" ? effectiveProfile.harness : null
6407
+ profile: canonicalDriverProfile,
6408
+ harness: boundBackend.backend === "sandbox" ? canonicalDriverProfile.harness : null
6107
6409
  };
6410
+ const turnStop = turnCap > 0 ? new AbortController() : void 0;
6411
+ const effectiveStopSignal = turnStop === void 0 ? stopSignal : stopSignal === void 0 ? turnStop.signal : AbortSignal.any([stopSignal, turnStop.signal]);
6108
6412
  const executor = baseFactory(spec, {
6109
6413
  signal: scope.signal,
6110
6414
  node: scopeOwnerExecutorNodeContext(scope),
6111
- seams: stopSignal === void 0 ? {} : { [bridgeStopSignalKey]: stopSignal }
6415
+ seams: {
6416
+ ...effectiveStopSignal === void 0 ? {} : { [bridgeStopSignalKey]: effectiveStopSignal },
6417
+ [bridgeRuntimeAttachmentsKey]: { [coordinationMcpAlias]: {
6418
+ transport: "http",
6419
+ url: coordinationMcpUrl
6420
+ } }
6421
+ }
6112
6422
  });
6113
6423
  activeExecutor = executor;
6114
6424
  let completed = false;
@@ -6192,25 +6502,21 @@ function driveHarnessFromBackend(backend, executionId, now = Date.now) {
6192
6502
  try {
6193
6503
  const declaration = runtimeOwnedExecutorMaterialization(executor);
6194
6504
  const executionBinding = runtimeOwnedExecutorExecutionBinding(executor);
6195
- const authoredProfileFromDriverExecution = (profile) => {
6196
- const { [coordinationMcpAlias]: _runtimeAttachment, ...authoredMcp } = profile.mcp ?? {};
6197
- const { mcp: _mcp, ...withoutMcp } = profile;
6198
- return agentProfileSchema.parse(Object.keys(authoredMcp).length > 0 ? {
6199
- ...withoutMcp,
6200
- mcp: authoredMcp
6201
- } : withoutMcp);
6202
- };
6203
6505
  if (pending === void 0 && (declaration === void 0 || executionBinding === void 0)) throw new ValidationError(`driveHarnessFromBackend: built-in runtime ${JSON.stringify(executor.runtime)} has no trusted materialization declaration or execution binding`);
6204
6506
  if (pending !== void 0) {
6205
6507
  if (pending.runtime !== executor.runtime || pending.binding.attemptId !== scopeOwnerExecutorNodeContext(scope).attemptId) throw new ValidationError("driveHarnessFromBackend: pending executor did not bind the kernel-minted attempt");
6206
- if (canonicalAgentProfileDigest(authoredProfileFromDriverExecution(pending.declaration.effectiveProfile)) !== canonicalAgentProfileDigest(canonicalDriverProfile)) throw new ValidationError("driveHarnessFromBackend: pending executor changed the authored AgentProfile before execution");
6508
+ if (canonicalAgentProfileDigest(pending.declaration.effectiveProfile) !== canonicalAgentProfileDigest(canonicalDriverProfile)) throw new ValidationError("driveHarnessFromBackend: pending executor changed the authored AgentProfile before execution");
6207
6509
  } else await publishMaterialization(declaration, executionBinding);
6208
6510
  if (executor.budgetExempt) throw new ValidationError(`driveHarnessFromBackend: runtime ${JSON.stringify(executor.runtime)} does not report usage and cannot drive a budgeted supervisor`);
6209
6511
  started = true;
6210
6512
  const run = executor.execute(task, scope.signal);
6211
6513
  if (isAsyncIterable(run)) {
6212
- for await (const event of run) if (event.kind === "iteration") await meterPending();
6213
- else pendingUsage.push(event);
6514
+ let turns = 0;
6515
+ for await (const event of run) if (event.kind === "iteration") {
6516
+ await meterPending();
6517
+ turns += 1;
6518
+ if (turnStop !== void 0 && turns >= turnCap && !turnStop.signal.aborted) turnStop.abort(`supervise: maxTurns ${turnCap} reached`);
6519
+ } else if (event.kind !== "progress") pendingUsage.push(event);
6214
6520
  await meterPending();
6215
6521
  const artifact = executor.resultArtifact();
6216
6522
  terminalAccountingCaptured = true;
@@ -6436,6 +6742,49 @@ function captureSuperviseOptions(opts) {
6436
6742
  ...otel === void 0 ? {} : { otel }
6437
6743
  });
6438
6744
  }
6745
+ /**
6746
+ * Record what a run-scoped cancel actually did, at the ONE place that observes the run's terminal
6747
+ * state: the `supervise()` settle path.
6748
+ *
6749
+ * The root manager writes `cancel_requested` when it issues the abort; only here is the run's own
6750
+ * outcome known. A run that ends `aborted` after that request reads `cancelled`; a run that
6751
+ * reached any other terminal state despite the request terminated nothing and reads `not_live`,
6752
+ * never a success. A request the run ended before applying is expired here too, so a reader can
6753
+ * tell run-over from in-progress and a stale request cannot outlive its run.
6754
+ */
6755
+ function recordRunCancellationOutcome(runDir, result, now) {
6756
+ if (runDir === void 0) return;
6757
+ const dir = resolve(runDir);
6758
+ const request = readRunCancelRequest(dir);
6759
+ if (request === void 0) return;
6760
+ const record = readRunCancellation(dir, request.operationId);
6761
+ if (record !== void 0 && record.effect !== "cancel_requested") return;
6762
+ const aborted = result.kind === "no-winner" && result.reason === "aborted";
6763
+ const observedAt = new Date(now()).toISOString();
6764
+ const base = {
6765
+ operationId: request.operationId,
6766
+ requestedAt: request.at,
6767
+ observedAt,
6768
+ ...request.reason === void 0 ? {} : { reason: request.reason }
6769
+ };
6770
+ if (record === void 0) {
6771
+ writeRunCancellation(dir, {
6772
+ ...base,
6773
+ effect: "not_live",
6774
+ detail: "run ended before the request was applied"
6775
+ });
6776
+ return;
6777
+ }
6778
+ writeRunCancellation(dir, aborted ? {
6779
+ ...base,
6780
+ effect: "cancelled",
6781
+ detail: "the run reached its terminal aborted state"
6782
+ } : {
6783
+ ...base,
6784
+ effect: "not_live",
6785
+ detail: `run settled ${result.kind === "winner" ? "winner" : result.reason} despite the abort request; nothing was terminated`
6786
+ });
6787
+ }
6439
6788
  /** A quarter of token and optional dollar capacity per worker; nested managers partition again. */
6440
6789
  /** A per-child budget may not exceed the conserved pool it is reserved from. */
6441
6790
  function assertPerWorkerWithinPool(perWorker, pool) {
@@ -6593,6 +6942,7 @@ function superviseInternal(profile, task, opts, testBrain) {
6593
6942
  await options.onCoordinationEvent?.(context, coordinationEventId(context, event), record);
6594
6943
  } : void 0;
6595
6944
  const managerBackend = options.driverBackend ?? options.backend;
6945
+ const spawnPreflight = options.backend?.backend === "bridge" ? bridgeSpawnPreflight(options.backend) : void 0;
6596
6946
  if (options.driveHarness && options.resolveDriveHarness) throw new ValidationError("supervise: provide driveHarness or resolveDriveHarness, not both");
6597
6947
  const driverMaterialization = Boolean(options.driveHarness || options.resolveDriveHarness) ? options.driveHarnessMaterialization ?? fullProfileMaterialization : managerBackend && automaticDriverBackendSupported(managerBackend) ? backendProfileMaterialization(managerBackend) : void 0;
6598
6948
  if (isExternalSupervisor(canonicalProfile) && !options.driveHarness && !options.resolveDriveHarness && (!managerBackend || !automaticDriverBackendSupported(managerBackend))) throw new ValidationError(`supervise: external supervisor profile.harness=${JSON.stringify(canonicalProfile.harness)} requires a local bridge driverBackend, an explicit driveHarness, or resolveDriveHarness with reachable coordination transport`);
@@ -6621,7 +6971,7 @@ function superviseInternal(profile, task, opts, testBrain) {
6621
6971
  return managerBackend && automaticDriverBackendSupported(managerBackend) ? driveHarnessFromBackend(managerBackend, externalExecutionId("supervised-manager", {
6622
6972
  runNamespace,
6623
6973
  ownerId: context.ownerId
6624
- }), options.now ?? Date.now) : void 0;
6974
+ }), options.now ?? Date.now, options.maxTurns) : void 0;
6625
6975
  };
6626
6976
  const rootDriveHarness = isExternalSupervisor(canonicalProfile) ? driveHarnessForOwner(freezeDetached({
6627
6977
  runId,
@@ -6712,6 +7062,7 @@ function superviseInternal(profile, task, opts, testBrain) {
6712
7062
  })) : void 0;
6713
7063
  if (isExternalSupervisor(authorized) && !nestedDriveHarness) throw new ValidationError(`supervise: authored external supervisor profile.harness=${JSON.stringify(authorized.harness)} requires a local bridge driverBackend, an explicit driveHarness, or resolveDriveHarness with reachable coordination transport`);
6714
7064
  assertProfileContract(authorized, isExternalSupervisor(authorized) ? driverMaterialization : promptModelProfileMaterialization, `supervise driver ${JSON.stringify(spawnContext.label)}`);
7065
+ if (managerBackend) assertBridgeProfileMaterializes(authorized, managerBackend, `supervise driver ${JSON.stringify(spawnContext.label)}`);
6715
7066
  const childFactory = makeRecursiveWorkerFor(authorized, childExecution.identity, depth + 1, ownerId);
6716
7067
  const nestedPerWorker = defaultPerWorker(spawnContext.budget);
6717
7068
  const authorizeNestedMessage = authorizeDownFor(authorized, depth + 1);
@@ -6740,6 +7091,8 @@ function superviseInternal(profile, task, opts, testBrain) {
6740
7091
  ...options.watchWorkers ? { watchWorkers: options.watchWorkers } : {},
6741
7092
  ...options.stallAfterMs !== void 0 ? { stallAfterMs: options.stallAfterMs } : {},
6742
7093
  ...options.continuityByProfile ? { continuityByProfile: options.continuityByProfile } : {},
7094
+ ...spawnPreflight ? { preflightSpawn: spawnPreflight } : {},
7095
+ ...options.peerMail ? { peerMail: options.peerMail } : {},
6743
7096
  ...options.stopRule ? { stopRule: options.stopRule } : {},
6744
7097
  ...options.onProgressStop ? { onProgressStop: options.onProgressStop } : {},
6745
7098
  ...options.maxTurns !== void 0 ? { maxTurns: options.maxTurns } : {},
@@ -6750,7 +7103,11 @@ function superviseInternal(profile, task, opts, testBrain) {
6750
7103
  onEvent: (_event, record) => log.append(runId, record, ownerId),
6751
7104
  loadPriorCoordination: () => log.load(runId, ownerId)
6752
7105
  } : {},
6753
- ...finalizer ? { finalizer } : {}
7106
+ ...finalizer ? { finalizer } : {},
7107
+ ...options.runDir === void 0 ? {} : {
7108
+ controlDir: resolve(options.runDir),
7109
+ controlScope: "subtree"
7110
+ }
6754
7111
  }), journal, childExecution.ref);
6755
7112
  };
6756
7113
  return makeRecursiveWorker;
@@ -6761,6 +7118,7 @@ function superviseInternal(profile, task, opts, testBrain) {
6761
7118
  const start = async () => {
6762
7119
  const priorCoordination = log ? await log.load(runId, rootOwnerId) : void 0;
6763
7120
  const authorizeRootMessage = authorizeDownFor(canonicalProfile, 1);
7121
+ const runControl = options.rootHandle ?? (options.runDir === void 0 ? void 0 : createRootHandle());
6764
7122
  const agentDeps = {
6765
7123
  blobs,
6766
7124
  makeWorkerAgent: workerFactory,
@@ -6774,6 +7132,8 @@ function superviseInternal(profile, task, opts, testBrain) {
6774
7132
  ...priorCoordination && (priorCoordination.questions.length > 0 || priorCoordination.findings.length > 0 || priorCoordination.continuations.length > 0 || priorCoordination.deliveryEvidence.length > 0) ? { priorCoordination } : {},
6775
7133
  ...finalizer ? { finalizer } : {},
6776
7134
  ...options.coordination ? { coordination: options.coordination } : {},
7135
+ ...spawnPreflight ? { preflightSpawn: spawnPreflight } : {},
7136
+ ...options.peerMail ? { peerMail: options.peerMail } : {},
6777
7137
  ...options.maxLiveWorkers !== void 0 ? { maxLiveWorkers: options.maxLiveWorkers } : {},
6778
7138
  ...options.router ? { router: options.router } : {},
6779
7139
  ...rootDriveHarness ? { driveHarness: rootDriveHarness } : {},
@@ -6802,7 +7162,10 @@ function superviseInternal(profile, task, opts, testBrain) {
6802
7162
  ...options.compaction ? { compaction: options.compaction } : {},
6803
7163
  ...options.driverRetry ? { driverRetry: options.driverRetry } : {},
6804
7164
  ...options.onDriverAttempt ? { onDriverAttempt: options.onDriverAttempt } : {},
6805
- ...options.runDir === void 0 ? {} : { controlDir: resolve(options.runDir) }
7165
+ ...options.runDir === void 0 || runControl === void 0 ? {} : {
7166
+ controlDir: resolve(options.runDir),
7167
+ abortRun: (reason) => runControl.abort(reason)
7168
+ }
6806
7169
  };
6807
7170
  const agent = testBrain === void 0 ? supervisorAgent(canonicalProfile, agentDeps) : supervisorAgentWithTestBrain(canonicalProfile, {
6808
7171
  ...agentDeps,
@@ -6816,7 +7179,7 @@ function superviseInternal(profile, task, opts, testBrain) {
6816
7179
  const recorder = spans;
6817
7180
  const hooks = recorder ? composeRuntimeHooks(options.hooks, recorder.hooks) : options.hooks;
6818
7181
  const supervisor = createSupervisor();
6819
- if (options.rootHandle) supervisor.attach(options.rootHandle);
7182
+ if (runControl !== void 0) supervisor.attach(runControl);
6820
7183
  const run = supervisor.run(agent, canonicalTask, {
6821
7184
  budget: options.budget,
6822
7185
  runId,
@@ -6842,6 +7205,7 @@ function superviseInternal(profile, task, opts, testBrain) {
6842
7205
  });
6843
7206
  const settle = async () => {
6844
7207
  const result = await run;
7208
+ recordRunCancellationOutcome(options.runDir, result, now);
6845
7209
  const rootProviderModel = ctx.resume === true ? rootProviderModelEvidence([]) : rootProviderModels.length > 0 ? rootProviderModelEvidence(rootProviderModels) : rootDriveHarness === void 0 ? rootProviderModelEvidence([]) : rootProviderModelEvidenceFromExecution(runtimeOwnedDriveHarnessProviderEvidence(rootDriveHarness));
6846
7210
  return {
6847
7211
  ...result,
@@ -6879,6 +7243,6 @@ function rootProviderModelEvidenceFromExecution(evidence) {
6879
7243
  return evidence ?? rootProviderModelEvidence([]);
6880
7244
  }
6881
7245
  //#endregion
6882
- export { DELEGATION_TRACE_MAX_BYTES as $, DELEGATION_HISTORY_INPUT_SCHEMA as A, defaultToolDetectors as At, DELEGATE_FEEDBACK_INPUT_SCHEMA as B, createMcpServer as C, plateauLength as Ct, createDelegationStatusHandler as D, createCoordinationTools as Dt, DELEGATION_STATUS_TOOL_NAME as E, canonicalFindingEvent as Et, DELEGATE_UI_AUDIT_INPUT_SCHEMA as F, createSupervisorSpanRecorder as Ft, DELEGATE_INPUT_SCHEMA as G, createDelegateFeedbackHandler as H, DELEGATE_UI_AUDIT_TOOL_NAME as I, gateOnDeliverable as It, validateDelegateArgs as J, DELEGATE_TOOL_NAME as K, createDelegateUiAuditHandler as L, mapExecutorResult as Lt, createDelegationHistoryHandler as M, createFileRunContext as Mt, validateDelegationHistoryArgs as N, createInMemoryRunContext as Nt, validateDelegationStatusArgs as O, normalizeAnalyzeOnSettle as Ot, DELEGATE_UI_AUDIT_DESCRIPTION as P, FileCoordinationLog as Pt, hashIdempotencyInput as Q, validateDelegateUiAuditArgs as R, createInProcessTransport as S, bestSoFar as St, DELEGATION_STATUS_INPUT_SCHEMA as T, DEFAULT_AWAIT_EVENT_TIMEOUT_MS as Tt, validateDelegateFeedbackArgs as U, DELEGATE_FEEDBACK_TOOL_NAME as V, DELEGATE_DESCRIPTION as W, delegate as X, defaultDelegateBudget as Y, DelegationTaskQueue as Z, promptHandle as _, noProgressFor as _t, assertCoordinationBinding as a, DelegationPersistenceError as at, classifyDriverFailure as b, anytimeReport as bt, supervisorAgentWithTestBrain as c, InMemoryDelegationStore as ct, delegatesWorkerBriefPrompt as d, driverAgent as dt, DELEGATION_TRACE_MAX_SPANS as et, dumbContinuationFailPrompt as f, finalizeBestDelivered as ft, naiveContinuationPrompt as g, createProgressTracker as gt, kernelPromptRegistry as h, anyOf as ht, workerFromBackend as i, createDelegationTraceCollector as it, DELEGATION_HISTORY_TOOL_NAME as j, watchTrace as jt, DELEGATION_HISTORY_DESCRIPTION as k, createEventBus as kt, analyzesFindingsReportPrompt as l, InMemoryFeedbackStore as lt, formatPromptHandle as m, allWorkersStalled as mt, supervise as n, capDelegationTrace as nt, resolveSupervisorProfile as o, DelegationStateCorruptError as ot, dumbContinuationPassPrompt as p, allOf as pt, createDelegateHandler as q, superviseWithTestBrain as r, composeLoopTraceEmitters as rt, supervisorAgent as s, FileDelegationStore as st, DEFAULT_AUTHORED_PROFILE_SECURITY_POLICY as t, buildDelegationTraceSpans as tt, createPromptRegistry as u, eventToSnapshot as ut, supervisorPolicyPrompt as v, plateau as vt, DELEGATION_STATUS_DESCRIPTION as w, renderAnytimeTable as wt, serveCoordinationMcp as x, areaUnderCurve as xt, DriverAttemptsExhaustedError as y, sampleFromSettled as yt, DELEGATE_FEEDBACK_DESCRIPTION as z };
7246
+ export { DELEGATION_TRACE_MAX_BYTES as $, DELEGATION_HISTORY_INPUT_SCHEMA as A, mapExecutorResult as At, DELEGATE_FEEDBACK_INPUT_SCHEMA as B, createMcpServer as C, plateauLength as Ct, createDelegationStatusHandler as D, FileCoordinationLog as Dt, DELEGATION_STATUS_TOOL_NAME as E, createInMemoryRunContext as Et, DELEGATE_UI_AUDIT_INPUT_SCHEMA as F, createEventBus as Ft, DELEGATE_INPUT_SCHEMA as G, createDelegateFeedbackHandler as H, DELEGATE_UI_AUDIT_TOOL_NAME as I, defaultToolDetectors as It, validateDelegateArgs as J, DELEGATE_TOOL_NAME as K, createDelegateUiAuditHandler as L, watchTrace as Lt, createDelegationHistoryHandler as M, canonicalFindingEvent as Mt, validateDelegationHistoryArgs as N, createCoordinationTools as Nt, validateDelegationStatusArgs as O, createSupervisorSpanRecorder as Ot, DELEGATE_UI_AUDIT_DESCRIPTION as P, normalizeAnalyzeOnSettle as Pt, hashIdempotencyInput as Q, validateDelegateUiAuditArgs as R, createInProcessTransport as S, bestSoFar as St, DELEGATION_STATUS_INPUT_SCHEMA as T, createFileRunContext as Tt, validateDelegateFeedbackArgs as U, DELEGATE_FEEDBACK_TOOL_NAME as V, DELEGATE_DESCRIPTION as W, delegate as X, defaultDelegateBudget as Y, DelegationTaskQueue as Z, promptHandle as _, noProgressFor as _t, assertCoordinationBinding as a, DelegationPersistenceError as at, classifyDriverFailure as b, anytimeReport as bt, supervisorAgentWithTestBrain as c, InMemoryDelegationStore as ct, delegatesWorkerBriefPrompt as d, driverAgent as dt, DELEGATION_TRACE_MAX_SPANS as et, dumbContinuationFailPrompt as f, finalizeBestDelivered as ft, naiveContinuationPrompt as g, createProgressTracker as gt, kernelPromptRegistry as h, anyOf as ht, workerFromBackend as i, createDelegationTraceCollector as it, DELEGATION_HISTORY_TOOL_NAME as j, DEFAULT_AWAIT_EVENT_TIMEOUT_MS as jt, DELEGATION_HISTORY_DESCRIPTION as k, gateOnDeliverable as kt, analyzesFindingsReportPrompt as l, InMemoryFeedbackStore as lt, formatPromptHandle as m, allWorkersStalled as mt, supervise as n, capDelegationTrace as nt, resolveSupervisorProfile as o, DelegationStateCorruptError as ot, dumbContinuationPassPrompt as p, allOf as pt, createDelegateHandler as q, superviseWithTestBrain as r, composeLoopTraceEmitters as rt, supervisorAgent as s, FileDelegationStore as st, DEFAULT_AUTHORED_PROFILE_SECURITY_POLICY as t, buildDelegationTraceSpans as tt, createPromptRegistry as u, eventToSnapshot as ut, supervisorPolicyPrompt as v, plateau as vt, DELEGATION_STATUS_DESCRIPTION as w, renderAnytimeTable as wt, serveCoordinationMcp as x, areaUnderCurve as xt, DriverAttemptsExhaustedError as y, sampleFromSettled as yt, DELEGATE_FEEDBACK_DESCRIPTION as z };
6883
7247
 
6884
- //# sourceMappingURL=supervise-hCgG6ROJ.js.map
7248
+ //# sourceMappingURL=supervise-BezbzKiJ.js.map