@tangle-network/agent-runtime 0.175.0 → 0.177.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/README.md +4 -0
  2. package/dist/{activation-CsdJWRXX.js → activation-XVI_z1g3.js} +3 -3
  3. package/dist/{activation-CsdJWRXX.js.map → activation-XVI_z1g3.js.map} +1 -1
  4. package/dist/agent.d.ts +1 -1
  5. package/dist/agent.js +2 -2
  6. package/dist/{authoring-DsLNInAz.js → authoring-bXmPNfCI.js} +2 -2
  7. package/dist/{authoring-DsLNInAz.js.map → authoring-bXmPNfCI.js.map} +1 -1
  8. package/dist/candidate-execution/index.js +4 -4
  9. package/dist/{candidate-execution-BbMbyhgC.js → candidate-execution-fBNUyWyE.js} +10 -4
  10. package/dist/candidate-execution-fBNUyWyE.js.map +1 -0
  11. package/dist/{conversation-7XGKoDM6.js → conversation-BREx2hi-.js} +3 -3
  12. package/dist/{conversation-7XGKoDM6.js.map → conversation-BREx2hi-.js.map} +1 -1
  13. package/dist/conversation.d.ts +1 -1
  14. package/dist/conversation.js +1 -1
  15. package/dist/coordination-driver-xO1tkxy-.js +3063 -0
  16. package/dist/coordination-driver-xO1tkxy-.js.map +1 -0
  17. package/dist/durable-file-g3YdfEh4.js +131 -0
  18. package/dist/durable-file-g3YdfEh4.js.map +1 -0
  19. package/dist/durable.d.ts +2 -2
  20. package/dist/durable.js +1 -1
  21. package/dist/{environment-provider-DWjbGH8W.d.ts → environment-provider-5E82esWE.d.ts} +6 -3
  22. package/dist/{environment-provider-Bn3652YU.js → environment-provider-DKyMWQJd.js} +121 -17
  23. package/dist/environment-provider-DKyMWQJd.js.map +1 -0
  24. package/dist/environment-provider.d.ts +1 -1
  25. package/dist/environment-provider.js +1 -1
  26. package/dist/{graph-BbeHgLXH.js → graph-DwJXa1Ph.js} +3 -3
  27. package/dist/graph-DwJXa1Ph.js.map +1 -0
  28. package/dist/graph.d.ts +3 -3
  29. package/dist/graph.js +4 -4
  30. package/dist/{improvement-cycle-DKGDCukL.js → improvement-cycle-eiSv6aH4.js} +5 -5
  31. package/dist/{improvement-cycle-DKGDCukL.js.map → improvement-cycle-eiSv6aH4.js.map} +1 -1
  32. package/dist/{index-CIhkH7mq.d.ts → index-BnAiQVLD.d.ts} +290 -29
  33. package/dist/{index-CFMB9ux3.d.ts → index-Bs0uhf-y.d.ts} +2 -2
  34. package/dist/{index-Dk5X9eKg.d.ts → index-C9XTLVP1.d.ts} +4 -4
  35. package/dist/index.d.ts +7 -7
  36. package/dist/index.js +13 -13
  37. package/dist/intelligence.d.ts +4 -4
  38. package/dist/intelligence.js +6 -6
  39. package/dist/kernel.d.ts +6 -6
  40. package/dist/kernel.js +13 -11
  41. package/dist/{knowledge-BaAX0HVu.js → knowledge-Bmtj_7u7.js} +5 -5
  42. package/dist/{knowledge-BaAX0HVu.js.map → knowledge-Bmtj_7u7.js.map} +1 -1
  43. package/dist/knowledge.d.ts +1 -1
  44. package/dist/knowledge.js +1 -1
  45. package/dist/{loop-runner-bin-DRYApMiu.d.ts → loop-runner-bin-ChS5deJ0.d.ts} +3 -3
  46. package/dist/{loop-runner-bin-D63Tytdk.js → loop-runner-bin-GTN5wqCt.js} +3 -3
  47. package/dist/{loop-runner-bin-D63Tytdk.js.map → loop-runner-bin-GTN5wqCt.js.map} +1 -1
  48. package/dist/loop-runner-bin.d.ts +1 -1
  49. package/dist/loop-runner-bin.js +1 -1
  50. package/dist/{materialization-Ct3I4wu3.js → materialization-CekWK6OO.js} +5 -3
  51. package/dist/materialization-CekWK6OO.js.map +1 -0
  52. package/dist/mcp/bin.js +3 -3
  53. package/dist/mcp/index.d.ts +3 -3
  54. package/dist/mcp/index.js +5 -4
  55. package/dist/mcp/index.js.map +1 -1
  56. package/dist/{openai-tools-BBgjWYNw.js → openai-tools-BBj1QGJp.js} +2 -2
  57. package/dist/{openai-tools-BBgjWYNw.js.map → openai-tools-BBj1QGJp.js.map} +1 -1
  58. package/dist/{prepare-BAyaNoZu.js → prepare-Tip-2ZoZ.js} +2 -2
  59. package/dist/{prepare-BAyaNoZu.js.map → prepare-Tip-2ZoZ.js.map} +1 -1
  60. package/dist/primeintellect/index.d.ts +1 -1
  61. package/dist/{protected-model-port-T2UWrDNQ.js → protected-model-port-Dd3a1SRm.js} +2 -2
  62. package/dist/{protected-model-port-T2UWrDNQ.js.map → protected-model-port-Dd3a1SRm.js.map} +1 -1
  63. package/dist/provision-supervisor-zA__ayUX.js +1020 -0
  64. package/dist/provision-supervisor-zA__ayUX.js.map +1 -0
  65. package/dist/{redact-DvLf4x1m.d.ts → redact-DkRLcSiE.d.ts} +3 -3
  66. package/dist/run-layout-Cd_l2XWV.js +645 -0
  67. package/dist/run-layout-Cd_l2XWV.js.map +1 -0
  68. package/dist/{runtime-D5u1M6T5.d.ts → runtime-B2eZ76kG.d.ts} +3 -3
  69. package/dist/{runtime-CojgP-hr.js → runtime-CJNlF5wM.js} +16 -1095
  70. package/dist/runtime-CJNlF5wM.js.map +1 -0
  71. package/dist/{spawn-journal-Tgvp5VS5.js → spawn-journal--N8Ten1q.js} +3 -1
  72. package/dist/{spawn-journal-Tgvp5VS5.js.map → spawn-journal--N8Ten1q.js.map} +1 -1
  73. package/dist/{stream-agent-turn-DxQ3GOuR.d.ts → stream-agent-turn-ByY8po9W.d.ts} +2 -2
  74. package/dist/{stream-agent-turn-CKIpCbSq.js → stream-agent-turn-CPj9SqVm.js} +2 -2
  75. package/dist/{stream-agent-turn-CKIpCbSq.js.map → stream-agent-turn-CPj9SqVm.js.map} +1 -1
  76. package/dist/{structural-rollout-BUYA61iV.js → structural-rollout-BeAUK7Ta.js} +4 -4
  77. package/dist/{structural-rollout-BUYA61iV.js.map → structural-rollout-BeAUK7Ta.js.map} +1 -1
  78. package/dist/{supervise-Ci0RfWQF.js → supervise-HsRKOlKY.js} +405 -3374
  79. package/dist/supervise-HsRKOlKY.js.map +1 -0
  80. package/dist/{supervisor-Bt1XLCVq.js → supervisor-ChLgoYhG.js} +1620 -33
  81. package/dist/supervisor-ChLgoYhG.js.map +1 -0
  82. package/dist/testing.d.ts +2 -2
  83. package/dist/testing.js +13 -12
  84. package/dist/testing.js.map +1 -1
  85. package/dist/{top-app-3rYfFPTW.js → top-app-DIBApkt1.js} +25 -9
  86. package/dist/top-app-DIBApkt1.js.map +1 -0
  87. package/dist/tui/bin.js +1 -1
  88. package/dist/tui/index.d.ts +9 -1
  89. package/dist/tui/index.js +3 -2
  90. package/dist/{types-DjVO-p7R.d.ts → types-a2ZZUl80.d.ts} +25 -2
  91. package/dist/{workspace-archive-B0Hqp2eJ.js → workspace-archive-DOZAeeKU.js} +2 -2
  92. package/dist/{workspace-archive-B0Hqp2eJ.js.map → workspace-archive-DOZAeeKU.js.map} +1 -1
  93. package/package.json +6 -4
  94. package/dist/candidate-execution-BbMbyhgC.js.map +0 -1
  95. package/dist/environment-provider-Bn3652YU.js.map +0 -1
  96. package/dist/graph-BbeHgLXH.js.map +0 -1
  97. package/dist/materialization-Ct3I4wu3.js.map +0 -1
  98. package/dist/run-layout-C2jgwsCq.js +0 -393
  99. package/dist/run-layout-C2jgwsCq.js.map +0 -1
  100. package/dist/runtime-CojgP-hr.js.map +0 -1
  101. package/dist/supervise-Ci0RfWQF.js.map +0 -1
  102. package/dist/supervisor-Bt1XLCVq.js.map +0 -1
  103. package/dist/top-app-3rYfFPTW.js.map +0 -1
@@ -1,3442 +1,472 @@
1
- import { C as runtimeOwnedScopeOwnerRuntime, S as runtimeOwnedPendingExecutorMaterialization, b as runtimeOwnedExecutorMaterialization, d as providerAttemptEvidence, f as recordRuntimeOwnedDriveHarnessProviderEvidence, i as attestRuntimeOwnedScopeOwner, s as inheritRuntimeOwnedExecutorAttestation, v as runtimeOwnedDriveHarnessProviderEvidence, x as runtimeOwnedExecutorProviderEvidence, y as runtimeOwnedExecutorExecutionBinding } from "./materialization-Ct3I4wu3.js";
2
- import { f as RuntimeRunStateError, i as ConfigError, m as ValidationError, o as NotFoundError, r as BackendTransportError, t as AgentEvalError$1 } from "./errors-CDZ8XsVj.js";
3
- import { n as detachedSnapshot, t as detachedFrozen } from "./snapshot-CTAf4uuA.js";
4
- import { b as workerTraceAnalysisStore, i as InMemorySpawnJournal, n as FileSpawnJournal, r as InMemoryResultBlobStore, t as FileResultBlobStore, x as contentAddress, y as parseWorkerToolTraceArtifact } from "./spawn-journal-Tgvp5VS5.js";
5
- import { _ as unmeteredSpend, r as chargedTokens, u as promptCacheTokenClasses } from "./util-D6ZEuBMi.js";
6
- import { i as writeAllBytes, n as parseCommittedJsonLines, r as prepareJsonlAppend, t as isNoEntError } from "./jsonl-file-CDfsCI5s.js";
1
+ import { C as runtimeOwnedScopeOwnerRuntime, S as runtimeOwnedPendingExecutorMaterialization, b as runtimeOwnedExecutorMaterialization, d as providerAttemptEvidence, f as recordRuntimeOwnedDriveHarnessProviderEvidence, i as attestRuntimeOwnedScopeOwner, s as inheritRuntimeOwnedExecutorAttestation, v as runtimeOwnedDriveHarnessProviderEvidence, x as runtimeOwnedExecutorProviderEvidence, y as runtimeOwnedExecutorExecutionBinding } from "./materialization-CekWK6OO.js";
2
+ import { f as RuntimeRunStateError, i as ConfigError, m as ValidationError, o as NotFoundError, r as BackendTransportError, t as AgentEvalError } from "./errors-CDZ8XsVj.js";
3
+ import { n as detachedSnapshot } from "./snapshot-CTAf4uuA.js";
4
+ import { x as contentAddress } from "./spawn-journal--N8Ten1q.js";
5
+ import { _ as unmeteredSpend } from "./util-D6ZEuBMi.js";
7
6
  import { a as concreteProfileModel, c as profileModelExecutionSettings, d as agentHarness, f as harnessRunsAgent, n as assertModelAllowed, o as enforceTokenLimits, r as assertProfileModelsAllowed, s as profileBridgeWireModel, t as assertExecutableAgentProfile } from "./model-policy-Sw4ywhtL.js";
8
- import { A as DEFAULT_SUCCESSFUL_SHUTDOWN_MS, B as captureReusableExecutorConfig, D as freeSlots, Dt as runBrainLoop, Et as routerBrain, F as bindReusableExecutorExecutionId, G as snapshotExecutorConfig, Ht as generateSpanId, I as bridgeAdmissionRead, It as buildLoopSpanNodes, Kt as toOtelAttributes, L as bridgeModelRouteRefusal, M as assertValidBudget, P as spendFromUsageEvents, R as bridgeRuntimeAttachmentsKey, Sn as unsupportedProfileDimensions, U as createExecutor, W as createExecutorRegistry, _ as recordScopeOwnerMaterialization, bn as renderUnsupported, c as pickBestDelivered, d as driverChild, dn as controlProfileMaterialization, dt as createPeerMailbox, f as withDriverExecutor, fn as defineProfileMaterializationContract, g as meterRuntimeOwnedProviderAttempt, gn as promptModelProfileMaterialization, gt as isTerminalNodeStatus, h as meterRuntimeOwnedAccounting, hn as promptControlProfileMaterialization, ht as isLiveNodeStatus, j as teardownExecutor, l as runFinalizer, m as deriveNodeExecutionIdentity, mn as profileMaterializationAxes$1, n as createSupervisor, o as bestDelivered, ot as createInbox, pn as fullProfileMaterialization, q as WORKER_TRACE_PROPAGATION, t as createRootHandle, u as runTree, un as assertProfileMaterialization, v as scopeOwnerExecutorNodeContext, wn as worktreeCliProfileMaterialization, z as bridgeStopSignalKey, zt as createOtelExporter } from "./supervisor-Bt1XLCVq.js";
7
+ import { $ as bridgeRuntimeAttachmentsKey, An as profileMaterializationAxes$1, Bn as worktreeCliProfileMaterialization, Dn as controlProfileMaterialization, En as assertProfileMaterialization, G as DEFAULT_SUCCESSFUL_SHUTDOWN_MS, Ht as routerBrain, In as renderUnsupported, K as teardownExecutor, Mn as promptModelProfileMaterialization, On as defineProfileMaterializationContract, Q as bridgeModelRouteRefusal, Qt as buildLoopSpanNodes, Rn as unsupportedProfileDimensions, X as bindReusableExecutorExecutionId, Y as spendFromUsageEvents, Z as bridgeAdmissionRead, _ as recordScopeOwnerMaterialization, cn as toOtelAttributes, ct as WORKER_TRACE_PROPAGATION, d as driverChild, et as bridgeStopSignalKey, g as meterRuntimeOwnedProviderAttempt, h as meterRuntimeOwnedAccounting, in as generateSpanId, it as createExecutor, jn as promptControlProfileMaterialization, kn as fullProfileMaterialization, l as runFinalizer, m as deriveNodeExecutionIdentity, n as createSupervisor, o as bestDelivered, ot as snapshotExecutorConfig, q as assertValidBudget, t as createRootHandle, tn as createOtelExporter, tt as captureReusableExecutorConfig, u as runTree, v as scopeOwnerExecutorNodeContext, xt as createInbox } from "./supervisor-ChLgoYhG.js";
9
8
  import { t as composeRuntimeHooks } from "./runtime-hooks-tXpAarhW.js";
10
- import { C as writeWorkerCancellation, S as writeRunCancellation, a as readRunCancelRequest, c as readWorkerCancellation, o as readRunCancellation, s as readWorkerCancelRequests } from "./run-layout-C2jgwsCq.js";
9
+ import { C as coordinationVerbNames, c as createProgressTracker, d as progressStop, r as driverAgent, v as createFileRunContext, w as createCoordinationTools, y as createInMemoryRunContext } from "./coordination-driver-xO1tkxy-.js";
10
+ import { k as writeRunCancellation, o as readRunCancelRequest, s as readRunCancellation } from "./run-layout-Cd_l2XWV.js";
11
11
  import { t as createStdioToolServer } from "./tool-server-DEmLr9YY.js";
12
12
  import { n as UI_LENSES } from "./substrate-B0TYNrXn.js";
13
13
  import { agentProfileSchema, canonicalAgentProfileDigest, canonicalCandidateDigest, validateAgentProfileSecurity } from "@tangle-network/agent-interface";
14
- import { argHash, errorStreakDetector, observeAll, repeatedActionDetector } from "@tangle-network/agent-eval";
15
14
  import { randomUUID } from "node:crypto";
16
- import path, { dirname, join, resolve } from "node:path";
15
+ import path, { dirname, resolve } from "node:path";
17
16
  import { isMaterializerHarness } from "@tangle-network/agent-profile-materialize";
18
17
  import { mkdir, readFile, rename, writeFile } from "node:fs/promises";
19
- import { readFileSync } from "node:fs";
20
18
  import { createServer } from "node:http";
21
19
  import { Readable, Writable } from "node:stream";
22
- //#region src/runtime/supervise/detector-monitor.ts
20
+ //#region src/runtime/supervise/completion-gate.ts
23
21
  /**
24
22
  *
25
- * The ONLINE analyst: watch a `TraceSource` and fold each tool span through agent-eval's published
26
- * streaming detector kernel (`repeatedActionDetector`/`errorStreakDetector` — the SAME kernel the
27
- * control loop folds), firing `onSignal` the moment a worker loops or error-storms. Substrate-
28
- * agnostic: it consumes spans from any source (owned router/bridge loop OR a sandbox box session),
29
- * never the raw tool seam. Detection logic + the failure taxonomy live in agent-eval; not reimplemented.
23
+ * The completion-oracle: **settled DELIVERED.**
24
+ *
25
+ * Foreman's one hard lesson (0/18 self-improvement deliverables) "done" must mean a check
26
+ * PASSED, not the agent's say-so. `gateOnDeliverable` wraps an `Executor` so its settlement
27
+ * is `valid` ONLY when the deliverable check passes. The child still RUNS and settles (its
28
+ * spend is conserved into the pool either way), but a child that ran WITHOUT delivering
29
+ * settles `valid:false` — so a keep-best driver never counts it as done, and a gate never
30
+ * inflates with self-judged wins.
31
+ *
32
+ * Dual-purpose by construction:
33
+ * - product: the agent fleet only advances on real, checked deliverables.
34
+ * - proof: the gate's `valid` is the honest settle — equal-k comparisons can't be gamed by an
35
+ * arm that "ran" without producing the artifact.
36
+ *
37
+ * The check is a DEPLOYABLE oracle — a test command, a state verifier, the commit0 judge —
38
+ * read off the child's output, never the model judging itself. A throwing check is
39
+ * fail-closed (not delivered), never a crash.
30
40
  *
31
41
  * @experimental
32
42
  */
33
- /** The default online panel for a tool-call pipe: a worker repeating the same call, or hammering
34
- * consecutive errors. (No-progress needs a domain progress-probe, so it is opt-in, not default.)
35
- *
36
- * Coverage note: `repeated-action` works for EVERY harness (it needs only tool name + args, which
37
- * every adapter provides). `error-streak` needs per-call status opencode carries it inline
38
- * (`state.status`, VALIDATED live), but claude-code/codex tool-call parts do NOT (their errors live
39
- * in separate result blocks not yet decoded), so error-streak is silent for those until result-block
40
- * decoding is added + live-validated. It is in the panel because it is correct where status exists. */
41
- function defaultToolDetectors() {
42
- return [repeatedActionDetector({ maxRepeated: 3 }), errorStreakDetector({ maxErrors: 3 })];
43
- }
44
- /** Subscribe to a `TraceSource` and run the streaming detectors over its live spans. Returns an
45
- * unsubscribe. A defensive `argHash` failure (circular args) never throws out of the side-channel. */
46
- function watchTrace(source, opts = {}) {
47
- const detectors = opts.detectors ?? defaultToolDetectors();
48
- return source.onSpan((span) => {
49
- let fingerprint;
43
+ /**
44
+ * Wrap an `Executor` so its settlement `valid` reflects the deliverable check, not the
45
+ * inner verdict. Handles both `execute` shapes (one-shot `Promise<ExecutorResult>` and
46
+ * streaming `AsyncIterable<UsageEvent>` + `resultArtifact()`); the check runs once the inner
47
+ * executor has produced its output. The inner `score` is preserved; only `valid` is gated.
48
+ */
49
+ function gateOnDeliverable(inner, deliverable) {
50
+ let gated;
51
+ const check = async (out, baseScore) => {
52
+ let delivered;
50
53
  try {
51
- fingerprint = `${span.toolName}|${argHash(span.args)}`;
54
+ delivered = await deliverable.check(out) === true;
52
55
  } catch {
53
- fingerprint = `${span.toolName}|<unhashable>`;
56
+ delivered = false;
54
57
  }
55
- const signals = observeAll(detectors, {
56
- actionFingerprint: fingerprint,
57
- ...span.status ? { status: span.status } : {},
58
- label: span.toolName
59
- });
60
- for (const s of signals) opts.onSignal?.(s, span);
61
- });
62
- }
63
- //#endregion
64
- //#region src/runtime/supervise/event-bus.ts
65
- /** Create the child→parent coordination bus: one typed pipe for settled outputs, questions, and analyst findings, with a priority-ordered pull queue and a pass-through subscribe lane.
66
- * @experimental In-process queue; durability is a transport swap that does not exist yet. */
67
- function createEventBus(now = Date.now) {
68
- const queue = [];
69
- const log = [];
70
- const subscribers = [];
71
- const byKind = {};
72
- const staged = /* @__PURE__ */ new WeakMap();
73
- let seq = 0;
74
- let published = 0;
75
- let pulled = 0;
76
- const matches = (r, kinds) => !kinds || kinds.includes(r.event.type);
77
- const bestIndex = (kinds) => {
78
- let best = -1;
79
- let bestPriority = Number.NEGATIVE_INFINITY;
80
- for (let i = 0; i < queue.length; i++) {
81
- const r = queue[i];
82
- if (!r || !matches(r, kinds)) continue;
83
- if (r.priority > bestPriority) {
84
- best = i;
85
- bestPriority = r.priority;
86
- }
58
+ return {
59
+ valid: delivered,
60
+ score: baseScore ?? (delivered ? 1 : 0)
61
+ };
62
+ };
63
+ /**
64
+ * Ask the delivery question once, from whatever the inner executor managed to produce.
65
+ *
66
+ * Fail-closed on the artifact being unavailable: an executor that never produced one delivered
67
+ * nothing, and leaving `gated` unset keeps the existing invalid-by-default reading.
68
+ */
69
+ const settleVerdict = async () => {
70
+ let art;
71
+ try {
72
+ art = inner.resultArtifact();
73
+ } catch {
74
+ return;
87
75
  }
88
- return best;
76
+ gated = await check(art.out, art.verdict?.score);
89
77
  };
90
- return {
91
- async publish(event, opts) {
92
- const record = staged.get(event) ?? {
93
- seq: seq++,
94
- at: now(),
95
- priority: opts?.priority ?? 0,
96
- event
97
- };
98
- staged.set(event, record);
99
- for (const handler of subscribers) await handler(record);
100
- staged.delete(event);
101
- if (opts?.queue !== false) queue.push(record);
102
- log.push(record);
103
- published += 1;
104
- byKind[event.type] = (byKind[event.type] ?? 0) + 1;
105
- return record;
106
- },
107
- pull(kinds) {
108
- const i = bestIndex(kinds);
109
- if (i < 0) return void 0;
110
- pulled++;
111
- return queue.splice(i, 1)[0]?.event;
112
- },
113
- subscribe(handler) {
114
- subscribers.push(handler);
115
- return () => {
116
- const i = subscribers.indexOf(handler);
117
- if (i >= 0) subscribers.splice(i, 1);
118
- };
119
- },
120
- pending(kinds) {
121
- return kinds ? queue.filter((r) => matches(r, kinds)).length : queue.length;
122
- },
123
- history() {
124
- return log;
78
+ return inheritRuntimeOwnedExecutorAttestation(inner, {
79
+ runtime: inner.runtime,
80
+ ...inner.budgetExempt !== void 0 ? { budgetExempt: inner.budgetExempt } : {},
81
+ ...inner.deliver ? { deliver: (m) => inner.deliver?.(m) } : {},
82
+ ...inner.progress ? { progress: () => inner.progress?.() } : {},
83
+ ...inner.traceSource ? { traceSource: () => inner.traceSource?.() } : {},
84
+ ...inner.metered ? { metered: () => inner.metered?.() } : {},
85
+ execute(task, signal) {
86
+ const r = inner.execute(task, signal);
87
+ if (isAsyncIterable$1(r)) return (async function* () {
88
+ try {
89
+ for await (const ev of r) yield ev;
90
+ } finally {
91
+ await settleVerdict();
92
+ }
93
+ })();
94
+ return (async () => {
95
+ let res;
96
+ try {
97
+ res = await r;
98
+ } catch (error) {
99
+ await settleVerdict();
100
+ throw error;
101
+ }
102
+ gated = await check(res.out, res.verdict?.score);
103
+ return {
104
+ ...res,
105
+ verdict: gated
106
+ };
107
+ })();
125
108
  },
126
- stats() {
109
+ teardown: (grace) => inner.teardown(grace),
110
+ resultArtifact() {
111
+ const art = inner.resultArtifact();
127
112
  return {
128
- published,
129
- pulled,
130
- byKind: { ...byKind }
113
+ ...art,
114
+ verdict: gated ?? art.verdict
131
115
  };
132
116
  }
133
- };
117
+ });
134
118
  }
135
- //#endregion
136
- //#region src/mcp/tools/coordination.ts
137
119
  /**
138
- *
139
- * MCP binding for a live `Scope`. A sandbox driver gets the same small verbs
140
- * the in-process driver has: spawn, observe, await, steer, ask/answer, analyze,
141
- * and stop. Settled outputs remain Scope artifacts; product code can project
142
- * them into any UI/report envelope it needs.
143
- *
144
- * @experimental
120
+ * Transform a Runtime executor's terminal artifact without losing its private
121
+ * profile-materialization attestation or altering its measured spend. This is
122
+ * the composition point for deterministic post-processing and grading; callers
123
+ * must not rebuild an Executor around a model transport merely to change `out`.
145
124
  */
146
- /** Where a question this driver cannot answer goes next. `answer_question` accepts these and
147
- * nothing else, so the decision type states them and nothing else. */
148
- const questionEscalationTargets = ["parent", "user"];
149
- const isQuestionEscalationTarget = (value) => questionEscalationTargets.includes(value);
150
- /** Producer-side cleanliness for the `finding` event. The findings payload is arbitrary analyst
151
- * output, the digest a subscriber computes (RFC 8785) throws on ANY `undefined` value — nested
152
- * included — and a throwing subscriber leaves the event invisible to EVERY subscriber. The
153
- * producer, not the digest, owns keeping the event canonical: an `undefined` payload is stripped
154
- * to key-absence, everything else is JSON round-tripped (nested `undefined` object values drop,
155
- * `undefined` array slots become `null`), and a payload JSON cannot represent at all (cycle,
156
- * BigInt, bare function) becomes a record OF that fact — degraded findings beat a vanished
157
- * event. */
158
- function canonicalFindingEvent(finding) {
159
- if (finding.findings === void 0) {
160
- const { findings: _absent, ...present } = finding;
161
- return present;
162
- }
163
- try {
164
- return {
165
- ...finding,
166
- findings: JSON.parse(JSON.stringify(finding.findings))
167
- };
168
- } catch (error) {
169
- return {
170
- ...finding,
171
- findings: { nonCanonicalFindings: error instanceof Error ? error.message : String(error) }
125
+ function mapExecutorResult(inner, map) {
126
+ let mapped;
127
+ const settle = async (result, task) => {
128
+ const transformed = await map(result, task);
129
+ mapped = {
130
+ outRef: transformed.outRef,
131
+ out: transformed.out,
132
+ ...transformed.verdict ? { verdict: transformed.verdict } : {},
133
+ spent: result.spent
172
134
  };
173
- }
174
- }
175
- /** Normalize the two spellings of an analyst-on-settle entry to the route form. */
176
- function normalizeAnalyzeOnSettle(entry) {
177
- return typeof entry === "string" ? { kind: entry } : entry;
178
- }
179
- /** Every cause at zero — a pre-flight publishes its whole ledger from the first read. */
180
- function emptyPreflightCounts() {
181
- return {
182
- "model-route": 0,
183
- "bridge-full": 0,
184
- "unmountable-tool": 0
135
+ return mapped;
185
136
  };
137
+ return inheritRuntimeOwnedExecutorAttestation(inner, {
138
+ runtime: inner.runtime,
139
+ ...inner.budgetExempt !== void 0 ? { budgetExempt: inner.budgetExempt } : {},
140
+ ...inner.deliver ? { deliver: (message) => inner.deliver?.(message) } : {},
141
+ ...inner.progress ? { progress: () => inner.progress?.() } : {},
142
+ ...inner.traceSource ? { traceSource: () => inner.traceSource?.() } : {},
143
+ ...inner.accounting ? { accounting: () => inner.accounting?.() } : {},
144
+ ...inner.metered ? { metered: () => inner.metered?.() } : {},
145
+ execute(task, signal) {
146
+ const execution = inner.execute(task, signal);
147
+ if (isAsyncIterable$1(execution)) return (async function* () {
148
+ for await (const event of execution) yield event;
149
+ await settle(inner.resultArtifact(), task);
150
+ })();
151
+ return (async () => settle(await execution, task))();
152
+ },
153
+ teardown: (grace) => inner.teardown(grace),
154
+ resultArtifact() {
155
+ if (!mapped) throw new Error("mapExecutorResult: resultArtifact() read before execute()");
156
+ return mapped;
157
+ }
158
+ });
186
159
  }
187
- /** Default ceiling for a single `await_event` block (ms). Chosen well under any reasonable remote
188
- * MCP client request timeout so the call returns a `pending` liveness snapshot instead of erroring;
189
- * the supervisor re-polls until the worker settles. */
190
- const DEFAULT_AWAIT_EVENT_TIMEOUT_MS = 15e3;
191
- /** The reserved coordination verb names — the complete set `createCoordinationTools` can emit
192
- * (the analyst pair is conditional but still reserved). A driver's extra WORK tools must not
193
- * collide with any of these, or it could no longer coordinate; callers validate eagerly against
194
- * this set so the conflict fails loud at construction, not buried in a swallowed `act()` throw. */
195
- const coordinationVerbNames = [
196
- "spawn_agent",
197
- "observe_agent",
198
- "steer_agent",
199
- "await_event",
200
- "list_questions",
201
- "answer_question",
202
- "ask_parent",
203
- "submit_result",
204
- "stop",
205
- "list_analysts",
206
- "run_analyst"
207
- ];
208
- /**
209
- * The `CoordinationEvent` kinds a driver may name in `await_event`. The pull queue carries the
210
- * UP-leg only: `steer` / `answer` / `instruction` / `delivery-attempt` are recorded `queue: false`
211
- * (history and subscribers, never pulled back), and `mail` is delivered to its addressee's inbox.
212
- *
213
- * Declared once because the tool advertises this list in its JSON Schema AND filters on it at
214
- * dispatch. Written twice, the two drift and the schema promises a kind the filter drops — a
215
- * driver then blocks on a queue that already holds its event.
216
- */
217
- const awaitableEventKinds = [
218
- "settled",
219
- "question",
220
- "finding"
221
- ];
222
- function isAwaitableEventKind(value) {
223
- return awaitableEventKinds.includes(value);
160
+ function isAsyncIterable$1(v) {
161
+ return v != null && typeof v[Symbol.asyncIterator] === "function";
224
162
  }
225
- const idArg = {
226
- type: "string",
227
- description: "The workerId returned by spawn_agent."
228
- };
163
+ //#endregion
164
+ //#region src/runtime/supervise/otel-spans.ts
229
165
  /**
230
- * Strip zod's object-KEY CODEC artifact from a derived JSON Schema, at every depth.
231
- *
232
- * `z.record(z.string(), …)` converts to `propertyNames: { type: 'string', pattern:
233
- * '^u(?:[0-9a-f]{4})*$' }` — a marker of the lossy key round-trip, not a constraint anything
234
- * enforces: `agentProfileSchema.safeParse({ name: 'r', tools: { bash: true, Read: false } })`
235
- * succeeds with those plain keys. Published verbatim it reads to a model as "every key must be a
236
- * run of hex quads", and the model either emits hex-encoded garbage keys or drops the field —
237
- * on `tools`, `permissions`, `metadata`, `mcp`, `mcp.*.env`, and `model.metadata`, which is most
238
- * of what a parent actually configures.
239
- *
240
- * Every `propertyNames` in the canonical profile schema is this artifact (25 of 25 at Interface
241
- * 0.40), and the canonical schema constrains no key by pattern, so dropping the keyword outright
242
- * loses nothing real and cannot be defeated by zod changing the encoding's exact regex.
166
+ * Supervisor tree OTLP spans. OPT-IN, off by default.
243
167
  *
244
- * Rebuilds rather than mutates: the canonical conversion output must stay untouched for callers
245
- * that compare against it.
246
- */
247
- const stripKeyCodecArtifacts = (node) => {
248
- if (Array.isArray(node)) return node.map(stripKeyCodecArtifacts);
249
- if (!node || typeof node !== "object") return node;
250
- return Object.fromEntries(Object.entries(node).filter(([key]) => key !== "propertyNames").map(([key, value]) => [key, stripKeyCodecArtifacts(value)]));
251
- };
252
- /** The canonical profile fields a spawning parent actually sets on a child, in the order a reader
253
- * needs them, each with the description published alongside it. Everything else stays legal to
254
- * pass — see {@link deriveSpawnProfileArg}.
168
+ * WHY. A supervised tree is legible today only by parsing this package's own spawn journal, so
169
+ * every other multi-agent shape on the machine (a coding-CLI's subagents, a pi fanout, ad-hoc tool
170
+ * parallelism) needs its own bespoke reader. A span carrying `parent_span_id` IS a tree, and any
171
+ * system can emit one — so emitting spans makes the supervisor readable by the same viewer as
172
+ * everything else, with no per-system reader.
255
173
  *
256
- * Why these eleven, with the measured numbers (serialized JSON bytes, Interface 0.40). The
257
- * canonical `properties` map is 10629 bytes, against 2424 bytes for every argument of every other
258
- * coordination tool COMBINED publishing it whole makes one parameter four times the rest of the
259
- * surface a driver re-reads each turn. Of the seven omitted fields (tags, connections, subagents,
260
- * hooks, modes, confidential, extensions) none is something a parent hands a worker; `permissions`
261
- * (321 bytes) IS, so it is published — a child that must not touch the network or the filesystem
262
- * is fenced there and nowhere else. `mcp` (2663 bytes) and `resources` (3122 bytes) are the two a
263
- * parent is least likely to author inline and were together 85% of the published cost, so they
264
- * carry a brief shape plus a description naming the full form instead of the canonical sub-tree. */
265
- const spawnProfileFields = [
266
- {
267
- name: "name",
268
- description: "Short identifier for this child, e.g. \"researcher\" or \"patch-writer\". A child normally sets it to its role; it shows up in traces and worker labels."
269
- },
270
- {
271
- name: "description",
272
- description: "One line saying what this child is for. Read by humans and by a parent listing its workers."
273
- },
274
- {
275
- name: "version",
276
- description: "Optional version string for this profile, so two revisions of the same role are distinguishable in evidence. A child usually omits it."
277
- },
278
- {
279
- name: "harness",
280
- description: "Which coding backend runs the child. Omit to inherit the run default; set it only when this child needs a specific backend (e.g. a long refactor on \"claude-code\")."
281
- },
282
- {
283
- name: "model",
284
- description: "Model routing: `default` model id, optional `small` for cheap sub-calls, `provider`, and `reasoningEffort`. A child typically sets `default` and `reasoningEffort` — raise effort for a hard reasoning task, lower it for bulk mechanical work."
285
- },
286
- {
287
- name: "prompt",
288
- description: "The child's standing instructions: `systemPrompt` (who it is and how it works) and `instructions` (durable rules). This is the role; the separate `task` argument carries what to do right now. A child with no systemPrompt has no role — always set one."
289
- },
290
- {
291
- name: "tools",
292
- description: "Per-tool on/off map keyed by tool name, e.g. `{ \"bash\": true, \"webfetch\": false }`. Omit to inherit the backend's default tool set; set it to narrow a child to the tools its task needs. Keys are plain tool names."
293
- },
294
- {
295
- name: "permissions",
296
- description: "Per-tool permission decisions: each tool name maps to \"allow\", \"deny\", or \"ask\", or to a nested map of finer-grained rules. This is where a parent fences a child it does not fully trust — a read-only child denies write and network tools here, not in `tools`."
297
- },
298
- {
299
- name: "mcp",
300
- description: "MCP tool servers to mount for the child, keyed by server name. Brief form: each value is either `{ transport: \"stdio\", command, args?, env?, cwd? }` or `{ transport: \"sse\" | \"http\", url, headers? }`. A child normally sets this only when it needs a tool server the task requires; the canonical AgentProfile schema carries the full form (secret-ref values for env/headers, per-server metadata) and governs validation.",
301
- brief: {
302
- type: "object",
303
- additionalProperties: { type: "object" }
304
- }
305
- },
306
- {
307
- name: "resources",
308
- description: "Files, tools, skills, and agents materialized into the child workspace before it starts. Brief form: `{ files?, tools?, skills?, agents? }`, each an array whose entries are `{ kind: \"inline\", name, content }` or `{ kind: \"github\", repository?, path, ref? }` (`files` entries wrap that as `{ path, resource, executable? }`). A child typically gets `files` for seed inputs and `skills` for a procedure it must follow; the canonical AgentProfile schema carries the full form and governs validation.",
309
- brief: {
310
- type: "object",
311
- properties: {
312
- files: {
313
- type: "array",
314
- items: { type: "object" }
315
- },
316
- tools: {
317
- type: "array",
318
- items: { type: "object" }
319
- },
320
- skills: {
321
- type: "array",
322
- items: { type: "object" }
323
- },
324
- agents: {
325
- type: "array",
326
- items: { type: "object" }
327
- }
328
- },
329
- additionalProperties: true
330
- }
331
- },
332
- {
333
- name: "metadata",
334
- description: "Free-form key/value bag carried with the profile (plain string keys). Use it for run bookkeeping a reader will want later; it does not change how the child executes."
335
- }
336
- ];
337
- /**
338
- * Build the published shape of `spawn_agent`'s `profile` argument from the canonical
339
- * `agentProfileSchema` conversion's `properties` map, so it cannot drift from the profile the
340
- * runtime materializes.
174
+ * WHAT IT IS NOT. This is telemetry, never the record of truth. The spawn journal remains the sole
175
+ * durable ledger for replay/resume and for cost; nothing here is read back, and a run whose export
176
+ * fails is unaffected in every observable way. The two data models are deliberately separate.
341
177
  *
342
- * DEGRADES, never throws. A canonical field that is absent renamed or removed upstream — is
343
- * simply omitted from the published shape, and a canonical schema that is no longer an object
344
- * publishes no properties at all. This function is reached from a statically-imported module, so a
345
- * throw here bricks `import '@tangle-network/agent-runtime/kernel'` for every consumer over an
346
- * upstream rename that costs them, at worst, one advisory field. The drift itself is still caught
347
- * loudly — as a test assertion in `tests/kernel/coordination.test.ts`, in CI, where it is our
348
- * problem rather than at a consumer's import, where it is theirs.
178
+ * HOW IT ATTACHES. This is a pure `RuntimeHooks` observer over the lifecycle events `Scope` ALREADY
179
+ * emits `agent.spawn` (a node opened), `agent.child` (a node settled), `agent.turn` (a driver
180
+ * inference turn was metered). It adds no event, mutates no journal, and changes no result. Because
181
+ * `Scope` re-seeds the same hooks into every nested scope (`makeNestedScopeSeam`), one observer sees
182
+ * the WHOLE recursion at arbitrary depth.
349
183
  *
350
- * Permissive on purpose: `additionalProperties: true` with no `required` list, so every canonical
351
- * field this shape omits stays legal to pass. This tool layer performs no profile validation.
184
+ * SPAN SHAPE. One span per supervised node, opened at spawn and closed at settle, parented to its
185
+ * parent node's span; the run root is the trace. Driver inference rides as an `LLM` child span under
186
+ * the node that metered it. Attributes reuse the vocabulary the rest of the stack already reads
187
+ * (`openinference.span.kind`, `agent.name`, `llm.token_count.*`, `llm.cost_usd`, `tangle.cost.usd`)
188
+ * — see `@tangle-network/agent-eval`'s `src/trace/attribute-vocabulary.ts`, which is the consumer.
352
189
  *
353
- * @internal exported for the drift and degradation tests; not part of the package's public API.
190
+ * UNKNOWN IS NEVER ZERO. `Spend.tokensKnown === false` / `usdKnown === false` mark work that
191
+ * HAPPENED with an unreported count. Those spans OMIT the token/cost attribute entirely and set
192
+ * `tangle.supervise.tokens_known` / `tangle.supervise.cost_known` to `false`, so a reader can never
193
+ * mistake an unmeasured turn for a free one.
354
194
  */
355
- function deriveSpawnProfileArg(canonicalProperties) {
356
- const published = [];
357
- for (const field of spawnProfileFields) {
358
- const canonical = canonicalProperties?.[field.name];
359
- if (canonical === void 0) continue;
360
- const shape = field.brief ?? stripKeyCodecArtifacts(canonical);
361
- published.push([field.name, {
362
- ...shape,
363
- description: field.description
364
- }]);
365
- }
366
- return {
367
- type: "object",
368
- description: "The child agent profile to run — this is the shape the DEFAULT worker seam accepts; a run wired with a custom makeWorkerAgent may accept a different one. The properties below are derived from the canonical AgentProfile schema and reduced to what a spawning parent sets (`mcp` and `resources` in a brief form); every other canonical field — tags, connections, subagents, hooks, modes, confidential, extensions — may still be passed. This tool does not validate the profile: the canonical AgentProfile schema governs validation downstream.",
369
- properties: Object.fromEntries(published),
370
- additionalProperties: true
195
+ /** OTEL status codes (`UNSET` / `OK` / `ERROR`) — the numeric wire values `OtelSpan.status` carries. */
196
+ const STATUS_UNSET = 0;
197
+ const STATUS_OK = 1;
198
+ const STATUS_ERROR = 2;
199
+ /** Longest string attribute value written from free-form detail, so an oversized turn payload
200
+ * cannot inflate a span. Identity/label attributes we control are never truncated. */
201
+ const MAX_DETAIL_CHARS = 256;
202
+ /**
203
+ * Build the span recorder for one supervised run, or `undefined` when no exporter resolves — the
204
+ * off-by-default path. A run that passes no `exporter` and no `exportConfig` never reaches this
205
+ * function at all; one that passes an `exportConfig` with no endpoint (and no env endpoint) gets
206
+ * `undefined` here, so "configured but unreachable" also costs nothing.
207
+ */
208
+ function createSupervisorSpanRecorder(opts) {
209
+ const exporter = opts.exporter ?? createOtelExporter(opts.exportConfig);
210
+ if (!exporter) return void 0;
211
+ const ownsExporter = opts.exporter === void 0;
212
+ const now = opts.now ?? Date.now;
213
+ const traceId = normalizeTraceId(opts.traceId, opts.runId);
214
+ const rootSpanId = generateSpanId();
215
+ const rootStartMs = now();
216
+ const base = {
217
+ "tangle.run.id": opts.runId,
218
+ "tangle.sessionId": opts.runId,
219
+ ...opts.attributes ?? {}
371
220
  };
372
- }
373
- spawnProfileFields.map((f) => f.name);
374
- let spawnProfileArgCache;
375
- /** The published `profile` shape, computed on FIRST tool-definition access and memoized — not at
376
- * module load. The conversion walks the whole canonical profile tree, and the strip walk rebuilds
377
- * it: 3.5ms on the first `createCoordinationTools`, 0.017ms on every later one (measured, Interface
378
- * 0.40, node 24). `src/runtime/index.ts` imports this module statically, so paying that at import
379
- * taxes every consumer of the kernel entrypoint — including the ones that never build a
380
- * coordination toolbox. The memo keeps it at once per process for the ones that do.
381
- *
382
- * Conversion choices. `io: 'input'` is the CALLER's view — pre-default, pre-transform — which is
383
- * what a spawning parent may pass, not what the runtime ends up holding. `unrepresentable: 'any'`
384
- * keeps the conversion total: the canonical schema contains transforms with no JSON Schema form,
385
- * and zod's default is to throw on them, which would leave the tool with no published shape. */
386
- function spawnProfileArg() {
387
- if (!spawnProfileArgCache) spawnProfileArgCache = detachedFrozen(deriveSpawnProfileArg(agentProfileSchema.toJSONSchema({
388
- io: "input",
389
- target: "draft-07",
390
- unrepresentable: "any"
391
- }).properties));
392
- return spawnProfileArgCache;
393
- }
394
- /** Build the driver's MCP tools over a live scope. */
395
- function createCoordinationTools(opts) {
396
- const deliverable = opts.deliverable;
397
- let stopped = false;
398
- let reason;
399
- let stopNotified = false;
400
- let submitted;
401
- let questionSeq = 0;
402
- const ledger = [];
403
- const questions = [...opts.priorQuestions ?? []];
404
- const questionPolicy = opts.questionPolicy ?? "auto";
405
- const notifyStop = () => {
406
- if (stopNotified) return;
407
- stopNotified = true;
408
- opts.onStop?.(reason);
221
+ /** Node id → its open span. Seeded with the run id ⇒ the root span, because a depth-0 spawn's
222
+ * `parentId` is the run id itself and every deeper spawn's is a real node id. */
223
+ const open = /* @__PURE__ */ new Map();
224
+ const spanIdOf = /* @__PURE__ */ new Map([[opts.runId, rootSpanId]]);
225
+ let finished = false;
226
+ /** Every export is best-effort: a throwing exporter must never reach the run. */
227
+ const emit = (span) => {
228
+ try {
229
+ exporter.exportSpan(span);
230
+ } catch {}
409
231
  };
410
- const completedKeys = /* @__PURE__ */ new Set();
411
- const keyByWorker = /* @__PURE__ */ new Map();
412
- const profileNameByWorker = /* @__PURE__ */ new Map();
413
- const liveHandles = /* @__PURE__ */ new Map();
414
- let unkeyedAssignmentOrdinal = nextUnkeyedAssignmentOrdinal(opts.scope);
415
- const preflightCounts = emptyPreflightCounts();
416
- for (const [key, prior] of opts.scope.resume?.keys ?? []) if (prior.state === "completed") completedKeys.add(key);
417
- const nodeForWorker = (id) => opts.scope.view.nodes.find((node) => node.id === id) ?? opts.scope.resume?.view.nodes.find((node) => node.id === id);
418
- const projectSettled = (settled, resumed = false) => {
419
- const node = nodeForWorker(settled.handle.id);
420
- const assignmentId = settled.handle.assignmentId ?? node?.assignmentId;
421
- const identity = settled.handle.identity ?? node?.identity;
422
- const materialization = settled.handle.materialization ?? node?.materialization;
423
- const executionBindings = settled.handle.executionBindings ?? node?.executionBindings;
424
- const settledAt = settled.settledAt ?? node?.settledAt;
425
- const trace = settled.trace ?? node?.trace ?? {
426
- status: "unavailable",
427
- reason: "legacy-settlement-without-trace-evidence"
428
- };
429
- const common = {
430
- id: settled.handle.id,
431
- ...assignmentId === void 0 ? {} : { assignmentId },
432
- ...identity === void 0 ? {} : { identity },
433
- ...materialization === void 0 ? {} : { materialization },
434
- ...executionBindings === void 0 ? {} : { executionBindings },
435
- ...settledAt === void 0 ? {} : { settledAt },
436
- trace,
437
- ...resumed ? { resumed: true } : {}
438
- };
439
- return detachedFrozen(settled.kind === "done" ? {
440
- ...common,
441
- status: "done",
442
- spent: settled.spent,
443
- ...settled.verdict?.score === void 0 ? {} : { score: settled.verdict.score },
444
- ...settled.verdict?.valid === void 0 ? {} : { valid: settled.verdict.valid },
445
- outRef: settled.outRef
446
- } : {
447
- ...common,
448
- status: "down",
449
- ...node?.spent === void 0 ? {} : { spent: node.spent },
450
- reason: settled.reason
451
- });
452
- };
453
- const resumedWorkers = [];
454
- for (const s of opts.scope.resume?.settled ?? []) {
455
- const worker = projectSettled(s, true);
456
- resumedWorkers.push(worker);
457
- ledger.push(worker);
458
- }
459
- const bus = createEventBus();
460
- if (opts.onEvent) {
461
- const cb = opts.onEvent;
462
- bus.subscribe((rec) => cb(rec.event, rec));
463
- }
464
- const resumeEvents = opts.replaySettlements ? resumedWorkers.map((worker) => detachedFrozen({
465
- type: "settled",
466
- worker
467
- })) : [];
468
- let resumeEventIndex = 0;
469
- let readyInFlight;
470
- const ready = () => {
471
- if (resumeEventIndex >= resumeEvents.length) return Promise.resolve();
472
- if (readyInFlight) return readyInFlight;
473
- readyInFlight = (async () => {
474
- while (resumeEventIndex < resumeEvents.length) {
475
- const event = resumeEvents[resumeEventIndex];
476
- if (!event) break;
477
- await bus.publish(event);
478
- resumeEventIndex += 1;
479
- }
480
- })().finally(() => {
481
- readyInFlight = void 0;
482
- });
483
- return readyInFlight;
484
- };
485
- const urgencyPriority = (u) => u === "blocks-run" ? 20 : u === "blocks-step" ? 10 : 0;
486
- const str = (v, field) => {
487
- if (typeof v !== "string" || v.length === 0) throw new Error(`coordination tools: "${field}" must be a non-empty string`);
488
- return v;
489
- };
490
- const obj = (raw) => {
491
- if (!raw || typeof raw !== "object") throw new Error("coordination tools: arguments must be an object");
492
- return raw;
493
- };
494
- const mergeBudget = (base, raw) => {
495
- if (!raw || typeof raw !== "object" || Array.isArray(raw)) throw new Error("coordination tools: \"budget\" must be an object");
496
- const o = raw;
497
- const field = (name) => {
498
- const v = o[name];
499
- if (v === void 0) return void 0;
500
- if (typeof v !== "number" || !Number.isFinite(v)) throw new Error(`coordination tools: "budget.${name}" must be a finite number`);
501
- return v;
502
- };
503
- const maxIterations = field("maxIterations");
504
- const maxTokens = field("maxTokens");
505
- const maxUsd = field("maxUsd");
506
- const deadlineMs = field("deadlineMs");
507
- const merged = {
508
- maxIterations: maxIterations ?? base.maxIterations,
509
- maxTokens: maxTokens ?? base.maxTokens,
510
- ...(maxUsd ?? base.maxUsd) === void 0 ? {} : { maxUsd: maxUsd ?? base.maxUsd },
511
- ...(deadlineMs ?? base.deadlineMs) === void 0 ? {} : { deadlineMs: deadlineMs ?? base.deadlineMs }
512
- };
513
- assertValidBudget(merged, "coordination tools: budget");
514
- return merged;
515
- };
516
- const level = (v) => {
517
- if (v === "worker" || v === "driver" || v === "loop") return v;
518
- throw new Error("coordination tools: \"level\" must be worker, driver, or loop");
519
- };
520
- const urgency = (v) => {
521
- if (v === "continue-without" || v === "blocks-step" || v === "blocks-run") return v;
522
- throw new Error("coordination tools: \"urgency\" must be continue-without, blocks-step, or blocks-run");
523
- };
524
- const commitSettled = (s, w) => {
525
- const settledKey = keyByWorker.get(s.handle.id);
526
- if (settledKey !== void 0 && s.kind === "done") completedKeys.add(settledKey);
527
- ledger.push(w);
528
- unwatchWorker(w.id);
529
- };
530
- let pendingSettlement;
531
- const analystRuns = /* @__PURE__ */ new Map();
532
- let analystRunOrdinal = 0;
533
- /** The `finding` event an analyst-agent settlement becomes: its settle OUTPUT is the findings
534
- * (a failed run publishes the failure as findings — degraded beats vanished). */
535
- const analystRunFinding = (run, settled) => detachedFrozen({
536
- type: "finding",
537
- finding: canonicalFindingEvent({
538
- fromWorker: run.sourceWorker,
539
- analyst: run.route.kind,
540
- findings: settled.kind === "done" ? settled.out : { analystRunFailed: settled.reason }
541
- })
542
- });
543
- /**
544
- * Spawn one analyst-AGENT run over a settled worker's evidence, through the SAME spawn
545
- * machinery a driver spawn uses (`scope.spawn` + `makeWorkerAgent`): the analyst's spend
546
- * reserves from the conserved pool, its node is journaled/traced like any worker, and a
547
- * node-pinning seam sees `context.analyst`. Its task is the route directive plus the settled
548
- * worker's persisted tool-trace spans. A refused spawn publishes a finding RECORDING the
549
- * refusal — observable, never silent — and must never take down the settlement path.
550
- */
551
- const spawnAnalystRun = async (route, worker) => {
552
- let spansText = "";
553
- let spanCount = 0;
554
- if (worker.trace.status === "available") try {
555
- const artifact = parseWorkerToolTraceArtifact(await opts.blobs.get(worker.trace.traceRef), worker.trace.traceRef);
556
- spanCount = artifact.spans.length;
557
- spansText = safeJsonText(artifact.spans);
558
- } catch {
559
- spansText = "";
560
- }
561
- const task = [
562
- ...route.directive === void 0 || route.directive.length === 0 ? [] : [route.directive],
563
- `Evidence — settled worker '${worker.id}' tool trace (${spanCount} spans):`,
564
- spansText.length === 0 ? "(no tool spans available)" : spansText
565
- ].join("\n\n");
566
- const assignmentId = `analyst:${route.kind}:o${analystRunOrdinal++}`;
567
- const label = `analyst:${route.kind}`;
568
- const context = Object.freeze({
569
- assignmentId,
570
- parentNodeId: opts.scope.view.root,
571
- budget: opts.perWorker,
572
- task,
573
- label,
574
- analyst: route.kind,
575
- continuity: "fresh"
576
- });
577
- let refusal;
578
- let spawnedId;
579
- try {
580
- const res = opts.scope.spawn(() => opts.makeWorkerAgent(route.agent, context), task, {
581
- budget: opts.perWorker,
582
- label,
583
- assignmentId
584
- });
585
- if (res.ok) spawnedId = res.handle.id;
586
- else refusal = String(res.reason);
587
- } catch (cause) {
588
- refusal = cause instanceof Error ? cause.message : String(cause);
589
- }
590
- if (spawnedId === void 0) {
591
- await bus.publish({
592
- type: "finding",
593
- finding: canonicalFindingEvent({
594
- fromWorker: worker.id,
595
- analyst: route.kind,
596
- findings: { analystSpawnRefused: refusal ?? "unknown" }
597
- })
598
- });
599
- return;
600
- }
601
- analystRuns.set(spawnedId, {
602
- route,
603
- sourceWorker: worker.id
604
- });
605
- watchWorker(spawnedId);
606
- };
607
- const workerRouteNames = (workerId) => {
608
- const names = /* @__PURE__ */ new Set();
609
- const profileName = profileNameByWorker.get(workerId);
610
- if (profileName !== void 0) names.add(profileName);
611
- const label = nodeForWorker(workerId)?.label;
612
- if (label !== void 0) names.add(label);
613
- return names;
614
- };
615
- /** The LIVE worker a route destination names, by profile name first, label second. */
616
- const liveWorkerIdNamed = (destination) => {
617
- const live = opts.scope.view.nodes.filter((node) => isLiveNodeStatus(node.status));
618
- return live.find((node) => profileNameByWorker.get(node.id) === destination)?.id ?? live.find((node) => node.label === destination)?.id;
619
- };
620
- const liveWorkerForNode = (name) => opts.scope.view.nodes.find((node) => isLiveNodeStatus(node.status) && profileNameByWorker.get(node.id) === name)?.id;
621
- const latestSettledWorkerForNode = (name) => {
622
- for (let i = ledger.length - 1; i >= 0; i -= 1) {
623
- const worker = ledger[i];
624
- if (profileNameByWorker.get(worker.id) === name) return worker.id;
625
- }
626
- };
627
- const nodeSpawnCount = (name) => {
628
- let count = 0;
629
- for (const profileName of profileNameByWorker.values()) if (profileName === name) count += 1;
630
- return count;
631
- };
632
- const parseContinuity = (v) => {
633
- if (v === void 0) return void 0;
634
- if (v === "fresh" || v === "resume") return v;
635
- throw new Error("coordination tools: \"continuity\" must be \"fresh\" or \"resume\"");
636
- };
637
- /**
638
- * Resolve the EFFECTIVE continuity of one spawn: the per-call request wins, else the profile
639
- * name's declared default, else `'fresh'`. Every refusal is loud and actionable:
640
- * - an EXPLICIT `'resume'` with no settled prior worker refuses (`resume-no-prior`) — the
641
- * DECLARED default degrades to `'fresh'` instead, so a resume edge's first traversal is
642
- * simply the first spawn;
643
- * - resume while a prior worker of the node is still LIVE refuses (`resume-while-live`) —
644
- * that is what steer is for, and the error says so;
645
- * - resume under a semantic `key` refuses (`resume-with-key`) — a key makes an assignment
646
- * run-once, resume explicitly runs the node again.
647
- */
648
- const resolveContinuity = (requested, profileName, key) => {
649
- const declared = profileName === void 0 ? "fresh" : opts.continuityByProfile?.[profileName] ?? "fresh";
650
- if (!(requested === "resume" || requested === void 0 && declared === "resume")) return { continuity: "fresh" };
651
- if (profileName === void 0 || profileName.length === 0) return {
652
- error: "resume-unnamed-profile",
653
- hint: "Resume targets a node by profile.name — the stable node identity — and this profile has none. Name the profile, or spawn fresh."
654
- };
655
- if (key !== void 0) return {
656
- error: "resume-with-key",
657
- hint: "A semantic key makes an assignment run-once (a completed key returns its committed result instead of running again); resume explicitly runs the node AGAIN. Drop the key to resume, or keep the key and spawn fresh."
658
- };
659
- const live = liveWorkerForNode(profileName);
660
- if (live !== void 0) return {
661
- error: "resume-while-live",
662
- hint: `Worker '${live}' on node '${profileName}' is still LIVE — resume re-attaches to a SETTLED session. To redirect the live worker, use steer_agent (that is the live-worker channel); to run a parallel sibling instead, pass continuity: 'fresh'.`
663
- };
664
- const prior = latestSettledWorkerForNode(profileName);
665
- if (prior === void 0) {
666
- if (requested === void 0) return { continuity: "fresh" };
667
- return {
668
- error: "resume-no-prior",
669
- hint: `Node '${profileName}' has no settled prior worker in this process to resume — resume continues a FINISHED session. Spawn the node fresh first (omit continuity or pass 'fresh').`
670
- };
671
- }
672
- return {
673
- continuity: "resume",
674
- resume: {
675
- ofWorker: prior,
676
- sequence: nodeSpawnCount(profileName) + 1
677
- }
678
- };
679
- };
680
- /**
681
- * Deliver one routed analyst finding to its destination worker through the SAME authorized
682
- * steer machinery a driver steer uses, so the delivery is recorded (`steer` event carrying
683
- * `analyst`) and its outcome is a fact. No live destination ⇒ a record-only failed steer —
684
- * observable, never a silent drop. A throw here must not kill the settle path: failures are
685
- * recorded on the bus (a `steer` with `delivered: false`) before being swallowed, with ONE
686
- * narrow exception — a bus that refuses the `delivery-attempt` record itself leaves only the
687
- * `instruction` receipt (an attempt with no outcome = explicitly unknown, per
688
- * recordDeliveryAttempt's own contract).
689
- */
690
- const deliverRoutedFinding = async (route, findings) => {
691
- const destination = route.to;
692
- const text = route.directive === void 0 || route.directive.length === 0 ? safeJsonText(findings) : `${route.directive}\n\n${safeJsonText(findings)}`;
693
- const targetId = liveWorkerIdNamed(destination);
694
- if (targetId === void 0) {
695
- await bus.publish({
696
- type: "steer",
697
- down: detachedFrozen({
698
- receiptId: randomUUID(),
699
- toWorker: destination,
700
- instruction: text,
701
- instructionDigest: canonicalCandidateDigest(text),
702
- delivered: false,
703
- outcome: "unknown-worker"
704
- }),
705
- analyst: route.kind
706
- }, { queue: false });
707
- return;
708
- }
709
- let instruction;
710
- try {
711
- instruction = authorizeInstruction("steer", targetId, text, false);
712
- await recordInstruction(instruction);
713
- } catch (cause) {
714
- try {
715
- await sendDown("steer", detachedFrozen({
716
- receiptId: randomUUID(),
717
- toWorker: targetId,
718
- instruction: text,
719
- instructionDigest: canonicalCandidateDigest(text),
720
- delivered: false,
721
- outcome: "runtime-error",
722
- error: cause instanceof Error ? cause.message : String(cause)
723
- }), route.kind);
724
- } catch {}
725
- return;
726
- }
727
- try {
728
- await attemptDelivery(instruction, {
729
- steer: instruction.instruction,
730
- interrupt: false
731
- }, { analyst: route.kind });
732
- } catch {}
733
- };
734
- const flushPendingSettlement = async () => {
735
- const pending = pendingSettlement;
736
- if (!pending) return false;
737
- await bus.publish(pending.event);
738
- if (pending.analystRun) {
739
- analystRuns.delete(pending.worker.id);
740
- unwatchWorker(pending.worker.id);
741
- pendingSettlement = void 0;
742
- const { route } = pending.analystRun;
743
- if (route.to !== void 0 && pending.event.type === "finding") await deliverRoutedFinding({
744
- kind: route.kind,
745
- to: route.to
746
- }, pending.event.finding.findings);
747
- return true;
748
- }
749
- commitSettled(pending.settled, pending.worker);
750
- pendingSettlement = void 0;
751
- if (pending.analyze && pending.worker.status === "done" && pending.worker.trace.status === "available" && opts.analyzeOnSettle?.length) {
752
- const routes = opts.analyzeOnSettle.map(normalizeAnalyzeOnSettle);
753
- const sourceNames = workerRouteNames(pending.worker.id);
754
- const applicable = routes.filter((route) => route.over === void 0 || route.over.some((name) => sourceNames.has(name)));
755
- const lensRoutes = applicable.filter((route) => route.agent === void 0);
756
- const agentRoutes = applicable.filter((route) => route.agent !== void 0);
757
- if (lensRoutes.length > 0 && opts.analysts) {
758
- const trace = await workerTraceAnalysisStore(pending.worker.trace, opts.blobs);
759
- for (const route of lensRoutes) {
760
- const findings = await opts.analysts.run(route.kind, trace);
761
- await bus.publish({
762
- type: "finding",
763
- finding: canonicalFindingEvent({
764
- fromWorker: pending.worker.id,
765
- analyst: route.kind,
766
- findings
767
- })
768
- });
769
- if (route.to !== void 0) await deliverRoutedFinding(route, findings);
770
- }
771
- }
772
- for (const route of agentRoutes) await spawnAnalystRun(route, pending.worker);
773
- }
774
- return true;
775
- };
776
- const drainSettlement = async () => {
777
- if (!pendingSettlement) {
778
- const settled = await opts.scope.next();
779
- if (!settled) return false;
780
- const worker = projectSettled(settled);
781
- const run = analystRuns.get(settled.handle.id);
782
- pendingSettlement = run ? {
783
- settled,
784
- worker,
785
- event: analystRunFinding(run, settled),
786
- analyze: false,
787
- analystRun: run
788
- } : {
789
- settled,
790
- worker,
791
- event: detachedFrozen({
792
- type: "settled",
793
- worker
794
- }),
795
- analyze: true
796
- };
797
- }
798
- return flushPendingSettlement();
799
- };
800
- const drainResolved = async () => {
801
- let drained = 0;
802
- for (;;) {
803
- if (!pendingSettlement) {
804
- const settled = await opts.scope.nextResolved();
805
- if (!settled) return drained;
806
- const worker = projectSettled(settled);
807
- const run = analystRuns.get(settled.handle.id);
808
- pendingSettlement = run ? {
809
- settled,
810
- worker,
811
- event: analystRunFinding(run, settled),
812
- analyze: false,
813
- analystRun: run
814
- } : {
815
- settled,
816
- worker,
817
- event: detachedFrozen({
818
- type: "settled",
819
- worker
820
- }),
821
- analyze: false
822
- };
823
- }
824
- await flushPendingSettlement();
825
- drained += 1;
826
- }
827
- };
828
- async function sendDown(type, down, questionIdOrAnalyst) {
829
- await bus.publish(type === "answer" ? {
830
- type,
831
- down,
832
- questionId: str(questionIdOrAnalyst, "questionId")
833
- } : questionIdOrAnalyst !== void 0 ? {
834
- type,
835
- down,
836
- analyst: questionIdOrAnalyst
837
- } : {
838
- type,
839
- down
840
- }, { queue: false });
841
- }
842
- const authorizeInstruction = (kind, workerId, instruction, interrupt, questionId) => {
843
- const workerIdentity = opts.scope.view.nodes.find((node) => node.id === workerId)?.identity;
844
- let authorizedInstruction = instruction;
845
- if (opts.authorizeDownMessage) {
846
- if (workerIdentity === void 0) throw new Error(`coordination tools: cannot authorize ${kind} for worker ${JSON.stringify(workerId)} without durable identity`);
847
- const decision = detachedFrozen(opts.authorizeDownMessage(detachedFrozen({
848
- kind,
849
- workerId,
850
- workerIdentity,
851
- instruction,
852
- interrupt,
853
- ...questionId !== void 0 ? { questionId } : {}
854
- })));
855
- if (typeof decision !== "object" || decision === null || Array.isArray(decision) || typeof decision.instruction !== "string" || decision.instruction.length === 0) throw new Error("coordination tools: authorizeDownMessage must return an instruction");
856
- authorizedInstruction = decision.instruction;
857
- }
858
- return detachedFrozen({
859
- receiptId: randomUUID(),
860
- kind,
861
- toWorker: workerId,
862
- instruction: authorizedInstruction,
863
- instructionDigest: canonicalCandidateDigest(authorizedInstruction),
864
- ...workerIdentity !== void 0 ? { workerIdentity } : {},
865
- interrupt,
866
- ...questionId !== void 0 ? { questionId } : {}
867
- });
868
- };
869
- /** Publish before `scope.send`: an awaited durable subscriber therefore commits the exact bytes
870
- * before the worker can observe them. */
871
- const recordInstruction = async (instruction) => {
872
- await bus.publish({
873
- type: "instruction",
874
- instruction
875
- }, { queue: false });
876
- };
877
- /** Commit delivery intent after the authorization receipt and before `Scope.send`. An attempt with
878
- * no matching outcome after a crash is explicitly unknown and must never be replayed. */
879
- const recordDeliveryAttempt = async (instruction) => {
880
- const attempt = detachedFrozen({
881
- receiptId: instruction.receiptId,
882
- kind: instruction.kind,
883
- toWorker: instruction.toWorker,
884
- instructionDigest: instruction.instructionDigest,
885
- interrupt: instruction.interrupt,
886
- ...instruction.questionId !== void 0 ? { questionId: instruction.questionId } : {}
887
- });
888
- await bus.publish({
889
- type: "delivery-attempt",
890
- attempt
891
- }, { queue: false });
892
- return attempt;
893
- };
894
- const deliveryOutcome = (workerId, delivered) => {
895
- if (delivered) return "delivered";
896
- if (opts.scope.signal.aborted) return "scope-stopped";
897
- const node = opts.scope.view.nodes.find((candidate) => candidate.id === workerId);
898
- if (!node) return "unknown-worker";
899
- if (!isLiveNodeStatus(node.status)) return "already-settled";
900
- return "runtime-has-no-inbox";
901
- };
902
- const attemptDelivery = async (instruction, message, origin) => {
903
- await recordDeliveryAttempt(instruction);
904
- let delivered = false;
905
- let outcome;
906
- let error;
907
- try {
908
- delivered = opts.scope.send(instruction.toWorker, message);
909
- outcome = deliveryOutcome(instruction.toWorker, delivered);
910
- } catch (cause) {
911
- outcome = "runtime-error";
912
- error = cause instanceof Error ? cause.message : String(cause);
913
- }
914
- const down = detachedFrozen({
915
- receiptId: instruction.receiptId,
916
- toWorker: instruction.toWorker,
917
- instruction: instruction.instruction,
918
- instructionDigest: instruction.instructionDigest,
919
- delivered,
920
- outcome,
921
- ...error !== void 0 ? { error } : {}
922
- });
923
- if (instruction.kind === "answer") await sendDown("answer", down, str(instruction.questionId, "questionId"));
924
- else await sendDown("steer", down, origin?.analyst);
925
- if (error !== void 0) throw new Error(`coordination tools: delivery failed: ${error}`);
926
- return down;
927
- };
928
- const projectEvent = (ev) => {
929
- if (ev.type === "settled") {
930
- const { id, status, ...evidence } = ev.worker;
931
- return {
932
- type: "settled",
933
- settled: id,
934
- status,
935
- ...evidence
936
- };
937
- }
938
- if (ev.type === "question") return {
939
- type: "question",
940
- question: ev.question
941
- };
942
- if (ev.type === "finding") return {
943
- type: "finding",
944
- ...ev.finding
945
- };
946
- if (ev.type === "answer") return {
947
- type: "answer",
948
- ...ev.down,
949
- questionId: ev.questionId
950
- };
951
- if (ev.type === "instruction") return {
952
- type: "instruction",
953
- ...ev.instruction
954
- };
955
- if (ev.type === "delivery-attempt") return {
956
- type: "delivery-attempt",
957
- ...ev.attempt
958
- };
959
- if (ev.type === "mail") return {
960
- type: "mail",
961
- ...ev.mail
962
- };
963
- return {
964
- type: ev.type,
965
- ...ev.down
966
- };
967
- };
968
- const peerMail = opts.peerMail ? createPeerMailbox({
969
- scope: opts.scope,
970
- publish: (mail) => bus.publish({
971
- type: "mail",
972
- mail
973
- }, { queue: false }).then(() => void 0),
974
- ...opts.peerMail.limits ? { limits: opts.peerMail.limits } : {}
975
- }) : void 0;
976
- const nextQuestionId = (from) => {
977
- for (;;) {
978
- const id = `${from}:q${questionSeq++}`;
979
- if (!questions.some((question) => question.id === id)) return id;
980
- }
981
- };
982
- const normalizeQuestion = (q, fallbackFrom) => {
983
- const from = str(q.from ?? fallbackFrom, "from");
984
- return {
985
- id: typeof q.id === "string" && q.id.length > 0 ? q.id : nextQuestionId(from),
986
- from,
987
- level: level(q.level),
988
- question: str(q.question, "question"),
989
- reason: str(q.reason, "reason"),
990
- ...q.options ? { options: q.options } : {},
991
- urgency: urgency(q.urgency)
992
- };
993
- };
994
- const addQuestion = (raw, fallbackFrom, decision) => {
995
- const q = normalizeQuestion(raw, fallbackFrom);
996
- const existing = questions.find((x) => x.id === q.id);
997
- if (existing) return {
998
- question: existing,
999
- added: false
1000
- };
1001
- const effectiveDecision = decision ?? (questionPolicy === "bubble" ? {
1002
- kind: "escalate",
1003
- to: "parent",
1004
- reason: "question policy bubbled to parent"
1005
- } : void 0);
1006
- const status = effectiveDecision?.kind === "answer" ? "answered" : effectiveDecision?.kind === "defer" ? "deferred" : effectiveDecision?.kind === "escalate" ? "escalated" : "open";
1007
- const record = {
1008
- ...q,
1009
- status,
1010
- openedAt: Date.now(),
1011
- ...effectiveDecision ? { decision: effectiveDecision } : {}
1012
- };
1013
- questions.push(record);
1014
- return {
1015
- question: record,
1016
- added: true
1017
- };
1018
- };
1019
- const emitNewQuestion = async (record) => {
1020
- if (record.added) await bus.publish({
1021
- type: "question",
1022
- question: record.question
1023
- }, { priority: urgencyPriority(record.question.urgency) });
1024
- return record.question;
1025
- };
1026
- const decideQuestion = (questionId, decision) => {
1027
- const idx = questions.findIndex((q) => q.id === questionId);
1028
- if (idx < 0) throw new Error(`unknown questionId ${JSON.stringify(questionId)}`);
1029
- const prior = questions[idx];
1030
- const status = decision.kind === "answer" ? "answered" : decision.kind === "defer" ? "deferred" : "escalated";
1031
- const next = {
1032
- ...prior,
1033
- status,
1034
- decision
1035
- };
1036
- questions[idx] = next;
1037
- return next;
1038
- };
1039
- const blockingQuestionsForStop = () => {
1040
- if (questionPolicy === "auto" || questionPolicy === "bubble") return [];
1041
- return questions.filter((q) => {
1042
- if (!(q.urgency === "blocks-step" || q.urgency === "blocks-run")) return false;
1043
- if (questionPolicy === "mustDecide") return q.status === "open";
1044
- return q.status !== "answered" && q.status !== "deferred";
1045
- });
1046
- };
1047
- const maxLiveWorkers = opts.maxLiveWorkers;
1048
- const localLiveWorkerCount = () => opts.scope.view.nodes.filter((n) => isLiveNodeStatus(n.status)).length;
1049
- const sharedWorkerCapacity = () => {
1050
- return opts.scope.workerCapacity;
1051
- };
1052
- const usesTreeWideLimit = () => {
1053
- const capacity = sharedWorkerCapacity();
1054
- return capacity !== void 0 && capacity.freeSlots !== null;
1055
- };
1056
- const liveWorkerCount = () => usesTreeWideLimit() ? sharedWorkerCapacity()?.live ?? localLiveWorkerCount() : localLiveWorkerCount();
1057
- const projectNodeEvidence = (node, resumed = false) => ({
1058
- id: node.id,
1059
- status: node.status,
1060
- ...node.assignmentId === void 0 ? {} : { assignmentId: node.assignmentId },
1061
- ...node.identity === void 0 ? {} : { identity: node.identity },
1062
- ...node.materialization === void 0 ? {} : { materialization: node.materialization },
1063
- ...node.executionBindings === void 0 ? {} : { executionBindings: node.executionBindings },
1064
- spent: node.spent,
1065
- ...node.settledAt === void 0 ? {} : { settledAt: node.settledAt },
1066
- ...node.outRef === void 0 ? {} : { outRef: node.outRef },
1067
- ...node.trace === void 0 ? {} : { trace: node.trace },
1068
- ...resumed ? { resumed: true } : {}
1069
- });
1070
- const liveSnapshot = () => opts.scope.view.nodes.filter((n) => isLiveNodeStatus(n.status)).map((n) => projectNodeEvidence(n));
1071
- const freeWorkerSlots = () => usesTreeWideLimit() ? sharedWorkerCapacity()?.freeSlots ?? null : freeSlots(localLiveWorkerCount(), maxLiveWorkers);
1072
- const readProgress = (id) => {
1073
- const scope = opts.scope;
1074
- if (typeof scope.progress !== "function") return void 0;
1075
- try {
1076
- return scope.progress(id, opts.stallAfterMs !== void 0 ? { stallAfterMs: opts.stallAfterMs } : {});
1077
- } catch {
1078
- return;
1079
- }
1080
- };
1081
- const watchers = /* @__PURE__ */ new Map();
1082
- const watchWorker = (id) => {
1083
- const watch = opts.watchWorkers;
1084
- if (!watch) return;
1085
- const scope = opts.scope;
1086
- if (typeof scope.traceSource !== "function") return;
1087
- let source;
1088
- try {
1089
- source = scope.traceSource(id);
1090
- } catch {
1091
- return;
1092
- }
1093
- if (!source) return;
1094
- const cap = watch.maxFindingsPerWorker ?? 3;
1095
- let raised = 0;
1096
- const unsub = watchTrace(source, {
1097
- ...watch.detectors ? { detectors: watch.detectors } : {},
1098
- onSignal: async (signal, span) => {
1099
- if (cap > 0 && raised >= cap) return;
1100
- raised += 1;
1101
- await bus.publish({
1102
- type: "finding",
1103
- finding: canonicalFindingEvent({
1104
- fromWorker: id,
1105
- analyst: `online:${signal.detector}`,
1106
- findings: {
1107
- detector: signal.detector,
1108
- severity: signal.severity,
1109
- reason: signal.reason,
1110
- streak: signal.streak,
1111
- ...signal.failureClass ? { failureClass: signal.failureClass } : {},
1112
- toolName: span.toolName,
1113
- at: span.endedAt,
1114
- progress: readProgress(id)
1115
- }
1116
- })
1117
- });
1118
- }
1119
- });
1120
- watchers.set(id, unsub);
1121
- };
1122
- const unwatchWorker = (id) => {
1123
- const unsub = watchers.get(id);
1124
- if (!unsub) return;
1125
- watchers.delete(id);
1126
- try {
1127
- unsub();
1128
- } catch {}
1129
- };
1130
- const awaitTimeoutMs = opts.awaitTimeoutMs ?? 15e3;
1131
- let inFlightDrain = null;
1132
- const ensureDrain = () => {
1133
- if (!inFlightDrain) inFlightDrain = drainSettlement().finally(() => {
1134
- inFlightDrain = null;
1135
- });
1136
- return inFlightDrain;
1137
- };
1138
- const raceDrainWithTimeout = async (drain) => {
1139
- if (awaitTimeoutMs <= 0) return { drained: await drain };
1140
- let timer;
1141
- const timeout = new Promise((resolve) => {
1142
- timer = setTimeout(() => resolve(void 0), awaitTimeoutMs);
1143
- if (typeof timer?.unref === "function") timer.unref();
1144
- });
1145
- try {
1146
- return await Promise.race([drain.then((drained) => ({ drained })), timeout]);
1147
- } finally {
1148
- if (timer) clearTimeout(timer);
1149
- }
1150
- };
1151
- const tools = [
1152
- {
1153
- name: "spawn_agent",
1154
- description: "Start a worker the driver will drive. `profile` is the worker or another driver; `task` is what it should do. Reserves budget from the conserved pool and fails closed. Pass an optional `budget` (per-field) to give a hard sub-task more than the default — it merges over the per-worker default; the conserved pool is still the hard fence. When a max-live-workers cap is set it also fails closed (`error: \"max-live-workers\"`) while that many workers are still in flight — settle or steer one before spawning another. Pass a `key` naming the assignment to make it run-once ACROSS restarts: a key that already completed returns the finished result (`resumed: \"completed\"` — no work re-runs, nothing is spent), a key whose prior attempt failed or was lost with a dead process spawns fresh and says so (`resumed: \"retried\" | \"lost\"`), and a key still running is refused (`error: \"duplicate-key\"`). Returns `freeSlots`: how many MORE workers you can start right now (`null` = uncapped). While `freeSlots > 0` there is idle capacity — call this again to fill it rather than waiting; parallel workers finish the run sooner than one at a time.",
1155
- inputSchema: {
1156
- type: "object",
1157
- properties: {
1158
- profile: spawnProfileArg(),
1159
- task: { description: "The task the worker should perform." },
1160
- label: {
1161
- type: "string",
1162
- description: "Optional trace label."
1163
- },
1164
- key: {
1165
- type: "string",
1166
- description: "Optional semantic name for this assignment (e.g. \"summarize-ch3\"). The same key never runs twice: completed keys return their committed result, even after a coordinator restart."
1167
- },
1168
- continuity: {
1169
- type: "string",
1170
- enum: ["fresh", "resume"],
1171
- description: "How this spawn continues the node's prior work. \"fresh\" (the default) starts a brand-new session. \"resume\" re-attaches to the node's most recent SETTLED worker: a NEW live worker is spawned whose session continues where that worker stopped (the backend receives the prior workerId and the resume sequence). Resume fails closed when the node has no settled prior worker (`error: \"resume-no-prior\"` — spawn it fresh first), while a prior worker of the node is still live (`error: \"resume-while-live\"` — steer_agent is the live-worker channel), and under a `key` (`error: \"resume-with-key\"` — keys are run-once, resume runs again). Omit to use the run's declared default for this profile name."
1172
- },
1173
- budget: {
1174
- type: "object",
1175
- description: "Optional per-spawn budget that merges over the per-worker default (per field). Only set the ceilings this sub-task needs raised; the conserved pool still fences.",
1176
- properties: {
1177
- maxIterations: {
1178
- type: "number",
1179
- minimum: 0
1180
- },
1181
- maxTokens: {
1182
- type: "number",
1183
- minimum: 0
1184
- },
1185
- maxUsd: {
1186
- type: "number",
1187
- minimum: 0
1188
- },
1189
- deadlineMs: {
1190
- type: "number",
1191
- minimum: 0
1192
- }
1193
- }
1194
- }
1195
- },
1196
- required: ["profile", "task"]
1197
- },
1198
- handler: async (raw) => {
1199
- const a = obj(raw);
1200
- const key = a.key === void 0 ? void 0 : str(a.key, "key");
1201
- if (!(key !== void 0 && completedKeys.has(key)) && !usesTreeWideLimit() && maxLiveWorkers !== void 0 && maxLiveWorkers > 0 && liveWorkerCount() >= maxLiveWorkers) return Promise.resolve({
1202
- error: "max-live-workers",
1203
- live: liveWorkerCount(),
1204
- freeSlots: freeWorkerSlots()
1205
- });
1206
- const parsedProfile = agentProfileSchema.safeParse(a.profile);
1207
- if (!parsedProfile.success) return Promise.resolve({
1208
- error: "invalid-profile",
1209
- issues: parsedProfile.error.issues.map((issue) => ({
1210
- path: issue.path.join("."),
1211
- message: issue.message
1212
- }))
1213
- });
1214
- const profile = detachedFrozen(parsedProfile.data);
1215
- const continuity = resolveContinuity(parseContinuity(a.continuity), profile.name, key);
1216
- if ("error" in continuity) return Promise.resolve({
1217
- error: continuity.error,
1218
- hint: continuity.hint,
1219
- live: liveWorkerCount(),
1220
- freeSlots: freeWorkerSlots()
1221
- });
1222
- const task = detachedFrozen(a.task);
1223
- const label = typeof a.label === "string" ? a.label : "worker";
1224
- if (opts.preflightSpawn) {
1225
- const preflightProfile = opts.resolveSpawnProfile ? detachedFrozen(opts.resolveSpawnProfile(profile)) : profile;
1226
- const refusal = await opts.preflightSpawn(preflightProfile, {
1227
- label,
1228
- ...key !== void 0 ? { key } : {},
1229
- task
1230
- });
1231
- if (refusal) {
1232
- preflightCounts[refusal.cause] += 1;
1233
- return {
1234
- error: "preflight-refused",
1235
- cause: refusal.cause,
1236
- detail: refusal.detail,
1237
- live: liveWorkerCount(),
1238
- freeSlots: freeWorkerSlots()
1239
- };
1240
- }
1241
- }
1242
- const budget = Object.freeze(a.budget === void 0 ? opts.perWorker : mergeBudget(opts.perWorker, a.budget));
1243
- const assignmentId = key !== void 0 ? `key:${key}` : `ordinal:${unkeyedAssignmentOrdinal++}`;
1244
- const peerMailUrl = peerMail?.mintCapability(assignmentId);
1245
- const context = Object.freeze({
1246
- assignmentId,
1247
- parentNodeId: opts.scope.view.root,
1248
- budget,
1249
- task,
1250
- label,
1251
- ...key !== void 0 ? { key } : {},
1252
- continuity: continuity.continuity,
1253
- ...continuity.continuity === "resume" ? { resume: continuity.resume } : {},
1254
- ...peerMailUrl !== void 0 ? { peerMailUrl } : {}
1255
- });
1256
- const res = opts.scope.spawn(() => opts.makeWorkerAgent(profile, context), task, {
1257
- budget,
1258
- label,
1259
- assignmentId,
1260
- ...key !== void 0 ? { key } : {}
1261
- });
1262
- if (res.ok && res.prior?.state === "completed") {
1263
- const s = res.prior.settled;
1264
- if (key !== void 0) completedKeys.add(key);
1265
- const { id, status, resumed: _resumed, ...evidence } = projectSettled(s);
1266
- return Promise.resolve({
1267
- workerId: id,
1268
- resumed: "completed",
1269
- status,
1270
- ...evidence,
1271
- live: liveWorkerCount(),
1272
- freeSlots: freeWorkerSlots()
1273
- });
1274
- }
1275
- if (res.ok) {
1276
- watchWorker(res.handle.id);
1277
- liveHandles.set(res.handle.id, res.handle);
1278
- peerMail?.bindCapability(assignmentId, res.handle.id);
1279
- if (key !== void 0) keyByWorker.set(res.handle.id, key);
1280
- if (typeof profile.name === "string" && profile.name.length > 0) profileNameByWorker.set(res.handle.id, profile.name);
1281
- }
1282
- const priorHistory = res.ok && res.prior !== void 0 && res.prior.state !== "completed" ? {
1283
- resumed: res.prior.state,
1284
- priorWorkerId: res.prior.priorId,
1285
- ...res.prior.state === "retried" ? { priorReason: res.prior.reason } : {}
1286
- } : {};
1287
- return Promise.resolve(res.ok ? {
1288
- workerId: res.handle.id,
1289
- assignmentId: res.handle.assignmentId ?? assignmentId,
1290
- ...res.handle.identity === void 0 ? {} : { identity: res.handle.identity },
1291
- ...res.handle.materialization === void 0 ? {} : { materialization: res.handle.materialization },
1292
- ...res.handle.executionBindings === void 0 ? {} : { executionBindings: res.handle.executionBindings },
1293
- continuity: continuity.continuity,
1294
- ...continuity.continuity === "resume" ? { resume: continuity.resume } : {},
1295
- live: liveWorkerCount(),
1296
- freeSlots: freeWorkerSlots(),
1297
- ...priorHistory
1298
- } : {
1299
- error: res.reason,
1300
- ...res.reason === "usd-unbudgeted" ? { hint: "This run's root budget declares no maxUsd, so a child budget naming maxUsd can never be admitted — at any amount. Retrying with a smaller maxUsd will fail identically. Spawn with a budget that omits maxUsd, or ask the caller to give the run a root maxUsd." } : {},
1301
- live: liveWorkerCount(),
1302
- freeSlots: freeWorkerSlots()
1303
- });
1304
- }
1305
- },
1306
- {
1307
- name: "observe_agent",
1308
- description: "Inspect a worker WHILE IT RUNS, not only after it finishes: status, spend so far, and `progress` — how long since it last did anything (`idleMs`), whether that counts as stalled, how many turns it has taken, the last tools/files it touched (`recentActivity`), what its executor CHANGED about the profile you gave it (`derived` — an MCP config it materialized, an extension it had to add), whether a steer can even reach it (`steerable`), and how many steers it has not yet read (`pendingMessages`). Returns the settled output artifact once it exists. Use this BEFORE steer_agent: a steer is only worth sending when the progress says the worker is on the wrong path or has stopped making any.",
1309
- inputSchema: {
1310
- type: "object",
1311
- properties: { workerId: idArg },
1312
- required: ["workerId"]
1313
- },
1314
- handler: async (raw) => {
1315
- const id = str(obj(raw).workerId, "workerId");
1316
- const node = opts.scope.view.nodes.find((n) => n.id === id);
1317
- if (!node) {
1318
- const resumed = opts.scope.resume?.view.nodes.find((n) => n.id === id);
1319
- if (!resumed) return { error: `unknown workerId ${JSON.stringify(id)}` };
1320
- const output = resumed.outRef ? await opts.blobs.get(resumed.outRef) : void 0;
1321
- return {
1322
- ...projectNodeEvidence(resumed, true),
1323
- outRef: resumed.outRef ?? null,
1324
- output: output ?? null,
1325
- progress: null
1326
- };
1327
- }
1328
- const output = node.outRef ? await opts.blobs.get(node.outRef) : void 0;
1329
- const progress = readProgress(id);
1330
- return {
1331
- ...projectNodeEvidence(node),
1332
- outRef: node.outRef ?? null,
1333
- output: output ?? null,
1334
- progress: progress ?? null
1335
- };
1336
- }
1337
- },
1338
- {
1339
- name: "steer_agent",
1340
- description: "Send a message DOWN to a still-LIVE worker (parent→child): a new instruction, a course correction, or a continuation. The worker drains it at its next step boundary — and before it may settle, so it cannot finish while a message it never read is pending. A worker that already settled is gone (returns delivered:false) — spawn a fresh one instead.",
1341
- inputSchema: {
1342
- type: "object",
1343
- properties: {
1344
- workerId: idArg,
1345
- instruction: {
1346
- type: "string",
1347
- description: "What the worker should do next."
1348
- },
1349
- interrupt: {
1350
- type: "boolean",
1351
- description: "true = forceful: abort the worker’s in-flight inference so it re-plans on the NEXT turn (a tool already mid-execution finishes first; only the owned tool-loop honors this). false/omitted = queued: it flushes at the next step boundary (and before it may settle)."
1352
- }
1353
- },
1354
- required: ["workerId", "instruction"]
1355
- },
1356
- handler: async (raw) => {
1357
- const a = obj(raw);
1358
- const workerId = str(a.workerId, "workerId");
1359
- const instruction = str(a.instruction, "instruction");
1360
- const interrupt = a.interrupt === true;
1361
- const authorized = authorizeInstruction("steer", workerId, instruction, interrupt);
1362
- await recordInstruction(authorized);
1363
- const delivery = await attemptDelivery(authorized, {
1364
- steer: authorized.instruction,
1365
- interrupt
1366
- });
1367
- if (delivery.delivered) return {
1368
- delivered: true,
1369
- progress: readProgress(workerId) ?? null
1370
- };
1371
- return {
1372
- delivered: false,
1373
- reason: delivery.outcome,
1374
- progress: readProgress(workerId) ?? null
1375
- };
1376
- }
1377
- },
1378
- {
1379
- name: "await_event",
1380
- description: "Wait for and pull the next message a worker, sub-driver, or analyst sent up — the unified inbox. An event is one of: a settled worker output ('settled'), a question needing your answer ('question', from ask_parent / the worker's ask-user), or a trace-analyst finding ('finding', from analyze-on-settle). Pass kinds:['settled'] for just the next finished worker; omit `kinds` to also receive questions and findings. Returns { idle: true } when nothing is queued and no workers are live. If a worker is still running when the wait elapses, returns { pending: true, live: [...] } (the workers still in flight) instead of blocking indefinitely — call await_event again to keep waiting; the settlement is not lost. Every reply carries `freeSlots`: how many more workers you can start right now (`null` = uncapped). A settled worker frees its slot, so `freeSlots > 0` means capacity is sitting idle — spawn into it before waiting again.",
1381
- inputSchema: {
1382
- type: "object",
1383
- properties: { kinds: {
1384
- type: "array",
1385
- items: {
1386
- type: "string",
1387
- enum: [...awaitableEventKinds]
1388
- },
1389
- description: "Restrict to these event kinds (any if omitted)."
1390
- } }
1391
- },
1392
- handler: async (raw) => {
1393
- const k = obj(raw).kinds;
1394
- const kinds = Array.isArray(k) ? k.filter(isAwaitableEventKind) : void 0;
1395
- let ev = bus.pull(kinds);
1396
- if (ev) return {
1397
- ...projectEvent(ev),
1398
- freeSlots: freeWorkerSlots()
1399
- };
1400
- const raced = await raceDrainWithTimeout(ensureDrain());
1401
- if (raced === void 0) return {
1402
- pending: true,
1403
- live: liveSnapshot(),
1404
- freeSlots: freeWorkerSlots()
1405
- };
1406
- ev = bus.pull(kinds);
1407
- if (!ev) return {
1408
- idle: !raced.drained,
1409
- freeSlots: freeWorkerSlots()
1410
- };
1411
- return {
1412
- ...projectEvent(ev),
1413
- freeSlots: freeWorkerSlots()
1414
- };
1415
- }
1416
- },
1417
- {
1418
- name: "list_questions",
1419
- description: "List questions raised by workers, drivers, or analysts. Blocking stop behavior follows questionPolicy.",
1420
- inputSchema: {
1421
- type: "object",
1422
- properties: {}
1423
- },
1424
- handler: () => Promise.resolve({ questions })
1425
- },
1426
- {
1427
- name: "answer_question",
1428
- description: "Record an answer, deferral, or escalation for a loop question.",
1429
- inputSchema: {
1430
- type: "object",
1431
- properties: {
1432
- questionId: { type: "string" },
1433
- answer: { type: "string" },
1434
- by: {
1435
- type: "string",
1436
- description: "Node id or \"user\"."
1437
- },
1438
- deferReason: { type: "string" },
1439
- escalateTo: {
1440
- type: "string",
1441
- enum: questionEscalationTargets
1442
- },
1443
- escalateReason: { type: "string" }
1444
- },
1445
- required: ["questionId"]
1446
- },
1447
- handler: async (raw) => {
1448
- const a = obj(raw);
1449
- const questionId = str(a.questionId, "questionId");
1450
- if (typeof a.answer === "string" && a.answer.length > 0) {
1451
- const answer = a.answer;
1452
- const pendingQuestion = questions.find((question) => question.id === questionId);
1453
- if (pendingQuestion === void 0) throw new Error(`unknown questionId ${JSON.stringify(questionId)}`);
1454
- const interrupt = pendingQuestion.urgency === "blocks-run" || pendingQuestion.urgency === "blocks-step";
1455
- const authorized = authorizeInstruction("answer", pendingQuestion.from, answer, interrupt, questionId);
1456
- await recordInstruction(authorized);
1457
- const delivery = await attemptDelivery(authorized, {
1458
- answer: authorized.instruction,
1459
- questionId,
1460
- interrupt
1461
- });
1462
- return {
1463
- question: delivery.delivered ? decideQuestion(questionId, {
1464
- kind: "answer",
1465
- answer: authorized.instruction,
1466
- by: typeof a.by === "string" && a.by.length > 0 ? a.by : "user"
1467
- }) : pendingQuestion,
1468
- delivered: delivery.delivered,
1469
- ...delivery.delivered ? {} : { reason: delivery.outcome }
1470
- };
1471
- }
1472
- if (typeof a.deferReason === "string" && a.deferReason.length > 0) return Promise.resolve({ question: decideQuestion(questionId, {
1473
- kind: "defer",
1474
- reason: a.deferReason
1475
- }) });
1476
- if (typeof a.escalateTo === "string" && a.escalateTo.length > 0) {
1477
- if (!isQuestionEscalationTarget(a.escalateTo)) throw new Error(`answer_question: escalateTo must be one of ${questionEscalationTargets.join(", ")}; received ${JSON.stringify(a.escalateTo)}`);
1478
- const escalateReason = typeof a.escalateReason === "string" && a.escalateReason.length > 0 ? a.escalateReason : "driver escalated";
1479
- return Promise.resolve({ question: decideQuestion(questionId, {
1480
- kind: "escalate",
1481
- to: a.escalateTo,
1482
- reason: escalateReason
1483
- }) });
1484
- }
1485
- throw new Error("answer_question: provide answer, deferReason, or escalateTo");
1486
- }
1487
- },
1488
- {
1489
- name: "ask_parent",
1490
- description: "Raise a question to the parent driver/Pi/user when this driver cannot decide.",
1491
- inputSchema: {
1492
- type: "object",
1493
- properties: {
1494
- from: { type: "string" },
1495
- level: {
1496
- type: "string",
1497
- enum: [
1498
- "worker",
1499
- "driver",
1500
- "loop"
1501
- ]
1502
- },
1503
- question: { type: "string" },
1504
- reason: { type: "string" },
1505
- urgency: {
1506
- type: "string",
1507
- enum: [
1508
- "continue-without",
1509
- "blocks-step",
1510
- "blocks-run"
1511
- ]
1512
- }
1513
- },
1514
- required: [
1515
- "from",
1516
- "level",
1517
- "question",
1518
- "reason",
1519
- "urgency"
1520
- ]
1521
- },
1522
- handler: async (raw) => {
1523
- const a = obj(raw);
1524
- const from = str(a.from, "from");
1525
- return { question: await emitNewQuestion(addQuestion({
1526
- from,
1527
- level: level(a.level),
1528
- question: str(a.question, "question"),
1529
- reason: str(a.reason, "reason"),
1530
- urgency: urgency(a.urgency)
1531
- }, from, {
1532
- kind: "escalate",
1533
- to: "parent",
1534
- reason: "asked parent"
1535
- })) };
1536
- }
1537
- },
1538
- ...deliverable ? [{
1539
- name: "submit_result",
1540
- description: [
1541
- "Submit the complete result to the injected independent check.",
1542
- "The first passing result is retained; stop work when accepted.",
1543
- ...deliverable.describe ? [`Expected result: ${deliverable.describe}`] : []
1544
- ].join(" "),
1545
- inputSchema: {
1546
- type: "object",
1547
- properties: { result: { description: "The complete result in the form requested by the task." } },
1548
- required: ["result"],
1549
- additionalProperties: false
1550
- },
1551
- handler: async (raw) => {
1552
- if (submitted) return {
1553
- accepted: true,
1554
- retained: "earlier-passing-result",
1555
- stop: true
1556
- };
1557
- const a = obj(raw);
1558
- if (!Object.hasOwn(a, "result")) throw new Error("submit_result: \"result\" is required");
1559
- const result = structuredClone(a.result);
1560
- let accepted = false;
1561
- try {
1562
- accepted = await deliverable.check(result) === true;
1563
- } catch {
1564
- accepted = false;
1565
- }
1566
- if (!accepted) return {
1567
- accepted: false,
1568
- stop: false
1569
- };
1570
- if (submitted) return {
1571
- accepted: true,
1572
- retained: "earlier-passing-result",
1573
- stop: true
1574
- };
1575
- submitted = Object.freeze({ result });
1576
- stopped = true;
1577
- reason = "result-accepted";
1578
- notifyStop();
1579
- return {
1580
- accepted: true,
1581
- retained: "this-result",
1582
- stop: true
1583
- };
1584
- }
1585
- }] : [],
1586
- {
1587
- name: "stop",
1588
- description: "Declare the run complete.",
1589
- inputSchema: {
1590
- type: "object",
1591
- properties: { reason: {
1592
- type: "string",
1593
- description: "Why you are stopping."
1594
- } }
1595
- },
1596
- handler: (raw) => {
1597
- const blocking = blockingQuestionsForStop();
1598
- if (blocking.length) return Promise.resolve({
1599
- stopped: false,
1600
- error: "unresolved-blocking-questions",
1601
- questions: blocking
1602
- });
1603
- stopped = true;
1604
- const r = obj(raw).reason;
1605
- reason = typeof r === "string" ? r : void 0;
1606
- notifyStop();
1607
- return Promise.resolve({ stopped: true });
1608
- }
1609
- }
1610
- ];
1611
- if (opts.analysts) {
1612
- tools.push({
1613
- name: "list_analysts",
1614
- description: "List trace-analyst lenses available to run over a settled worker.",
1615
- inputSchema: {
1616
- type: "object",
1617
- properties: {}
1618
- },
1619
- handler: () => Promise.resolve({ analysts: opts.analysts?.kinds })
1620
- });
1621
- tools.push({
1622
- name: "run_analyst",
1623
- description: "Apply an analyst lens to a settled worker trace.",
1624
- inputSchema: {
1625
- type: "object",
1626
- properties: {
1627
- kind: {
1628
- type: "string",
1629
- description: "The analyst kind id."
1630
- },
1631
- workerId: idArg
1632
- },
1633
- required: ["kind", "workerId"]
1634
- },
1635
- handler: async (raw) => {
1636
- const a = obj(raw);
1637
- const id = str(a.workerId, "workerId");
1638
- const node = nodeForWorker(id);
1639
- if (!node) return { error: `unknown workerId ${JSON.stringify(id)}` };
1640
- if (isLiveNodeStatus(node.status)) return { error: `worker ${JSON.stringify(id)} has not settled — no trace to analyze yet` };
1641
- const trace = ledger.find((worker) => worker.id === id)?.trace ?? node.trace ?? {
1642
- status: "unavailable",
1643
- reason: "legacy-settlement-without-trace-evidence"
1644
- };
1645
- let store;
1646
- try {
1647
- store = await workerTraceAnalysisStore(trace, opts.blobs);
1648
- } catch (error) {
1649
- return {
1650
- error: error instanceof Error ? error.message : String(error),
1651
- trace
1652
- };
1653
- }
1654
- return { findings: await opts.analysts?.run(str(a.kind, "kind"), store) };
1655
- }
1656
- });
1657
- }
1658
- const abortWorker = (ref, reason) => {
1659
- const live = opts.scope.view.nodes.filter((node) => isLiveNodeStatus(node.status));
1660
- const target = live.find((node) => node.id === ref) ?? live.find((node) => profileNameByWorker.get(node.id) === ref) ?? live.find((node) => node.label === ref);
1661
- if (target === void 0) return void 0;
1662
- const handle = liveHandles.get(target.id);
1663
- if (handle === void 0) return void 0;
1664
- handle.abort(reason);
1665
- return {
1666
- id: target.id,
1667
- label: target.label
1668
- };
1669
- };
1670
- return {
1671
- tools,
1672
- ready,
1673
- history: () => bus.history(),
1674
- raiseFinding: (finding) => bus.publish({
1675
- type: "finding",
1676
- finding: canonicalFindingEvent(finding)
1677
- }).then(() => void 0),
1678
- stats: () => opts.preflightSpawn === void 0 ? bus.stats() : {
1679
- ...bus.stats(),
1680
- preflight: { ...preflightCounts }
1681
- },
1682
- isStopped: () => stopped,
1683
- stopReason: () => reason,
1684
- submittedResult: () => submitted,
1685
- settled: () => ledger,
1686
- questions: () => questions,
1687
- drainResolved,
1688
- abortWorker,
1689
- ...peerMail ? { peerMail } : {}
1690
- };
1691
- }
1692
- function nextUnkeyedAssignmentOrdinal(scope) {
1693
- let next = 0;
1694
- const views = [scope.resume?.view, scope.view];
1695
- for (const view of views) {
1696
- if (view === void 0) continue;
1697
- for (const node of view.nodes) {
1698
- const match = /^ordinal:(\d+)$/.exec(node.assignmentId ?? "");
1699
- if (match === null) continue;
1700
- const ordinal = Number(match[1]);
1701
- if (!Number.isSafeInteger(ordinal)) throw new Error(`coordination: durable assignment id '${node.assignmentId}' exceeds the safe ordinal range`);
1702
- next = Math.max(next, ordinal + 1);
1703
- }
1704
- }
1705
- if (!Number.isSafeInteger(next)) throw new Error("coordination: durable assignment ordinal space is exhausted");
1706
- return next;
1707
- }
1708
- /** Stringify a findings payload for a routed delivery; never throws (a cyclic payload degrades to
1709
- * its String form rather than killing the settle path). */
1710
- function safeJsonText(value) {
1711
- if (typeof value === "string") return value;
1712
- try {
1713
- return JSON.stringify(value) ?? String(value);
1714
- } catch {
1715
- return String(value);
1716
- }
1717
- }
1718
- //#endregion
1719
- //#region src/runtime/supervise/completion-gate.ts
1720
- /**
1721
- *
1722
- * The completion-oracle: **settled ⟺ DELIVERED.**
1723
- *
1724
- * Foreman's one hard lesson (0/18 self-improvement deliverables) — "done" must mean a check
1725
- * PASSED, not the agent's say-so. `gateOnDeliverable` wraps an `Executor` so its settlement
1726
- * is `valid` ONLY when the deliverable check passes. The child still RUNS and settles (its
1727
- * spend is conserved into the pool either way), but a child that ran WITHOUT delivering
1728
- * settles `valid:false` — so a keep-best driver never counts it as done, and a gate never
1729
- * inflates with self-judged wins.
1730
- *
1731
- * Dual-purpose by construction:
1732
- * - product: the agent fleet only advances on real, checked deliverables.
1733
- * - proof: the gate's `valid` is the honest settle — equal-k comparisons can't be gamed by an
1734
- * arm that "ran" without producing the artifact.
1735
- *
1736
- * The check is a DEPLOYABLE oracle — a test command, a state verifier, the commit0 judge —
1737
- * read off the child's output, never the model judging itself. A throwing check is
1738
- * fail-closed (not delivered), never a crash.
1739
- *
1740
- * @experimental
1741
- */
1742
- /**
1743
- * Wrap an `Executor` so its settlement `valid` reflects the deliverable check, not the
1744
- * inner verdict. Handles both `execute` shapes (one-shot `Promise<ExecutorResult>` and
1745
- * streaming `AsyncIterable<UsageEvent>` + `resultArtifact()`); the check runs once the inner
1746
- * executor has produced its output. The inner `score` is preserved; only `valid` is gated.
1747
- */
1748
- function gateOnDeliverable(inner, deliverable) {
1749
- let gated;
1750
- const check = async (out, baseScore) => {
1751
- let delivered;
1752
- try {
1753
- delivered = await deliverable.check(out) === true;
1754
- } catch {
1755
- delivered = false;
1756
- }
1757
- return {
1758
- valid: delivered,
1759
- score: baseScore ?? (delivered ? 1 : 0)
1760
- };
1761
- };
1762
- /**
1763
- * Ask the delivery question once, from whatever the inner executor managed to produce.
1764
- *
1765
- * Fail-closed on the artifact being unavailable: an executor that never produced one delivered
1766
- * nothing, and leaving `gated` unset keeps the existing invalid-by-default reading.
1767
- */
1768
- const settleVerdict = async () => {
1769
- let art;
1770
- try {
1771
- art = inner.resultArtifact();
1772
- } catch {
1773
- return;
1774
- }
1775
- gated = await check(art.out, art.verdict?.score);
1776
- };
1777
- return inheritRuntimeOwnedExecutorAttestation(inner, {
1778
- runtime: inner.runtime,
1779
- ...inner.budgetExempt !== void 0 ? { budgetExempt: inner.budgetExempt } : {},
1780
- ...inner.deliver ? { deliver: (m) => inner.deliver?.(m) } : {},
1781
- ...inner.progress ? { progress: () => inner.progress?.() } : {},
1782
- ...inner.traceSource ? { traceSource: () => inner.traceSource?.() } : {},
1783
- ...inner.metered ? { metered: () => inner.metered?.() } : {},
1784
- execute(task, signal) {
1785
- const r = inner.execute(task, signal);
1786
- if (isAsyncIterable$1(r)) return (async function* () {
1787
- try {
1788
- for await (const ev of r) yield ev;
1789
- } finally {
1790
- await settleVerdict();
1791
- }
1792
- })();
1793
- return (async () => {
1794
- let res;
1795
- try {
1796
- res = await r;
1797
- } catch (error) {
1798
- await settleVerdict();
1799
- throw error;
1800
- }
1801
- gated = await check(res.out, res.verdict?.score);
1802
- return {
1803
- ...res,
1804
- verdict: gated
1805
- };
1806
- })();
1807
- },
1808
- teardown: (grace) => inner.teardown(grace),
1809
- resultArtifact() {
1810
- const art = inner.resultArtifact();
1811
- return {
1812
- ...art,
1813
- verdict: gated ?? art.verdict
1814
- };
1815
- }
1816
- });
1817
- }
1818
- /**
1819
- * Transform a Runtime executor's terminal artifact without losing its private
1820
- * profile-materialization attestation or altering its measured spend. This is
1821
- * the composition point for deterministic post-processing and grading; callers
1822
- * must not rebuild an Executor around a model transport merely to change `out`.
1823
- */
1824
- function mapExecutorResult(inner, map) {
1825
- let mapped;
1826
- const settle = async (result, task) => {
1827
- const transformed = await map(result, task);
1828
- mapped = {
1829
- outRef: transformed.outRef,
1830
- out: transformed.out,
1831
- ...transformed.verdict ? { verdict: transformed.verdict } : {},
1832
- spent: result.spent
1833
- };
1834
- return mapped;
1835
- };
1836
- return inheritRuntimeOwnedExecutorAttestation(inner, {
1837
- runtime: inner.runtime,
1838
- ...inner.budgetExempt !== void 0 ? { budgetExempt: inner.budgetExempt } : {},
1839
- ...inner.deliver ? { deliver: (message) => inner.deliver?.(message) } : {},
1840
- ...inner.progress ? { progress: () => inner.progress?.() } : {},
1841
- ...inner.traceSource ? { traceSource: () => inner.traceSource?.() } : {},
1842
- ...inner.accounting ? { accounting: () => inner.accounting?.() } : {},
1843
- ...inner.metered ? { metered: () => inner.metered?.() } : {},
1844
- execute(task, signal) {
1845
- const execution = inner.execute(task, signal);
1846
- if (isAsyncIterable$1(execution)) return (async function* () {
1847
- for await (const event of execution) yield event;
1848
- await settle(inner.resultArtifact(), task);
1849
- })();
1850
- return (async () => settle(await execution, task))();
1851
- },
1852
- teardown: (grace) => inner.teardown(grace),
1853
- resultArtifact() {
1854
- if (!mapped) throw new Error("mapExecutorResult: resultArtifact() read before execute()");
1855
- return mapped;
1856
- }
1857
- });
1858
- }
1859
- function isAsyncIterable$1(v) {
1860
- return v != null && typeof v[Symbol.asyncIterator] === "function";
1861
- }
1862
- //#endregion
1863
- //#region src/runtime/supervise/otel-spans.ts
1864
- /**
1865
- * Supervisor tree → OTLP spans. OPT-IN, off by default.
1866
- *
1867
- * WHY. A supervised tree is legible today only by parsing this package's own spawn journal, so
1868
- * every other multi-agent shape on the machine (a coding-CLI's subagents, a pi fanout, ad-hoc tool
1869
- * parallelism) needs its own bespoke reader. A span carrying `parent_span_id` IS a tree, and any
1870
- * system can emit one — so emitting spans makes the supervisor readable by the same viewer as
1871
- * everything else, with no per-system reader.
1872
- *
1873
- * WHAT IT IS NOT. This is telemetry, never the record of truth. The spawn journal remains the sole
1874
- * durable ledger for replay/resume and for cost; nothing here is read back, and a run whose export
1875
- * fails is unaffected in every observable way. The two data models are deliberately separate.
1876
- *
1877
- * HOW IT ATTACHES. This is a pure `RuntimeHooks` observer over the lifecycle events `Scope` ALREADY
1878
- * emits — `agent.spawn` (a node opened), `agent.child` (a node settled), `agent.turn` (a driver
1879
- * inference turn was metered). It adds no event, mutates no journal, and changes no result. Because
1880
- * `Scope` re-seeds the same hooks into every nested scope (`makeNestedScopeSeam`), one observer sees
1881
- * the WHOLE recursion at arbitrary depth.
1882
- *
1883
- * SPAN SHAPE. One span per supervised node, opened at spawn and closed at settle, parented to its
1884
- * parent node's span; the run root is the trace. Driver inference rides as an `LLM` child span under
1885
- * the node that metered it. Attributes reuse the vocabulary the rest of the stack already reads
1886
- * (`openinference.span.kind`, `agent.name`, `llm.token_count.*`, `llm.cost_usd`, `tangle.cost.usd`)
1887
- * — see `@tangle-network/agent-eval`'s `src/trace/attribute-vocabulary.ts`, which is the consumer.
1888
- *
1889
- * UNKNOWN IS NEVER ZERO. `Spend.tokensKnown === false` / `usdKnown === false` mark work that
1890
- * HAPPENED with an unreported count. Those spans OMIT the token/cost attribute entirely and set
1891
- * `tangle.supervise.tokens_known` / `tangle.supervise.cost_known` to `false`, so a reader can never
1892
- * mistake an unmeasured turn for a free one.
1893
- */
1894
- /** OTEL status codes (`UNSET` / `OK` / `ERROR`) — the numeric wire values `OtelSpan.status` carries. */
1895
- const STATUS_UNSET = 0;
1896
- const STATUS_OK = 1;
1897
- const STATUS_ERROR = 2;
1898
- /** Longest string attribute value written from free-form detail, so an oversized turn payload
1899
- * cannot inflate a span. Identity/label attributes we control are never truncated. */
1900
- const MAX_DETAIL_CHARS = 256;
1901
- /**
1902
- * Build the span recorder for one supervised run, or `undefined` when no exporter resolves — the
1903
- * off-by-default path. A run that passes no `exporter` and no `exportConfig` never reaches this
1904
- * function at all; one that passes an `exportConfig` with no endpoint (and no env endpoint) gets
1905
- * `undefined` here, so "configured but unreachable" also costs nothing.
1906
- */
1907
- function createSupervisorSpanRecorder(opts) {
1908
- const exporter = opts.exporter ?? createOtelExporter(opts.exportConfig);
1909
- if (!exporter) return void 0;
1910
- const ownsExporter = opts.exporter === void 0;
1911
- const now = opts.now ?? Date.now;
1912
- const traceId = normalizeTraceId(opts.traceId, opts.runId);
1913
- const rootSpanId = generateSpanId();
1914
- const rootStartMs = now();
1915
- const base = {
1916
- "tangle.run.id": opts.runId,
1917
- "tangle.sessionId": opts.runId,
1918
- ...opts.attributes ?? {}
1919
- };
1920
- /** Node id → its open span. Seeded with the run id ⇒ the root span, because a depth-0 spawn's
1921
- * `parentId` is the run id itself and every deeper spawn's is a real node id. */
1922
- const open = /* @__PURE__ */ new Map();
1923
- const spanIdOf = /* @__PURE__ */ new Map([[opts.runId, rootSpanId]]);
1924
- let finished = false;
1925
- /** Every export is best-effort: a throwing exporter must never reach the run. */
1926
- const emit = (span) => {
1927
- try {
1928
- exporter.exportSpan(span);
1929
- } catch {}
1930
- };
1931
- const span = (spanId, parentSpanId, name, startMs, endMs, attrs, status, message) => ({
1932
- traceId,
1933
- spanId,
1934
- ...parentSpanId ? { parentSpanId } : {},
1935
- name,
1936
- kind: 1,
1937
- startTimeUnixNano: msToNano(startMs),
1938
- endTimeUnixNano: msToNano(Math.max(startMs, endMs)),
1939
- attributes: toOtelAttributes(attrs),
1940
- status: {
1941
- code: status,
1942
- ...message ? { message } : {}
1943
- }
1944
- });
1945
- function onSpawn(event) {
1946
- const p = record(event.payload);
1947
- const childId = str(p.childId);
1948
- if (!childId) return;
1949
- const label = str(p.label) ?? "node";
1950
- const runtime = str(p.runtime);
1951
- const isWait = runtime === "wait";
1952
- const attrs = {
1953
- ...base,
1954
- "openinference.span.kind": isWait ? "CHAIN" : "AGENT",
1955
- "agent.name": label,
1956
- "tangle.supervise.node.id": childId,
1957
- "tangle.supervise.node.label": label,
1958
- "tangle.supervise.node.kind": isWait ? "wait" : "agent",
1959
- "tangle.supervise.tree.root": event.runId
1960
- };
1961
- if (event.parentId) attrs["tangle.supervise.node.parent_id"] = event.parentId;
1962
- if (runtime) attrs["tangle.supervise.node.runtime"] = runtime;
1963
- if (typeof p.depth === "number") attrs["tangle.supervise.node.depth"] = p.depth;
1964
- if (typeof event.stepIndex === "number") attrs["tangle.supervise.node.ordinal"] = event.stepIndex;
1965
- if (p.resumed === true) attrs["tangle.supervise.node.resumed"] = true;
1966
- assignBudget(attrs, p.budget);
1967
- const spanId = generateSpanId();
1968
- spanIdOf.set(childId, spanId);
1969
- open.set(childId, {
1970
- spanId,
1971
- parentSpanId: event.parentId && spanIdOf.get(event.parentId) || rootSpanId,
1972
- name: label,
1973
- startMs: event.timestamp,
1974
- attrs
1975
- });
1976
- }
1977
- function onSettled(event) {
1978
- const p = record(event.payload);
1979
- const childId = str(p.childId);
1980
- if (!childId) return;
1981
- const node = open.get(childId);
1982
- if (!node) return;
1983
- open.delete(childId);
1984
- const status = str(p.status);
1985
- const down = status === "down";
1986
- const attrs = {
1987
- ...node.attrs,
1988
- "tangle.supervise.node.status": status ?? "done"
1989
- };
1990
- if (typeof event.stepIndex === "number") attrs["tangle.supervise.node.seq"] = event.stepIndex;
1991
- if (typeof p.outRef === "string") attrs["tangle.supervise.node.out_ref"] = p.outRef;
1992
- if (typeof p.valid === "boolean") attrs["tangle.supervise.verdict.valid"] = p.valid;
1993
- if (typeof p.score === "number") attrs["tangle.supervise.verdict.score"] = p.score;
1994
- if (down) {
1995
- attrs["error.type"] = p.infra === true ? "infra" : "child-down";
1996
- const reason = str(p.reason);
1997
- if (reason) attrs["error.message"] = truncate(reason);
1998
- if (typeof p.infra === "boolean") attrs["tangle.supervise.node.infra"] = p.infra;
1999
- }
2000
- const wokeBy = str(record(p.wait).settled);
2001
- if (wokeBy) attrs["tangle.supervise.wait.settled"] = wokeBy;
2002
- assignSpend(attrs, p.spent);
2003
- emit(span(node.spanId, node.parentSpanId, node.name, node.startMs, event.timestamp, attrs, down ? STATUS_ERROR : STATUS_OK, down ? str(p.reason) ?? "child down" : void 0));
2004
- }
2005
- function onTurn(event) {
2006
- const p = record(event.payload);
2007
- const parentId = event.parentId;
2008
- const parentSpanId = parentId && spanIdOf.get(parentId) || rootSpanId;
2009
- const attrs = {
2010
- ...base,
2011
- "openinference.span.kind": "LLM",
2012
- "inference.observation_kind": "LLM",
2013
- "tangle.supervise.node.kind": "inference"
2014
- };
2015
- if (parentId) attrs["tangle.supervise.node.id"] = parentId;
2016
- for (const [key, value] of Object.entries(p)) {
2017
- if (key === "spend") continue;
2018
- if (key === "driver" && typeof value === "string") {
2019
- attrs["agent.name"] = value;
2020
- attrs["inference.agent_name"] = value;
2021
- continue;
2022
- }
2023
- if (key === "model" && typeof value === "string") {
2024
- attrs["llm.model_name"] = value;
2025
- continue;
2026
- }
2027
- if (Array.isArray(value)) {
2028
- const scalars = value.filter((v) => typeof v === "string" || typeof v === "number");
2029
- if (scalars.length > 0) attrs[`tangle.supervise.turn.${key}`] = truncate(scalars.join(","));
2030
- continue;
2031
- }
2032
- if (typeof value === "string") attrs[`tangle.supervise.turn.${key}`] = truncate(value);
2033
- else if (typeof value === "number" || typeof value === "boolean") attrs[`tangle.supervise.turn.${key}`] = value;
2034
- }
2035
- const spend = assignSpend(attrs, p.spend);
2036
- const endMs = event.timestamp + (spend?.ms ?? 0);
2037
- emit(span(generateSpanId(), parentSpanId, "gen_ai.client.inference", event.timestamp, endMs, attrs, STATUS_OK));
2038
- }
2039
- return {
2040
- hooks: { onEvent(event) {
2041
- if (finished) return;
2042
- try {
2043
- if (event.phase !== "after") return;
2044
- if (event.target === "agent.spawn") onSpawn(event);
2045
- else if (event.target === "agent.child") onSettled(event);
2046
- else if (event.target === "agent.turn") onTurn(event);
2047
- } catch {}
2048
- } },
2049
- traceId,
2050
- rootSpanId,
2051
- workerTrace(spawningNodeId) {
2052
- return {
2053
- traceId,
2054
- parentSpanId: spanIdOf.get(spawningNodeId) ?? rootSpanId
2055
- };
2056
- },
2057
- async finish(outcome) {
2058
- if (finished) return;
2059
- finished = true;
2060
- const endMs = now();
2061
- try {
2062
- for (const [nodeId, node] of open) emit(span(node.spanId, node.parentSpanId, node.name, node.startMs, endMs, {
2063
- ...node.attrs,
2064
- "tangle.supervise.node.status": "unsettled",
2065
- "tangle.supervise.node.settled": false,
2066
- "tangle.supervise.node.id": nodeId
2067
- }, STATUS_UNSET));
2068
- open.clear();
2069
- emit(span(rootSpanId, opts.parentSpanId, "supervisor.run", rootStartMs, endMs, rootAttrs(base, opts.agentName ?? "supervisor", outcome), rootStatus(outcome), rootMessage(outcome)));
2070
- await exporter.flush();
2071
- if (ownsExporter) await exporter.shutdown();
2072
- } catch {}
2073
- }
2074
- };
2075
- }
2076
- function rootAttrs(base, agentName, outcome) {
2077
- const attrs = {
2078
- ...base,
2079
- "openinference.span.kind": "AGENT",
2080
- "inference.observation_kind": "AGENT",
2081
- "agent.name": agentName,
2082
- "inference.agent_name": agentName,
2083
- "tangle.supervise.node.kind": "root"
2084
- };
2085
- const result = outcome?.result;
2086
- if (result) {
2087
- attrs["tangle.supervise.result"] = result.kind;
2088
- if (result.kind === "no-winner") {
2089
- attrs["tangle.supervise.reason"] = result.reason;
2090
- attrs["tangle.supervise.down_count"] = result.downCount;
2091
- if (result.reason === "driver-failed") {
2092
- attrs["error.type"] = result.error.name;
2093
- attrs["error.message"] = truncate(result.error.message);
2094
- }
2095
- }
2096
- assignSpend(attrs, result.spentTotal);
2097
- }
2098
- if (outcome?.error !== void 0) {
2099
- attrs["tangle.supervise.result"] = "error";
2100
- const err = outcome.error;
2101
- attrs["error.type"] = err instanceof Error ? err.name : typeof err;
2102
- attrs["error.message"] = truncate(err instanceof Error ? err.message : String(err));
2103
- }
2104
- return attrs;
2105
- }
2106
- function rootStatus(outcome) {
2107
- if (outcome?.error !== void 0) return STATUS_ERROR;
2108
- if (!outcome?.result) return STATUS_UNSET;
2109
- return outcome.result.kind === "winner" ? STATUS_OK : STATUS_ERROR;
2110
- }
2111
- function rootMessage(outcome) {
2112
- if (outcome?.error !== void 0) return truncate(outcome.error instanceof Error ? outcome.error.message : String(outcome.error));
2113
- const result = outcome?.result;
2114
- return result && result.kind === "no-winner" ? result.reason : void 0;
2115
- }
2116
- /**
2117
- * Write a `Spend` onto a span, honouring the "a missing measurement is never zero" invariant: a
2118
- * channel marked NOT known contributes no number at all and instead flags itself, so nothing
2119
- * downstream can sum an unmeasured turn as free. Returns the spend it read (for the duration).
2120
- */
2121
- function assignSpend(attrs, value) {
2122
- if (!isRecord(value)) return void 0;
2123
- const spend = value;
2124
- const tokens = isRecord(spend.tokens) ? spend.tokens : void 0;
2125
- if (spend.tokensKnown === false) attrs["tangle.supervise.tokens_known"] = false;
2126
- else if (tokens) {
2127
- if (typeof tokens.input === "number") attrs["llm.token_count.prompt"] = tokens.input;
2128
- if (typeof tokens.output === "number") attrs["llm.token_count.completion"] = tokens.output;
2129
- }
2130
- if (spend.usdKnown === false) attrs["tangle.supervise.cost_known"] = false;
2131
- else if (typeof spend.usd === "number" && Number.isFinite(spend.usd)) {
2132
- attrs["llm.cost_usd"] = spend.usd;
2133
- attrs["tangle.cost.usd"] = spend.usd;
2134
- }
2135
- if (typeof spend.iterations === "number") attrs["tangle.supervise.iterations"] = spend.iterations;
2136
- if (typeof spend.ms === "number" && spend.ms > 0) attrs["tangle.supervise.duration_ms"] = spend.ms;
2137
- return spend;
2138
- }
2139
- function assignBudget(attrs, value) {
2140
- if (!isRecord(value)) return;
2141
- const budget = value;
2142
- if (typeof budget.maxTokens === "number") attrs["tangle.supervise.budget.max_tokens"] = budget.maxTokens;
2143
- if (typeof budget.maxIterations === "number") attrs["tangle.supervise.budget.max_iterations"] = budget.maxIterations;
2144
- if (typeof budget.maxUsd === "number") attrs["tangle.supervise.budget.max_usd"] = budget.maxUsd;
2145
- }
2146
- /**
2147
- * A trace id must be 32 hex characters. A caller-supplied one is used verbatim when it already is;
2148
- * anything else (and the default) is DERIVED from the run id by content address — deterministic, so
2149
- * a resumed run rejoins the trace its first process opened rather than forking a new one.
2150
- */
2151
- function normalizeTraceId(traceId, runId) {
2152
- if (traceId && /^[0-9a-f]{32}$/i.test(traceId)) return traceId.toLowerCase();
2153
- return contentAddress(traceId ?? runId).slice(7, 39);
2154
- }
2155
- function msToNano(ms) {
2156
- return (BigInt(Number.isFinite(ms) ? Math.floor(ms) : 0) * 1000000n).toString();
2157
- }
2158
- function isRecord(value) {
2159
- return typeof value === "object" && value !== null && !Array.isArray(value);
2160
- }
2161
- function record(value) {
2162
- return isRecord(value) ? value : {};
2163
- }
2164
- function str(value) {
2165
- return typeof value === "string" && value.length > 0 ? value : void 0;
2166
- }
2167
- function truncate(value) {
2168
- return value.length > MAX_DETAIL_CHARS ? `${value.slice(0, MAX_DETAIL_CHARS)}…` : value;
2169
- }
2170
- //#endregion
2171
- //#region src/runtime/supervise/coordination-log.ts
2172
- /**
2173
- * Durable side-log for coordination evidence the spawn journal does not own: questions, analyst
2174
- * findings, answer decisions, authorized continuation receipts, delivery-attempt markers, and
2175
- * delivery outcomes. A durable run
2176
- * (`supervise({ runDir })`) appends them as they publish and loads them on resume, so a restarted
2177
- * coordinator retains the exact evidence produced by prior processes.
2178
- *
2179
- * Answer down-events also fold status on load: a question answered before the crash reloads as
2180
- * `answered`, not as a re-blocking `open`. Settled events are skipped (the spawn journal is their
2181
- * ledger). A receipt followed by an attempt but no outcome proves the process died in the delivery
2182
- * window; that outcome remains unknown and no prior instruction is auto-delivered.
2183
- *
2184
- * JSONL, one fsynced record per event, keyed by `runId` — several runs may share one log file
2185
- * exactly as they share one spawn-journal file.
2186
- *
2187
- * @experimental
2188
- */
2189
- /** Persist prior context plus exact continuation authorization, attempt, and result evidence.
2190
- * Settlements have their own journal. */
2191
- function persisted(event) {
2192
- return event.type !== "settled";
2193
- }
2194
- /** FS-backed `CoordinationLog`: append-only JSONL, fsynced per record. */
2195
- var FileCoordinationLog = class {
2196
- path;
2197
- appendTail = Promise.resolve();
2198
- constructor(path) {
2199
- this.path = path;
2200
- }
2201
- async append(runId, record, ownerId) {
2202
- if (!persisted(record.event)) return;
2203
- const append = this.appendTail.then(() => this.appendRecord(runId, record, ownerId));
2204
- this.appendTail = append.catch(() => void 0);
2205
- return append;
2206
- }
2207
- async appendRecord(runId, busRecord, ownerId) {
2208
- const fs = await import("node:fs/promises");
2209
- const path = await import("node:path");
2210
- await fs.mkdir(path.dirname(this.path), { recursive: true });
2211
- const record = {
2212
- runId,
2213
- ...ownerId !== void 0 ? { ownerId } : {},
2214
- ...busRecord
2215
- };
2216
- const needsSeparator = await prepareJsonlAppend(this.path);
2217
- const fh = await fs.open(this.path, "a");
2218
- try {
2219
- await writeAllBytes(fh, `${needsSeparator ? "\n" : ""}${JSON.stringify(record)}\n`);
2220
- await fh.sync();
2221
- } finally {
2222
- await fh.close();
2223
- }
2224
- }
2225
- async load(runId, ownerId) {
2226
- const fs = await import("node:fs/promises");
2227
- let text;
2228
- try {
2229
- text = await fs.readFile(this.path, "utf8");
2230
- } catch (err) {
2231
- if (isNoEntError(err)) return emptyPriorCoordination(ownerId);
2232
- throw err;
2233
- }
2234
- const byId = /* @__PURE__ */ new Map();
2235
- const findings = [];
2236
- const continuations = [];
2237
- const deliveryEvidence = [];
2238
- const mail = [];
2239
- const records = [];
2240
- let legacySeq = 0;
2241
- for (const stored of parseCommittedJsonLines(text, this.path)) {
2242
- if (stored.runId !== runId) continue;
2243
- if (ownerId !== void 0 && stored.ownerId !== ownerId) continue;
2244
- const record = "seq" in stored ? {
2245
- seq: stored.seq,
2246
- at: stored.at,
2247
- priority: stored.priority,
2248
- event: stored.event
2249
- } : {
2250
- seq: legacySeq++,
2251
- at: Date.parse(stored.at),
2252
- priority: 0,
2253
- event: stored.event
2254
- };
2255
- records.push(record);
2256
- const ev = record.event;
2257
- if (ev.type === "delivery-attempt" || ev.type === "steer" || ev.type === "answer") deliveryEvidence.push(ev);
2258
- if (ev.type === "question") byId.set(ev.question.id, ev.question);
2259
- else if (ev.type === "finding") findings.push(ev.finding);
2260
- else if (ev.type === "answer") {
2261
- const prior = byId.get(ev.questionId);
2262
- if (prior && ev.down.delivered) byId.set(ev.questionId, {
2263
- ...prior,
2264
- status: "answered",
2265
- decision: {
2266
- kind: "answer",
2267
- answer: ev.down.instruction,
2268
- by: "prior-run"
2269
- }
2270
- });
2271
- } else if (ev.type === "instruction") continuations.push(ev.instruction);
2272
- else if (ev.type === "mail") mail.push(ev.mail);
232
+ const span = (spanId, parentSpanId, name, startMs, endMs, attrs, status, message) => ({
233
+ traceId,
234
+ spanId,
235
+ ...parentSpanId ? { parentSpanId } : {},
236
+ name,
237
+ kind: 1,
238
+ startTimeUnixNano: msToNano(startMs),
239
+ endTimeUnixNano: msToNano(Math.max(startMs, endMs)),
240
+ attributes: toOtelAttributes(attrs),
241
+ status: {
242
+ code: status,
243
+ ...message ? { message } : {}
2273
244
  }
2274
- return {
2275
- ...ownerId !== void 0 ? { ownerId } : {},
2276
- questions: [...byId.values()],
2277
- findings,
2278
- continuations,
2279
- deliveryEvidence,
2280
- mail,
2281
- records
245
+ });
246
+ function onSpawn(event) {
247
+ const p = record(event.payload);
248
+ const childId = str(p.childId);
249
+ if (!childId) return;
250
+ const label = str(p.label) ?? "node";
251
+ const runtime = str(p.runtime);
252
+ const isWait = runtime === "wait";
253
+ const attrs = {
254
+ ...base,
255
+ "openinference.span.kind": isWait ? "CHAIN" : "AGENT",
256
+ "agent.name": label,
257
+ "tangle.supervise.node.id": childId,
258
+ "tangle.supervise.node.label": label,
259
+ "tangle.supervise.node.kind": isWait ? "wait" : "agent",
260
+ "tangle.supervise.tree.root": event.runId
2282
261
  };
2283
- }
2284
- };
2285
- function emptyPriorCoordination(ownerId) {
2286
- return {
2287
- ...ownerId !== void 0 ? { ownerId } : {},
2288
- questions: [],
2289
- findings: [],
2290
- continuations: [],
2291
- deliveryEvidence: [],
2292
- mail: [],
2293
- records: []
2294
- };
2295
- }
2296
- //#endregion
2297
- //#region src/runtime/supervise/run-context.ts
2298
- /**
2299
- *
2300
- * `createInMemoryRunContext` — the one-call bundle of the in-memory stores a
2301
- * `createSupervisor().run(root, task, opts)` needs: a fresh `InMemorySpawnJournal`
2302
- * (the event-sourced spawn log), a fresh `InMemoryResultBlobStore` (the
2303
- * content-addressed `outRef` payload store the driver's `observe`/`finalize` reads
2304
- * settled outputs through), and a fresh `createExecutorRegistry()` (the open
2305
- * `AgentSpec → Executor` resolver).
2306
- *
2307
- * It exists to kill the boilerplate every offline/local supervised run repeats by
2308
- * hand — three constructors threaded into `SupervisorOpts` — and to single-source the
2309
- * ONE wiring invariant that is easy to get wrong: when the root is the recursive
2310
- * `driverAgent` LLM-driver brain AND it may spawn DRIVER children (agents
2311
- * driving agents), the registry MUST be wrapped with `withDriverExecutor` so a
2312
- * `role: 'driver'` child resolves to the nested-scope executor — and that SAME blob
2313
- * store MUST be the one passed to `driverAgent({ blobs })`, or the driver
2314
- * reads from a different store than the scope writes to. Pass `{ withDriver: true }`
2315
- * and reuse the returned `blobs` for both.
2316
- *
2317
- * The spread shape matches `SupervisorOpts` exactly, so the call site reads:
2318
- * const run = createInMemoryRunContext()
2319
- * await createSupervisor().run(root, task, { budget, runId, ...run })
2320
- *
2321
- * @experimental
2322
- */
2323
- /**
2324
- * Build a fresh in-memory run context. Every call returns NEW stores (no shared global
2325
- * state between runs), so two runs never cross-contaminate their journals/blobs.
2326
- */
2327
- function createInMemoryRunContext(opts = {}) {
2328
- const base = createExecutorRegistry();
2329
- return {
2330
- journal: new InMemorySpawnJournal(),
2331
- blobs: new InMemoryResultBlobStore(),
2332
- executors: opts.withDriver ? withDriverExecutor(base) : base
2333
- };
2334
- }
2335
- /**
2336
- * Build a DURABLE run context: the spawn journal and the result blobs are file-backed (fsynced
2337
- * per append/write) under `dir`, and the context carries `resume: true` so spreading it into
2338
- * `SupervisorOpts` makes the supervisor `loadTree`-first. A run that dies mid-flight therefore
2339
- * resumes when it is re-run with the SAME `runId` and the SAME `dir`: the committed children come
2340
- * back on `Scope.resume` (rehydrated by `replaySpawnTree`) instead of being re-executed.
2341
- *
2342
- * Layout: `${dir}/spawn-journal.jsonl` (one JSONL record per event), `${dir}/blobs/` (one
2343
- * content-addressed JSON file per settled result), and `${dir}/coordination-log.jsonl`
2344
- * (questions, findings, answer decisions, and authorized continuation receipts retained as
2345
- * evidence). The directory is created on first write.
2346
- *
2347
- * Opt-in by construction — `createInMemoryRunContext()` is unchanged and stays the default, so no
2348
- * existing consumer writes to disk or resumes unless it asks for this.
2349
- */
2350
- function createFileRunContext(dir, opts = {}) {
2351
- const base = createExecutorRegistry();
2352
- return {
2353
- journal: new FileSpawnJournal(`${dir}/spawn-journal.jsonl`),
2354
- blobs: new FileResultBlobStore(`${dir}/blobs`),
2355
- executors: opts.withDriver ? withDriverExecutor(base) : base,
2356
- resume: true,
2357
- coordinationLog: new FileCoordinationLog(`${dir}/coordination-log.jsonl`)
2358
- };
2359
- }
2360
- //#endregion
2361
- //#region src/runtime/anytime.ts
2362
- /**
2363
- * The best-so-far fold — the ONE definition of "how good was the run after k results", shared by
2364
- * the post-run anytime report below and by the LIVE progress-based stop rules
2365
- * (`supervise/stop-rules.ts`). Given the observed objective per settled result in order, it returns
2366
- * the running maximum. A result with no objective (`undefined` — it failed, or it was never
2367
- * scored) carries the previous best forward rather than resetting it.
2368
- *
2369
- * It is extracted rather than duplicated on purpose: a stop rule that decides a run has plateaued
2370
- * must agree, number for number, with the report that later says whether stopping was right.
2371
- */
2372
- function bestSoFar(values) {
2373
- const out = [];
2374
- let best = 0;
2375
- for (const v of values) {
2376
- if (typeof v === "number" && v > best) best = v;
2377
- out.push(best);
2378
- }
2379
- return out;
2380
- }
2381
- /** Mean of a best-so-far curve — the anytime AUC when the curve is normalized to [0,1]. Higher =
2382
- * the run climbed earlier. Shared with the stop rules so "improving" means one thing. */
2383
- function areaUnderCurve(curve) {
2384
- if (curve.length === 0) return 0;
2385
- return curve.reduce((s, v) => s + v, 0) / curve.length;
2386
- }
2387
- /**
2388
- * How many trailing entries of a best-so-far curve are within `minDelta` of the curve's value
2389
- * `window` steps back — i.e. the length of the current PLATEAU, in settles. `0` means the most
2390
- * recent settle improved the best by more than `minDelta`.
2391
- *
2392
- * The plateau math the live stop rules read. Defined here, beside the report that measures whether
2393
- * the plateau was real, so there is exactly one notion of "not improving".
2394
- */
2395
- function plateauLength(curve, minDelta) {
2396
- if (curve.length === 0) return 0;
2397
- const last = curve[curve.length - 1];
2398
- let i = curve.length - 1;
2399
- while (i > 0 && last - curve[i - 1] <= minDelta) i -= 1;
2400
- return curve.length - 1 - i;
2401
- }
2402
- const median = (xs) => {
2403
- if (xs.length === 0) return null;
2404
- const s = [...xs].sort((a, b) => a - b);
2405
- const mid = Math.floor(s.length / 2);
2406
- return s.length % 2 === 1 ? s[mid] : (s[mid - 1] + s[mid]) / 2;
2407
- };
2408
- /** Derive anytime metrics from waterfall spans. `targets` are the satisficing score
2409
- * bars (default [1] = fully resolved; COCO-style multi-target: [0.5, 0.8, 1]);
2410
- * `targetFor` overrides the bar per task (task-specific satisfaction) — when set, the
2411
- * per-task bar replaces every entry of `targets` for that task. */
2412
- function anytimeReport(spans, opts) {
2413
- const targets = opts?.targets ?? [1];
2414
- const byRun = /* @__PURE__ */ new Map();
2415
- for (const s of spans) {
2416
- if (!s.label.startsWith("shot:")) continue;
2417
- const list = byRun.get(s.runId) ?? [];
2418
- list.push(s);
2419
- byRun.set(s.runId, list);
2420
- }
2421
- const perTask = [];
2422
- for (const [runId, shots] of byRun) {
2423
- const m = runId.match(/^agentic:(.+):(.+)$/);
2424
- const strategy = m?.[1] ?? runId;
2425
- const taskId = m?.[2] ?? runId;
2426
- const ordered = [...shots].sort((a, b) => (a.endMs ?? a.startMs) - (b.endMs ?? b.startMs));
2427
- const t0 = Math.min(...ordered.map((s) => s.startMs));
2428
- const taskTargets = opts?.targetFor ? [opts.targetFor(taskId)] : targets;
2429
- const bests = bestSoFar(ordered.map((s) => typeof s.score === "number" ? s.score : void 0));
2430
- let cumUsd = 0;
2431
- const points = [];
2432
- const hits = {};
2433
- for (const t of taskTargets) hits[String(t)] = null;
2434
- for (const [i, s] of ordered.entries()) {
2435
- cumUsd += s.usd;
2436
- const best = bests[i];
2437
- const elapsedMs = (s.endMs ?? s.startMs) - t0;
2438
- points.push({
2439
- elapsedMs,
2440
- cumUsd,
2441
- best
2442
- });
2443
- for (const t of taskTargets) if (hits[String(t)] === null && best >= t) hits[String(t)] = {
2444
- ms: elapsedMs,
2445
- shots: points.length,
2446
- usd: cumUsd
2447
- };
2448
- }
2449
- perTask.push({
2450
- taskId,
2451
- strategy,
2452
- points,
2453
- hits
262
+ if (event.parentId) attrs["tangle.supervise.node.parent_id"] = event.parentId;
263
+ if (runtime) attrs["tangle.supervise.node.runtime"] = runtime;
264
+ if (typeof p.depth === "number") attrs["tangle.supervise.node.depth"] = p.depth;
265
+ if (typeof event.stepIndex === "number") attrs["tangle.supervise.node.ordinal"] = event.stepIndex;
266
+ if (p.resumed === true) attrs["tangle.supervise.node.resumed"] = true;
267
+ assignBudget(attrs, p.budget);
268
+ const spanId = generateSpanId();
269
+ spanIdOf.set(childId, spanId);
270
+ open.set(childId, {
271
+ spanId,
272
+ parentSpanId: event.parentId && spanIdOf.get(event.parentId) || rootSpanId,
273
+ name: label,
274
+ startMs: event.timestamp,
275
+ attrs
2454
276
  });
2455
277
  }
2456
- const byStrategy = /* @__PURE__ */ new Map();
2457
- for (const t of perTask) {
2458
- const list = byStrategy.get(t.strategy) ?? [];
2459
- list.push(t);
2460
- byStrategy.set(t.strategy, list);
2461
- }
2462
- const perStrategy = [];
2463
- for (const [strategy, tasks] of byStrategy) {
2464
- const totalMs = tasks.reduce((s, t) => s + (t.points[t.points.length - 1]?.elapsedMs ?? 0), 0);
2465
- const totalUsd = tasks.reduce((s, t) => s + (t.points[t.points.length - 1]?.cumUsd ?? 0), 0);
2466
- const maxShots = Math.max(0, ...tasks.map((t) => t.points.length));
2467
- const curveByShot = [];
2468
- for (let i = 0; i < maxShots; i += 1) {
2469
- const vals = tasks.map((t) => t.points[Math.min(i, t.points.length - 1)].best);
2470
- curveByShot.push(vals.reduce((s, v) => s + v, 0) / vals.length);
2471
- }
2472
- const auc = areaUnderCurve(curveByShot);
2473
- const summaryTargets = opts?.targetFor ? [NaN] : targets;
2474
- for (const t of summaryTargets) {
2475
- const key = (taskCurve) => opts?.targetFor ? Object.values(taskCurve.hits)[0] ?? null : taskCurve.hits[String(t)] ?? null;
2476
- const reached = tasks.filter((x) => key(x) !== null);
2477
- perStrategy.push({
2478
- strategy,
2479
- target: t,
2480
- tasks: tasks.length,
2481
- reachedTarget: reached.length,
2482
- medianTttMs: median(reached.map((x) => key(x).ms)),
2483
- medianShotsToTarget: median(reached.map((x) => key(x).shots)),
2484
- ertMs: reached.length > 0 ? totalMs / reached.length : null,
2485
- erUsd: reached.length > 0 ? totalUsd / reached.length : null,
2486
- curveByShot,
2487
- auc
2488
- });
2489
- }
2490
- }
2491
- perStrategy.sort((a, b) => a.strategy.localeCompare(b.strategy) || a.target - b.target);
2492
- return {
2493
- targets,
2494
- perTask,
2495
- perStrategy
2496
- };
2497
- }
2498
- /** One row per (strategy, satisficing target): the shareable time-to-satisfactory table. */
2499
- function renderAnytimeTable(report) {
2500
- const lines = [`anytime metrics · satisficing targets [${report.targets.join(", ")}] · ERT = Σ all wall-time / #successes (COCO)`, "strategy ≥tgt reach med-TTT med-shots ERT(all-in) $/success AUC curve"];
2501
- for (const s of report.perStrategy) {
2502
- const curve = s.curveByShot.map((v) => "▁▂▃▄▅▆▇█"[Math.min(7, Math.floor(v * 8))]).join("");
2503
- const tgt = Number.isNaN(s.target) ? "task" : s.target.toFixed(2);
2504
- lines.push(`${s.strategy.padEnd(19)} ${tgt.padStart(4)} ${String(s.reachedTarget).padStart(4)}/${String(s.tasks).padEnd(3)} ${s.medianTttMs === null ? " —" : `${(s.medianTttMs / 1e3).toFixed(1).padStart(6)}s`} ${s.medianShotsToTarget === null ? " —" : String(s.medianShotsToTarget).padStart(5)} ${s.ertMs === null ? " —" : `${(s.ertMs / 1e3).toFixed(1).padStart(9)}s`} ${s.erUsd === null ? " —" : `$${s.erUsd.toFixed(4)}`} ${s.auc.toFixed(2)} ${curve}`);
2505
- }
2506
- return lines.join("\n");
2507
- }
2508
- //#endregion
2509
- //#region src/runtime/supervise/stop-rules.ts
2510
- /**
2511
- *
2512
- * PROGRESS-BASED STOP RULES — end a long-horizon run for the right reason.
2513
- *
2514
- * Every existing bound is a CEILING: iterations, tokens, dollars, an absolute deadline, a turn
2515
- * cap. A ceiling answers "may this run continue?" and never "is this run still getting anywhere?".
2516
- * So a supervision tree that stopped learning at settle 4 keeps buying workers until it hits a
2517
- * wall — the run ends on exhaustion, and the operator cannot tell a run that finished from a run
2518
- * that ran out.
2519
- *
2520
- * A stop rule reads the run's own PROGRESS and decides. Three signals feed it:
2521
- * - the objective curve over settled work (best-so-far, from `anytime.ts` — see below),
2522
- * - the LIVE worker feed (`WorkerProgress`: `idleMs`, `stalled`, `turns`, `tokens`),
2523
- * - tree-level shape (how many are in flight, how many are waiting, when the last settle landed).
2524
- *
2525
- * ── Two boundaries this module holds deliberately ───────────────────────────────────────────────
2526
- *
2527
- * ENFORCEMENT lives here; THRESHOLDS do not. "Stop after 5 settles with no improvement" is a
2528
- * judgment about a domain — how noisy its scores are, how expensive a worker is, how much a late
2529
- * breakthrough is worth. That belongs to the caller (a loop, a bench, a product). Every rule below
2530
- * takes its thresholds as required options with no hidden defaults for the numbers that decide;
2531
- * the module ships the MECHANISM and refuses to ship the judgment.
2532
- *
2533
- * A stop rule can only ADD a stop, never remove one. The driver evaluates the hard ceilings
2534
- * (`poolStarved`, `deadlinePassed`, abort, the driver's own stop) FIRST and independently; the
2535
- * rule is consulted only when they all say "continue". So no rule can talk a run past its budget.
2536
- *
2537
- * ── What is reused, not re-derived ──────────────────────────────────────────────────────────────
2538
- *
2539
- * `anytime.ts` already computed best-so-far curves, their area, and plateau detection — but only
2540
- * after a run was over, from waterfall spans. Rather than write a second copy for the live path,
2541
- * `bestSoFar` / `areaUnderCurve` / `plateauLength` were extracted there and are imported here. A
2542
- * stop rule that calls a run plateaued therefore agrees, number for number, with the report that
2543
- * later judges whether stopping was right.
2544
- *
2545
- * @experimental
2546
- */
2547
- const CONTINUE = { stop: false };
2548
- /** Build the settled-work ledger a `StopRule` decides from: record each settlement (idempotent by
2549
- * id) and materialize a `ProgressView` combining the best-so-far curve with the live worker feed. */
2550
- function createProgressTracker(opts = {}) {
2551
- const now = opts.now ?? Date.now;
2552
- const requireDelivered = opts.requireDelivered ?? true;
2553
- const minImprovement = opts.minImprovement ?? 0;
2554
- const seen = /* @__PURE__ */ new Set();
2555
- const recorded = [];
2556
- return {
2557
- record(sample) {
2558
- if (seen.has(sample.id)) return false;
2559
- seen.add(sample.id);
2560
- recorded.push(sample);
2561
- return true;
2562
- },
2563
- samples: () => [...recorded],
2564
- view(scope, viewOpts) {
2565
- const curve = bestSoFar(recorded.map((s) => requireDelivered && !s.delivered ? void 0 : s.objective));
2566
- const best = curve.length > 0 ? curve[curve.length - 1] : 0;
2567
- let lastImprovementAt = 0;
2568
- let lastImprovementIdx = -1;
2569
- let prev = 0;
2570
- for (const [i, value] of curve.entries()) {
2571
- if (value - prev > minImprovement) {
2572
- lastImprovementAt = recorded[i].at;
2573
- lastImprovementIdx = i;
2574
- }
2575
- prev = value;
2576
- }
2577
- const treeView = scope?.view;
2578
- const workers = [];
2579
- if (scope && treeView) for (const node of treeView.nodes) {
2580
- if (isTerminalNodeStatus(node.status)) continue;
2581
- if (node.status === "waiting") continue;
2582
- const p = scope.progress(node.id, viewOpts?.stallAfterMs !== void 0 ? {
2583
- now: now(),
2584
- stallAfterMs: viewOpts.stallAfterMs
2585
- } : { now: now() });
2586
- if (p) workers.push(p);
2587
- }
2588
- return {
2589
- now: now(),
2590
- settles: recorded.length,
2591
- delivered: recorded.filter((s) => s.delivered).length,
2592
- curve,
2593
- best,
2594
- auc: areaUnderCurve(curve),
2595
- lastSettleAt: recorded.length > 0 ? recorded[recorded.length - 1].at : 0,
2596
- lastImprovementAt,
2597
- settlesSinceImprovement: lastImprovementIdx < 0 ? recorded.length : recorded.length - 1 - lastImprovementIdx,
2598
- workers,
2599
- inFlight: treeView?.inFlight ?? 0,
2600
- waiting: treeView?.waiting ?? 0
2601
- };
2602
- },
2603
- evaluate(rule, scope, viewOpts) {
2604
- return rule(this.view(scope, viewOpts));
2605
- }
2606
- };
2607
- }
2608
- /** Build a `ProgressSample` from a scope settlement. The objective is the verdict score and
2609
- * `delivered` is the verdict's `valid` — the SAME single delivery signal `finalizeBestDelivered`
2610
- * and `defaultSelectWinner` use, so "progress" and "winner" cannot disagree. */
2611
- function sampleFromSettled(settled, at) {
2612
- if (settled.kind === "down") return {
2613
- id: settled.handle.id,
2614
- at,
2615
- delivered: false
2616
- };
2617
- return {
2618
- id: settled.handle.id,
2619
- at,
2620
- ...settled.verdict?.score !== void 0 ? { objective: settled.verdict.score } : {},
2621
- delivered: settled.verdict?.valid === true
2622
- };
2623
- }
2624
- /**
2625
- * "Nothing new has happened." Fires when the run has produced no new settled work for `ms`, or no
2626
- * IMPROVEMENT over the last `settles` settlements.
2627
- *
2628
- * A tree whose only remaining nodes are armed WAITS is exempt from the time bound: a run waiting
2629
- * on CI is not a run that stopped making progress, and killing it there would defeat mechanic C.
2630
- */
2631
- function noProgressFor(opts) {
2632
- if (opts.ms === void 0 && opts.settles === void 0) throw new ValidationError("noProgressFor: set at least one of { ms, settles }");
2633
- if (opts.ms !== void 0 && opts.ms <= 0) throw new ValidationError("noProgressFor: ms must be > 0");
2634
- if (opts.settles !== void 0 && opts.settles < 1) throw new ValidationError("noProgressFor: settles must be >= 1");
2635
- const minSettles = opts.minSettles ?? 1;
2636
- return (v) => {
2637
- if (v.settles < minSettles) return CONTINUE;
2638
- if (opts.settles !== void 0 && v.settlesSinceImprovement >= opts.settles) return {
2639
- stop: true,
2640
- reason: `no-progress: ${v.settlesSinceImprovement} settles with no improvement (limit ${opts.settles}), best=${v.best}`
2641
- };
2642
- if (opts.ms !== void 0 && v.waiting === 0 && v.lastSettleAt > 0) {
2643
- const idle = v.now - v.lastSettleAt;
2644
- if (idle >= opts.ms) return {
2645
- stop: true,
2646
- reason: `no-progress: ${idle}ms since the last settlement (limit ${opts.ms}ms)`
2647
- };
2648
- }
2649
- return CONTINUE;
2650
- };
2651
- }
2652
- /**
2653
- * "The objective has stopped climbing." Fires when the best-so-far curve has risen by no more than
2654
- * `minDelta` across the last `window` settlements.
2655
- *
2656
- * Built on `anytime.plateauLength` — the same plateau math the post-run anytime report uses, so a
2657
- * rule that stops a run and a report that grades the decision cannot disagree about whether the
2658
- * run was flat.
2659
- */
2660
- function plateau(opts) {
2661
- if (!Number.isInteger(opts.window) || opts.window < 1) throw new ValidationError("plateau: window must be a positive integer");
2662
- if (!Number.isFinite(opts.minDelta) || opts.minDelta < 0) throw new ValidationError("plateau: minDelta must be >= 0");
2663
- const minSettles = opts.minSettles ?? opts.window;
2664
- return (v) => {
2665
- if (v.settles < minSettles) return CONTINUE;
2666
- const flat = plateauLength(v.curve, opts.minDelta);
2667
- if (flat >= opts.window) return {
2668
- stop: true,
2669
- reason: `plateau: best-so-far rose <= ${opts.minDelta} over the last ${flat} settles (window ${opts.window}), best=${v.best}, auc=${v.auc.toFixed(3)}`
2670
- };
2671
- return CONTINUE;
2672
- };
2673
- }
2674
- /**
2675
- * "Everyone is stuck." Fires when every live worker reads `stalled` — no metered activity for
2676
- * longer than the stall threshold — and none of the tree is merely waiting.
2677
- *
2678
- * `stalled` is a derived read at observation time, never a background watchdog; this rule only
2679
- * reads it. A tree with armed waits never fires: waiting is not stalling.
2680
- */
2681
- function allWorkersStalled(opts = {}) {
2682
- const minWorkers = opts.minWorkers ?? 1;
2683
- return (v) => {
2684
- if (v.waiting > 0) return CONTINUE;
2685
- if (v.workers.length < minWorkers) return CONTINUE;
2686
- if (!v.workers.every((w) => w.stalled)) return CONTINUE;
2687
- const worst = Math.max(...v.workers.map((w) => w.idleMs));
2688
- return {
2689
- stop: true,
2690
- reason: `all-stalled: ${v.workers.length} live workers idle, worst ${worst}ms`
2691
- };
2692
- };
2693
- }
2694
- /** Stop when ANY rule stops — the ordinary composition (each rule is a separate reason to end). */
2695
- function anyOf(...rules) {
2696
- return (v) => {
2697
- for (const rule of rules) {
2698
- const d = rule(v);
2699
- if (d.stop) return d;
2700
- }
2701
- return CONTINUE;
2702
- };
2703
- }
2704
- /** Stop only when EVERY rule stops — for a conservative gate that needs corroboration. */
2705
- function allOf(...rules) {
2706
- if (rules.length === 0) throw new ValidationError("allOf: needs at least one rule");
2707
- return (v) => {
2708
- const reasons = [];
2709
- for (const rule of rules) {
2710
- const d = rule(v);
2711
- if (!d.stop) return CONTINUE;
2712
- reasons.push(d.reason);
2713
- }
2714
- return {
2715
- stop: true,
2716
- reason: reasons.join(" AND ")
2717
- };
2718
- };
2719
- }
2720
- /**
2721
- * Evaluate a rule against the run's settled work — the ONE evaluator both supervisor arms call.
2722
- *
2723
- * The router arm calls it before each driver inference turn; the harness arm calls it on each
2724
- * worker settle. Ordering is the contract in both: the hard ceilings (`poolStarved`,
2725
- * `deadlinePassed`, abort, the driver's own stop) are checked first and independently, so a stop
2726
- * rule can only ever ADD a stop — it can never keep a run alive past a budget it has exhausted.
2727
- *
2728
- * Folding the whole roster each call is idempotent by worker id, so it costs O(settled) and never
2729
- * double-counts. `settledAt` carries the instant the ledger recorded a settlement; `now()` is the
2730
- * fallback resolution a per-turn guard has.
2731
- */
2732
- function progressStop(tracker, rule, ledger, scope, now, stallAfterMs) {
2733
- for (const w of ledger.settled()) tracker.record({
2734
- id: w.id,
2735
- at: w.settledAt ?? now(),
2736
- ...w.score !== void 0 ? { objective: w.score } : {},
2737
- delivered: w.status === "done" && w.valid === true
2738
- });
2739
- return tracker.evaluate(rule, scope, stallAfterMs !== void 0 ? { stallAfterMs } : void 0);
2740
- }
2741
- //#endregion
2742
- //#region src/runtime/supervise/coordination-driver.ts
2743
- /**
2744
- *
2745
- * `driverAgent` — the driver's BRAIN.
2746
- *
2747
- * The recursive driver-executor (`driver-executor.ts`) runs a driver `Agent.act` inside a
2748
- * nested `Scope`; this is the intelligent `act`: it mounts the coordination MCP verbs
2749
- * (`createCoordinationTools`) over that scope and runs an LLM tool-loop, so the driver
2750
- * REASONS — spawn / observe / steer / await / stop — about how to drive its children,
2751
- * instead of running a fixed script. Each turn: ask the driver LLM for tool calls, run them
2752
- * against the live scope, fold the results back, repeat until the driver stops (no tool
2753
- * calls) or the turn cap forces a keep-best finalize.
2754
- *
2755
- * Recursion composes through `makeWorkerAgent`: `spawn_agent` resolves a `profile` to a
2756
- * worker LEAF or — when the profile is a driver — a `driverChild` wrapping ANOTHER
2757
- * `driverAgent` over its own nested scope (see `driver-executor.ts`). So an agent
2758
- * drives an agent that drives an agent, each an LLM tool-loop, all on one conserved-budget
2759
- * tree.
2760
- *
2761
- * Two seams are INJECTED so the loop runs offline with no creds and stays decoupled:
2762
- * - `brain` (`ToolLoopChat`) — one driver-LLM turn over the canonical tool-loop seam; a test
2763
- * drives a scripted mock, production passes the router's tool-calling (`routerBrain`), a
2764
- * sandboxed harness drives the verbs as MCP tools. The same seam every tool-loop uses.
2765
- * - `systemPrompt` — the driver's stance (the agent-eval worker-driver prompt / the prompt
2766
- * generator). Injected, never hardcoded — the prompt is a pluggable role.
2767
- *
2768
- * @experimental
2769
- */
2770
- /** The default chapter-close prompt: the brain summarizes its OWN progress for its future self before
2771
- * the detailed history is dropped. Emphasis on PENDING work — the part a too-eager chapter-close
2772
- * loses (the coding-burn counter-finding: closing after one fix leaves integration bugs uncircled). */
2773
- const distillInstruction = "CONTEXT COMPACTION. Your detailed turn-by-turn history is about to be discarded to free your context window. Write a COMPLETE, compact handoff note for your future self so you can keep going without it. Cover: (1) what you have accomplished; (2) every worker you spawned and its current status/result; (3) what subtasks remain unfinished, failing, or unverified — be specific and exhaustive here, this is the part you must not lose; (4) your immediate next action. Do not call any tools; respond with the note only.";
2774
- /** Factual ground truth for the digest — the live worker roster from Scope plus the delivered-result
2775
- * ledger, independent of whatever the brain's prose summary captures. */
2776
- function summarizeRoster(view, settled) {
2777
- if (view.nodes.length === 0) return "Workers in current live scope: none yet.";
2778
- const settledById = new Map(settled.map((w) => [w.id, w]));
2779
- const lines = view.nodes.map((node) => formatRosterNode(node, settledById.get(node.id)));
2780
- return `Workers in current live scope (ground truth from the run, ${view.nodes.length} total, ${view.inFlight} in flight):\n${lines.join("\n")}`;
2781
- }
2782
- function formatRosterNode(node, settled) {
2783
- const result = settled?.status === "done" ? `, delivered=${settled.valid ?? false}${settled.score !== void 0 ? `, score=${settled.score}` : ""}${settled.outRef ? `, outRef=${settled.outRef}` : ""}` : settled?.status === "down" ? `, reason=${settled.reason ?? "unknown"}` : node.outRef ? `, outRef=${node.outRef}` : "";
2784
- return `- ${node.id}: ${node.status}, label=${node.label}, runtime=${node.runtime}${result}`;
2785
- }
2786
- /** Spawn-progress is impossible: the pool can't afford another worker AND nothing is in flight to
2787
- * await. A long-horizon driver bounded by the conserved pool stops here instead of spinning (the
2788
- * in-loop budget guard the turn cap alone never provided). Checks BOTH conserved channels: tokens
2789
- * (can't afford a worker) and usd (a usd-capped pool whose ceiling the driver's own metered
2790
- * inference has drained — `meter` debits usd, so without this a huge-token/small-usd pool would
2791
- * overspend usd up to the turn tripwire). */
2792
- function poolStarved(scope, perWorker) {
2793
- const b = scope.budget;
2794
- if (scope.view.inFlight > 0 || scope.view.waiting > 0) return false;
2795
- const tokenStarved = b.tokensLeft < perWorker.maxTokens;
2796
- const iterationStarved = b.iterationsLeft <= 0;
2797
- const usdStarved = b.usdCapped && (b.usdLeft <= 0 || perWorker.maxUsd !== void 0 && b.usdLeft < perWorker.maxUsd);
2798
- return tokenStarved || iterationStarved || usdStarved;
2799
- }
2800
- /** The absolute wall-clock deadline (when the root set one) has passed. */
2801
- function deadlinePassed(scope, now) {
2802
- const b = scope.budget;
2803
- return b.deadlineMs > 0 && now() >= b.deadlineMs;
2804
- }
2805
- /** The USD-denominated members of {@link PromptCacheUsage} — the schema, not a guess. Every
2806
- * other known member (`readTokens`, `writeTokens`, `missTokens`) is a token COUNT. */
2807
- const PROMPT_CACHE_USD_FIELDS = /* @__PURE__ */ new Set(["readSavingsUsd"]);
2808
- /** Dollar amounts above this are provider nonsense, not evidence. The old all-integer rule
2809
- * rejected them as a side effect; keeping an explicit ceiling preserves that protection
2810
- * without pretending a dollar amount is an integer. */
2811
- const MAX_PROMPT_CACHE_USD = 1e6;
2812
- /**
2813
- * Validate provider-reported prompt-cache evidence.
2814
- *
2815
- * Prompt-cache carries two kinds of number and they obey different rules: token COUNTS are
2816
- * integers, and USD amounts are fractional by nature. Applying the count rule to a dollar
2817
- * field refuses every provider that reports cache savings in dollars — a healthy router
2818
- * response carrying `readSavingsUsd: 0.0034` failed the driver outright before this split.
2819
- *
2820
- * Classification is schema-first: a field named in {@link PromptCacheUsage} is validated by
2821
- * what that member IS. `promptCache` is an open record (the sandbox path forwards provider
2822
- * fields verbatim), so an unknown field falls back to the `usd` name-suffix convention —
2823
- * documented here as the contract a provider must follow to report dollars.
2824
- *
2825
- * Returns the refusal, or `undefined` when the evidence is acceptable.
2826
- */
2827
- function validateDriverPromptCache(promptCache) {
2828
- for (const [field, value] of Object.entries(promptCache ?? {})) {
2829
- if (typeof value !== "number") continue;
2830
- if (PROMPT_CACHE_USD_FIELDS.has(field) || /usd$/i.test(field)) {
2831
- if (!Number.isFinite(value) || value < 0 || value > MAX_PROMPT_CACHE_USD) return new ValidationError(`driverAgent: prompt-cache field ${JSON.stringify(field)} must be a non-negative finite number of dollars`);
2832
- continue;
2833
- }
2834
- if (!Number.isSafeInteger(value) || value < 0) return new ValidationError(`driverAgent: prompt-cache field ${JSON.stringify(field)} must be a non-negative safe integer`);
2835
- }
2836
- }
2837
- /** The journal file `createFileRunContext` writes inside the run directory. The acknowledger
2838
- * reads it as EVIDENCE for the terminated-descendants set — the nested trees of a cancelled
2839
- * lead journal their terminal records there before the lead settles into this scope. */
2840
- const SPAWN_JOURNAL_FILE = "spawn-journal.jsonl";
2841
- /**
2842
- * The worker-cancel ACKNOWLEDGER — the runtime-side half of `run-layout`'s `cancelWorker`
2843
- * contract, run from the coordination driver's turn loop (one cancellation-inbox read per turn,
2844
- * no new process, no poller, no extra lifetime). Every manager with a `controlDir` mounts one;
2845
- * OWNERSHIP keeps them from colliding: a request naming a node id is owned by the manager whose
2846
- * own id is that node's parent, and a label/profile-name reference is owned by the `'run'`-scoped
2847
- * (root) manager only — so exactly one acknowledger can ever apply one operation.
2848
- *
2849
- * Two-phase, honestly reported: `cancel_requested` is written the moment a live worker's abort is
2850
- * issued (through the per-child abort chain the scope already owns, so siblings are untouched);
2851
- * `cancelled` is written only when that worker's settlement is DELIVERED on the settle path with
2852
- * a terminal `down`, and then the record names every subtree node id proven terminated. A worker
2853
- * that already settled — or that settles `done` despite the abort — records `not_live`; a
2854
- * reference matching nothing this manager owns stays pending (`cancelWorker` reports it
2855
- * `unknown`). No path reports success for a missing worker.
2856
- *
2857
- * Expiry is run end, not a clock: `finish()` (after the final post-drain pass) writes `not_live`
2858
- * for every owned request never applied and `unknown` for an issued abort whose settle the run
2859
- * ended too soon to observe. A pending request can therefore never outlive its run and abort a
2860
- * future spawn that happens to reuse a label.
2861
- *
2862
- * Idempotency is a lookup, in-process and across processes: an operation with a durable
2863
- * acknowledgement is returned as-is and never re-applied.
2864
- */
2865
- function createCancelAcknowledger(deps) {
2866
- const tracked = /* @__PURE__ */ new Map();
2867
- let runTracked;
2868
- const abortIssuedAt = /* @__PURE__ */ new Map();
2869
- const iso = () => new Date(deps.now()).toISOString();
2870
- const write = (record) => {
2871
- writeWorkerCancellation(deps.dir, record);
2872
- tracked.set(record.operationId, record);
2873
- };
2874
- /** `ref` is exactly one of THIS manager's direct-child node ids (`${ownerId}:s<seq>`). */
2875
- const directChildId = (ref) => ref.startsWith(`${deps.ownerId}:s`) && /^s\d+$/.test(ref.slice(deps.ownerId.length + 1));
2876
- /** Whether this acknowledger owns `ref`. A node id deeper in this subtree belongs to the nested
2877
- * manager that parents it; anything that is not a node id under this manager is a
2878
- * label/profile-name reference, owned by the `'run'`-scoped manager alone. */
2879
- const owned = (ref) => {
2880
- if (directChildId(ref)) return true;
2881
- if (deps.controlScope !== "run") return false;
2882
- return !ref.startsWith(`${deps.ownerId}:`) && ref !== deps.ownerId;
2883
- };
2884
- const deliveredTerminal = (id) => {
2885
- const row = deps.coord.settled().find((w) => w.id === id);
2886
- if (row !== void 0) return row.status;
2887
- const node = deps.scope.view.nodes.find((n) => n.id === id);
2888
- if (node === void 0) return void 0;
2889
- if (node.status === "done") return "done";
2890
- if (node.status === "failed" || node.status === "cancelled") return "down";
2891
- };
2892
- const apply = (request) => {
2893
- const aborted = deps.coord.abortWorker(request.worker, request.reason ?? "cancel requested");
2894
- const base = {
2895
- operationId: request.operationId,
2896
- worker: request.worker,
2897
- requestedAt: request.at,
2898
- observedAt: iso(),
2899
- ...request.reason === void 0 ? {} : { reason: request.reason }
278
+ function onSettled(event) {
279
+ const p = record(event.payload);
280
+ const childId = str(p.childId);
281
+ if (!childId) return;
282
+ const node = open.get(childId);
283
+ if (!node) return;
284
+ open.delete(childId);
285
+ const status = str(p.status);
286
+ const down = status === "down";
287
+ const attrs = {
288
+ ...node.attrs,
289
+ "tangle.supervise.node.status": status ?? "done"
2900
290
  };
2901
- if (aborted !== void 0) {
2902
- abortIssuedAt.set(request.operationId, base.observedAt);
2903
- write({
2904
- ...base,
2905
- effect: "cancel_requested",
2906
- workerId: aborted.id,
2907
- detail: `abort issued to live worker '${aborted.label}' (${aborted.id}); termination not yet proven`,
2908
- terminated: []
2909
- });
2910
- return;
291
+ if (typeof event.stepIndex === "number") attrs["tangle.supervise.node.seq"] = event.stepIndex;
292
+ if (typeof p.outRef === "string") attrs["tangle.supervise.node.out_ref"] = p.outRef;
293
+ if (typeof p.valid === "boolean") attrs["tangle.supervise.verdict.valid"] = p.valid;
294
+ if (typeof p.score === "number") attrs["tangle.supervise.verdict.score"] = p.score;
295
+ if (down) {
296
+ attrs["error.type"] = p.infra === true ? "infra" : "child-down";
297
+ const reason = str(p.reason);
298
+ if (reason) attrs["error.message"] = truncate(reason);
299
+ if (typeof p.infra === "boolean") attrs["tangle.supervise.node.infra"] = p.infra;
2911
300
  }
2912
- const goneId = deps.scope.view.nodes.find((n) => (n.id === request.worker || n.label === request.worker) && isTerminalNodeStatus(n.status))?.id ?? deps.coord.settled().find((w) => w.id === request.worker)?.id;
2913
- if (goneId !== void 0) write({
301
+ const wokeBy = str(record(p.wait).settled);
302
+ if (wokeBy) attrs["tangle.supervise.wait.settled"] = wokeBy;
303
+ assignSpend(attrs, p.spent);
304
+ emit(span(node.spanId, node.parentSpanId, node.name, node.startMs, event.timestamp, attrs, down ? STATUS_ERROR : STATUS_OK, down ? str(p.reason) ?? "child down" : void 0));
305
+ }
306
+ function onTurn(event) {
307
+ const p = record(event.payload);
308
+ const parentId = event.parentId;
309
+ const parentSpanId = parentId && spanIdOf.get(parentId) || rootSpanId;
310
+ const attrs = {
2914
311
  ...base,
2915
- effect: "not_live",
2916
- workerId: goneId,
2917
- detail: `worker '${goneId}' had already settled before this operation was applied`,
2918
- terminated: []
2919
- });
2920
- };
2921
- /** The proven-terminated set for one record: the worker plus every subtree id with a terminal
2922
- * journal record at/after the abort was issued. Union with what the record already names, so
2923
- * the set only ever grows (a late teardown journal adds; nothing removes). */
2924
- const provenTerminated = (record, workerId) => {
2925
- const since = abortIssuedAt.get(record.operationId) ?? record.observedAt;
2926
- return [.../* @__PURE__ */ new Set([
2927
- ...record.terminated,
2928
- workerId,
2929
- ...terminatedDescendants(deps.dir, workerId, since)
2930
- ])].sort();
2931
- };
2932
- const reconcile = (record) => {
2933
- const workerId = record.workerId;
2934
- if (workerId === void 0) return;
2935
- const terminal = deliveredTerminal(workerId);
2936
- if (terminal === void 0) return;
2937
- if (terminal === "down") {
2938
- write({
2939
- ...record,
2940
- effect: "cancelled",
2941
- observedAt: iso(),
2942
- terminated: provenTerminated(record, workerId),
2943
- detail: `worker '${workerId}' reached a terminal down state on the settle path`
2944
- });
2945
- return;
2946
- }
2947
- write({
2948
- ...record,
2949
- effect: "not_live",
2950
- observedAt: iso(),
2951
- terminated: [],
2952
- detail: `worker '${workerId}' settled done despite the abort request; nothing was terminated`
2953
- });
2954
- };
2955
- /** Re-scan a `cancelled` record while the manager still turns: a descendant whose teardown
2956
- * journals after the lead's settle joins the set on a later pass instead of being lost. Only
2957
- * a grown set is re-written; the window needs the in-process abort instant, so a record a
2958
- * PRIOR process closed stays as that process proved it. */
2959
- const regrow = (record) => {
2960
- const workerId = record.workerId;
2961
- if (workerId === void 0 || !abortIssuedAt.has(record.operationId)) return;
2962
- const terminated = provenTerminated(record, workerId);
2963
- if (terminated.length > record.terminated.length) write({
2964
- ...record,
2965
- observedAt: iso(),
2966
- terminated
2967
- });
2968
- };
2969
- /**
2970
- * The RUN-scoped request: seen once, `cancel_requested` written the moment the run's cascading
2971
- * abort is issued through the one controller the run already has. The `supervise()` settle path
2972
- * records what the run then actually did — this manager cannot observe its own tree's terminal
2973
- * state from inside `act`.
2974
- *
2975
- * Applied only at a TURN boundary, never on the final post-drain pass: by then the driver has
2976
- * finished and drained, so a root abort could only void work that is already delivered. A
2977
- * request that arrives that late expires in `finish()` instead — it terminated nothing.
2978
- */
2979
- const passRun = () => {
2980
- if (deps.controlScope !== "run" || deps.abortRun === void 0) return;
2981
- const request = readRunCancelRequest(deps.dir);
2982
- if (request === void 0) return;
2983
- if (runTracked !== void 0) return;
2984
- const prior = readRunCancellation(deps.dir, request.operationId);
2985
- if (prior !== void 0) {
2986
- runTracked = prior;
2987
- return;
2988
- }
2989
- const record = {
2990
- operationId: request.operationId,
2991
- effect: "cancel_requested",
2992
- requestedAt: request.at,
2993
- observedAt: iso(),
2994
- ...request.reason === void 0 ? {} : { reason: request.reason },
2995
- detail: "root abort issued to the whole run; termination not yet proven"
312
+ "openinference.span.kind": "LLM",
313
+ "inference.observation_kind": "LLM",
314
+ "tangle.supervise.node.kind": "inference"
2996
315
  };
2997
- writeRunCancellation(deps.dir, record);
2998
- runTracked = record;
2999
- deps.abortRun(request.reason ?? "run cancel requested");
3000
- };
3001
- const pass = (phase) => {
3002
- if (phase === "turn") passRun();
3003
- for (const request of readWorkerCancelRequests(deps.dir)) {
3004
- if (!owned(request.worker)) continue;
3005
- let record = tracked.get(request.operationId);
3006
- if (record === void 0) {
3007
- record = readWorkerCancellation(deps.dir, request.operationId);
3008
- if (record !== void 0) tracked.set(request.operationId, record);
316
+ if (parentId) attrs["tangle.supervise.node.id"] = parentId;
317
+ for (const [key, value] of Object.entries(p)) {
318
+ if (key === "spend") continue;
319
+ if (key === "driver" && typeof value === "string") {
320
+ attrs["agent.name"] = value;
321
+ attrs["inference.agent_name"] = value;
322
+ continue;
323
+ }
324
+ if (key === "model" && typeof value === "string") {
325
+ attrs["llm.model_name"] = value;
326
+ continue;
3009
327
  }
3010
- if (record === void 0) {
3011
- apply(request);
328
+ if (Array.isArray(value)) {
329
+ const scalars = value.filter((v) => typeof v === "string" || typeof v === "number");
330
+ if (scalars.length > 0) attrs[`tangle.supervise.turn.${key}`] = truncate(scalars.join(","));
3012
331
  continue;
3013
332
  }
3014
- if (record.effect === "cancel_requested") reconcile(record);
3015
- else if (record.effect === "cancelled") regrow(record);
333
+ if (typeof value === "string") attrs[`tangle.supervise.turn.${key}`] = truncate(value);
334
+ else if (typeof value === "number" || typeof value === "boolean") attrs[`tangle.supervise.turn.${key}`] = value;
3016
335
  }
3017
- };
336
+ const spend = assignSpend(attrs, p.spend);
337
+ const endMs = event.timestamp + (spend?.ms ?? 0);
338
+ emit(span(generateSpanId(), parentSpanId, "gen_ai.client.inference", event.timestamp, endMs, attrs, STATUS_OK));
339
+ }
3018
340
  return {
3019
- pass,
3020
- finish() {
3021
- const runRequest = deps.controlScope === "run" && deps.abortRun !== void 0 ? readRunCancelRequest(deps.dir) : void 0;
3022
- if (runRequest !== void 0 && readRunCancellation(deps.dir, runRequest.operationId) === void 0) writeRunCancellation(deps.dir, {
3023
- operationId: runRequest.operationId,
3024
- effect: "not_live",
3025
- requestedAt: runRequest.at,
3026
- observedAt: iso(),
3027
- ...runRequest.reason === void 0 ? {} : { reason: runRequest.reason },
3028
- detail: "run ended before the request was applied"
3029
- });
3030
- for (const request of readWorkerCancelRequests(deps.dir)) {
3031
- if (!owned(request.worker)) continue;
3032
- const record = tracked.get(request.operationId) ?? readWorkerCancellation(deps.dir, request.operationId);
3033
- if (record === void 0) {
3034
- write({
3035
- operationId: request.operationId,
3036
- worker: request.worker,
3037
- effect: "not_live",
3038
- requestedAt: request.at,
3039
- observedAt: iso(),
3040
- ...request.reason === void 0 ? {} : { reason: request.reason },
3041
- detail: "run ended before the request was applied",
3042
- terminated: []
3043
- });
3044
- continue;
3045
- }
3046
- if (record.effect === "cancel_requested") write({
3047
- ...record,
3048
- effect: "unknown",
3049
- observedAt: iso(),
3050
- detail: "abort issued; run ended before termination was observed"
3051
- });
3052
- }
341
+ hooks: { onEvent(event) {
342
+ if (finished) return;
343
+ try {
344
+ if (event.phase !== "after") return;
345
+ if (event.target === "agent.spawn") onSpawn(event);
346
+ else if (event.target === "agent.child") onSettled(event);
347
+ else if (event.target === "agent.turn") onTurn(event);
348
+ } catch {}
349
+ } },
350
+ traceId,
351
+ rootSpanId,
352
+ workerTrace(spawningNodeId) {
353
+ return {
354
+ traceId,
355
+ parentSpanId: spanIdOf.get(spawningNodeId) ?? rootSpanId
356
+ };
357
+ },
358
+ async finish(outcome) {
359
+ if (finished) return;
360
+ finished = true;
361
+ const endMs = now();
362
+ try {
363
+ for (const [nodeId, node] of open) emit(span(node.spanId, node.parentSpanId, node.name, node.startMs, endMs, {
364
+ ...node.attrs,
365
+ "tangle.supervise.node.status": "unsettled",
366
+ "tangle.supervise.node.settled": false,
367
+ "tangle.supervise.node.id": nodeId
368
+ }, STATUS_UNSET));
369
+ open.clear();
370
+ emit(span(rootSpanId, opts.parentSpanId, "supervisor.run", rootStartMs, endMs, rootAttrs(base, opts.agentName ?? "supervisor", outcome), rootStatus(outcome), rootMessage(outcome)));
371
+ await exporter.flush();
372
+ if (ownsExporter) await exporter.shutdown();
373
+ } catch {}
3053
374
  }
3054
375
  };
3055
376
  }
3056
- /**
3057
- * Subtree node ids with a terminal `down`/`cancelled` journal record at or after `sinceIso` —
3058
- * the abort-issue instant (the acknowledger's own `observedAt` on the `cancel_requested` record,
3059
- * runtime clock), never the client's `requestedAt` — read from the durable spawn journal beside
3060
- * the run layout. The set is proven at acknowledgement time and is approximate about post-abort
3061
- * causation: a descendant that died of its OWN cause after the abort was issued is
3062
- * indistinguishable from the cascade and may be included; one whose teardown journals late joins
3063
- * on a later acknowledger pass; a teardown journal still absent when the run ends is absent from
3064
- * the set. Ids are hierarchical (`parent:sN`), so `${nodeId}:` prefixes exactly the subtree.
3065
- * Tolerant of a missing or partially-written journal: evidence that cannot be read names fewer
3066
- * nodes, never wrong ones.
3067
- */
3068
- function terminatedDescendants(dir, nodeId, sinceIso) {
3069
- let raw;
3070
- try {
3071
- raw = readFileSync(join(dir, SPAWN_JOURNAL_FILE), "utf8");
3072
- } catch {
3073
- return [];
3074
- }
3075
- const prefix = `${nodeId}:`;
3076
- const ids = /* @__PURE__ */ new Set();
3077
- for (const line of raw.split("\n")) {
3078
- const trimmed = line.trim();
3079
- if (!trimmed) continue;
3080
- let parsed;
3081
- try {
3082
- parsed = JSON.parse(trimmed);
3083
- } catch {
3084
- continue;
377
+ function rootAttrs(base, agentName, outcome) {
378
+ const attrs = {
379
+ ...base,
380
+ "openinference.span.kind": "AGENT",
381
+ "inference.observation_kind": "AGENT",
382
+ "agent.name": agentName,
383
+ "inference.agent_name": agentName,
384
+ "tangle.supervise.node.kind": "root"
385
+ };
386
+ const result = outcome?.result;
387
+ if (result) {
388
+ attrs["tangle.supervise.result"] = result.kind;
389
+ if (result.kind === "no-winner") {
390
+ attrs["tangle.supervise.reason"] = result.reason;
391
+ attrs["tangle.supervise.down_count"] = result.downCount;
392
+ if (result.reason === "driver-failed") {
393
+ attrs["error.type"] = result.error.name;
394
+ attrs["error.message"] = truncate(result.error.message);
395
+ }
3085
396
  }
3086
- if (parsed.kind !== "event" || parsed.event === void 0) continue;
3087
- const event = parsed.event;
3088
- if (!(event.kind === "settled" && event.status === "down" || event.kind === "cancelled")) continue;
3089
- if (typeof event.id !== "string" || !event.id.startsWith(prefix)) continue;
3090
- if (typeof event.at !== "string" || event.at < sinceIso) continue;
3091
- ids.add(event.id);
397
+ assignSpend(attrs, result.spentTotal);
398
+ }
399
+ if (outcome?.error !== void 0) {
400
+ attrs["tangle.supervise.result"] = "error";
401
+ const err = outcome.error;
402
+ attrs["error.type"] = err instanceof Error ? err.name : typeof err;
403
+ attrs["error.message"] = truncate(err instanceof Error ? err.message : String(err));
3092
404
  }
3093
- return [...ids].sort();
405
+ return attrs;
406
+ }
407
+ function rootStatus(outcome) {
408
+ if (outcome?.error !== void 0) return STATUS_ERROR;
409
+ if (!outcome?.result) return STATUS_UNSET;
410
+ return outcome.result.kind === "winner" ? STATUS_OK : STATUS_ERROR;
411
+ }
412
+ function rootMessage(outcome) {
413
+ if (outcome?.error !== void 0) return truncate(outcome.error instanceof Error ? outcome.error.message : String(outcome.error));
414
+ const result = outcome?.result;
415
+ return result && result.kind === "no-winner" ? result.reason : void 0;
3094
416
  }
3095
417
  /**
3096
- * Build the intelligent recursive driver. Its `act` is the LLM tool-loop; spawn it as a
3097
- * `driverChild` (`driver-executor.ts`) to run it inside a nested scope, recursively.
418
+ * Write a `Spend` onto a span, honouring the "a missing measurement is never zero" invariant: a
419
+ * channel marked NOT known contributes no number at all and instead flags itself, so nothing
420
+ * downstream can sum an unmeasured turn as free. Returns the spend it read (for the duration).
3098
421
  */
3099
- function driverAgent(opts) {
3100
- if (typeof opts.brain !== "function") throw new ValidationError("driverAgent: opts.brain must be a function");
3101
- if ((opts.extraTools?.length ?? 0) > 0 && typeof opts.executeExtraTool !== "function") throw new ValidationError("driverAgent: extraTools requires executeExtraTool (how to run a work-tool call)");
3102
- if ((opts.analyzeOnSettle ?? []).map(normalizeAnalyzeOnSettle).some((route) => route.agent === void 0) && !opts.analysts) throw new ValidationError("driverAgent: analyzeOnSettle requires analysts (the lens registry the kinds resolve against)");
3103
- const reserved = new Set(coordinationVerbNames);
3104
- for (const tool of opts.nodeTools ?? []) {
3105
- if (reserved.has(tool.name)) throw new ValidationError(`driverAgent: node tool "${tool.name}" collides with a coordination verb or another node tool`);
3106
- reserved.add(tool.name);
422
+ function assignSpend(attrs, value) {
423
+ if (!isRecord(value)) return void 0;
424
+ const spend = value;
425
+ const tokens = isRecord(spend.tokens) ? spend.tokens : void 0;
426
+ if (spend.tokensKnown === false) attrs["tangle.supervise.tokens_known"] = false;
427
+ else if (tokens) {
428
+ if (typeof tokens.input === "number") attrs["llm.token_count.prompt"] = tokens.input;
429
+ if (typeof tokens.output === "number") attrs["llm.token_count.completion"] = tokens.output;
3107
430
  }
3108
- for (const t of opts.extraTools ?? []) {
3109
- if (reserved.has(t.name)) throw new ValidationError(`driverAgent: extra work tool "${t.name}" collides with a coordination verb or node tool`);
3110
- reserved.add(t.name);
431
+ if (spend.usdKnown === false) attrs["tangle.supervise.cost_known"] = false;
432
+ else if (typeof spend.usd === "number" && Number.isFinite(spend.usd)) {
433
+ attrs["llm.cost_usd"] = spend.usd;
434
+ attrs["tangle.cost.usd"] = spend.usd;
3111
435
  }
3112
- if (opts.maxTurns !== void 0 && opts.maxTurns < 0) throw new ValidationError("driverAgent: maxTurns must be >= 0 (0 lifts the turn cap; bounds become the conserved pool + deadline + abort)");
3113
- const maxTurns = opts.maxTurns ?? 16;
3114
- const now = opts.now ?? Date.now;
3115
- const inbox = opts.inbox ?? createInbox();
3116
- return {
3117
- name: opts.name,
3118
- deliver(message) {
3119
- return inbox.deliver(message);
3120
- },
3121
- async act(task, scope) {
3122
- const coord = createCoordinationTools({
3123
- scope,
3124
- blobs: opts.blobs,
3125
- makeWorkerAgent: opts.makeWorkerAgent,
3126
- ...opts.authorizeDownMessage ? { authorizeDownMessage: opts.authorizeDownMessage } : {},
3127
- perWorker: opts.perWorker,
3128
- ...opts.deliverable ? { deliverable: opts.deliverable } : {},
3129
- ...opts.maxLiveWorkers !== void 0 ? { maxLiveWorkers: opts.maxLiveWorkers } : {},
3130
- ...opts.analysts ? { analysts: opts.analysts } : {},
3131
- ...opts.analyzeOnSettle ? { analyzeOnSettle: opts.analyzeOnSettle } : {},
3132
- ...opts.watchWorkers ? { watchWorkers: opts.watchWorkers } : {},
3133
- ...opts.stallAfterMs !== void 0 ? { stallAfterMs: opts.stallAfterMs } : {},
3134
- ...opts.continuityByProfile ? { continuityByProfile: opts.continuityByProfile } : {},
3135
- ...opts.preflightSpawn ? { preflightSpawn: opts.preflightSpawn } : {},
3136
- ...opts.resolveSpawnProfile ? { resolveSpawnProfile: opts.resolveSpawnProfile } : {},
3137
- ...opts.onEvent ? { onEvent: opts.onEvent } : {},
3138
- ...opts.replaySettlements ? { replaySettlements: true } : {},
3139
- ...opts.priorCoordination?.questions.length ? { priorQuestions: opts.priorCoordination.questions } : {}
3140
- });
3141
- await coord.ready();
3142
- opts.onCoordinationTools?.(coord.tools);
3143
- const acknowledger = opts.controlDir === void 0 ? void 0 : createCancelAcknowledger({
3144
- dir: opts.controlDir,
3145
- coord,
3146
- scope,
3147
- now,
3148
- ownerId: scope.view.root,
3149
- controlScope: opts.controlScope ?? "run",
3150
- ...opts.abortRun ? { abortRun: opts.abortRun } : {}
3151
- });
3152
- for (const w of scope.resume?.waits ?? []) {
3153
- const rearmed = scope.wait(w.spec, { label: w.label });
3154
- if (!rearmed.ok) throw new RuntimeRunStateError(`driverAgent: cannot re-arm resumed wait '${w.label}' (${rearmed.reason})`);
3155
- }
3156
- const byName = new Map([...coord.tools, ...opts.nodeTools ?? []].map((t) => [t.name, t]));
3157
- const toolSpecs = [
3158
- ...coord.tools.map((t) => ({
3159
- type: "function",
3160
- function: {
3161
- name: t.name,
3162
- description: t.description,
3163
- parameters: t.inputSchema
3164
- }
3165
- })),
3166
- ...(opts.nodeTools ?? []).map((t) => ({
3167
- type: "function",
3168
- function: {
3169
- name: t.name,
3170
- description: t.description,
3171
- parameters: t.inputSchema
3172
- }
3173
- })),
3174
- ...(opts.extraTools ?? []).map((t) => ({
3175
- type: "function",
3176
- function: {
3177
- name: t.name,
3178
- description: t.description,
3179
- parameters: t.parameters
3180
- }
3181
- }))
3182
- ];
3183
- const system = typeof opts.systemPrompt === "function" ? opts.systemPrompt(task) : opts.systemPrompt;
3184
- const tracker = opts.stopRule ? createProgressTracker({ now }) : void 0;
3185
- let progressStopReason;
3186
- let driverTurn = 0;
3187
- let driverCall = 0;
3188
- const meteredBrain = async (messages, tools, detail) => {
3189
- let res;
3190
- const call = driverCall;
3191
- driverCall += 1;
3192
- const callContext = Object.freeze({
3193
- signal: scope.signal,
3194
- callId: `${scope.view.root}:brain:${crypto.randomUUID()}`,
3195
- correlationId: scope.view.root
3196
- });
3197
- try {
3198
- res = await opts.brain(messages, tools, callContext);
3199
- } catch (error) {
3200
- opts.onProviderModel?.(void 0);
3201
- await meterRuntimeOwnedProviderAttempt(scope, unmeteredSpend(0), providerAttemptEvidence(void 0), {
3202
- driver: opts.name,
3203
- inferenceFailed: true,
3204
- call,
3205
- callId: callContext.callId,
3206
- correlationId: callContext.correlationId,
3207
- ...detail
3208
- });
3209
- throw error;
3210
- }
3211
- let evidenceError;
3212
- opts.onProviderModel?.(res.model);
3213
- if (opts.expectedModel !== void 0) {
3214
- if (res.model === void 0) evidenceError = new ValidationError(`driverAgent: Router response omitted model identity; expected ${JSON.stringify(opts.expectedModel)}`);
3215
- else if (res.model !== opts.expectedModel) evidenceError = new ValidationError(`driverAgent: Router response reported model ${JSON.stringify(res.model)}; expected ${JSON.stringify(opts.expectedModel)}`);
3216
- }
3217
- if (res.transportAttempts !== void 0 && (!Number.isSafeInteger(res.transportAttempts) || res.transportAttempts < 1)) evidenceError = new ValidationError("driverAgent: transportAttempts must be a positive safe integer when reported");
3218
- evidenceError = validateDriverPromptCache(res.promptCache) ?? evidenceError;
3219
- const trustedCost = res.costProvenance === "provider-receipt" || res.costProvenance === "billing-receipt";
3220
- const cacheUsage = promptCacheTokenClasses(res.usage?.input, res.promptCache);
3221
- await meterRuntimeOwnedProviderAttempt(scope, {
3222
- iterations: 0,
3223
- tokens: {
3224
- input: res.usage?.input ?? 0,
3225
- output: res.usage?.output ?? 0,
3226
- ...cacheUsage
3227
- },
3228
- ...res.usage === void 0 ? { tokensKnown: false } : {},
3229
- usd: trustedCost ? res.costUsd ?? 0 : 0,
3230
- ...trustedCost && res.costUsd !== void 0 ? {} : { usdKnown: false },
3231
- ms: 0
3232
- }, providerAttemptEvidence(res.model), {
3233
- driver: opts.name,
3234
- call,
3235
- callId: callContext.callId,
3236
- correlationId: callContext.correlationId,
3237
- toolCalls: (res.toolCalls ?? []).map((c) => c.name),
3238
- ...res.model !== void 0 ? { model: res.model } : {},
3239
- ...res.transportAttempts !== void 0 ? { transportAttempts: res.transportAttempts } : {},
3240
- ...res.usage?.reasoning !== void 0 ? { reasoningTokens: res.usage.reasoning } : {},
3241
- ...res.promptCache !== void 0 ? { promptCache: res.promptCache } : {},
3242
- ...res.usageUnknown === true ? { streamUsageMissing: true } : {},
3243
- ...res.costProvenance === "catalog-estimate" ? { estimatedCostUsd: res.costUsd } : {},
3244
- ...detail
3245
- });
3246
- if (evidenceError !== void 0) throw evidenceError;
3247
- return res;
3248
- };
3249
- const chat = async (messages, tools) => {
3250
- const res = await meteredBrain(messages, tools, {
3251
- kind: "driver-inference",
3252
- turn: driverTurn
3253
- });
3254
- driverTurn += 1;
3255
- return res;
3256
- };
3257
- const compaction = opts.compaction ? {
3258
- thresholdTokens: opts.compaction.thresholdTokens,
3259
- distill: opts.compaction.distill ?? (async (msgs) => {
3260
- const roster = summarizeRoster(scope.view, coord.settled());
3261
- try {
3262
- const narrative = ((await meteredBrain([...msgs, {
3263
- role: "user",
3264
- content: distillInstruction
3265
- }], [], {
3266
- kind: "driver-compaction",
3267
- compactingTurn: driverTurn
3268
- })).content ?? "").trim();
3269
- return narrative ? `${roster}\n\n## Progress notes\n${narrative}` : roster;
3270
- } catch (e) {
3271
- return `${roster}\n\n## Progress notes\nSummary unavailable: ${errMessage(e)}`;
3272
- }
3273
- }),
3274
- ...opts.compaction.onCompact ? { onCompact: opts.compaction.onCompact } : {},
3275
- ...opts.compaction.preserveHead !== void 0 ? { preserveHead: opts.compaction.preserveHead } : {},
3276
- ...opts.compaction.estimateTokens ? { estimateTokens: opts.compaction.estimateTokens } : {}
3277
- } : void 0;
3278
- await runBrainLoop({
3279
- chat,
3280
- tools: toolSpecs,
3281
- ...compaction ? { compaction } : {},
3282
- execute: async (name, args) => {
3283
- if (opts.executeExtraTool) {
3284
- const worked = await runExtraTool(opts.executeExtraTool, name, args);
3285
- if (worked !== null && worked !== void 0) return worked;
3286
- }
3287
- const tool = byName.get(name);
3288
- return safeJson(tool ? await runTool(tool, args) : { error: `unknown tool: ${name}` });
3289
- },
3290
- initialMessages: [
3291
- {
3292
- role: "system",
3293
- content: system
3294
- },
3295
- {
3296
- role: "user",
3297
- content: stringifyTask(task)
3298
- },
3299
- ...scope.resume ? [{
3300
- role: "user",
3301
- content: resumeBrief(scope.resume, opts.priorCoordination)
3302
- }] : hasPriorCoordination(opts.priorCoordination) ? [{
3303
- role: "user",
3304
- content: priorCoordinationBrief(opts.priorCoordination)
3305
- }] : []
3306
- ],
3307
- maxTurns,
3308
- hooks: {
3309
- beforeTurn: (_turn, messages) => {
3310
- acknowledger?.pass("turn");
3311
- const pending = inbox.drain();
3312
- if (pending.length > 0) messages.push({
3313
- role: "user",
3314
- content: inbox.fold(pending)
3315
- });
3316
- },
3317
- stopBefore: () => {
3318
- if (coord.isStopped() || scope.signal.aborted || poolStarved(scope, opts.perWorker) || deadlinePassed(scope, now)) return true;
3319
- if (!opts.stopRule || !tracker) return false;
3320
- const decision = progressStop(tracker, opts.stopRule, coord, scope, now, opts.stallAfterMs);
3321
- if (!decision.stop) return false;
3322
- if (progressStopReason === void 0) {
3323
- progressStopReason = decision.reason;
3324
- opts.onProgressStop?.(decision.reason);
3325
- }
3326
- return true;
3327
- }
3328
- }
3329
- });
3330
- await coord.drainResolved();
3331
- acknowledger?.pass("final");
3332
- acknowledger?.finish();
3333
- const submitted = coord.submittedResult();
3334
- if (submitted) return submitted.result;
3335
- return runFinalizer(opts.finalizer ?? bestDelivered, {
3336
- settled: coord.settled(),
3337
- blobs: opts.blobs,
3338
- tree: runTree(scope),
3339
- budget: scope.budget
3340
- });
3341
- }
3342
- };
436
+ if (typeof spend.iterations === "number") attrs["tangle.supervise.iterations"] = spend.iterations;
437
+ if (typeof spend.ms === "number" && spend.ms > 0) attrs["tangle.supervise.duration_ms"] = spend.ms;
438
+ return spend;
439
+ }
440
+ function assignBudget(attrs, value) {
441
+ if (!isRecord(value)) return;
442
+ const budget = value;
443
+ if (typeof budget.maxTokens === "number") attrs["tangle.supervise.budget.max_tokens"] = budget.maxTokens;
444
+ if (typeof budget.maxIterations === "number") attrs["tangle.supervise.budget.max_iterations"] = budget.maxIterations;
445
+ if (typeof budget.maxUsd === "number") attrs["tangle.supervise.budget.max_usd"] = budget.maxUsd;
3343
446
  }
3344
447
  /**
3345
- * The factual context a resumed driver starts from everything the durable stores prove about
3346
- * the prior process(es): committed settlements, per-key states (completed / lost / failed),
3347
- * re-armed waits, carried-over questions/findings/continuation receipts, and spend already paid.
3348
- * Injected as the brain's first user-context on a resumed run so it continues from unresolved work;
3349
- * old continuation receipts are evidence and are never auto-delivered.
448
+ * A trace id must be 32 hex characters. A caller-supplied one is used verbatim when it already is;
449
+ * anything else (and the default) is DERIVED from the run id by content address — deterministic, so
450
+ * a resumed run rejoins the trace its first process opened rather than forking a new one.
3350
451
  */
3351
- function resumeBrief(resume, prior) {
3352
- const lines = [
3353
- "RESUME: this run continues a prior coordinator process. Its committed work is restored",
3354
- "below and already counts toward the deliverable — do NOT redo it. Continue from the",
3355
- "unresolved work only.",
3356
- "",
3357
- `Committed workers (${resume.settled.length}):`
3358
- ];
3359
- if (resume.settled.length === 0) lines.push("- none");
3360
- for (const s of resume.settled) lines.push(s.kind === "done" ? `- ${s.handle.id} (${s.handle.label}): done, score=${s.verdict?.score ?? 0}, valid=${s.verdict?.valid ?? false}, outRef=${s.outRef}` : `- ${s.handle.id} (${s.handle.label}): down, reason=${s.reason}`);
3361
- const byState = (state) => [...resume.keys].filter(([, v]) => v.state === state);
3362
- const completed = byState("completed");
3363
- const lost = byState("in-doubt");
3364
- const failed = byState("down");
3365
- if (completed.length > 0) lines.push("", "COMPLETED keys — spawn_agent with the same key returns the finished result, spending nothing:", ...completed.map(([k, v]) => `- ${k} → ${v.id} (${v.label})`));
3366
- if (lost.length > 0) lines.push("", "Keys LOST in flight with the prior process — this is the unresolved work; spawn_agent with the same key starts a fresh attempt:", ...lost.map(([k, v]) => `- ${k} (prior attempt ${v.id}, ${v.label})`));
3367
- if (failed.length > 0) lines.push("", "Keys whose prior attempt FAILED (settled down) — spawn_agent with the same key retries:", ...failed.map(([k, v]) => `- ${k} (prior attempt ${v.id}, ${v.label})`));
3368
- if (resume.waits.length > 0) lines.push("", "Pending waits RE-ARMED on their original deadlines (they settle through await_event):", ...resume.waits.map((w) => `- ${w.label} (${w.spec.kind})`));
3369
- appendPriorCoordination(lines, prior);
3370
- const spent = resume.priorSpend;
3371
- lines.push("", "Budget the run ALREADY spent before this process (it counts toward the run total):", `- child work: tokens charged=${chargedTokens(spent.childWork.tokens)} (in=${spent.childWork.tokens.input} out=${spent.childWork.tokens.output}), usd=${spent.childWork.usd}, iterations=${spent.childWork.iterations}`, `- driver inference: tokens charged=${chargedTokens(spent.driverInference.tokens)} (in=${spent.driverInference.tokens.input} out=${spent.driverInference.tokens.output}), usd=${spent.driverInference.usd}`);
3372
- return lines.join("\n");
3373
- }
3374
- function hasPriorCoordination(prior) {
3375
- return prior !== void 0 && (prior.questions.length > 0 || prior.findings.length > 0 || prior.continuations.length > 0 || prior.deliveryEvidence.length > 0);
3376
- }
3377
- function priorCoordinationBrief(prior) {
3378
- const lines = [
3379
- "PRIOR COORDINATION EVIDENCE: this logical supervisor ran in an earlier process.",
3380
- "Use the evidence below as context. Never auto-deliver an old continuation; issue a new",
3381
- "authorized instruction only when current live state still warrants it."
3382
- ];
3383
- appendPriorCoordination(lines, prior);
3384
- return lines.join("\n");
3385
- }
3386
- function appendPriorCoordination(lines, prior) {
3387
- const openQuestions = (prior?.questions ?? []).filter((q) => q.status === "open" || q.status === "escalated");
3388
- if (openQuestions.length > 0) lines.push("", "Questions carried over, still undecided (answer_question decides them; list_questions shows all):", ...openQuestions.map((q) => `- [${q.id}] from=${q.from}, urgency=${q.urgency}: ${q.question}`));
3389
- if ((prior?.findings.length ?? 0) > 0) lines.push("", "Analyst findings from the prior process:", ...(prior?.findings ?? []).map((f) => `- ${f.analyst} on ${f.fromWorker}: ${safeJson(f.findings)}`));
3390
- if ((prior?.continuations.length ?? 0) > 0) {
3391
- const attempts = new Set((prior?.deliveryEvidence ?? []).filter((event) => event.type === "delivery-attempt").map((event) => event.attempt.receiptId));
3392
- const outcomes = new Map((prior?.deliveryEvidence ?? []).filter((event) => event.type === "steer" || event.type === "answer").map((event) => [event.down.receiptId, event.down.outcome]));
3393
- lines.push("", "Authorized continuations committed by the prior process (evidence only; never replayed automatically):", ...(prior?.continuations ?? []).map((continuation) => {
3394
- const delivery = outcomes.get(continuation.receiptId) ?? (attempts.has(continuation.receiptId) ? "unknown-after-crash" : "not-attempted-before-crash");
3395
- return `- receipt=${continuation.receiptId}, ${continuation.kind} → ${continuation.toWorker}, instruction=${continuation.instructionDigest}, delivery=${delivery}`;
3396
- }));
3397
- }
3398
- }
3399
- /** Run a work tool. A throw is data to the driver (it can recover next turn), not a crash — fold
3400
- * the error back as a string result. null/undefined passes through (the caller treats it as "not
3401
- * handled" and falls to the coordination dispatch). */
3402
- async function runExtraTool(execute, name, args) {
3403
- try {
3404
- return await execute(name, args);
3405
- } catch (e) {
3406
- return `error: ${e instanceof Error ? e.message : String(e)}`;
3407
- }
452
+ function normalizeTraceId(traceId, runId) {
453
+ if (traceId && /^[0-9a-f]{32}$/i.test(traceId)) return traceId.toLowerCase();
454
+ return contentAddress(traceId ?? runId).slice(7, 39);
3408
455
  }
3409
- async function runTool(tool, args) {
3410
- try {
3411
- return await tool.handler(args);
3412
- } catch (e) {
3413
- return { error: e instanceof Error ? e.message : String(e) };
3414
- }
456
+ function msToNano(ms) {
457
+ return (BigInt(Number.isFinite(ms) ? Math.floor(ms) : 0) * 1000000n).toString();
3415
458
  }
3416
- /** Keep-best finalize under the completion-oracle: return the highest-scoring DELIVERED child's
3417
- * output (settled `done` AND `valid` its deliverable check passed). Returns undefined when no
3418
- * child delivered — an honest "the driver produced nothing", never a high-scoring result that
3419
- * ran without passing its check (Foreman's 0/18 lesson). `valid` is the single delivery signal,
3420
- * matching `defaultSelectWinner`'s valid-first rule; the oracle just doesn't fall back to an
3421
- * unchecked best-effort. The same argmax as the `bestDelivered` finalizer (`pickBestDelivered`);
3422
- * this direct form serves callers that hold a bare ledger + blob store. */
3423
- async function finalizeBestDelivered(settled, blobs) {
3424
- const best = pickBestDelivered(settled.filter((w) => w.status === "done" && w.valid === true));
3425
- if (best === void 0) return void 0;
3426
- return best.outRef ? await blobs.get(best.outRef) : void 0;
459
+ function isRecord(value) {
460
+ return typeof value === "object" && value !== null && !Array.isArray(value);
3427
461
  }
3428
- function stringifyTask(task) {
3429
- return typeof task === "string" ? task : safeJson(task);
462
+ function record(value) {
463
+ return isRecord(value) ? value : {};
3430
464
  }
3431
- function safeJson(v) {
3432
- try {
3433
- return JSON.stringify(v) ?? String(v);
3434
- } catch {
3435
- return String(v);
3436
- }
465
+ function str(value) {
466
+ return typeof value === "string" && value.length > 0 ? value : void 0;
3437
467
  }
3438
- function errMessage(e) {
3439
- return e instanceof Error ? e.message : String(e);
468
+ function truncate(value) {
469
+ return value.length > MAX_DETAIL_CHARS ? `${value.slice(0, MAX_DETAIL_CHARS)}…` : value;
3440
470
  }
3441
471
  //#endregion
3442
472
  //#region src/mcp/feedback-store.ts
@@ -3498,7 +528,7 @@ function eventToSnapshot(event) {
3498
528
  *
3499
529
  * @stable
3500
530
  */
3501
- var DelegationStateCorruptError = class extends AgentEvalError$1 {
531
+ var DelegationStateCorruptError = class extends AgentEvalError {
3502
532
  constructor(message, options) {
3503
533
  super("validation", message, options);
3504
534
  }
@@ -3511,7 +541,7 @@ var DelegationStateCorruptError = class extends AgentEvalError$1 {
3511
541
  *
3512
542
  * @stable
3513
543
  */
3514
- var DelegationPersistenceError = class extends AgentEvalError$1 {
544
+ var DelegationPersistenceError = class extends AgentEvalError {
3515
545
  constructor(message, options) {
3516
546
  super("config", message, options);
3517
547
  }
@@ -5388,7 +2418,7 @@ function classifyDriverFailure(error, signal) {
5388
2418
  return "terminal";
5389
2419
  }
5390
2420
  if (error instanceof ValidationError || error instanceof ConfigError || error instanceof RuntimeRunStateError) return "terminal";
5391
- if (error instanceof AgentEvalError$1) return "terminal";
2421
+ if (error instanceof AgentEvalError) return "terminal";
5392
2422
  return "transient";
5393
2423
  }
5394
2424
  /** The budget's own verdict on whether another attempt may run at all. */
@@ -7170,7 +4200,8 @@ function superviseInternal(profile, task, opts, testBrain) {
7170
4200
  ...options.signal ? { signal: options.signal } : {},
7171
4201
  ...hooks ? { hooks } : {},
7172
4202
  ...recorder ? { workerTrace: recorder.workerTrace } : {},
7173
- ...recorder && traceUnpropagated ? { workerTraceUnpropagated: traceUnpropagated } : {}
4203
+ ...recorder && traceUnpropagated ? { workerTraceUnpropagated: traceUnpropagated } : {},
4204
+ ...options.runDir === void 0 ? {} : { interactiveBindingDir: resolve(options.runDir) }
7174
4205
  });
7175
4206
  const settle = async () => {
7176
4207
  const result = await run;
@@ -7212,6 +4243,6 @@ function rootProviderModelEvidenceFromExecution(evidence) {
7212
4243
  return evidence ?? rootProviderModelEvidence([]);
7213
4244
  }
7214
4245
  //#endregion
7215
- export { hashIdempotencyInput as $, DELEGATION_HISTORY_INPUT_SCHEMA as A, gateOnDeliverable as At, DELEGATE_FEEDBACK_DESCRIPTION as B, createMcpServer as C, bestSoFar as Ct, createDelegationStatusHandler as D, createInMemoryRunContext as Dt, DELEGATION_STATUS_TOOL_NAME as E, createFileRunContext as Et, DELEGATE_UI_AUDIT_DESCRIPTION as F, normalizeAnalyzeOnSettle as Ft, DELEGATE_DESCRIPTION as G, DELEGATE_FEEDBACK_TOOL_NAME as H, DELEGATE_UI_AUDIT_INPUT_SCHEMA as I, questionEscalationTargets as It, createDelegateHandler as J, DELEGATE_INPUT_SCHEMA as K, DELEGATE_UI_AUDIT_TOOL_NAME as L, createEventBus as Lt, createDelegationHistoryHandler as M, DEFAULT_AWAIT_EVENT_TIMEOUT_MS as Mt, validateDelegationHistoryArgs as N, canonicalFindingEvent as Nt, validateDelegationStatusArgs as O, FileCoordinationLog as Ot, delegationProfiles as P, createCoordinationTools as Pt, DelegationTaskQueue as Q, createDelegateUiAuditHandler as R, defaultToolDetectors as Rt, createInProcessTransport as S, areaUnderCurve as St, DELEGATION_STATUS_INPUT_SCHEMA as T, renderAnytimeTable as Tt, createDelegateFeedbackHandler as U, DELEGATE_FEEDBACK_INPUT_SCHEMA as V, validateDelegateFeedbackArgs as W, defaultDelegateBudget as X, validateDelegateArgs as Y, delegate as Z, promptHandle as _, createProgressTracker as _t, assertCoordinationBinding as a, createDelegationTraceCollector as at, classifyDriverFailure as b, sampleFromSettled as bt, supervisorAgentWithTestBrain as c, FileDelegationStore as ct, delegatesWorkerBriefPrompt as d, eventToSnapshot as dt, DELEGATION_TRACE_MAX_BYTES as et, dumbContinuationFailPrompt as f, driverAgent as ft, naiveContinuationPrompt as g, anyOf as gt, kernelPromptRegistry as h, allWorkersStalled as ht, workerFromBackend as i, composeLoopTraceEmitters as it, DELEGATION_HISTORY_TOOL_NAME as j, mapExecutorResult as jt, DELEGATION_HISTORY_DESCRIPTION as k, createSupervisorSpanRecorder as kt, analyzesFindingsReportPrompt as l, InMemoryDelegationStore as lt, formatPromptHandle as m, allOf as mt, supervise as n, buildDelegationTraceSpans as nt, resolveSupervisorProfile as o, DelegationPersistenceError as ot, dumbContinuationPassPrompt as p, finalizeBestDelivered as pt, DELEGATE_TOOL_NAME as q, superviseWithTestBrain as r, capDelegationTrace as rt, supervisorAgent as s, DelegationStateCorruptError as st, DEFAULT_AUTHORED_PROFILE_SECURITY_POLICY as t, DELEGATION_TRACE_MAX_SPANS as tt, createPromptRegistry as u, InMemoryFeedbackStore as ut, supervisorPolicyPrompt as v, noProgressFor as vt, DELEGATION_STATUS_DESCRIPTION as w, plateauLength as wt, serveCoordinationMcp as x, anytimeReport as xt, DriverAttemptsExhaustedError as y, plateau as yt, validateDelegateUiAuditArgs as z, watchTrace as zt };
4246
+ export { hashIdempotencyInput as $, DELEGATION_HISTORY_INPUT_SCHEMA as A, DELEGATE_FEEDBACK_DESCRIPTION as B, createMcpServer as C, createDelegationStatusHandler as D, DELEGATION_STATUS_TOOL_NAME as E, DELEGATE_UI_AUDIT_DESCRIPTION as F, DELEGATE_DESCRIPTION as G, DELEGATE_FEEDBACK_TOOL_NAME as H, DELEGATE_UI_AUDIT_INPUT_SCHEMA as I, createDelegateHandler as J, DELEGATE_INPUT_SCHEMA as K, DELEGATE_UI_AUDIT_TOOL_NAME as L, createDelegationHistoryHandler as M, validateDelegationHistoryArgs as N, validateDelegationStatusArgs as O, delegationProfiles as P, DelegationTaskQueue as Q, createDelegateUiAuditHandler as R, createInProcessTransport as S, DELEGATION_STATUS_INPUT_SCHEMA as T, createDelegateFeedbackHandler as U, DELEGATE_FEEDBACK_INPUT_SCHEMA as V, validateDelegateFeedbackArgs as W, defaultDelegateBudget as X, validateDelegateArgs as Y, delegate as Z, promptHandle as _, assertCoordinationBinding as a, createDelegationTraceCollector as at, classifyDriverFailure as b, supervisorAgentWithTestBrain as c, FileDelegationStore as ct, delegatesWorkerBriefPrompt as d, eventToSnapshot as dt, DELEGATION_TRACE_MAX_BYTES as et, dumbContinuationFailPrompt as f, createSupervisorSpanRecorder as ft, naiveContinuationPrompt as g, kernelPromptRegistry as h, workerFromBackend as i, composeLoopTraceEmitters as it, DELEGATION_HISTORY_TOOL_NAME as j, DELEGATION_HISTORY_DESCRIPTION as k, analyzesFindingsReportPrompt as l, InMemoryDelegationStore as lt, formatPromptHandle as m, mapExecutorResult as mt, supervise as n, buildDelegationTraceSpans as nt, resolveSupervisorProfile as o, DelegationPersistenceError as ot, dumbContinuationPassPrompt as p, gateOnDeliverable as pt, DELEGATE_TOOL_NAME as q, superviseWithTestBrain as r, capDelegationTrace as rt, supervisorAgent as s, DelegationStateCorruptError as st, DEFAULT_AUTHORED_PROFILE_SECURITY_POLICY as t, DELEGATION_TRACE_MAX_SPANS as tt, createPromptRegistry as u, InMemoryFeedbackStore as ut, supervisorPolicyPrompt as v, DELEGATION_STATUS_DESCRIPTION as w, serveCoordinationMcp as x, DriverAttemptsExhaustedError as y, validateDelegateUiAuditArgs as z };
7216
4247
 
7217
- //# sourceMappingURL=supervise-Ci0RfWQF.js.map
4248
+ //# sourceMappingURL=supervise-HsRKOlKY.js.map