@tangle-network/agent-runtime 0.196.0 → 0.198.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/README.md +1 -2
  2. package/dist/{activation-BxMZybuo.js → activation-CwdO0ml2.js} +2 -2
  3. package/dist/{activation-BxMZybuo.js.map → activation-CwdO0ml2.js.map} +1 -1
  4. package/dist/{improve-DZs0KXhK.d.ts → activation-DFRTurvU.d.ts} +99 -5
  5. package/dist/agent.d.ts +1 -1
  6. package/dist/agent.js +2 -2
  7. package/dist/candidate-execution/index.d.ts +370 -2
  8. package/dist/candidate-execution/index.js +828 -4
  9. package/dist/candidate-execution/index.js.map +1 -0
  10. package/dist/{coordination-driver-zkCrfEbS.js → coordination-driver-Cf638-6q.js} +2 -5
  11. package/dist/{coordination-driver-zkCrfEbS.js.map → coordination-driver-Cf638-6q.js.map} +1 -1
  12. package/dist/{delegate-D7sonS78.js → delegate-Bez07Dlh.js} +2 -2
  13. package/dist/{delegate-D7sonS78.js.map → delegate-Bez07Dlh.js.map} +1 -1
  14. package/dist/durable.d.ts +110 -6
  15. package/dist/durable.js +404 -47
  16. package/dist/durable.js.map +1 -1
  17. package/dist/{graph-BuS-NxXD.js → graph-DBk3Unto.js} +3 -3
  18. package/dist/{graph-BuS-NxXD.js.map → graph-DBk3Unto.js.map} +1 -1
  19. package/dist/{improvement-cycle-Tw5nYT5R.js → improvement-cycle-Bma8TJs-.js} +8 -10
  20. package/dist/{improvement-cycle-Tw5nYT5R.js.map → improvement-cycle-Bma8TJs-.js.map} +1 -1
  21. package/dist/{index-CYOJsxSg.d.ts → index-D8LWK3Ha.d.ts} +6 -7
  22. package/dist/index.d.ts +694 -16
  23. package/dist/index.js +1780 -17
  24. package/dist/index.js.map +1 -1
  25. package/dist/intelligence.d.ts +3 -7
  26. package/dist/intelligence.js +5 -6
  27. package/dist/intelligence.js.map +1 -1
  28. package/dist/kernel.d.ts +2 -5
  29. package/dist/kernel.js +8 -10
  30. package/dist/{loop-runner-bin-XKMc3Ld-.js → loop-runner-bin-Bd3dHQuw.js} +3 -3
  31. package/dist/{loop-runner-bin-XKMc3Ld-.js.map → loop-runner-bin-Bd3dHQuw.js.map} +1 -1
  32. package/dist/{loop-runner-bin-DwrSB04m.d.ts → loop-runner-bin-Bn-0o_4k.d.ts} +3 -3
  33. package/dist/loop-runner-bin.d.ts +1 -1
  34. package/dist/loop-runner-bin.js +1 -1
  35. package/dist/mcp/bin.js +3 -3
  36. package/dist/mcp/index.d.ts +2 -3
  37. package/dist/mcp/index.js +4 -5
  38. package/dist/mcp/index.js.map +1 -1
  39. package/dist/{prepare-DDGp0-rW.js → prepare-CAO1yXov.js} +2407 -2407
  40. package/dist/prepare-CAO1yXov.js.map +1 -0
  41. package/dist/{protected-model-port-DxFN8DLS.js → protected-model-port-B5avcRiQ.js} +2 -2
  42. package/dist/{protected-model-port-DxFN8DLS.js.map → protected-model-port-B5avcRiQ.js.map} +1 -1
  43. package/dist/{provision-supervisor-CpMMShE_.js → provision-supervisor-C97q5NN_.js} +3 -6
  44. package/dist/{provision-supervisor-CpMMShE_.js.map → provision-supervisor-C97q5NN_.js.map} +1 -1
  45. package/dist/{redact-Cbl2O-4N.js → redact-CVyxp2AU.js} +5232 -2525
  46. package/dist/redact-CVyxp2AU.js.map +1 -0
  47. package/dist/{runtime-DqRieJ6e.js → runtime-BrTBBSfC.js} +427 -23
  48. package/dist/runtime-BrTBBSfC.js.map +1 -0
  49. package/dist/{server-BjB6ay4N.js → server-CO4bvAwe.js} +4 -4
  50. package/dist/{server-BjB6ay4N.js.map → server-CO4bvAwe.js.map} +1 -1
  51. package/dist/{types-DFLZMaeh.d.ts → stream-agent-turn-CLOQr497.d.ts} +1986 -6
  52. package/dist/{structural-rollout-DPbZWgEm.js → structural-rollout-D__lmWXe.js} +1101 -8
  53. package/dist/structural-rollout-D__lmWXe.js.map +1 -0
  54. package/dist/{supervise-D9aNi8_f.js → supervise-KSe4JEcg.js} +30 -7
  55. package/dist/supervise-KSe4JEcg.js.map +1 -0
  56. package/dist/testing.d.ts +2 -2
  57. package/dist/testing.js +13 -13
  58. package/dist/tui/index.d.ts +1 -1
  59. package/dist/tui/index.js +1 -1
  60. package/dist/{workspace-archive-Ybomp7AN.js → workspace-archive-BMOnloFf.js} +3 -3
  61. package/dist/{workspace-archive-Ybomp7AN.js.map → workspace-archive-BMOnloFf.js.map} +1 -1
  62. package/package.json +4 -31
  63. package/dist/activation-DyWB0K6E.d.ts +0 -98
  64. package/dist/authored-code-URmkdgjv.js +0 -37
  65. package/dist/authored-code-URmkdgjv.js.map +0 -1
  66. package/dist/candidate-execution-nvqVIMyS.js +0 -829
  67. package/dist/candidate-execution-nvqVIMyS.js.map +0 -1
  68. package/dist/conversation-BxJ0SIBM.js +0 -1363
  69. package/dist/conversation-BxJ0SIBM.js.map +0 -1
  70. package/dist/conversation.d.ts +0 -2
  71. package/dist/conversation.js +0 -2
  72. package/dist/environment-provider-1fKZh2zl.js +0 -2281
  73. package/dist/environment-provider-1fKZh2zl.js.map +0 -1
  74. package/dist/environment-provider-B-I2jlQy.d.ts +0 -143
  75. package/dist/environment-provider.d.ts +0 -2
  76. package/dist/environment-provider.js +0 -2
  77. package/dist/graph.d.ts +0 -753
  78. package/dist/graph.js +0 -2111
  79. package/dist/graph.js.map +0 -1
  80. package/dist/index-CUosKU4N.d.ts +0 -372
  81. package/dist/index-D9mb6fn2.d.ts +0 -691
  82. package/dist/index-ZnxSe6iK.d.ts +0 -138
  83. package/dist/jsonl-file-BEpaEYjT.js +0 -141
  84. package/dist/jsonl-file-BEpaEYjT.js.map +0 -1
  85. package/dist/knowledge-CPzT_Ve7.js +0 -428
  86. package/dist/knowledge-CPzT_Ve7.js.map +0 -1
  87. package/dist/knowledge.d.ts +0 -2
  88. package/dist/knowledge.js +0 -2
  89. package/dist/materialization-Cy0oM8tb.js +0 -672
  90. package/dist/materialization-Cy0oM8tb.js.map +0 -1
  91. package/dist/prepare-DDGp0-rW.js.map +0 -1
  92. package/dist/primeintellect/index.d.ts +0 -218
  93. package/dist/primeintellect/index.js +0 -739
  94. package/dist/primeintellect/index.js.map +0 -1
  95. package/dist/redact-Cbl2O-4N.js.map +0 -1
  96. package/dist/runtime-0xNaV6TJ.d.ts +0 -1699
  97. package/dist/runtime-DqRieJ6e.js.map +0 -1
  98. package/dist/stream-agent-turn-Dt5mZpc3.js +0 -1103
  99. package/dist/stream-agent-turn-Dt5mZpc3.js.map +0 -1
  100. package/dist/stream-agent-turn-urHpmO_Z.d.ts +0 -160
  101. package/dist/structural-rollout-DPbZWgEm.js.map +0 -1
  102. package/dist/supervise-D9aNi8_f.js.map +0 -1
@@ -1,9 +1,12 @@
1
- import { J as AgentTaskStatus, U as AgentRuntimeEvent, at as RuntimeStreamEvent, v as LoopTokenUsage, y as LoopTraceEmitter } from "./types-Q0PMagdm.js";
1
+ import { E as SandboxClient, J as AgentTaskStatus, U as AgentRuntimeEvent, Y as BackendErrorDetail, at as RuntimeStreamEvent, i as ExecCtx, k as Validator, v as LoopTokenUsage, y as LoopTraceEmitter } from "./types-Q0PMagdm.js";
2
2
  import { l as RuntimeHooks } from "./runtime-hooks-Bj6wJHlH.js";
3
- import { AgentExactRunControlRef, AgentInteractiveSession, AgentInteractiveSessionPromptAcknowledgement, AgentInteractiveSessionPromptCommand, AgentInteractiveSessionRef, AgentInteractiveSessionStart, AgentNativeContextContinuationOptions, AgentNativeContextContinuationResult, AgentProfile, AgentSessionStatus, ChildTaskEvent, ContextTransferRequest, HarnessType, InteractionAcknowledgement, InteractionRequest, InteractionResponseCommand, NativeContextBoundaryProof, NativeContextContinuationRequest, NativeContextContinuationTurn, RuntimeEventEnvelope, Sha256Digest } from "@tangle-network/agent-interface";
3
+ import { AgentExactRunControlRef, AgentInteractiveSession, AgentInteractiveSessionPromptAcknowledgement, AgentInteractiveSessionPromptCommand, AgentInteractiveSessionRef, AgentInteractiveSessionStart, AgentNativeContextContinuationOptions, AgentNativeContextContinuationResult, AgentProfile, AgentProfileMcpServer, AgentProfileValidationResult, AgentSessionStatus, ChildTaskEvent, ContextTransferRequest, HarnessType, InteractionAcknowledgement, InteractionRequest, InteractionResponseCommand, NativeContextBoundaryProof, NativeContextContinuationRequest, NativeContextContinuationTurn, ReasoningEffort, RuntimeEventEnvelope, Sha256Digest, StreamEvent } from "@tangle-network/agent-interface";
4
4
  import { ControlEvalResult, DefaultVerdict, KnowledgeReadinessReport, KnowledgeRequirement, ToolSpan } from "@tangle-network/agent-eval";
5
- import { BackendType } from "@tangle-network/sandbox";
6
- import { AgentEnvironment, AgentEnvironmentCapabilities, AgentEnvironmentProvider, AgentTurnInput, AgentTurnResult, CreateAgentEnvironmentInput } from "@tangle-network/agent-interface/environment-provider";
5
+ import { AgentRunOutcome } from "@tangle-network/sandbox/runtime";
6
+ import { ChildProcess } from "node:child_process";
7
+ import { WorkspacePlanReceipt } from "@tangle-network/agent-profile-materialize";
8
+ import { BackendType, CreateSandboxOptions, PromptOptions, Sandbox, SandboxEvent, SandboxInstance } from "@tangle-network/sandbox";
9
+ import { AgentEnvironment, AgentEnvironment as AgentEnvironment$1, AgentEnvironmentCapabilities, AgentEnvironmentCapabilities as AgentEnvironmentCapabilities$1, AgentEnvironmentEvent, AgentEnvironmentEvent as AgentEnvironmentEvent$1, AgentEnvironmentProvider, AgentEnvironmentProvider as AgentEnvironmentProvider$1, AgentEnvironmentQuery, AgentEnvironmentStatus, AgentEnvironmentSummary, AgentProfileRef, AgentProfileRef as AgentProfileRef$1, AgentSession, AgentSessionRef, AgentSessionStatus as AgentSessionStatus$1, AgentTurnInput, AgentTurnInput as AgentTurnInput$1, AgentTurnResult, AgentTurnResult as AgentTurnResult$1, CheckpointRef, CheckpointRequest, CreateAgentEnvironmentInput, CreateAgentEnvironmentInput as CreateAgentEnvironmentInput$1, ExecRequest, ExecResult, ForkRequest, PlacementInfo, ResourceRequest, WorkspaceRequest } from "@tangle-network/agent-interface/environment-provider";
7
10
  //#region src/runtime/retained-run-types.d.ts
8
11
  /** Cursor plus runtime sequence needed to continue one ordered replay. @stable */
9
12
  interface RetainedRunReplayPoint {
@@ -2620,5 +2623,1982 @@ interface WidenGate<Out> {
2620
2623
  readonly judgeExempt?: boolean;
2621
2624
  }
2622
2625
  //#endregion
2623
- export { SupervisorOpts as $, RetainedRunEnvironmentAdmission as $n, sanitizeAgentRuntimeEvent as $t, ProviderModelAttemptEvidence as A, RecoverRetainedInteractiveRunOptions as An, OtelExporter as At, Scope as B, RecoverRetainedRunIntentOptions as Bn, loopEventToOtelSpan as Bt, MaterializedModelIdentity as C, DEFAULT_STALL_AFTER_MS as Cn, EvalRunGeneration as Ct, NodeSnapshot as D, createActivityLog as Dn, LoopSpanNode as Dt, NodeId as E, WorkerProgress as En, INTELLIGENCE_WIRE_VERSION as Et, RootHandle as F, StartRetainedInteractiveRunOptions as Fn, buildRuntimeEventOtelSpans as Ft, SpawnPrior as G, RetainedInteractiveIntentAdmission as Gn, RuntimeStreamEventCollector as Gt, SpawnEvent as H, RecoverRetainedRunResult as Hn, padTraceId as Ht, RootMaterialization as I, NativeContextContinuationExecution as In, createOpenInferenceFileExporter as It, SpendChannel as J, RetainedRunAdmissionHook as Jn, RuntimeTelemetryOptions as Jt, SpawnRejection as K, RetainedInteractiveStartedAdmission as Kn, RuntimeStreamEventSink as Kt, RootProviderModelEvidence as L, NativeContextContinuationHandle as Ln, createOtelExporter as Lt, ResultBlobStore as M, RetainedInteractiveEnvironmentInput as Mn, RuntimeEventOtelOptions as Mt, ResumedKeyState as N, RetainedInteractiveRunHandle as Nn, buildLoopOtelSpans as Nt, NodeStatus as O, readWorkerProgress as On, OtelAttribute as Ot, ResumedWork as P, RetainedInteractiveStartMaterial as Pn, buildLoopSpanNodes as Pt, Supervisor as Q, RetainedRunEffect as Qn, createRuntimeStreamEventCollector as Qt, RootSignal as R, NativeContextContinuationInput as Rn, exportEvalRuns as Rt, MaterializedExecutionIdentity as S, ActivityNote as Sn, EvalRunEvent as St, NodeExecutionIdentity as T, ScopeProgressInput as Tn, EvalRunsExportResult as Tt, SpawnJournal as U, RetainedInteractiveAdmission as Un, toOtelAttributes as Ut, Settled as V, RecoverRetainedRunOptions as Vn, padSpanId as Vt, SpawnOpts as W, RetainedInteractiveEnvironmentAdmission as Wn, RuntimeEventCollector as Wt, SteerableRootHandle as X, RetainedRunCancellation as Xn, SanitizedKnowledgeRequirement as Xt, SpendGap as Y, RetainedRunCancelOptions as Yn, SanitizedKnowledgeReadinessReport as Yt, SupervisedResult as Z, RetainedRunDispatchedAdmission as Zn, createRuntimeEventCollector as Zt, ExecutorRegistry as _, TraceSource as _n, TraceContext as _t, DefaultVerdict as a, WaitProbeRegistry as an, RetainedRunStartMaterial as ar, WaitOpts as at, ExecutorToolCall as b, sandboxSessionTraceSource as bn, readTraceContextFromEnv as bt, ExecutorAccounting as c, createWaitProbes as cn, StartRetainedRunOptions as cr, WorkerInteractiveUnavailableReason as ct, ExecutorContext as d, timerAt as dn, WorkerTraceResolver as dt, sanitizeKnowledgeReadinessReport as en, RetainedRunEventOptions as er, TokenUsageProvenance as et, ExecutorExecutionBinding as f, validateWaitSpec as fn, WorkerTraceSeamCarrier as ft, ExecutorProgressEvent as g, ToolStepInput as gn, workerTraceSeamKey as gt, ExecutorNodeContext as h, SessionTraceBox as hn, workerTraceHeaders as ht, Budget as i, WaitProbe as in, RetainedRunSnapshot as ir, UsageEvent as it, ProviderModelExecutionEvidence as j, RetainedInteractiveAdmissionHook as jn, OtelSpan as jt, ProfileMaterializationReceipt as k, ReconnectRetainedInteractiveRunOptions as kn, OtelExportConfig as kt, ExecutorCancellation as l, isWaitOutcome as ln, WorkerTraceEvidence as lt, ExecutorMaterialization as m, SessionMessageLike as mn, workerTraceEnv as mt, AgentExecutionRef as n, PendingWait as nn, RetainedRunIntentAdmission as nr, UnconfirmedTeardown as nt, ExecutionBindingReceipt as o, WaitRejection as on, RetainedRunTurnInput as or, WidenGate as ot, ExecutorFactory as p, waitUntil as pn, readWorkerTraceContext as pt, Spend as q, RetainedRunAdmission as qn, RuntimeStreamEventSummary as qt, AgentSpec as r, WaitOutcome as rn, RetainedRunReplayPoint as rr, UnknownMaterializationReason as rt, Executor as s, WaitSpec as sn, StartRetainedRunInEnvironmentOptions as sr, WorkerInteractiveSession as st, Agent as t, sanitizeRuntimeStreamEvent as tn, RetainedRunHandle as tr, TreeView as tt, ExecutorCancellationRequest as u, pollFor as un, WorkerTraceUnavailableReason as ut, ExecutorResult as v, createPushTraceSource as vn, createPropagatingTraceEmitter as vt, NoWinnerError as w, ExecutorProgress as wn, EvalRunsExportConfig as wt, Handle as x, ActivityLog as xn, traceContextToEnv as xt, ExecutorTeardownWarning as y, decodeToolPart as yn, mergeTraceEnv as yt, Runtime as z, ReconnectRetainedRunOptions as zn, generateSpanId as zt };
2624
- //# sourceMappingURL=types-DFLZMaeh.d.ts.map
2626
+ //#region src/redact.d.ts
2627
+ /**
2628
+ *
2629
+ * Redaction for values that may leave the Runtime process. The default scrubs
2630
+ * common leak classes (API keys, bearer tokens, emails, private keys) from
2631
+ * strings and walks nested objects and arrays. A customer with domain-specific
2632
+ * PII supplies their own `redact` hook.
2633
+ *
2634
+ * This is intentionally narrower than `src/sanitize.ts` (which redacts the
2635
+ * runtime's *event envelope* field-by-field): here the value is opaque
2636
+ * customer payload, so the scrub is value-shaped, not schema-shaped.
2637
+ *
2638
+ * @experimental
2639
+ */
2640
+ /** A redactor maps an arbitrary trace value to a safe-to-export value. Pure;
2641
+ * must not throw on cyclic input (the default tolerates cycles). */
2642
+ type Redactor = (value: unknown) => unknown;
2643
+ /**
2644
+ * The built-in redactor. Walks objects and arrays; replaces values under
2645
+ * secret-bearing keys wholesale; scrubs in-value patterns from every string.
2646
+ * Cycle-safe (a seen-set short-circuits self-referential payloads to
2647
+ * `'[circular]'`), depth-bounded, and total — never throws on customer input.
2648
+ */
2649
+ declare function defaultRedactor(value: unknown): unknown;
2650
+ /**
2651
+ * Resolve the redactor a client uses. A caller-supplied hook handles
2652
+ * domain-specific values first, then the built-in scrubber still removes
2653
+ * common credentials and email addresses. Returning `false` is the explicit
2654
+ * opt-out for already-reviewed public values.
2655
+ */
2656
+ declare function resolveRedactor(redact: Redactor | false | undefined): Redactor;
2657
+ //#endregion
2658
+ //#region src/mcp/protocol.d.ts
2659
+ /**
2660
+ * Shared wire contracts for the in-process stdio MCP servers.
2661
+ *
2662
+ * Keeping these types in one module prevents the delegation and generic tool
2663
+ * servers from accepting subtly different JSON-RPC messages.
2664
+ *
2665
+ * @experimental
2666
+ */
2667
+ /** A callable MCP tool exposed by either stdio server. @experimental */
2668
+ interface McpToolDescriptor {
2669
+ name: string;
2670
+ description: string;
2671
+ inputSchema: Record<string, unknown>;
2672
+ handler: (raw: unknown) => Promise<unknown>;
2673
+ }
2674
+ /** Stdio-shaped transport used by the shared JSON-RPC server implementation. @experimental */
2675
+ interface McpTransport {
2676
+ input: NodeJS.ReadableStream;
2677
+ output: NodeJS.WritableStream;
2678
+ }
2679
+ /** One JSON-RPC 2.0 request or notification. @experimental */
2680
+ interface JsonRpcMessage {
2681
+ jsonrpc: '2.0';
2682
+ id?: number | string | null;
2683
+ method: string;
2684
+ params?: unknown;
2685
+ }
2686
+ /** One JSON-RPC 2.0 response. @experimental */
2687
+ interface JsonRpcResponse {
2688
+ jsonrpc: '2.0';
2689
+ id: number | string | null;
2690
+ result?: unknown;
2691
+ error?: {
2692
+ code: number;
2693
+ message: string;
2694
+ data?: unknown;
2695
+ };
2696
+ }
2697
+ //#endregion
2698
+ //#region src/runtime/supervise/peer-mail.d.ts
2699
+ /**
2700
+ * What one envelope IS, typed so a reader can act on it without parsing prose.
2701
+ *
2702
+ * - `ask` — request a fact the sender lacks; expects an `answer`.
2703
+ * - `tell` — share a result; MUST carry evidence refs.
2704
+ * - `challenge` — dispute a peer's claim; MUST cite the refs of the claim it disputes.
2705
+ * - `answer` — reply to an `ask` or a `challenge`.
2706
+ */
2707
+ type PeerMailKind = 'ask' | 'tell' | 'challenge' | 'answer';
2708
+ /** One admitted peer message. `threadId` is the root mail's id; `depth` is 0 for a root mail and
2709
+ * one more than its parent for a reply, which is what the reply-depth cap counts. */
2710
+ interface PeerMailEnvelope {
2711
+ readonly mailId: string;
2712
+ readonly threadId: string;
2713
+ readonly depth: number;
2714
+ /** The bound sender — resolved from the capability, never from a tool argument. */
2715
+ readonly from: string;
2716
+ readonly to: string;
2717
+ readonly kind: PeerMailKind;
2718
+ readonly subject: string;
2719
+ readonly body: string;
2720
+ /** Evidence the receiver can re-check for itself. Required for `tell` and `challenge`. */
2721
+ readonly evidenceRefs: ReadonlyArray<string>;
2722
+ /** The mail id this replies to. Never a coordination question id — a peer cannot address the
2723
+ * parent's answer channel. */
2724
+ readonly replyTo?: string;
2725
+ readonly at: number;
2726
+ }
2727
+ /** Why an attempt did not reach a sibling. Each value is a fact the sender can read and act on. */
2728
+ type PeerMailRefusal = 'sender-unbound' | 'self-addressed' | 'send-quota-exhausted' | 'mailbox-full' | 'thread-depth-exceeded' | 'thread-stopped' | 'unknown-reply-target' | 'evidence-required' | 'subject-too-large' | 'body-too-large' | 'forged-authority' | 'unknown-worker' | 'already-settled' | 'worker-has-no-inbox' | 'scope-stopped' | 'runtime-error';
2729
+ type PeerMailOutcome = 'delivered' | PeerMailRefusal;
2730
+ /** The audit record for one attempt — published whether it delivered or was refused, because a
2731
+ * refused attempt is exactly what a parent auditing a channel needs to see. */
2732
+ interface PeerMailEvent {
2733
+ readonly envelope: PeerMailEnvelope;
2734
+ readonly delivered: boolean;
2735
+ readonly outcome: PeerMailOutcome;
2736
+ /** Canonical digest of the exact admitted body, so a later claim can name the bytes it read. */
2737
+ readonly bodyDigest: string;
2738
+ readonly error?: string;
2739
+ }
2740
+ /** Hard bounds. Every one fails closed with a refusal the sender can read. */
2741
+ interface PeerMailLimits {
2742
+ /** Mail one worker may attempt to send for the whole run. */
2743
+ readonly maxSentPerWorker: number;
2744
+ /** Mail one worker may receive for the whole run. */
2745
+ readonly maxInboxPerWorker: number;
2746
+ /** Total admitted body bytes one worker may receive for the whole run. */
2747
+ readonly maxInboxBytesPerWorker: number;
2748
+ /** Maximum reply depth; a root mail is depth 0, so `2` allows ask → answer → answer. */
2749
+ readonly maxThreadDepth: number;
2750
+ readonly maxBodyBytes: number;
2751
+ readonly maxSubjectBytes: number;
2752
+ }
2753
+ /** Bounds chosen so a peer channel cannot become the dominant cost of a run: eight sends and
2754
+ * sixteen receives per worker, 32 KiB of received body, and a reply chain that terminates. */
2755
+ declare const DEFAULT_PEER_MAIL_LIMITS: PeerMailLimits;
2756
+ /**
2757
+ * Phrases that mark the run's AUTHORITY in a folded prompt. A peer that writes one of these is
2758
+ * trying to speak as the supervisor, so intake refuses the envelope outright.
2759
+ *
2760
+ * The render-time fence in the inbox is the second half of this defence and neither half is
2761
+ * sufficient alone: a fence loses to a body that closes it, and an intake filter loses to a body
2762
+ * that invents a new authority phrase. Together they make forgery mechanically detectable and give
2763
+ * the standing prompt one concrete boundary to bind to. Neither makes a model OBEY a boundary.
2764
+ */
2765
+ declare const AUTHORITY_MARKERS: ReadonlyArray<string>;
2766
+ /** The wire property carrying an envelope to a worker inbox. Deliberately its OWN discriminant:
2767
+ * reusing `steer`/`answer` would let a peer mint a message on the parent's channels. */
2768
+ declare const PEER_MAIL_WIRE_KEY = "mail";
2769
+ /** The tool names a mail capability endpoint serves. It serves NOTHING else. */
2770
+ declare const peerMailVerbNames: readonly ["send_mail", "read_mail"];
2771
+ /** What a worker sees when it reads its own mailbox. */
2772
+ interface PeerMailReadout {
2773
+ /** The reading worker's own id, so a worker can address a reply correctly. */
2774
+ readonly you: string;
2775
+ /** Every envelope admitted to this worker so far, oldest first. */
2776
+ readonly inbox: ReadonlyArray<PeerMailEnvelope>;
2777
+ /** Live siblings this worker may write to (itself excluded). Without this a worker knows no
2778
+ * peer's id and the channel is unusable. */
2779
+ readonly peers: ReadonlyArray<{
2780
+ readonly workerId: string;
2781
+ readonly label: string;
2782
+ }>;
2783
+ readonly sent: number;
2784
+ /** Sends still allowed, or `null` when this run set no send quota. */
2785
+ readonly sendQuotaLeft: number | null;
2786
+ readonly limits: PeerMailLimits;
2787
+ }
2788
+ interface PeerMailSendInput {
2789
+ readonly to: unknown;
2790
+ readonly kind: unknown;
2791
+ readonly subject: unknown;
2792
+ readonly body: unknown;
2793
+ readonly evidenceRefs?: unknown;
2794
+ readonly replyTo?: unknown;
2795
+ }
2796
+ interface PeerMailbox {
2797
+ readonly limits: PeerMailLimits;
2798
+ /**
2799
+ * Publish the base URL of the capability listener once it has a port. Until it is set no spawn
2800
+ * receives a mail endpoint: a capability nobody can reach is not worth handing out, and a URL
2801
+ * built from an unassigned port would be a lie.
2802
+ */
2803
+ setEndpoint(baseUrl: string): void;
2804
+ /** Mint (idempotently, per assignment) the capability URL for one spawn. Undefined before the
2805
+ * listener has published its endpoint. */
2806
+ mintCapability(assignmentId: string): string | undefined;
2807
+ /** Bind a minted capability to the concrete worker the spawn produced. Until this runs the
2808
+ * capability can send nothing. */
2809
+ bindCapability(assignmentId: string, workerId: string): void;
2810
+ /** Resolve the capability path segment carried in a request URL. */
2811
+ hasCapability(capabilityId: string): boolean;
2812
+ /** The two tools a single capability serves, with the sender closed over. */
2813
+ tools(capabilityId: string): McpToolDescriptor[];
2814
+ send(capabilityId: string, input: PeerMailSendInput): Promise<PeerMailEvent>;
2815
+ read(capabilityId: string): PeerMailReadout;
2816
+ /** The parent's control: refuse every further mail on one thread. Returns false when the thread
2817
+ * was already stopped. Mail already delivered is not recalled — this stops the next reply. */
2818
+ stopThread(threadId: string): boolean;
2819
+ /** Every attempt in order — delivered and refused alike. */
2820
+ history(): ReadonlyArray<PeerMailEvent>;
2821
+ }
2822
+ interface PeerMailboxOptions {
2823
+ readonly scope: Scope<unknown>;
2824
+ /** Publish one attempt as a coordination event. Awaited, so a durable subscriber commits the
2825
+ * record before the sender learns the outcome. */
2826
+ readonly publish: (event: PeerMailEvent) => Promise<void>;
2827
+ readonly limits?: Partial<PeerMailLimits>;
2828
+ readonly now?: () => number;
2829
+ }
2830
+ /** True when `text` carries a phrase reserved for the run's authority. Case-insensitive, because
2831
+ * the render is read by a model and case is not what distinguishes an instruction. */
2832
+ declare function claimsAuthority(text: string): boolean;
2833
+ /** True when `value` is an envelope this runtime produced. The worker inbox parses with this, so a
2834
+ * malformed or partial wire object is discarded rather than rendered as a peer message. */
2835
+ declare function isPeerMailEnvelope(value: unknown): value is PeerMailEnvelope;
2836
+ /** Create the run's post office. One per manager scope; the manager's siblings are its addresses. */
2837
+ declare function createPeerMailbox(opts: PeerMailboxOptions): PeerMailbox;
2838
+ /**
2839
+ * The two tools ONE capability serves. `capabilityId` is closed over and `from` is not a parameter,
2840
+ * so the endpoint a worker holds can only ever speak as that worker. The descriptions carry the
2841
+ * authority rule, because the receiving model reads them as part of the channel's contract.
2842
+ */
2843
+ declare function peerMailTools(mailbox: PeerMailbox, capabilityId: string): McpToolDescriptor[];
2844
+ //#endregion
2845
+ //#region src/runtime/router-retry-policy.d.ts
2846
+ /** Exact retry controls accepted at `AgentProfile.model.metadata.retry`. */
2847
+ interface RouterRetryPolicy {
2848
+ /** Total attempts, including the first request. */
2849
+ readonly maxAttempts?: number;
2850
+ /** Delay before the second attempt. Later delays grow exponentially. */
2851
+ readonly initialBackoffMs?: number;
2852
+ /** Maximum delay between attempts. */
2853
+ readonly maxBackoffMs?: number;
2854
+ /** Symmetric random variation around each delay, from 0 through 1. */
2855
+ readonly jitter?: number;
2856
+ /** HTTP statuses that may be retried. */
2857
+ readonly retryStatuses?: ReadonlyArray<number>;
2858
+ /** Deadline for receiving one attempt's response headers. Zero disables it. */
2859
+ readonly requestTimeoutMs?: number;
2860
+ }
2861
+ //#endregion
2862
+ //#region src/runtime/tool-loop.d.ts
2863
+ /** Provider-neutral conversation record accepted by a tool-loop brain. */
2864
+ type ToolLoopMessageRecord = Record<string, unknown>;
2865
+ /** One provider-neutral tool request emitted by a tool-loop model. */
2866
+ interface ToolLoopToolCall {
2867
+ id: string;
2868
+ name: string;
2869
+ /** Raw JSON arguments emitted by the model. */
2870
+ arguments: string;
2871
+ }
2872
+ /** Runtime-owned identity and cancellation for one logical inference call. The wrapper is frozen
2873
+ * before dispatch; a transport may observe the signal but cannot replace the authority it names. */
2874
+ interface ToolLoopCallContext {
2875
+ readonly signal: AbortSignal;
2876
+ readonly callId: string;
2877
+ readonly correlationId: string;
2878
+ }
2879
+ /** One inference turn over the running conversation + the tool specs → the model's text, any
2880
+ * tool calls, and token usage. The seam every brain satisfies. */
2881
+ type ToolLoopChat = (messages: ReadonlyArray<ToolLoopMessageRecord>, tools: ReadonlyArray<ToolSpec>, context?: ToolLoopCallContext) => Promise<{
2882
+ content?: string | null;
2883
+ toolCalls: ToolLoopToolCall[];
2884
+ usage?: {
2885
+ input: number;
2886
+ output: number;
2887
+ reasoning?: number;
2888
+ };
2889
+ /** Dollar value reported for the turn. It is not billed spend unless provenance says so. */
2890
+ costUsd?: number;
2891
+ costProvenance?: 'provider-receipt' | 'billing-receipt' | 'catalog-estimate';
2892
+ /** The turn ran but its usage was not reported when the transport EXPECTED one (the streamed
2893
+ * router transport asks for usage and this says it never arrived). A metering caller records an
2894
+ * unknown turn on it; `runBrainLoop` itself ignores it. */
2895
+ usageUnknown?: true;
2896
+ /** Provider-observed model identity. Profile-bound callers validate it before accepting output. */
2897
+ model?: string;
2898
+ /** Provider-reported prompt-cache evidence; missing fields remain missing. */
2899
+ promptCache?: Readonly<Record<string, number | string>>;
2900
+ /** Physical HTTP/injected-transport attempts spent by this one logical call. */
2901
+ transportAttempts?: number;
2902
+ }>;
2903
+ /** Self-compaction — bound the loop's OWN context window the way a fresh-respawn (dumb-Ralph) loop
2904
+ * does, but in place. A stateless chat API re-sends the WHOLE running conversation every turn, so an
2905
+ * agent that accumulates dozens of turns of tool results re-bills its entire transcript on every
2906
+ * inference — the context-overflow-one-level-up that the conserved budget pool cannot fix. With
2907
+ * compaction set, once the conversation exceeds `thresholdTokens` the accumulated middle (every prior
2908
+ * assistant turn + tool result) is distilled into ONE compact progress note and the conversation is
2909
+ * reset to `[...head, digest]`: the preserved head (system + the original task) survives, the stale
2910
+ * turn-by-turn history does not. The model keeps deciding; it stops re-billing the whole transcript.
2911
+ * Fires at a CLEAN turn boundary (after a turn's tool results are folded in, before the next
2912
+ * inference) so it never orphans an assistant `tool_calls` from its `tool` replies. */
2913
+ interface ToolLoopCompaction {
2914
+ /** Compact once the estimated token count of the conversation exceeds this. */
2915
+ readonly thresholdTokens: number;
2916
+ /** Distill the conversation into a compact progress note that REPLACES the middle. Receives the
2917
+ * full conversation (so it can summarize everything done so far); returns the digest string. */
2918
+ readonly distill: (messages: ReadonlyArray<ToolLoopMessageRecord>) => Promise<string> | string;
2919
+ /** Leading messages preserved verbatim (system + the original task). Default 2. */
2920
+ readonly preserveHead?: number;
2921
+ /** Token estimator over the conversation. Default ≈ chars/4 (incl. tool-call arguments). */
2922
+ readonly estimateTokens?: (messages: ReadonlyArray<ToolLoopMessageRecord>) => number;
2923
+ /** Notified each time a compaction fires — for observability/metering. */
2924
+ readonly onCompact?: (info: {
2925
+ turn: number;
2926
+ beforeTokens: number;
2927
+ afterTokens: number;
2928
+ }) => void;
2929
+ }
2930
+ /** Public supervisor-facing compaction config: same knobs as the primitive, but `distill` is optional
2931
+ * because the supervisor has a default digest that combines a brain note with live worker state. */
2932
+ type ToolLoopCompactionOptions = Omit<ToolLoopCompaction, 'distill'> & {
2933
+ readonly distill?: ToolLoopCompaction['distill'];
2934
+ };
2935
+ //#endregion
2936
+ //#region src/runtime/router-client.d.ts
2937
+ /**
2938
+ * Connection details for Runtime's Router-backed executors.
2939
+ *
2940
+ * This is deliberately transport-only: model, prompt, tools, generation settings, and retry
2941
+ * policy belong to the exact executable `AgentProfile` consumed by `streamAgentTurn`.
2942
+ */
2943
+ interface RouterTransportConfig {
2944
+ routerBaseUrl: string;
2945
+ routerKey: string;
2946
+ /** Injectable OpenAI-compatible transport for offline execution. */
2947
+ complete?: (body: Record<string, unknown>, request?: {
2948
+ readonly headers: Readonly<Record<string, string>>;
2949
+ readonly signal?: AbortSignal;
2950
+ }) => Promise<unknown>;
2951
+ }
2952
+ /**
2953
+ * Private request configuration used by Runtime's Router adapter.
2954
+ *
2955
+ * Do not export this through a package entry point. Public callers execute a concrete
2956
+ * `AgentProfile` through `createExecutor` + `streamAgentTurn`; only Runtime may lower that profile
2957
+ * into these provider request fields.
2958
+ */
2959
+ interface RouterConfig extends RouterTransportConfig {
2960
+ model: string;
2961
+ /** Exact retry controls lowered from `AgentProfile.model.metadata.retry`. */
2962
+ retry?: RouterRetryPolicy;
2963
+ /**
2964
+ * Optional ceiling for one completion, forwarded as `max_tokens`.
2965
+ *
2966
+ * A REASONING model spends this budget on hidden thinking BEFORE it emits a visible token, so
2967
+ * the default can truncate one mid-thought and return no content at all — observed live with a
2968
+ * model that spent 8,188 of the 8,192 on reasoning and answered with nothing. Raise it for a
2969
+ * thinking model; the ceiling belongs to the router and model a caller chose, which is why it
2970
+ * lives here rather than on one call site.
2971
+ */
2972
+ maxTokens?: number;
2973
+ /**
2974
+ * Optional ceiling on TOTAL completion tokens — visible answer plus hidden reasoning —
2975
+ * forwarded as `max_completion_tokens`. Distinct from `maxTokens`: a reasoning model can spend
2976
+ * an entire `max_tokens` budget on hidden thinking, so only this field bounds what the provider
2977
+ * bills for one completion.
2978
+ */
2979
+ maxCompletionTokens?: number;
2980
+ /**
2981
+ * Take the tool-calling completion over SSE instead of one buffered POST. Off by default —
2982
+ * `routerChatWithTools` never streams, and every existing caller keeps the buffered transport
2983
+ * byte for byte.
2984
+ *
2985
+ * Why it exists: a buffered POST holds one connection idle for the WHOLE completion, and a
2986
+ * supervisor turn is the longest completion in the system. An intermediary gateway with an
2987
+ * idle-read timeout kills that connection mid-completion (the 524/503 family). A streamed
2988
+ * response puts bytes on the wire from the first generated token on, so the connection is only
2989
+ * idle through prefill. It does NOT shorten prefill, so a gateway whose deadline is
2990
+ * time-to-FIRST-byte is unaffected; only an idle-timeout gateway is.
2991
+ *
2992
+ * Mutually exclusive with `complete`: the injected transport returns one parsed JSON body and has
2993
+ * no stream to read, so setting both throws rather than silently taking the buffered path.
2994
+ *
2995
+ * WHICH PATHS CAN OPT IN. This flag is read in exactly one place (the private `chatWithTools` transport switch), so
2996
+ * every entry point that takes a caller-supplied `RouterConfig` honors it: `routerBrain`,
2997
+ * `routerToolLoop`, and `supervisorAgent` (which spreads `deps.router` into the brain's config —
2998
+ * the supervisor turn this exists for). Two production call sites build a `RouterConfig` literal
2999
+ * from their own options and therefore CANNOT express it today: the bench strategy's
3000
+ * `routerToolLoop` config in `strategy.ts` and the local sandbox client's `routerBrain` config in
3001
+ * `local-sandbox-client.ts`. Neither drives a supervisor-length turn; setting `stream` on a
3002
+ * config handed to either has no path to reach them, and they stay buffered.
3003
+ */
3004
+ stream?: boolean;
3005
+ }
3006
+ interface ToolSpec {
3007
+ type: 'function';
3008
+ function: {
3009
+ name: string;
3010
+ description?: string;
3011
+ parameters: unknown;
3012
+ };
3013
+ }
3014
+ //#endregion
3015
+ //#region src/mcp/local-harness.d.ts
3016
+ /**
3017
+ * Local coding harness available inside the sandbox — a narrowing of the shared `HarnessType`
3018
+ * vocabulary, NOT a private spelling of it. The harness id is `claude-code`; `claude` is the
3019
+ * EXECUTABLE name and lives only in the `command` field below. Keeping one vocabulary is what
3020
+ * lets a `LocalHarness` be handed straight to the profile materializer and the capability table
3021
+ * with no translation step.
3022
+ */
3023
+ type LocalHarness = Extract<HarnessType, 'claude-code' | 'codex' | 'opencode' | 'pi'>;
3024
+ /** Every local harness, in table order — the one list `AGENT_RUNTIME_LOCAL_HARNESSES` and any
3025
+ * other harness enumeration reads, so adding a row above is the only edit a new harness needs. */
3026
+ declare const LOCAL_HARNESSES: ReadonlyArray<LocalHarness>;
3027
+ /** The harness a caller gets when it expresses no preference. A composition-root default, not a
3028
+ * capability claim: one constant so the several entry points cannot drift apart. */
3029
+ declare const DEFAULT_LOCAL_HARNESS: LocalHarness;
3030
+ /** The CLI binary a harness id runs. The two are NOT the same string (`claude-code` runs `claude`),
3031
+ * so anything spawning a harness — a version probe, a login check — reads it from here rather than
3032
+ * passing the harness id as a command. */
3033
+ declare function localHarnessExecutable(harness: LocalHarness): string;
3034
+ /**
3035
+ * Whether the harness's native control can express this reasoning effort. Admission checks read
3036
+ * this so a profile the invocation would later refuse is rejected BEFORE any workspace state is
3037
+ * created, against the same table that emits the argv.
3038
+ */
3039
+ declare function harnessSupportsReasoningEffort(harness: LocalHarness, reasoningEffort: ReasoningEffort): boolean;
3040
+ /** @experimental */
3041
+ interface RunLocalHarnessOptions {
3042
+ harness: LocalHarness;
3043
+ /** Working directory for the subprocess (typically a worktree path). */
3044
+ cwd: string;
3045
+ /** Prompt forwarded as the harness CLI's task argument. */
3046
+ taskPrompt: string;
3047
+ /**
3048
+ * Pre-built command + args (e.g. from `harnessInvocation` so the full authored
3049
+ * `AgentProfile` — systemPrompt + model — reaches the harness). When set it OVERRIDES the
3050
+ * default prompt-only `buildArgs(taskPrompt)` path; `command` defaults to the harness's
3051
+ * default binary when only `args` is supplied. When absent the legacy prompt-only shape
3052
+ * is used unchanged.
3053
+ */
3054
+ invocation?: {
3055
+ command?: string;
3056
+ args: ReadonlyArray<string>;
3057
+ };
3058
+ /** Allow autonomous edits without an interactive approval gate, using whichever bypass argv the
3059
+ * harness declares. Use only when `cwd` is an isolated candidate worktree. */
3060
+ dangerouslySkipPermissions?: boolean;
3061
+ /** Isolate Codex from ambient configuration/instructions and require JSONL token usage.
3062
+ * The invocation should come from `harnessInvocation(..., { codexReproducible: true })`. */
3063
+ codexReproducible?: boolean;
3064
+ /** Absolute host paths that reproducible Codex must not read. The normalized set is compiled
3065
+ * into the controlled permission profile and its digest is returned in execution evidence. */
3066
+ codexReadDeniedPaths?: ReadonlyArray<string>;
3067
+ /** Optional wall-clock kill deadline (ms). Omit it for no timer. A positive value sends
3068
+ * SIGTERM on expiry. */
3069
+ timeoutMs?: number;
3070
+ /** Newest stdout/stderr bytes retained per stream. Default 64 MiB. */
3071
+ maxOutputBytes?: number;
3072
+ /** Caller cancellation. SIGTERM is sent on abort. */
3073
+ signal?: AbortSignal;
3074
+ /** Override env (defaults to inheriting from the parent). */
3075
+ env?: NodeJS.ProcessEnv;
3076
+ /**
3077
+ * Test seam — inject a custom spawner so unit tests can mock the
3078
+ * subprocess without touching the OS. Defaults to node's `child_process.spawn`.
3079
+ */
3080
+ spawn?: (command: string, args: ReadonlyArray<string>, opts: {
3081
+ cwd: string;
3082
+ env: NodeJS.ProcessEnv;
3083
+ stdio: 'pipe';
3084
+ detached: boolean;
3085
+ }) => ChildProcess;
3086
+ /** Test seam for locating the native Codex executable before it is staged in the worktree. */
3087
+ resolveCodexExecutable?: (command: string, env: NodeJS.ProcessEnv) => Promise<string>;
3088
+ }
3089
+ /**
3090
+ * Exact aggregate usage emitted by Codex's terminal `turn.completed` JSONL event.
3091
+ *
3092
+ * `cachedInputTokens` is a part of `inputTokens` and `reasoningOutputTokens` is a part of
3093
+ * `outputTokens`; neither adds to the total it describes. `cacheWriteInputTokens` is optional
3094
+ * because the codex CLI reports it and a provider-normalized capture omits it, and an absent
3095
+ * counter must stay absent rather than become a zero that claims no cache write was measured.
3096
+ */
3097
+ interface CodexTokenUsage {
3098
+ inputTokens: number;
3099
+ cachedInputTokens: number;
3100
+ outputTokens: number;
3101
+ reasoningOutputTokens: number;
3102
+ cacheWriteInputTokens?: number;
3103
+ }
3104
+ /** Isolation settings asserted before a reproducible Codex run is allowed to start. */
3105
+ interface CodexExecutionPolicy {
3106
+ sessionPersistence: 'ephemeral';
3107
+ userConfig: false;
3108
+ rules: false;
3109
+ projectInstructions: false;
3110
+ skillInstructions: false;
3111
+ appInstructions: false;
3112
+ toolSuggestions: false;
3113
+ multiAgentInstructions: false;
3114
+ sandbox: 'workspace-write';
3115
+ permissionProfile: 'agent_runtime_reproducible';
3116
+ approvalPolicy: 'never';
3117
+ shellNetwork: false;
3118
+ webSearch: false;
3119
+ serviceTier: 'default';
3120
+ shellEnvironment: 'core-filtered';
3121
+ loginShell: false;
3122
+ credentialsReadable: false;
3123
+ hostHomeReadable: false;
3124
+ procEnvironment: 'private-sanitized';
3125
+ sensitiveEnvironmentNamesVisible: false;
3126
+ parentRepoRead: false;
3127
+ gitMetadata: false;
3128
+ temporaryDirectory: 'workspace-private';
3129
+ stagedExecutable: 'static-elf-read-only';
3130
+ callerReadDeniedPaths: 'enforced';
3131
+ containerSockets: false;
3132
+ }
3133
+ /** Zero-model-call evidence for the exact Codex process about to run. */
3134
+ interface CodexExecutionEvidence {
3135
+ cliVersion: string;
3136
+ executableSha256: string;
3137
+ /** SHA-256 of the exact composed prompt argument proved present in the rendered prompt. */
3138
+ requestedPromptSha256: string;
3139
+ effectivePromptSha256: string;
3140
+ nonPromptArgsSha256: string;
3141
+ controlledConfigSha256: string;
3142
+ /** Sorted normalized paths compiled into the permission profile. */
3143
+ readDeniedPaths: string[];
3144
+ readDeniedPathsSha256: string;
3145
+ readDeniedPathCount: number;
3146
+ policy: CodexExecutionPolicy;
3147
+ }
3148
+ /** @experimental */
3149
+ interface LocalHarnessResult {
3150
+ /** OS exit code. `null` when killed before exit. */
3151
+ exitCode: number | null;
3152
+ /** Concatenated stdout. */
3153
+ stdout: string;
3154
+ /** Concatenated stderr. */
3155
+ stderr: string;
3156
+ /** Set when the process exited via signal (timeout / abort). */
3157
+ killedBySignal: NodeJS.Signals | null;
3158
+ /** Wall-clock duration ms (spawn → exit). */
3159
+ durationMs: number;
3160
+ /** Set when timeoutMs elapsed before exit. */
3161
+ timedOut: boolean;
3162
+ /**
3163
+ * Set when the caller's AbortSignal fired before this result settled.
3164
+ * Optional so injected runners and stored results from older releases remain valid.
3165
+ */
3166
+ aborted?: boolean;
3167
+ /** Present for a reproducible Codex run; parsed from the real terminal JSONL event. */
3168
+ usage?: CodexTokenUsage;
3169
+ /** Present for reproducible Codex runs; generated and checked before model execution. */
3170
+ evidence?: CodexExecutionEvidence;
3171
+ }
3172
+ /**
3173
+ * Spawn a local coding harness CLI as a subprocess + collect its output.
3174
+ *
3175
+ * NOT responsible for parsing the harness's output or extracting a diff —
3176
+ * the in-process executor's `streamPrompt` orchestrates `git diff` against
3177
+ * the worktree after this resolves. This function is intentionally narrow:
3178
+ * spawn, wait, capture, return.
3179
+ *
3180
+ * Fails loud — throws when:
3181
+ * - `cwd` doesn't exist (subprocess emits ENOENT; surfaced as Error)
3182
+ * - the harness binary is not on PATH (ENOENT)
3183
+ * - the caller signal was already aborted before process launch
3184
+ *
3185
+ * Does NOT throw when:
3186
+ * - the subprocess exits non-zero (`result.exitCode` carries the code)
3187
+ * - a non-reproducible subprocess is aborted / timed out (`result.aborted` /
3188
+ * `result.timedOut` carries the reason even when a TERM-aware child exits zero)
3189
+ *
3190
+ * Reproducible Codex additionally requires a terminal usage event. If cancellation
3191
+ * prevents that event, this rejects with `CodexExecutionDiagnosticError` instead of
3192
+ * returning an incomplete reproducibility receipt.
3193
+ *
3194
+ * @experimental
3195
+ */
3196
+ declare function runLocalHarness(options: RunLocalHarnessOptions): Promise<LocalHarnessResult>;
3197
+ /**
3198
+ * Parse and validate the one terminal usage event emitted by `codex exec --json`.
3199
+ *
3200
+ * The JSONL framing is this surface's own; the usage RECORD is read by `parseCodexUsageRecord`,
3201
+ * the one codex usage reader the sandbox decoder also calls, so both surfaces hold the same field
3202
+ * policy and the same two cross-field invariants.
3203
+ */
3204
+ declare function parseCodexTokenUsage(stdout: string): CodexTokenUsage;
3205
+ //#endregion
3206
+ //#region src/mcp/worktree.d.ts
3207
+ /**
3208
+ *
3209
+ * Git worktree helpers for the in-process delegation executor. Each
3210
+ * delegation runs in its own worktree so multiple parallel harness
3211
+ * subprocesses (claude / codex / opencode in a 3-way fanout) don't clobber
3212
+ * each other's edits on the shared workspace.
3213
+ *
3214
+ * Worktrees live under `<repoRoot>/.agent-worktrees/<runId>/`. After the
3215
+ * harness exits + the diff is captured, the worktree is removed.
3216
+ *
3217
+ * All operations spawn `git` via `child_process.spawn` synchronously
3218
+ * (via a `runGit` helper). Stays narrow on purpose: no commits, no rebases.
3219
+ * Diff capture stages all changes (`git add -A`) into the ephemeral worktree's
3220
+ * index so created (untracked) files appear in the `--cached` diff.
3221
+ *
3222
+ * @experimental
3223
+ */
3224
+ /** @experimental */
3225
+ interface WorktreeHandle {
3226
+ /** Absolute path to the worktree directory. */
3227
+ path: string;
3228
+ /** SHA the worktree was created at. */
3229
+ baseSha: string;
3230
+ /** Branch name created for this worktree (typically `delegate/<runId>`). */
3231
+ branch: string;
3232
+ }
3233
+ /** @experimental */
3234
+ interface CreateWorktreeOptions {
3235
+ /** Absolute path to the main git checkout. */
3236
+ repoRoot: string;
3237
+ /** Unique id for the worktree path + branch. Use the delegation run id. */
3238
+ runId: string;
3239
+ /** Parent directory the worktree lives under. Defaults to `.agent-worktrees`. */
3240
+ variantsDir?: string;
3241
+ /** Override the base ref (default `HEAD`). */
3242
+ baseRef?: string;
3243
+ /** Test seam — inject a custom git runner. */
3244
+ runGit?: GitRunner;
3245
+ }
3246
+ /** @experimental */
3247
+ interface DiffOptions {
3248
+ /** Worktree to diff. */
3249
+ worktree: WorktreeHandle;
3250
+ /** What to compare against. Default `worktree.baseSha`. */
3251
+ baseRef?: string;
3252
+ /**
3253
+ * Repository-relative input paths to omit from the captured worker patch.
3254
+ * Paths are passed to Git with literal exclusion magic, so profile-provided
3255
+ * `*`, `?`, `[` and `:` characters can never expand into broader pathspecs.
3256
+ */
3257
+ excludePaths?: ReadonlyArray<string>;
3258
+ /** Test seam. */
3259
+ runGit?: GitRunner;
3260
+ }
3261
+ /** @experimental */
3262
+ interface DiffResult {
3263
+ patch: string;
3264
+ stats: {
3265
+ filesChanged: number;
3266
+ insertions: number;
3267
+ deletions: number;
3268
+ };
3269
+ }
3270
+ /** @experimental */
3271
+ interface RemoveWorktreeOptions {
3272
+ worktree: WorktreeHandle;
3273
+ repoRoot: string;
3274
+ /** Force removal even if dirty (default true; the loser of a fanout has uncommitted changes). */
3275
+ force?: boolean;
3276
+ /** Test seam. */
3277
+ runGit?: GitRunner;
3278
+ }
3279
+ /** Pluggable git runner (sync) — replaceable in tests. */
3280
+ type GitRunner = (args: ReadonlyArray<string>, opts: {
3281
+ cwd: string;
3282
+ }) => {
3283
+ stdout: string;
3284
+ stderr: string;
3285
+ exitCode: number;
3286
+ };
3287
+ /** Checkout a fresh git worktree for a delegation run on a new branch under `variantsDir`. @experimental */
3288
+ declare function createWorktree(options: CreateWorktreeOptions): Promise<WorktreeHandle>;
3289
+ /** Stage worker changes and return the diff + shortstat, excluding declared input paths. @experimental */
3290
+ declare function captureWorktreeDiff(options: DiffOptions): Promise<DiffResult>;
3291
+ /**
3292
+ * Remove a git worktree and delete its branch. Already-removed paths are harmless; every other
3293
+ * Git failure rejects so callers cannot report a worktree as destroyed when cleanup failed.
3294
+ * @experimental
3295
+ */
3296
+ declare function removeWorktree(options: RemoveWorktreeOptions): Promise<void>;
3297
+ //#endregion
3298
+ //#region src/mcp/worktree-harness.d.ts
3299
+ /** Outcome of one verification command run in the worktree (test or typecheck). */
3300
+ interface WorktreeCommandResult {
3301
+ /** The shell command line that was run. */
3302
+ command: string;
3303
+ /** Did the command exit 0? The PASS signal a deliverable gate / coder output reads. */
3304
+ passed: boolean;
3305
+ /** OS exit code, or `null` when killed before exit. */
3306
+ exitCode: number | null;
3307
+ /** Combined stdout+stderr (capped) — surfaced in traces for diagnosis. */
3308
+ output: string;
3309
+ }
3310
+ /** Proof of the profile inputs delivered before the worker process started. */
3311
+ interface WorktreeProfileMaterializationReceipt {
3312
+ /** Digest of the exact materializer plan: files, modes, environment, flags, and unsupported rows. */
3313
+ workspacePlanDigest: string;
3314
+ /** Repository-relative profile input files written into the worker worktree. */
3315
+ writtenPaths: string[];
3316
+ /** Must be empty on a successful run because this path fails closed. */
3317
+ unsupported: WorkspacePlanReceipt['unsupported'];
3318
+ /** Environment variable names added to the worker process. Values remain out of telemetry. */
3319
+ environmentNames: string[];
3320
+ /** Exact additional CLI arguments emitted by the materializer. */
3321
+ flags: string[];
3322
+ /** `resources.instructions` bypasses native project files so reproducible Codex cannot drop it. */
3323
+ resourceInstructions: {
3324
+ delivery: 'none' | 'invocation-prompt';
3325
+ sha256: string | null;
3326
+ byteLength: number;
3327
+ };
3328
+ }
3329
+ /** The canonical result of one worktree-harness run, projected by each port to its own shape. */
3330
+ interface WorktreeHarnessResult {
3331
+ /** The branch the worktree was cut on (`delegate/<runId>`). */
3332
+ branch: string;
3333
+ /** `git diff` of the worktree against its base — the unified patch the harness produced. */
3334
+ patch: string;
3335
+ /** Shortstat-derived change counts. */
3336
+ stats: {
3337
+ filesChanged: number;
3338
+ insertions: number;
3339
+ deletions: number;
3340
+ };
3341
+ /**
3342
+ * Exact profile materialization applied before the harness launched.
3343
+ * Absent on transports that cannot return a materializer receipt; never fabricated.
3344
+ */
3345
+ profileMaterialization?: WorktreeProfileMaterializationReceipt;
3346
+ /** The harness subprocess outcome. */
3347
+ harness: {
3348
+ name: LocalHarness | 'bridge';
3349
+ exitCode: number | null;
3350
+ timedOut: boolean;
3351
+ killedBySignal: NodeJS.Signals | null;
3352
+ durationMs: number;
3353
+ stdout: string;
3354
+ stderr: string;
3355
+ /** Exact Codex JSONL usage when reproducible mode is enabled. */
3356
+ usage?: CodexTokenUsage;
3357
+ /** Installed CLI version captured immediately before execution. */
3358
+ cliVersion?: string;
3359
+ /** SHA-256 of the native Codex executable staged read-only in the candidate worktree. */
3360
+ executableSha256?: string;
3361
+ /** SHA-256 of the exact composed prompt argument proved present in Codex's rendered prompt. */
3362
+ requestedPromptSha256?: string;
3363
+ /** SHA-256 of `codex debug prompt-input` output for the exact isolated prompt. */
3364
+ effectivePromptSha256?: string;
3365
+ /** SHA-256 of the exact executable + argv with prompt content replaced by `<PROMPT>`. */
3366
+ nonPromptArgsSha256?: string;
3367
+ /** SHA-256 of the isolated config that fixes permissions and shell environment. */
3368
+ controlledConfigSha256?: string;
3369
+ /** SHA-256 of the normalized caller-supplied host read-denial paths. */
3370
+ readDeniedPathsSha256?: string;
3371
+ /** Sorted normalized caller-supplied host read-denial paths. */
3372
+ readDeniedPaths?: string[];
3373
+ /** Number of normalized caller-supplied host read-denial paths. */
3374
+ readDeniedPathCount?: number;
3375
+ /** Explicit isolation claims checked before model execution. */
3376
+ executionPolicy?: CodexExecutionPolicy;
3377
+ };
3378
+ /** Verification signals derived in the live worktree (present only when commands were given). */
3379
+ checks?: {
3380
+ tests?: WorktreeCommandResult;
3381
+ typecheck?: WorktreeCommandResult;
3382
+ };
3383
+ }
3384
+ /** The single shell-command-in-worktree runner seam (replaces the per-executor copies). */
3385
+ type WorktreeCheckRunner = (opts: {
3386
+ command: string;
3387
+ cwd: string;
3388
+ timeoutMs: number;
3389
+ signal?: AbortSignal;
3390
+ }) => Promise<{
3391
+ exitCode: number | null;
3392
+ output: string;
3393
+ }>;
3394
+ /** The canonical result of one in-place harness run. The edits are the DIRECTORY, not a patch:
3395
+ * the caller supplied the workspace and reads it directly. */
3396
+ interface InPlaceHarnessResult {
3397
+ /** The directory the harness ran in, exactly as supplied. */
3398
+ workspacePath: string;
3399
+ /** Exact profile materialization applied before the harness launched, and removed after it. */
3400
+ profileMaterialization: WorktreeProfileMaterializationReceipt;
3401
+ /** The harness subprocess outcome. */
3402
+ harness: WorktreeHarnessResult['harness'];
3403
+ }
3404
+ //#endregion
3405
+ //#region src/runtime/harness-usage.d.ts
3406
+ /**
3407
+ * One harness's own token-usage report for one turn, in the runtime's field names.
3408
+ *
3409
+ * `input` is the provider's TOTAL prompt count and `output` is its TOTAL completion count.
3410
+ * The other three counters CLASSIFY a part of one of those totals; none of them adds to it.
3411
+ * `cachedInput` and `cacheWriteInput` classify `input`, which is the convention
3412
+ * `promptCacheTokenClasses` (`util.ts`) folds: `freshInput = input - cacheRead - cacheWrite`.
3413
+ * `reasoningOutput` classifies `output`.
3414
+ *
3415
+ * A counter the harness does not report stays absent, because a zero would claim the harness
3416
+ * measured none.
3417
+ */
3418
+ interface HarnessUsage {
3419
+ /** The harness family whose adapter produced this report. */
3420
+ readonly harness: HarnessType;
3421
+ /** Total prompt tokens the provider charged for the turn, the cached ones included. */
3422
+ readonly input: number;
3423
+ /** Total completion tokens the provider charged for the turn, the reasoning ones included. */
3424
+ readonly output: number;
3425
+ /** The part of `input` the provider served from its prompt cache. */
3426
+ readonly cachedInput?: number;
3427
+ /** The part of `input` the provider wrote into its prompt cache. */
3428
+ readonly cacheWriteInput?: number;
3429
+ /** The part of `output` the model spent on reasoning. Never added to `output`. */
3430
+ readonly reasoningOutput?: number;
3431
+ }
3432
+ /**
3433
+ * Decode a sandbox event with one harness's adapter, or `undefined` when the event carries no
3434
+ * harness-native usage.
3435
+ *
3436
+ * A NAMED harness reads with that harness's adapter only, and a named harness with no adapter
3437
+ * reports nothing. It never falls through to another harness's adapter: a different harness's
3438
+ * `turn.completed` decoded as codex would either drop the counters codex does not name or fail on
3439
+ * a field codex requires, and both answers would be about the wrong harness. The composite over
3440
+ * every registered adapter runs only when the caller cannot name the harness.
3441
+ *
3442
+ * Throws `ValidationError` when an adapter recognizes the event as its harness's usage carrier and
3443
+ * cannot read the numbers.
3444
+ */
3445
+ declare function decodeHarnessUsage(event: SandboxEvent, harness?: HarnessType): HarnessUsage | undefined;
3446
+ //#endregion
3447
+ //#region src/runtime/codex-rollout-store.d.ts
3448
+ /** Who wrote one rollout, exactly as its own `session_meta` states it. Nothing here is inferred. */
3449
+ interface CodexRolloutIdentity {
3450
+ /** The rollout's own thread id (`session_meta.payload.id`). */
3451
+ readonly sessionId: string;
3452
+ /** The thread this one was spawned or forked from, when it was. */
3453
+ readonly parentThreadId?: string;
3454
+ /** The thread whose rows are prepended into this file, when this file is a fork. */
3455
+ readonly forkedFromId?: string;
3456
+ /** True when `thread_source` reads `subagent`: a harness-native child, invisible to the journal. */
3457
+ readonly nativeChild: boolean;
3458
+ /** The child's own path in the harness's agent tree (`/root/c1_b_grid`), when it has one. */
3459
+ readonly agentPath?: string;
3460
+ /** The harness's own nickname for the child ("Turing"), when it has one. */
3461
+ readonly agentNickname?: string;
3462
+ /** Spawn depth the harness recorded. `1` is a direct child of the seat. */
3463
+ readonly depth?: number;
3464
+ /** The working directory the session ran in, used to attribute a store to a workspace. */
3465
+ readonly cwd?: string;
3466
+ /** The codex build that wrote it. */
3467
+ readonly cliVersion?: string;
3468
+ /** When the session itself started, from its own `session_meta` timestamp. */
3469
+ readonly startedAtMs?: number;
3470
+ }
3471
+ /** How this reader isolated the session's own rows from the parent rows prepended to its file. */
3472
+ type CodexForkBoundary =
3473
+ /** Not a fork: every row in the file belongs to this session. */
3474
+ {
3475
+ readonly kind: 'whole-file';
3476
+ } |
3477
+ /** A fork whose own first turn was isolated, and by which rule. */
3478
+ {
3479
+ readonly kind: 'resolved';
3480
+ readonly rule: 'history-start-ordinal' | 'turn-is-session' | 'turn-uuid-v7' | 'turn-start-time';
3481
+ /** The `turn_id` of the session's own first turn. */
3482
+ readonly turnId?: string;
3483
+ /** Rows credited to the parent and excluded from `own`. */
3484
+ readonly inheritedTurns: number;
3485
+ } |
3486
+ /** A fork this reader could not isolate. `own` is absent; nothing may be charged. */
3487
+ {
3488
+ readonly kind: 'unresolved';
3489
+ readonly reason: string;
3490
+ };
3491
+ /** One turn of one session, with the counters it added to the session's cumulative total. */
3492
+ interface CodexRolloutTurn {
3493
+ readonly turnId?: string;
3494
+ readonly startedAtMs?: number;
3495
+ readonly usage: HarnessUsage;
3496
+ }
3497
+ /** One rollout file, read. */
3498
+ interface CodexRolloutSession {
3499
+ readonly identity: CodexRolloutIdentity;
3500
+ readonly boundary: CodexForkBoundary;
3501
+ /**
3502
+ * The session's OWN spend — the cumulative delta from its fork boundary to its last report.
3503
+ * ABSENT when the boundary is unresolved: an unattributable number must not be charged.
3504
+ */
3505
+ readonly own?: HarnessUsage;
3506
+ /** The session's own turns, newest last. Empty when the file reported no usage. */
3507
+ readonly turns: readonly CodexRolloutTurn[];
3508
+ /**
3509
+ * The file's final cumulative `total_token_usage`, kept ONLY as the diagnostic that shows how
3510
+ * far a naive file total is from the truth. Never charge this.
3511
+ */
3512
+ readonly fileCumulativeInput: number;
3513
+ readonly fileCumulativeOutput: number;
3514
+ }
3515
+ /** What one incremental read of a store observed. */
3516
+ interface CodexStoreDelta {
3517
+ /** Spend by sessions that are NOT native children — the seat's own turns. */
3518
+ readonly seat: HarnessUsage;
3519
+ /** Spend by `thread_source: subagent` sessions — the harness-native children. */
3520
+ readonly native: HarnessUsage;
3521
+ /** Sessions whose fork boundary could not be isolated, so their spend is absent, not zero. */
3522
+ readonly unresolved: ReadonlyArray<{
3523
+ readonly sessionId: string;
3524
+ readonly reason: string;
3525
+ }>;
3526
+ /**
3527
+ * Every session this read touched, for evidence. Each one states its WHOLE own spend and turn
3528
+ * list, which is not the same number as `seat` / `native`: those two carry only what this read
3529
+ * newly observed.
3530
+ */
3531
+ readonly sessions: readonly CodexRolloutSession[];
3532
+ }
3533
+ /** A store reader that credits each turn once: it tails only the bytes appended since the last read. */
3534
+ interface CodexRolloutStoreReader {
3535
+ /**
3536
+ * Read everything appended since the previous call and attribute it.
3537
+ *
3538
+ * The FIRST call establishes the baseline. Call it before the first turn so pre-existing rows are
3539
+ * consumed and credited to nothing; every later call returns exactly that turn's spend.
3540
+ */
3541
+ read(): Promise<CodexStoreDelta>;
3542
+ }
3543
+ /** Where a harness keeps its own session store, and which workspace may be credited from it. */
3544
+ interface CodexRolloutStoreRef {
3545
+ /**
3546
+ * Absolute path to the harness home the CLI writes into — `CODEX_HOME`, or `$HOME/.codex`.
3547
+ * This MUST be the run's own isolated store. Pointing it at an ambient host store credits one
3548
+ * run with another run's files, which is the exact defect this reader exists to end.
3549
+ */
3550
+ readonly root: string;
3551
+ /**
3552
+ * Credit only sessions whose recorded `cwd` is this path or below it. Absent credits every
3553
+ * session under `root`, which is correct only for a store no other run writes to.
3554
+ */
3555
+ readonly workspaceRoot?: string;
3556
+ }
3557
+ /**
3558
+ * Read one rollout's rows into a session record.
3559
+ *
3560
+ * `rows` is the file's JSON values in file order. Pass the whole file to read a completed session;
3561
+ * the store reader passes appended slices and carries the identity forward itself.
3562
+ */
3563
+ declare function readCodexRolloutSession(rows: Iterable<unknown>): CodexRolloutSession | undefined;
3564
+ /**
3565
+ * Open an incremental reader over a codex store.
3566
+ *
3567
+ * Nothing is read until `read()` is called, and every read is bounded by the bytes appended since
3568
+ * the previous one, so a 695MB rollout is scanned once rather than once per turn.
3569
+ */
3570
+ declare function createCodexRolloutStoreReader(ref: CodexRolloutStoreRef): CodexRolloutStoreReader;
3571
+ /** Sum two usage reports on every counter both of them state. */
3572
+ declare function addHarnessUsage(left: HarnessUsage, right: HarnessUsage): HarnessUsage;
3573
+ /** True when a report states any spend at all. */
3574
+ declare function harnessUsageIsEmpty(usage: HarnessUsage): boolean;
3575
+ //#endregion
3576
+ //#region src/runtime/tangle-sandbox-exact-process-provider.d.ts
3577
+ type SandboxControlClient = Pick<Sandbox, 'create' | 'get' | 'list'>;
3578
+ interface CreateTangleSandboxExactProcessProviderOptions {
3579
+ name?: string;
3580
+ }
3581
+ /**
3582
+ * Adapt Tangle Sandbox's managed control runtime to Runtime's exact-process provider.
3583
+ *
3584
+ * The adapter deliberately exposes no ordinary agent environment: an exact experiment
3585
+ * must start a fresh Sandbox with no managed agent and launch its declared argv directly.
3586
+ */
3587
+ declare function createTangleSandboxExactProcessProvider(client: SandboxControlClient, options?: CreateTangleSandboxExactProcessProviderOptions): AgentEnvironmentProvider;
3588
+ //#endregion
3589
+ //#region src/runtime/environment-provider.d.ts
3590
+ /** Provider object or registry name accepted by runtime provider adapters.
3591
+ * @experimental */
3592
+ type AgentEnvironmentProviderRef = AgentEnvironmentProvider | string;
3593
+ /** In-memory registry for named `AgentEnvironmentProvider` instances.
3594
+ * @experimental */
3595
+ interface AgentEnvironmentProviderRegistry {
3596
+ register(provider: AgentEnvironmentProvider, options?: {
3597
+ replace?: boolean;
3598
+ }): void;
3599
+ has(name: string): boolean;
3600
+ get(name: string): AgentEnvironmentProvider | undefined;
3601
+ require(name: string): AgentEnvironmentProvider;
3602
+ names(): string[];
3603
+ providers(): AgentEnvironmentProvider[];
3604
+ capabilities(name: string): Promise<AgentEnvironmentCapabilities>;
3605
+ }
3606
+ /** Create a registry that resolves provider names to concrete provider instances.
3607
+ * @experimental */
3608
+ declare function createAgentEnvironmentProviderRegistry(providers?: Iterable<AgentEnvironmentProvider>): AgentEnvironmentProviderRegistry;
3609
+ /** Resolve a provider instance or registry name, failing loudly when a name is unknown.
3610
+ * @experimental */
3611
+ declare function resolveAgentEnvironmentProvider(provider: AgentEnvironmentProviderRef, registry?: AgentEnvironmentProviderRegistry): AgentEnvironmentProvider;
3612
+ /** Options for exposing an `AgentEnvironmentProvider` through the legacy sandbox client port.
3613
+ * @experimental */
3614
+ interface ProviderAsSandboxClientOptions {
3615
+ defaults?: Partial<CreateAgentEnvironmentInput>;
3616
+ requireTerminalEvent?: boolean;
3617
+ /** Require declared live continuation plus concrete session controls. */
3618
+ requireSession?: boolean;
3619
+ mapCreateOptions?: (options: CreateSandboxOptions | undefined) => Partial<CreateAgentEnvironmentInput>;
3620
+ }
3621
+ /** Adapt a neutral environment provider to the `SandboxClient` interface used by existing loop paths.
3622
+ * @experimental */
3623
+ declare function providerAsSandboxClient(provider: AgentEnvironmentProvider, options?: ProviderAsSandboxClientOptions): SandboxClient;
3624
+ /** Options for wrapping the current Tangle sandbox client as an environment provider.
3625
+ * @experimental */
3626
+ interface SandboxClientProviderOptions {
3627
+ name?: string;
3628
+ defaultBackend?: BackendType;
3629
+ capabilities?: AgentEnvironmentCapabilities | (() => AgentEnvironmentCapabilities | Promise<AgentEnvironmentCapabilities>);
3630
+ validateProfile?: (profile: AgentProfileRef) => AgentProfileValidationResult | Promise<AgentProfileValidationResult>;
3631
+ /** Resolve a named profile before calling Sandbox, which accepts inline profiles only. */
3632
+ resolveProfile?: (profileId: string) => AgentProfile | Promise<AgentProfile>;
3633
+ mapCreateInput?: (input: CreateAgentEnvironmentInput) => CreateSandboxOptions;
3634
+ }
3635
+ /**
3636
+ * Adapt a `SandboxClient` into the shared `AgentEnvironmentProvider` contract.
3637
+ * The provider declares the public SDK contract before it creates an environment.
3638
+ * Each environment exposes interactive methods only when its deployment declares every required capability.
3639
+ * @experimental */
3640
+ declare function sandboxClientAsProvider(client: SandboxClient, options?: SandboxClientProviderOptions): AgentEnvironmentProvider;
3641
+ /**
3642
+ * What one provider-executed turn settles on: the visible answer plus the complete event archive
3643
+ * the environment streamed. It is the value a `ProviderExecutorOptions.validator` scores.
3644
+ *
3645
+ * @experimental
3646
+ */
3647
+ interface ProviderLeafOut {
3648
+ content: string;
3649
+ events: AgentEnvironmentEvent[];
3650
+ }
3651
+ /**
3652
+ * Per-run Sandbox prompt options for the provider path — the same field, the same name, and the
3653
+ * same kernel-owned exclusions as `ExecCtx.promptOptions` on the sandbox path.
3654
+ *
3655
+ * The kernel owns `sessionId` and `signal`, so neither is declarable: a caller-chosen session id
3656
+ * would make every worker share one server session, and the abort channel belongs to the run.
3657
+ * `model` is excluded too, and for a different reason: this executor's materialization record
3658
+ * names the model from `AgentProfile`, so a turn-level override would make the record state a
3659
+ * model the provider did not run. Declare the instrument on `AgentProfile.model`.
3660
+ *
3661
+ * Everything else is the per-call configuration a portable profile cannot carry. `backend` is the
3662
+ * load-bearing one: `backend.model.authMode` plus `authFiles` is how a caller-owned subscription
3663
+ * seat reaches the harness inside the environment. Runtime lowers these onto the turn with the one
3664
+ * mapper it already uses in the other direction, so a sandbox-shaped provider reads them from
3665
+ * `AgentTurnInput.providerOptions.backend` exactly as it reads a sandbox box's prompt options.
3666
+ *
3667
+ * @experimental
3668
+ */
3669
+ type ProviderPromptOptions = Omit<PromptOptions, 'model' | 'sessionId' | 'signal'>;
3670
+ /** Options for running a provider as a supervise-mode executor.
3671
+ * @experimental */
3672
+ interface ProviderExecutorOptions {
3673
+ defaults?: Partial<CreateAgentEnvironmentInput>;
3674
+ runtime?: Runtime;
3675
+ destroyOnSettle?: boolean;
3676
+ requireTerminalEvent?: boolean;
3677
+ /**
3678
+ * Per-run prompt options merged UNDER every streamed turn: a mapped turn's own field wins, and
3679
+ * the runtime's abort signal is applied last. `providerOptions` merges one level, so a
3680
+ * `taskToTurn` that sets its own provider option cannot silently drop the session credential
3681
+ * declared here.
3682
+ */
3683
+ promptOptions?: ProviderPromptOptions;
3684
+ /**
3685
+ * OPT-IN executable score for this worker, with the SAME contract the sandbox seam's validator
3686
+ * has: `validate` runs while the environment is still alive, so `ValidationCtx.box` can read
3687
+ * files and run commands in the environment it is scoring. Every other supervised hook fires
3688
+ * after teardown and can only read the artifact.
3689
+ *
3690
+ * The verdict becomes the settled artifact's verdict. Absent, nothing changes and the leaf falls
3691
+ * back to its own settle verdict.
3692
+ */
3693
+ validator?: Validator<ProviderLeafOut>;
3694
+ /** Transform only the profile sent to `provider.create`. The original profile
3695
+ * remains the input to `taskToTurn`, so execution-only normalization cannot
3696
+ * rewrite the caller's task mapping. */
3697
+ profileForCreate?: (profile: AgentProfile) => AgentProfile;
3698
+ taskToTurn?: (task: unknown, specProfile: AgentProfile) => AgentTurnInput;
3699
+ }
3700
+ /** Adapt an environment provider into an `ExecutorFactory` for `createExecutor`.
3701
+ *
3702
+ * `createExecutor({ backend: 'provider', provider })` is the composition most callers want; it
3703
+ * builds this factory and injects the seam. See `examples/provider-executor/`.
3704
+ *
3705
+ * Still `@experimental`: the entry point that consumes it, `createExecutor`, carries no stability
3706
+ * tag and is therefore experimental by default, so a stable promise here would be reachable only
3707
+ * through an experimental symbol.
3708
+ *
3709
+ * @experimental */
3710
+ declare function providerAsExecutor(provider: AgentEnvironmentProvider, options?: ProviderExecutorOptions): ExecutorFactory<unknown>;
3711
+ //#endregion
3712
+ //#region src/runtime/key-provider.d.ts
3713
+ /** Resolve named secrets. The ONE seam every secret store adapts to. */
3714
+ interface KeyProvider {
3715
+ /** The value for `name`, or `undefined` when this provider does not hold it. */
3716
+ get(name: string): Promise<string | undefined>;
3717
+ }
3718
+ /** The env-backed provider: reads the (dotenvx-loaded) process env. Empty /
3719
+ * whitespace-only values count as absent — fail loud, not with a blank key. */
3720
+ declare function envKeyProvider(env?: Record<string, string | undefined>): KeyProvider;
3721
+ /** The `AgentProfileMcpServer.metadata` key the declarative secret-env map
3722
+ * rides under: `{ ENV_VAR_NAME: 'PROVIDER_KEY_NAME' }`. Names only — values
3723
+ * are resolved at materialize time and never stored. */
3724
+ declare const mcpSecretEnvMetadataKey = "secretEnv";
3725
+ /** Read (and validate) a server entry's declared secret-env map, if any.
3726
+ * Malformed metadata throws — a half-declared secret must never half-boot. */
3727
+ declare function secretEnvOfMcpServer(server: AgentProfileMcpServer): Record<string, string> | undefined;
3728
+ /**
3729
+ * Resolve a declared secret-env map into the real env entries for a server
3730
+ * spawn. Fail-closed: no provider or a missing key throws, naming the KEY
3731
+ * NAME only (the value never appears in any message). `label` names the
3732
+ * server for the error (e.g. `profile.mcp['exa']`).
3733
+ */
3734
+ declare function resolveSecretEnv(secretEnv: Record<string, string>, keys: KeyProvider | undefined, label: string): Promise<Record<string, string>>;
3735
+ /** The spawn-ready strings for one stdio MCP server: profile config values
3736
+ * resolved, secrets separated so the client can redact them. */
3737
+ interface ResolvedMcpServerLaunch {
3738
+ args?: string[];
3739
+ /** Public env, safe to appear in diagnostics. */
3740
+ env?: Record<string, string>;
3741
+ /** Resolved secret env. Reaches only the child process; redacted everywhere else. */
3742
+ protectedEnv?: Record<string, string>;
3743
+ }
3744
+ /**
3745
+ * Resolve a profile MCP server's `args`/`env` config values (interface ≥0.40
3746
+ * `AgentProfileConfigValue`) plus the legacy `metadata.secretEnv` channel into
3747
+ * the plain strings a spawn needs.
3748
+ *
3749
+ * Rules, all fail-closed:
3750
+ * - `args` must be public values. A secret-ref in argv is refused: argv is
3751
+ * readable by every host process (/proc/PID/cmdline) and outside the
3752
+ * protected-value redaction channel, so a secret there cannot be contained.
3753
+ * - `env` secret-refs resolve through the KeyProvider (missing provider or key
3754
+ * throws, naming the KEY NAME only) and land in `protectedEnv`.
3755
+ * - An env var declared secret on BOTH channels (env secret-ref and
3756
+ * metadata.secretEnv) is ambiguous configuration and throws.
3757
+ * - A public `env` entry shadowed by a legacy metadata secret keeps the
3758
+ * pre-0.40 spawn precedence: the secret value wins in the child env.
3759
+ */
3760
+ declare function resolveMcpServerLaunch(server: AgentProfileMcpServer, keys: KeyProvider | undefined, label: string): Promise<ResolvedMcpServerLaunch>;
3761
+ //#endregion
3762
+ //#region src/runtime/sandbox-events.d.ts
3763
+ /** The provider/model the platform reports it actually bound to a turn, when it reports one.
3764
+ * `source` is the platform's own account of where that choice came from — `environment` means
3765
+ * the platform chose, not the request. */
3766
+ interface SandboxServedBackend {
3767
+ readonly provider?: string;
3768
+ readonly model?: string;
3769
+ readonly source?: string;
3770
+ }
3771
+ /**
3772
+ * Read the served execution identity off one Sandbox event.
3773
+ *
3774
+ * The platform reports `effectiveBackend` on `execution.started` and again on the terminal
3775
+ * event (`@tangle-network/sandbox`, `EffectiveBackend`). Absence returns `undefined`
3776
+ * and must stay unknown — a request is not a receipt, so nothing here may be inferred from
3777
+ * what was asked for.
3778
+ */
3779
+ declare function sandboxEventServedBackend(event: SandboxEvent): SandboxServedBackend | undefined;
3780
+ /**
3781
+ * Fail the execution when the platform reports serving a model other than the exact one asked for.
3782
+ *
3783
+ * Measured motive (agent-runtime#892, live infrastructure 2026-08-17): 6 of 6 boxes whose profile
3784
+ * declared `zai-coding-plan/glm-5.2` reported
3785
+ * `{"provider":"openai-compat","model":"deepseek/deepseek-v4-flash","source":"environment"}`,
3786
+ * while the materialization receipt recorded the declared model as `status: "known"`. Sending
3787
+ * `backend.model` makes that substitution unlikely; only reading the report back makes it
3788
+ * detectable. A run that cannot say which model produced its evidence must not settle as one
3789
+ * that can.
3790
+ *
3791
+ * Silent when the platform reports no served model: unobserved stays unobserved.
3792
+ */
3793
+ declare function assertSandboxServedModel(event: SandboxEvent, expected: {
3794
+ readonly provider?: string;
3795
+ readonly model?: string;
3796
+ } | undefined): void;
3797
+ /**
3798
+ * Extract a `RuntimeStreamEvent`-shaped `llm_call` from a sandbox event when
3799
+ * the event carries usage/cost data. Returns `undefined` for non-cost events
3800
+ * so the kernel can iterate the full stream without branching.
3801
+ *
3802
+ * Pure by contract: it never throws on a failed run. The terminal truth
3803
+ * boundary is the public Sandbox outcome tracker, applied after the complete
3804
+ * stream. Post-hoc readers — {@link sumSandboxUsage}, the
3805
+ * analyst trace store, the chat projection — must stay able to read a failed
3806
+ * turn's events, which is when reading them matters most.
3807
+ *
3808
+ * Canonical cost-carrying types observed in the wild:
3809
+ * - `llm_call` — `data: { model, tokensIn, tokensOut, costUsd, ... }`
3810
+ * - `message.completed` / `result` — `data: { usage: { inputTokens,
3811
+ * outputTokens, totalCostUsd? } }`
3812
+ * - `cost.usage` / `usage` — same shape under a dedicated type
3813
+ *
3814
+ * Numeric coercion is strict: `Number.isFinite` gates every accumulator write
3815
+ * so a sentinel `NaN` from a misbehaving backend cannot poison the ledger.
3816
+ */
3817
+ declare function extractLlmCallEvent(event: SandboxEvent, agentRunName: string): (RuntimeStreamEvent & {
3818
+ type: 'llm_call';
3819
+ }) | undefined;
3820
+ /**
3821
+ * Per-turn usage accounting over BOTH the canonical events and the harness-native ones.
3822
+ *
3823
+ * Some harnesses report a turn's tokens only inside their own event (`harness-usage.ts`), and a
3824
+ * stream may carry that report AND a canonical usage event for the same turn. Crediting both
3825
+ * counts one turn twice, so this ledger holds the precedence rule: a canonical usage event WINS,
3826
+ * and a harness-native report is credited only for a turn in which no canonical usage arrived.
3827
+ *
3828
+ * The harness-native report is held until the turn ends, because it can arrive before the
3829
+ * canonical answer is known — codex emits `turn.completed` ahead of the terminal transport
3830
+ * events. Call {@link SandboxUsageLedger.observe} for every event of a turn, then
3831
+ * {@link SandboxUsageLedger.settleTurn} once at the turn boundary; settling also resets the
3832
+ * ledger for the next turn, so one ledger serves a whole multi-turn session.
3833
+ *
3834
+ * The ledger never throws on a receipt it cannot read: `observe` returns a receipt with
3835
+ * `tokensKnown: false` and `tokensUnknownReason`, so one policy serves every consumer.
3836
+ */
3837
+ interface SandboxUsageLedger {
3838
+ /** Account one event. Returns the canonical usage receipt to credit now, if the event is one. */
3839
+ observe(event: SandboxEvent, agentRunName: string): (RuntimeStreamEvent & {
3840
+ type: 'llm_call';
3841
+ }) | undefined;
3842
+ /** End the turn. Returns the held harness-native receipt when no canonical usage arrived. */
3843
+ settleTurn(agentRunName: string): (RuntimeStreamEvent & {
3844
+ type: 'llm_call';
3845
+ }) | undefined;
3846
+ }
3847
+ /** A {@link SandboxUsageLedger} for one worker. Pass the worker's harness to decode with that
3848
+ * harness's adapter; omit it to try every registered adapter. */
3849
+ declare function createSandboxUsageLedger(harness?: HarnessType): SandboxUsageLedger;
3850
+ /**
3851
+ * Sum the token usage + USD cost of a sandbox turn's events — the one honest way to meter an
3852
+ * `openSandboxRun` cell. Folds a {@link SandboxUsageLedger} over the stream, so it reads usage off
3853
+ * EVERY backend event shape — the canonical events plus a harness that reports usage only in its
3854
+ * own event — and a `runProfileMatrix` dispatch can report it to `ctx.cost`:
3855
+ *
3856
+ * receipt: (turn) => {
3857
+ * const u = sumSandboxUsage(turn.events)
3858
+ * return { model, inputTokens: u.input, outputTokens: u.output,
3859
+ * ...(u.tokensKnown === false ? { usageUnknown: true } : {}),
3860
+ * ...(u.usdKnown !== false && u.costUsd > 0 ? { actualCostUsd: u.costUsd } : {}),
3861
+ * ...(u.usdKnown === false ? { costUnknown: true } : {}),
3862
+ * ...(u.estimatedCostUsd !== undefined ? { estimatedCostUsd: u.estimatedCostUsd } : {}) }
3863
+ * }
3864
+ *
3865
+ * Without this a cell reads `{tokens:0, cost:0}` and the backend-integrity guard correctly aborts the
3866
+ * matrix as a stub. `agentRunName` is the fallback model label for cost-only events (default `'agent'`).
3867
+ *
3868
+ * Pure by contract, like the ledger it folds: it never throws. A harness receipt the ledger cannot
3869
+ * read leaves the result at `tokensKnown: false` with `tokensUnknownReason` carrying the decode
3870
+ * message — an unreadable receipt is a different fact from a turn that reported no usage, and a
3871
+ * post-hoc reader that threw would lose the whole failed turn it exists to report.
3872
+ */
3873
+ declare function sumSandboxUsage(events: readonly SandboxEvent[], agentRunName?: string): {
3874
+ input: number;
3875
+ output: number;
3876
+ costUsd: number;
3877
+ tokensKnown?: false;
3878
+ usdKnown?: false;
3879
+ estimatedCostUsd?: number;
3880
+ tokensUnknownReason?: string;
3881
+ };
3882
+ /**
3883
+ * Cross-event state for {@link mapSandboxToolEvent}. Sandbox backends emit a
3884
+ * tool invocation as MANY `message.part.updated` frames on the same call id
3885
+ * (pending → running → completed), so faithful projection needs per-call
3886
+ * status memory: one `tool_call` on first sighting, at most one `tool_result`
3887
+ * on the terminal transition, nothing on intermediate re-frames. Create one
3888
+ * state per turn via {@link createSandboxToolPartState}.
3889
+ *
3890
+ * @experimental
3891
+ */
3892
+ interface SandboxToolPartState {
3893
+ /** Last seen status per tool call id. A terminal status is sticky — later
3894
+ * frames on a settled call project to nothing. */
3895
+ statusByCall: Map<string, string>;
3896
+ /** Sequence for synthesized call ids when an event carries none. */
3897
+ seq: number;
3898
+ }
3899
+ /**
3900
+ * Fresh per-turn {@link SandboxToolPartState} for {@link mapSandboxToolEvent} — an
3901
+ * empty call-status map so each turn projects tool frames independently.
3902
+ *
3903
+ * @experimental
3904
+ */
3905
+ declare function createSandboxToolPartState(): SandboxToolPartState;
3906
+ /**
3907
+ * Project one `SandboxEvent` onto the `tool_call` / `tool_result` variants of
3908
+ * `RuntimeStreamEvent` — the tool-part projection `mapSandboxEvent`
3909
+ * deliberately does NOT perform. Opt-in and additive: `mapSandboxEvent`'s
3910
+ * default vocabulary (text/reasoning deltas + `llm_call`) is unchanged;
3911
+ * consumers that need the tool surface (chat UIs rendering tool activity)
3912
+ * compose this projector alongside it — `streamAgentTurn` does exactly that
3913
+ * under its `preserveToolParts` option.
3914
+ *
3915
+ * Handled shapes (observed on the opencode / claude-code sandbox backends):
3916
+ * - `message.part.updated` with `part.type === 'tool'` — stateful: a
3917
+ * `tool_call` on the call id's first frame (args from `state.input` or
3918
+ * `state.metadata.input`), a `tool_result` when the status transitions to
3919
+ * `completed` (result from `state.output` / `metadata.output`) or to a
3920
+ * terminal failure (result is `{ error, status, output? }` — the error
3921
+ * surfaced in-band, never dropped).
3922
+ * - bare `tool*` event types (`tool.call`, `tool_result`, …) — stateless:
3923
+ * `*result*` types project to `tool_result`, the rest to `tool_call`.
3924
+ *
3925
+ * Returns `[]` for every non-tool event.
3926
+ *
3927
+ * @experimental
3928
+ */
3929
+ declare function mapSandboxToolEvent(event: SandboxEvent, state: SandboxToolPartState): (RuntimeStreamEvent & {
3930
+ type: 'tool_call' | 'tool_result';
3931
+ })[];
3932
+ /**
3933
+ * Project one `SandboxEvent` onto the `RuntimeStreamEvent` chat-UX vocabulary,
3934
+ * for runtimes that bridge a sandbox `streamPrompt` into the
3935
+ * `AgentRuntime.act` streaming contract. Returns `undefined` for events that
3936
+ * have no faithful projection — the raw stream is preserved separately for the
3937
+ * `OutputAdapter`, so an unmapped event never loses data.
3938
+ *
3939
+ * Mapped (the task-optional incremental variants — no synthesized task
3940
+ * lifecycle, no guessed tool-part shapes):
3941
+ * - `message.part.updated` text part → `text_delta`
3942
+ * - `message.part.updated` reasoning/thinking part → `reasoning_delta`
3943
+ * - cost-bearing events → `llm_call` (shared with the ledger extractor)
3944
+ *
3945
+ * Tool parts are deliberately NOT mapped here (unchanged default) — compose
3946
+ * {@link mapSandboxToolEvent} alongside when a consumer needs them.
3947
+ *
3948
+ * The opencode backend emits incremental text as
3949
+ * `{ type: 'message.part.updated', data: { part: { type, text }, delta } }`;
3950
+ * `delta` is the increment, `part.text` the running accumulation.
3951
+ */
3952
+ declare function mapSandboxEvent(event: SandboxEvent, opts?: {
3953
+ agentRunName?: string;
3954
+ }): RuntimeStreamEvent | undefined;
3955
+ /**
3956
+ * Project one `SandboxEvent` onto Runtime's executor progress vocabulary: incremental text and
3957
+ * reasoning, tool calls and results, and an interaction request. It composes the existing
3958
+ * projections ({@link mapSandboxEvent}, {@link mapSandboxToolEvent}, and the canonical Agent
3959
+ * Interface decode) so every sandbox-shaped executor publishes live output through one reader.
3960
+ * Usage-bearing events project to nothing here — accounting stays on the `tokens`/`cost`
3961
+ * channels.
3962
+ *
3963
+ * Pass one {@link SandboxToolPartState} per turn so a multi-frame tool call yields one call and
3964
+ * at most one result.
3965
+ *
3966
+ * @experimental
3967
+ */
3968
+ declare function sandboxProgressEvents(event: SandboxEvent, state: SandboxToolPartState): ExecutorProgressEvent[];
3969
+ //#endregion
3970
+ //#region src/runtime/sandbox-executor-output.d.ts
3971
+ /**
3972
+ * What a settled turn produced, as an explicit marker.
3973
+ *
3974
+ * `text` carries the byte length of the answer, `empty` says a text-bearing terminal event was
3975
+ * observed and carried nothing, and `absent` says no text-bearing event was observed at all.
3976
+ * The three are distinct on purpose: an empty settle blob used to be indistinguishable from lost
3977
+ * output, so a reader could not tell a box that produced nothing from one whose answer never
3978
+ * arrived.
3979
+ */
3980
+ type SandboxOutputMarker = {
3981
+ readonly kind: 'text';
3982
+ readonly bytes: number;
3983
+ } | {
3984
+ readonly kind: 'empty';
3985
+ } | {
3986
+ readonly kind: 'absent';
3987
+ };
3988
+ /** Parsed output of one Sandbox executor turn. */
3989
+ interface SandboxLeafOut {
3990
+ events: SandboxEvent[];
3991
+ /** The observed answer. `undefined` when no text-bearing event was observed — never `''`. */
3992
+ content: string | undefined;
3993
+ /** Explicit account of what the turn produced. */
3994
+ output: SandboxOutputMarker;
3995
+ /**
3996
+ * Provider and model the platform reported serving this turn, when it reported one. Absent means
3997
+ * the platform said nothing; it is never filled from the request, because a request is not a
3998
+ * receipt.
3999
+ */
4000
+ servedBackend?: SandboxServedBackend;
4001
+ toolCalls?: ExecutorToolCall[];
4002
+ outcome?: AgentRunOutcome;
4003
+ }
4004
+ //#endregion
4005
+ //#region src/runtime/supervise/inbox.d.ts
4006
+ /** A message from the run's AUTHORITY — the parent driver. These two kinds carry instruction. */
4007
+ interface AuthorityInboxMessage {
4008
+ readonly kind: 'steer' | 'answer';
4009
+ readonly text: string;
4010
+ /** Forceful messages abort the in-flight turn; queued ones wait for the boundary flush. */
4011
+ readonly interrupt: boolean;
4012
+ /** Present for an `answer` — the question id it resolves. */
4013
+ readonly questionId?: string;
4014
+ }
4015
+ /** A message from a SIBLING worker. Information, never instruction — the parent stays the only
4016
+ * authority over this worker's task. */
4017
+ interface PeerInboxMessage {
4018
+ readonly kind: 'mail';
4019
+ readonly text: string;
4020
+ /** Always false. Peer mail is queued by construction; see this file's header. */
4021
+ readonly interrupt: false;
4022
+ readonly envelope: PeerMailEnvelope;
4023
+ }
4024
+ type InboxMessage = AuthorityInboxMessage | PeerInboxMessage;
4025
+ interface Inbox {
4026
+ /** The `Executor.deliver` implementation. Returns false when the raw message is malformed and
4027
+ * therefore was not queued; callers must not acknowledge a message this inbox discarded. */
4028
+ deliver(msg: unknown): boolean;
4029
+ /** Remove and return all pending messages (the flush). */
4030
+ drain(): InboxMessage[];
4031
+ pending(): number;
4032
+ /** Pending messages from the run's AUTHORITY only. This is what the pre-settle fence counts:
4033
+ * a worker may not finish while a steer or answer it never read is queued, but peer mail must
4034
+ * never be able to hold a finished worker open. */
4035
+ pendingAuthority(): number;
4036
+ /** Open a fresh per-turn interrupt signal; a later forceful `deliver` aborts it. The loop links
4037
+ * this into the signal it passes to its inference call, then re-plans when it fires. */
4038
+ freshInterrupt(): AbortSignal;
4039
+ /** Render drained messages as ONE operator turn to fold into the worker's conversation. */
4040
+ fold(messages: ReadonlyArray<InboxMessage>): string;
4041
+ }
4042
+ /** Create the worker-side inbox for the down-leg: the driver's `steer_agent` / `answer_question` messages and a sibling's peer mail queue here, and the worker's loop drains them at step boundaries and before settle. */
4043
+ declare function createInbox(): Inbox;
4044
+ //#endregion
4045
+ //#region src/runtime/supervise/sandbox-session.d.ts
4046
+ /** Ceiling on continuation turns. Turn 0 is the task; every later turn is a folded steer, so
4047
+ * this bounds how many times a supervisor may redirect ONE worker before it must respawn. */
4048
+ declare const DEFAULT_SANDBOX_STEERING_MAX_TURNS = 24;
4049
+ /** Opt-in configuration for the steerable sandbox worker (`SandboxSeam.steering`). Absent, the
4050
+ * sandbox executor keeps its historical single-shot `runAgentRounds` composition verbatim. */
4051
+ interface SandboxSteeringOptions {
4052
+ /** Max turns for one worker (turn 0 + folded steers). Default {@link DEFAULT_SANDBOX_STEERING_MAX_TURNS}. */
4053
+ readonly maxTurns?: number;
4054
+ /** How many recent tool/turn notes `progress()` reports. Default 12. */
4055
+ readonly activityWindow?: number;
4056
+ /** Per-turn wall-clock ceiling; the turn's stream is aborted when it elapses. */
4057
+ readonly turnTimeoutMs?: number;
4058
+ }
4059
+ /** What the steerable session exposes to its executor: the usage stream plus the live reads. */
4060
+ interface SteerableSandboxSession {
4061
+ /** Drive the worker to settlement. `signal` is the spawn-scoped abort handed to `execute`. */
4062
+ stream(task: unknown, signal: AbortSignal): AsyncIterable<UsageEvent>;
4063
+ progress(): ExecutorProgress;
4064
+ /** Ask the box to stop the running execution on this exact session and report what it answered. */
4065
+ cancel(request: ExecutorCancellationRequest): Promise<ExecutorCancellation>;
4066
+ traceSource(): TraceSource;
4067
+ artifact(): {
4068
+ outRef: string;
4069
+ out: unknown;
4070
+ verdict?: DefaultVerdict;
4071
+ spent: Spend;
4072
+ } | undefined;
4073
+ teardown(): Promise<void>;
4074
+ }
4075
+ interface SteerableSandboxArgs {
4076
+ readonly controller: AbortController;
4077
+ readonly profile: AgentProfile;
4078
+ readonly harness: BackendType;
4079
+ readonly sandboxClient: SandboxClient;
4080
+ readonly inbox: Inbox;
4081
+ readonly taskToPrompt: (task: unknown) => string;
4082
+ readonly options?: SandboxSteeringOptions;
4083
+ readonly loopCtx?: Partial<Omit<ExecCtx, 'sandboxClient' | 'signal'>>;
4084
+ /**
4085
+ * Inherited `TRACE_ID` / `PARENT_SPAN_ID` for the box, merged into `CreateSandboxOptions.env` so
4086
+ * the remote worker's own spans join the supervisor's trace under the spawning node's span.
4087
+ * Absent when the run records no spans — the create options are then untouched.
4088
+ */
4089
+ readonly traceEnv?: Record<string, string>;
4090
+ readonly contentRef: (prefix: string, value: unknown) => string;
4091
+ readonly now?: () => number;
4092
+ }
4093
+ /** One steerable sandbox worker. The returned session is inert until `stream()` is drained. */
4094
+ declare function createSteerableSandboxSession(args: SteerableSandboxArgs): SteerableSandboxSession;
4095
+ //#endregion
4096
+ //#region src/runtime/supervise/runtime.d.ts
4097
+ /**
4098
+ * Router/inline transport seam. The profile owns model, prompt, and generation behavior.
4099
+ */
4100
+ interface RouterSeam {
4101
+ routerBaseUrl: string;
4102
+ routerKey: string;
4103
+ /** Injectable transport for offline/local execution; still passes through Runtime metering. */
4104
+ complete?: RouterConfig['complete'];
4105
+ /** When present, return one turn's requested tool calls without executing them. */
4106
+ tools?: ReadonlyArray<ToolSpec>;
4107
+ }
4108
+ /**
4109
+ * Sandbox executor seam. The `sandboxClient` the composed `runAgentRounds` creates
4110
+ * boxes through, plus the optional trace/run/lineage wiring forwarded into the
4111
+ * loop. `lineage` is opaque here (PR #150's `RunAgentRoundsOptions.lineage`): forwarded
4112
+ * forward-compatibly, never inspected — this executor does NOT reinvent
4113
+ * checkpoint/fork.
4114
+ */
4115
+ interface SandboxSeam {
4116
+ sandboxClient: SandboxClient;
4117
+ /** Forwarded into the composed `runAgentRounds`'s `ctx` (trace emitter, run handle, etc.). */
4118
+ loopCtx?: Partial<Omit<ExecCtx, 'sandboxClient' | 'signal'>>;
4119
+ /** PR #150 `RunAgentRoundsOptions.lineage` passthrough — opaque; forwarded, not parsed. */
4120
+ lineage?: unknown;
4121
+ /** Hard cap on the composed loop's iterations. The budget pool reserves against
4122
+ * the spawn `Budget.maxIterations`; this is the leaf's own ceiling. Default 1. */
4123
+ maxIterations?: number;
4124
+ /**
4125
+ * OPT-IN executable score for this worker. Forwarded to the composed
4126
+ * `runAgentRounds` as its `validator`, so the kernel calls `validate` while the
4127
+ * iteration's box is still alive: `ValidationCtx.box` is a LIVE `SandboxInstance`
4128
+ * and the check can run commands or read files in the container it is scoring.
4129
+ * Every other supervised hook fires after teardown and can only read the artifact.
4130
+ *
4131
+ * The resulting verdict becomes the winner's verdict, which this executor already
4132
+ * surfaces on its `ExecutorResult`. Absent, nothing changes: the loop runs
4133
+ * unscored and the leaf falls back to its own settle verdict.
4134
+ *
4135
+ * Not representable with `steering` — a steerable session is a multi-turn session
4136
+ * on one box, not a `runAgentRounds` composition, so the pair is rejected instead
4137
+ * of silently dropping the score.
4138
+ */
4139
+ validator?: Validator<SandboxLeafOut>;
4140
+ /**
4141
+ * OPT-IN: run this worker as a multi-turn, STEERABLE session instead of the historical
4142
+ * single-shot `runAgentRounds` composition. Setting it gives the sandbox worker an `Executor.deliver`
4143
+ * inbox (so `Scope.send` / `steer_agent` actually reach it), a live tool-activity trace, and a
4144
+ * `progress()` read — turning the default cloud worker from something a supervisor can only
4145
+ * wait on into something it can watch and correct.
4146
+ *
4147
+ * Absent, nothing changes: the same `runAgentRounds` leaf, no inbox, `steer_agent` still reports
4148
+ * `delivered:false`. Opt-in because a steerable worker holds ONE box across several turns,
4149
+ * which is a different resource profile from a fire-and-forget shot.
4150
+ */
4151
+ steering?: SandboxSteeringOptions;
4152
+ }
4153
+ /**
4154
+ * UNMETERED CLI subprocess seam. `bin` + `args` describe the process to spawn.
4155
+ *
4156
+ * READ THIS BEFORE CHOOSING `backend: 'cli'`. This backend pipes a prompt to a subprocess's stdin
4157
+ * and reads its stdout. It has no usage receipt of any kind, so it reports its spend with
4158
+ * `Spend.tokensKnown: false`: the work is recorded, its `{0,0}` tokens and `$0` are a FLOOR rather
4159
+ * than a measurement, and a ceiling priced from either is a ceiling that cannot fire. The executor
4160
+ * is also `budgetExempt: true`, which is why `driveHarnessFromBackend` refuses it outright rather
4161
+ * than pretending to budget it.
4162
+ *
4163
+ * If you need a metered harness worker, use `backend: 'bridge'` (a cli-bridge session, which
4164
+ * reports the harness's real per-turn tokens and cost) or `backend: 'cli-worktree'` with
4165
+ * `codexReproducible`. Reach for this seam only when the subprocess genuinely is not an inference
4166
+ * agent, or when you have accepted that its cost is invisible.
4167
+ *
4168
+ * `args` is argv for a LOCAL, in-process spawn under this process's own privileges. It is not a
4169
+ * remote channel and nothing forwards it over a wire.
4170
+ */
4171
+ interface CliSeam {
4172
+ bin: string;
4173
+ args?: string[];
4174
+ /** Extra environment for the subprocess (merged over `process.env`). */
4175
+ env?: Record<string, string>;
4176
+ /** Working directory for the subprocess. */
4177
+ cwd?: string;
4178
+ }
4179
+ /**
4180
+ * cli-worktree seam. A supervisor-authored `AgentProfile` driving a local coding-harness CLI
4181
+ * (claude / codex / opencode) on its own git worktree — the leaf `createWorktreeCliExecutor`
4182
+ * named as data. `repoRoot` is transport data; `AgentProfile.harness` selects the CLI.
4183
+ * `taskPrompt` remains an optional direct-call fallback for callers that execute with `undefined`.
4184
+ * The authored
4185
+ * `profile.prompt.systemPrompt` + `profile.model.default` reach the harness via the §1.5
4186
+ * `harnessInvocation` mapper. Everything else mirrors `WorktreeCliExecutorOptions`.
4187
+ */
4188
+ interface CliWorktreeSeam {
4189
+ repoRoot: string;
4190
+ taskPrompt?: string;
4191
+ runId?: string;
4192
+ baseRef?: string;
4193
+ harnessTimeoutMs?: number;
4194
+ /** Isolated, network-off Codex execution with terminal JSONL usage capture. */
4195
+ codexReproducible?: boolean;
4196
+ /** Absolute host paths denied to reproducible Codex. */
4197
+ codexReadDeniedPaths?: ReadonlyArray<string>;
4198
+ testCmd?: string;
4199
+ typecheckCmd?: string;
4200
+ checkTimeoutMs?: number;
4201
+ checkOutputCap?: number;
4202
+ budgetExempt?: boolean;
4203
+ /** Live cli-bridge transport inside the worktree. When set, the worktree leaf accepts
4204
+ * `deliver()` messages and resumes the same bridge session in this worktree cwd. */
4205
+ bridge?: CliWorktreeBridgeSeam;
4206
+ /** Test seam — forwarded to worktree helpers. */
4207
+ runGit?: GitRunner;
4208
+ /** Test seam — forwarded to verification checks. */
4209
+ runCommand?: WorktreeCheckRunner;
4210
+ }
4211
+ /**
4212
+ * cli-in-place seam. A supervisor-authored `AgentProfile` driving a local coding-harness CLI
4213
+ * (claude-code / codex / opencode / pi) on a workspace the CALLER supplies — the leaf
4214
+ * `createInPlaceCliExecutor` named as data. `workspacePath` is transport data;
4215
+ * `AgentProfile.harness` selects the CLI, and the authored `profile.prompt.systemPrompt` +
4216
+ * `profile.model.default` reach the harness via the §1.5 `harnessInvocation` mapper.
4217
+ *
4218
+ * READ THIS BEFORE CHOOSING BETWEEN THIS AND `cli-worktree`. They differ in ONE thing, and it is
4219
+ * the thing that decides which one a caller wants:
4220
+ *
4221
+ * - `cli-worktree` cuts a git worktree of its OWN off `repoRoot`, runs the harness there,
4222
+ * returns the captured patch, and removes the worktree at teardown. The directory it was
4223
+ * given is never edited. That is correct for a fanout of N candidate authors that must not
4224
+ * clobber each other, and for a caller whose deliverable IS the patch.
4225
+ * - `cli-in-place` runs the harness in `workspacePath` itself. The edits stay in that directory
4226
+ * after the call, so the NEXT call sees them. That is what a caller needs when the workspace
4227
+ * has to persist between calls — a multi-shot author resuming on top of its own edits
4228
+ * (`agenticGenerator`), or a candidate directory the caller commits itself.
4229
+ *
4230
+ * Because the workspace persists, so would the profile inputs this path materializes into it. They
4231
+ * are removed before the call returns, so the directory a caller inspects afterwards holds the
4232
+ * harness's own edits and nothing else, and a `git status` over it answers "did the author change
4233
+ * anything" rather than "did Runtime write a settings file".
4234
+ *
4235
+ * There is no reproducible-Codex mode here: that mode stages an executable and a write probe INTO
4236
+ * its working directory, which a caller-owned workspace is not the place for. Use `cli-worktree`
4237
+ * with `codexReproducible` when the isolated, metered Codex run is what you want.
4238
+ */
4239
+ interface CliInPlaceSeam {
4240
+ /** Absolute path to the EXISTING directory the harness edits. Runtime never creates, cleans, or
4241
+ * removes it. */
4242
+ workspacePath: string;
4243
+ taskPrompt?: string;
4244
+ harnessTimeoutMs?: number;
4245
+ /** Test seam — inject the harness runner so unit tests script a `LocalHarnessResult`. */
4246
+ runHarness?: typeof runLocalHarness;
4247
+ }
4248
+ interface CliWorktreeBridgeSeam {
4249
+ bridgeUrl: string;
4250
+ bridgeBearer: string;
4251
+ /** Caller-owned deadline for each bridge turn. Runtime enforces it locally and sends the
4252
+ * same value in `execution.timeoutMs` so cli-bridge cannot substitute its own cutoff. */
4253
+ timeoutMs?: number;
4254
+ /** Stable cli-bridge session id. Defaults to `bridge-worktree-${runId}`. */
4255
+ sessionId?: string;
4256
+ /** Transport reconnects allowed after the first POST. Default 3; set 0 to disable. */
4257
+ maxReconnects?: number;
4258
+ }
4259
+ /**
4260
+ * cli-bridge seam. A local OpenAI-compatible bridge that fronts harness CLIs
4261
+ * (claude-code / opencode / kimi / pi) behind one HTTP surface. The spawned
4262
+ * `AgentProfile` is the sole harness/provider/model and behavioral authority and
4263
+ * is forwarded verbatim per request; this seam carries transport data only.
4264
+ *
4265
+ * The executor opens a resumable cli-bridge session. `sessionId` identifies the
4266
+ * harness conversation across turns; each turn also receives its own durable run id.
4267
+ * A dropped HTTP reader reattaches to that exact run and explicit cancel is the only
4268
+ * operation allowed to stop it. Omit `sessionId` and the executor mints one per spawn.
4269
+ *
4270
+ * ── HOW TO CONTROL WHAT THE HARNESS LOADS (there is no argv field, by design) ──
4271
+ *
4272
+ * A worker often needs the harness started in a KNOWN state — no ambient extensions, skills,
4273
+ * context files, or prompt templates — because ambient state is how a paired experiment silently
4274
+ * loses its pairing: an installed extension that persists memory across runs carries arm A's state
4275
+ * into arm B, and nothing reports it.
4276
+ *
4277
+ * That is what the spawned `AgentProfile` is FOR. `agent_profile`
4278
+ * rides every request verbatim, and cli-bridge maps it onto each harness's own native controls:
4279
+ *
4280
+ * - Materializing any profile at all already starts the harness isolated from ambient
4281
+ * workspace state — for pi that is `--no-context-files --no-skills --no-prompt-templates`,
4282
+ * applied to every request that carries an `agent_profile`.
4283
+ * - `AgentProfile.extensions.<harness>` is the named, per-harness control channel. An explicit
4284
+ * `extensions: { pi: { load: [] } }` disables ambient extension discovery outright
4285
+ * (pi's `--no-extensions`); listing package names loads exactly those and nothing else.
4286
+ * - `permissions` / `tools` / `mcp` map onto the harness's native tool and server controls.
4287
+ *
4288
+ * A caller therefore does NOT need to hand-roll an `Executor` to isolate a harness run, and the
4289
+ * profile expressing it stays portable: the same declaration means the same thing on a different
4290
+ * harness, whereas an argv string means nothing anywhere else.
4291
+ *
4292
+ * WHY NOT A GENERAL ARGV PASSTHROUGH. `bridgeUrl` addresses a process-spawning server. Forwarding
4293
+ * an arbitrary argv array to it would let any caller holding a bearer token choose the flags of a
4294
+ * process on the bridge host — which for real harness CLIs includes flags that load code from a
4295
+ * path, read a file into the prompt, redirect the working directory, or turn off the isolation the
4296
+ * bridge applies. cli-bridge deliberately confines workers (a filesystem jail and deny-by-default
4297
+ * network egress), and every one of those confinements is expressed as spawn configuration, so an
4298
+ * argv channel is a channel for unwinding them. It would also break this executor's own contract:
4299
+ * the durable-run replay protocol, session pinning, and streaming mode are all argv the bridge
4300
+ * owns, and a caller-supplied duplicate silently wins or corrupts the parse. The structured profile
4301
+ * channel is validated, per-harness, portable, and refuses controls it does not understand — keep
4302
+ * new harness capability there.
4303
+ */
4304
+ interface BridgeSeam {
4305
+ bridgeUrl: string;
4306
+ bridgeBearer: string;
4307
+ /**
4308
+ * Optional request-scoped model credential.
4309
+ *
4310
+ * The key name is portable configuration. The provider is a live service and is intentionally
4311
+ * not serialised. Runtime resolves both values immediately before every bridge POST and sends
4312
+ * them only to a loopback bridge through private request headers.
4313
+ */
4314
+ modelCredential?: BridgeModelCredential;
4315
+ /** Optional working directory forwarded to cli-bridge and persisted with the session. */
4316
+ cwd?: string;
4317
+ /**
4318
+ * The harness's OWN on-disk session store, read as a spend receipt.
4319
+ *
4320
+ * cli-bridge forwards no token usage for a codex worker, so a turn whose provider counters exist
4321
+ * only in codex's rollout meters `{0, 0}` with `tokensKnown: false`. Measured on one live seat
4322
+ * (discovery#80): 9 of 9 `metered` events read zero while 27,320,482 codex tokens sat in the same
4323
+ * run directory, 1,453,948 of them belonging to harness-native children the journal never saw.
4324
+ *
4325
+ * Naming the store here turns those rows into evidence. The executor tails it once per turn and
4326
+ * credits the DELTA, so each turn is charged once, and it reports the counters with
4327
+ * `provenance: 'harness-store'` so a reader can tell a disk receipt from a stream receipt.
4328
+ *
4329
+ * The path must be the run's OWN isolated store. An ambient host store credits this run with
4330
+ * another run's files, and `workspaceRoot` is the structural guard against it.
4331
+ */
4332
+ harnessStore?: BridgeHarnessStore;
4333
+ /** Caller-owned deadline for each bridge turn. Runtime enforces it locally and sends the
4334
+ * same value in `execution.timeoutMs` so the bridge-owned process follows the same policy. */
4335
+ timeoutMs?: number;
4336
+ /** Stable, caller-owned cli-bridge session id for harness-side resume. Defaults
4337
+ * to a freshly minted per-spawn id so each worker is its own resumable session. */
4338
+ sessionId?: string;
4339
+ /** Transport reconnects allowed after the first POST. Default 3; set 0 to disable. */
4340
+ maxReconnects?: number;
4341
+ /** Newest-last activity window `progress()` reports. Default 12. */
4342
+ activityWindow?: number;
4343
+ }
4344
+ /**
4345
+ * A harness's own session store on the bridge host, named so the runtime may read it.
4346
+ *
4347
+ * Only `codex` has a reader today. Any other harness is REFUSED rather than read with codex's
4348
+ * decoder: a different harness's file decoded as a codex rollout would either drop counters it does
4349
+ * not name or credit a number that is about the wrong wire shape.
4350
+ */
4351
+ interface BridgeHarnessStore extends CodexRolloutStoreRef {
4352
+ /** The harness family that wrote the store. */
4353
+ readonly harness: HarnessType;
4354
+ }
4355
+ /** A live, request-scoped model credential reference for a local cli-bridge. */
4356
+ interface BridgeModelCredential {
4357
+ /** Provider key name for the scoped model token. */
4358
+ key: string;
4359
+ /** Provider key name for the exact scoped HTTPS model gateway URL. */
4360
+ baseUrlKey: string;
4361
+ /** Live credential service. Runtime retains this reference through reusable captures. */
4362
+ provider: KeyProvider;
4363
+ }
4364
+ /**
4365
+ * Generic environment provider executor config. External packages implement
4366
+ * `AgentEnvironmentProvider`; this built-in wrapper lets `createExecutor` consume them as backend
4367
+ * data while preserving the existing usage channel. Runtime depends on no provider package, so a
4368
+ * Tangle provider and a hand-written one compose identically. Worked wiring:
4369
+ * `examples/provider-executor/`.
4370
+ *
4371
+ * Everything a create needs travels on `CreateAgentEnvironmentInput` through
4372
+ * {@link ProviderExecutorOptions.defaults}; everything one turn needs travels on
4373
+ * {@link ProviderExecutorOptions.promptOptions}. Wrapping the provider's own client to reach a
4374
+ * field is what this seam exists to replace: the wrapper is invisible to Runtime, so its options
4375
+ * are absent from every record the run produces.
4376
+ *
4377
+ * READINESS IS THE PROVIDER'S CONTRACT. `provider.create` resolves with an environment that can
4378
+ * take a turn, so this seam streams straight into it and adds no readiness wait of its own. The
4379
+ * sandbox seam's `acquireSandbox` exists because a raw `SandboxClient.create` returns before the
4380
+ * box is ready; a second poll here would hide a provider that does not honor the contract, and
4381
+ * that provider is an upstream defect to report rather than a race to paper over.
4382
+ */
4383
+ interface ProviderSeam extends ProviderExecutorOptions {
4384
+ provider: AgentEnvironmentProvider$1 | string;
4385
+ registry?: AgentEnvironmentProviderRegistry;
4386
+ /**
4387
+ * Compose the provider through the existing steerable sandbox session.
4388
+ * The exact profile must name its harness, and the provider must expose live
4389
+ * continuation plus session controls. The provider still owns environment
4390
+ * creation and session semantics.
4391
+ */
4392
+ steering?: SandboxSteeringOptions;
4393
+ }
4394
+ /**
4395
+ * Router seam WITH tool use — the tool-using router backend. Same direct
4396
+ * OpenAI-compatible endpoint as `RouterSeam`, but each turn passes `tools`; when
4397
+ * the model emits tool_calls they run via `executeToolCall` ON THIS HOST and the
4398
+ * results fold back as `tool` messages, repeating until the model answers without
4399
+ * a tool or `maxTurns` is hit. A real agentic loop, OFF-BOX — no sandbox, so it
4400
+ * is unaffected by a box's egress allowlist. One turn = one completion = the
4401
+ * equal-compute unit. `executeToolCall` receives the task so per-task tool
4402
+ * surfaces (e.g. a gym keyed by task) can dispatch correctly.
4403
+ */
4404
+ interface RouterToolsSeam {
4405
+ routerBaseUrl: string;
4406
+ routerKey: string;
4407
+ complete?: RouterConfig['complete'];
4408
+ tools: ReadonlyArray<ToolSpec>;
4409
+ executeToolCall: (name: string, args: Record<string, unknown>, task: unknown) => Promise<string>;
4410
+ /** Exact conversation to continue. Runtime validates its system message against the profile. */
4411
+ initialMessages?: ReadonlyArray<Readonly<Record<string, unknown>>>;
4412
+ /** Observe the detached final conversation for session persistence. */
4413
+ onMessages?: (messages: ReadonlyArray<Readonly<Record<string, unknown>>>) => void | Promise<void>;
4414
+ /** Online observer of each tool step — the seam a `DetectorMonitor` taps to watch the live pipe
4415
+ * (raise a `finding` when the worker loops/errors). Called after every tool call resolves, with
4416
+ * real per-call wall-clock (`startedAt`/`endedAt`/`durationMs`) so a push `TraceSource` can carry
4417
+ * non-zero span durations onto the unified timeline. */
4418
+ onToolStep?: (step: {
4419
+ toolName: string;
4420
+ args: Record<string, unknown>;
4421
+ status: 'ok' | 'error';
4422
+ startedAt?: number;
4423
+ endedAt?: number;
4424
+ durationMs?: number;
4425
+ }) => void;
4426
+ }
4427
+ /**
4428
+ * The leaf `createWorktreeCliExecutor` as a backend-as-data factory: a supervisor-authored
4429
+ * `AgentProfile` driving claude / codex / opencode on its own worktree. `budgetExempt` like
4430
+ * the other CLI leaves; the authored systemPrompt + model reach the harness via §1.5.
4431
+ */
4432
+ declare const cliWorktreeExecutor: ExecutorFactory<unknown>;
4433
+ /**
4434
+ * The leaf `createInPlaceCliExecutor` as a backend-as-data factory: a supervisor-authored
4435
+ * `AgentProfile` driving a local coding CLI in the workspace the caller supplied, so its edits are
4436
+ * still there for the next spawn. `budgetExempt` like the other CLI leaves; the authored
4437
+ * systemPrompt + model reach the harness via §1.5.
4438
+ */
4439
+ declare const cliInPlaceExecutor: ExecutorFactory<unknown>;
4440
+ /**
4441
+ * Config for {@link createExecutor}: the backend is DATA — the cost dial a profile,
4442
+ * an experiment config, or a replay journal can name — not an import choice. Each
4443
+ * variant carries its backend's seam.
4444
+ */
4445
+ type ExecutorConfig = ({
4446
+ backend: 'router';
4447
+ } & RouterSeam) | ({
4448
+ backend: 'router-tools';
4449
+ } & RouterToolsSeam) | ({
4450
+ backend: 'bridge';
4451
+ } & BridgeSeam) | ({
4452
+ backend: 'cli';
4453
+ } & CliSeam) | ({
4454
+ backend: 'cli-worktree';
4455
+ } & CliWorktreeSeam) | ({
4456
+ backend: 'cli-in-place';
4457
+ } & CliInPlaceSeam) | ({
4458
+ backend: 'provider';
4459
+ } & ProviderSeam) | ({
4460
+ backend: 'sandbox';
4461
+ } & SandboxSeam);
4462
+ /**
4463
+ * The single built-in executor factory. Picks a leaf backend by data (`config.backend`),
4464
+ * injects the matching seam, and delegates to that backend's built-in implementation.
4465
+ * The `Executor` port stays OPEN: bring-your-own agents implement `Executor` directly, while Scope
4466
+ * or `createExecutorRegistry` still parses and seals their exact profile before use. Use this instead of a
4467
+ * per-vendor adapter or a closed `inline|sandbox|cli` switch — those bypass the
4468
+ * `UsageEvent` reporting channel.
4469
+ */
4470
+ declare function createExecutor(config: ExecutorConfig): ExecutorFactory<unknown>;
4471
+ /**
4472
+ * The open resolver/registry. Pre-registers the three built-ins under their
4473
+ * runtime tags (`'router'`, `'sandbox'`, `'cli'`) and accepts `register(name,
4474
+ * factory)` for any additional runtime. A BYO `AgentSpec.executor` has highest routing precedence
4475
+ * after the same exact-profile intake validation. Registration + BYO remain open extension points.
4476
+ *
4477
+ * `resolve` precedence (frozen in `ExecutorRegistry`): a BYO `spec.executorFactory` →
4478
+ * `spec.executor` → `harness === null` → the `'router'` factory; else a registered factory for the
4479
+ * harness-derived runtime (`'sandbox'` for any `BackendType`); else fail loud.
4480
+ */
4481
+ declare function createExecutorRegistry(): ExecutorRegistry;
4482
+ //#endregion
4483
+ //#region src/runtime/stream-agent-turn.d.ts
4484
+ /**
4485
+ * The execution substrate one turn runs on — a closed discriminated union over
4486
+ * the three stream surfaces the runtime already owns.
4487
+ *
4488
+ * @stable
4489
+ */
4490
+ type AgentTurnBackend = {
4491
+ /** A Runtime-owned executor factory materialized from this exact canonical profile. */
4492
+ kind: 'executor';
4493
+ factory: ExecutorFactory<unknown>;
4494
+ /** Exact canonical identity materialized by the executor. */
4495
+ profile: AgentProfile;
4496
+ /** Model label stamped on cost-only `llm_call` events. Default `'agent'`. */
4497
+ agentRunName?: string;
4498
+ };
4499
+ /** @stable */
4500
+ interface StreamAgentTurnOptions {
4501
+ /** Caller-initiated cancellation. Terminates the stream with `final.status: 'aborted'`. */
4502
+ signal?: AbortSignal;
4503
+ /**
4504
+ * Wall-clock deadline for the whole turn in ms. An expired deadline aborts
4505
+ * the backend and terminates the stream with `final.status: 'failed'`
4506
+ * (a blown deadline is a turn failure, not a caller cancellation).
4507
+ */
4508
+ timeoutMs?: number;
4509
+ /** Stable logical paid-call id, forwarded as the provider idempotency key and retained in evidence. */
4510
+ callId?: string;
4511
+ /** Caller trace tag retained in evidence and forwarded when the transport supports it. */
4512
+ correlationId?: string;
4513
+ /**
4514
+ * Opt-in tool-part projection for box and executor backends: sandbox tool
4515
+ * parts additionally surface in-stream as
4516
+ * `tool_call` / `tool_result` events (`mapSandboxToolEvent`), so a consumer
4517
+ * rendering tool activity needs no bespoke sandbox-event parser. Default
4518
+ * off — the stream vocabulary existing consumers see is unchanged. No-op
4519
+ * for the `chat` kind (its backend emits `RuntimeStreamEvent`s directly,
4520
+ * tool events included when the backend produces them).
4521
+ */
4522
+ preserveToolParts?: boolean;
4523
+ /**
4524
+ * Raw-event tap for box-kind backends: called (and awaited) with every
4525
+ * unmapped `SandboxEvent` BEFORE it is projected, so a consumer can read
4526
+ * parts the chat-UX projection drops (part ids, step markers, custom
4527
+ * backend events) without forking the mapper. Purely observational — it
4528
+ * cannot alter the mapped stream. Never called for the `chat` kind, which
4529
+ * has no sandbox events.
4530
+ */
4531
+ onRawEvent?: (event: SandboxEvent) => void | Promise<void>;
4532
+ }
4533
+ /**
4534
+ * Metered usage of one turn, summed over every cost-bearing event the backend
4535
+ * emitted. `input`/`output` are token counts and are accompanied by
4536
+ * `tokensKnown: false` when the backend did not report them. `costUsd`/`model`
4537
+ * are present only when the backend actually reported them.
4538
+ *
4539
+ * @stable
4540
+ */
4541
+ interface AgentTurnUsage {
4542
+ input: number;
4543
+ output: number;
4544
+ /** Present when a real turn ran but the provider did not report token usage. */
4545
+ tokensKnown?: false;
4546
+ costUsd?: number;
4547
+ /** Present when Runtime could not prove the full dollar amount. */
4548
+ usdKnown?: false;
4549
+ /** Separately-labelled local/catalog estimate; never billed spend. */
4550
+ estimatedCostUsd?: number;
4551
+ /** Provider-reported prompt-cache fields; absent fields remain unknown. */
4552
+ promptCache?: Readonly<Record<string, number | string>>;
4553
+ /** Provider-reported reasoning-token subset of output, when available. */
4554
+ reasoningTokens?: number;
4555
+ model?: string;
4556
+ }
4557
+ /**
4558
+ * A drained turn: the terminal summary plus every event the stream yielded.
4559
+ * `status`/`error` mirror the terminal `final` event so a failed or aborted
4560
+ * turn stays inspectable without re-scanning `events`.
4561
+ *
4562
+ * @stable
4563
+ */
4564
+ interface CollectedAgentTurn {
4565
+ finalText: string;
4566
+ /** Exact terminal artifact output from a Runtime-owned executor. */
4567
+ output?: unknown;
4568
+ usage: AgentTurnUsage;
4569
+ /** Exact underlying transport calls when the Runtime-owned executor reports them. */
4570
+ transportAttempts?: number;
4571
+ toolCalls: Array<{
4572
+ id?: string;
4573
+ name: string;
4574
+ arguments: string;
4575
+ }>;
4576
+ events: RuntimeStreamEvent[];
4577
+ status: AgentTaskStatus;
4578
+ error?: BackendErrorDetail;
4579
+ /** Public Sandbox outcome, when the turn ran through a Sandbox stream or executor. */
4580
+ sandboxOutcome?: AgentRunOutcome;
4581
+ }
4582
+ /**
4583
+ * Run ONE agent turn on any backend kind and stream its events. Yields the
4584
+ * `RuntimeStreamEvent` vocabulary incrementally and always ends with a `final`
4585
+ * event carrying the turn's text and usage (`metadata.tokenUsage`,
4586
+ * `metadata.costUsd?`, `metadata.model?`) — on success, failure, abort, and
4587
+ * timeout alike. The generator never throws; failures surface in-band as
4588
+ * `backend_error` + `final` with a typed `error` detail.
4589
+ *
4590
+ * @stable
4591
+ */
4592
+ declare function streamAgentTurn(backend: AgentTurnBackend, input: AgentTurnInput, opts?: StreamAgentTurnOptions): AsyncGenerator<RuntimeStreamEvent>;
4593
+ /**
4594
+ * Drain a `streamAgentTurn` stream (or any `RuntimeStreamEvent` stream that
4595
+ * honors its terminal contract) into the turn summary plus the full event
4596
+ * list. Fail-loud: throws when the stream ends without a terminal `final`
4597
+ * event — a stream that violates the contract must not read as an empty turn.
4598
+ *
4599
+ * @stable
4600
+ */
4601
+ declare function collectAgentTurn(stream: AsyncIterable<RuntimeStreamEvent>): Promise<CollectedAgentTurn>;
4602
+ //#endregion
4603
+ export { resolveSecretEnv as $, RetainedRunEnvironmentAdmission as $a, sanitizeAgentRuntimeEvent as $i, defaultRedactor as $n, SupervisorOpts as $r, WorktreeHarnessResult as $t, Inbox as A, RecoverRetainedInteractiveRunOptions as Aa, OtelExporter as Ai, AUTHORITY_MARKERS as An, ProviderModelAttemptEvidence as Ar, providerAsExecutor as At, createSandboxToolPartState as B, RecoverRetainedRunIntentOptions as Ba, loopEventToOtelSpan as Bi, PeerMailSendInput as Bn, Scope as Br, CodexRolloutStoreReader as Bt, createExecutorRegistry as C, DEFAULT_STALL_AFTER_MS as Ca, EvalRunGeneration as Ci, ToolSpec as Cn, MaterializedModelIdentity as Cr, ProviderExecutorOptions as Ct, SteerableSandboxSession as D, createActivityLog as Da, LoopSpanNode as Di, ToolLoopCompactionOptions as Dn, NodeSnapshot as Dr, SandboxClientProviderOptions as Dt, SteerableSandboxArgs as E, WorkerProgress as Ea, INTELLIGENCE_WIRE_VERSION as Ei, ToolLoopCompaction as En, NodeId as Er, ResourceRequest as Et, SandboxOutputMarker as F, StartRetainedInteractiveRunOptions as Fa, buildRuntimeEventOtelSpans as Fi, PeerMailKind as Fn, RootHandle as Fr, SandboxControlClient as Ft, sandboxEventServedBackend as G, RetainedInteractiveIntentAdmission as Ga, RuntimeStreamEventCollector as Gi, isPeerMailEnvelope as Gn, SpawnPrior as Gr, createCodexRolloutStoreReader as Gt, extractLlmCallEvent as H, RecoverRetainedRunResult as Ha, padTraceId as Hi, PeerMailboxOptions as Hn, SpawnEvent as Hr, CodexRolloutTurn as Ht, SandboxServedBackend as I, NativeContextContinuationExecution as Ia, createOpenInferenceFileExporter as Ii, PeerMailLimits as In, RootMaterialization as Ir, createTangleSandboxExactProcessProvider as It, KeyProvider as J, RetainedRunAdmissionHook as Ja, RuntimeTelemetryOptions as Ji, JsonRpcMessage as Jn, SpendChannel as Jr, HarnessUsage as Jt, sandboxProgressEvents as K, RetainedInteractiveStartedAdmission as Ka, RuntimeStreamEventSink as Ki, peerMailTools as Kn, SpawnRejection as Kr, harnessUsageIsEmpty as Kt, SandboxToolPartState as L, NativeContextContinuationHandle as La, createOtelExporter as Li, PeerMailOutcome as Ln, RootProviderModelEvidence as Lr, CodexForkBoundary as Lt, PeerInboxMessage as M, RetainedInteractiveEnvironmentInput as Ma, RuntimeEventOtelOptions as Mi, PEER_MAIL_WIRE_KEY as Mn, ResultBlobStore as Mr, resolveAgentEnvironmentProvider as Mt, createInbox as N, RetainedInteractiveRunHandle as Na, buildLoopOtelSpans as Ni, PeerMailEnvelope as Nn, ResumedKeyState as Nr, sandboxClientAsProvider as Nt, createSteerableSandboxSession as O, readWorkerProgress as Oa, OtelAttribute as Oi, ToolLoopMessageRecord as On, NodeStatus as Or, WorkspaceRequest as Ot, SandboxLeafOut as P, RetainedInteractiveStartMaterial as Pa, buildLoopSpanNodes as Pi, PeerMailEvent as Pn, ResumedWork as Pr, CreateTangleSandboxExactProcessProviderOptions as Pt, resolveMcpServerLaunch as Q, RetainedRunEffect as Qa, createRuntimeStreamEventCollector as Qi, Redactor as Qn, Supervisor as Qr, WorktreeCommandResult as Qt, SandboxUsageLedger as R, NativeContextContinuationInput as Ra, exportEvalRuns as Ri, PeerMailReadout as Rn, RootSignal as Rr, CodexRolloutIdentity as Rt, createExecutor as S, ActivityNote as Sa, EvalRunEvent as Si, RouterTransportConfig as Sn, MaterializedExecutionIdentity as Sr, ProviderAsSandboxClientOptions as St, SandboxSteeringOptions as T, ScopeProgressInput as Ta, EvalRunsExportResult as Ti, ToolLoopChat as Tn, NodeExecutionIdentity as Tr, ProviderPromptOptions as Tt, mapSandboxEvent as U, RetainedInteractiveAdmission as Ua, toOtelAttributes as Ui, claimsAuthority as Un, SpawnJournal as Ur, CodexStoreDelta as Ut, createSandboxUsageLedger as V, RecoverRetainedRunOptions as Va, padSpanId as Vi, PeerMailbox as Vn, Settled as Vr, CodexRolloutStoreRef as Vt, mapSandboxToolEvent as W, RetainedInteractiveEnvironmentAdmission as Wa, RuntimeEventCollector as Wi, createPeerMailbox as Wn, SpawnOpts as Wr, addHarnessUsage as Wt, envKeyProvider as X, RetainedRunCancellation as Xa, SanitizedKnowledgeRequirement as Xi, McpToolDescriptor as Xn, SteerableRootHandle as Xr, InPlaceHarnessResult as Xt, ResolvedMcpServerLaunch as Y, RetainedRunCancelOptions as Ya, SanitizedKnowledgeReadinessReport as Yi, JsonRpcResponse as Yn, SpendGap as Yr, decodeHarnessUsage as Yt, mcpSecretEnvMetadataKey as Z, RetainedRunDispatchedAdmission as Za, createRuntimeEventCollector as Zi, McpTransport as Zn, SupervisedResult as Zr, WorktreeCheckRunner as Zt, RouterSeam as _, TraceSource as _a, TraceContext as _i, RunLocalHarnessOptions as _n, ExecutorRegistry as _r, CreateAgentEnvironmentInput$1 as _t, StreamAgentTurnOptions as a, WaitProbeRegistry as aa, WaitOpts as ai, RemoveWorktreeOptions as an, RetainedRunStartMaterial as ao, DefaultVerdict as ar, AgentEnvironmentProviderRef as at, cliInPlaceExecutor as b, sandboxSessionTraceSource as ba, readTraceContextFromEnv as bi, parseCodexTokenUsage as bn, ExecutorToolCall as br, ForkRequest as bt, BridgeHarnessStore as c, createWaitProbes as ca, WorkerInteractiveUnavailableReason as ci, createWorktree as cn, StartRetainedRunOptions as co, ExecutorAccounting as cr, AgentEnvironmentStatus as ct, CliInPlaceSeam as d, timerAt as da, WorkerTraceResolver as di, CodexExecutionPolicy as dn, ExecutorContext as dr, AgentSession as dt, sanitizeKnowledgeReadinessReport as ea, TokenUsageProvenance as ei, WorktreeProfileMaterializationReceipt as en, RetainedRunEventOptions as eo, resolveRedactor as er, secretEnvOfMcpServer as et, CliSeam as f, validateWaitSpec as fa, WorkerTraceSeamCarrier as fi, CodexTokenUsage as fn, ExecutorExecutionBinding as fr, AgentSessionRef as ft, ProviderSeam as g, ToolStepInput as ga, workerTraceSeamKey as gi, LocalHarnessResult as gn, ExecutorProgressEvent as gr, CheckpointRequest as gt, ExecutorConfig as h, SessionTraceBox as ha, workerTraceHeaders as hi, LocalHarness as hn, ExecutorNodeContext as hr, CheckpointRef as ht, CollectedAgentTurn as i, WaitProbe as ia, UsageEvent as ii, GitRunner as in, RetainedRunSnapshot as io, Budget as ir, AgentEnvironmentProvider$1 as it, InboxMessage as j, RetainedInteractiveAdmissionHook as ja, OtelSpan as ji, DEFAULT_PEER_MAIL_LIMITS as jn, ProviderModelExecutionEvidence as jr, providerAsSandboxClient as jt, AuthorityInboxMessage as k, ReconnectRetainedInteractiveRunOptions as ka, OtelExportConfig as ki, ToolLoopToolCall as kn, ProfileMaterializationReceipt as kr, createAgentEnvironmentProviderRegistry as kt, BridgeModelCredential as l, isWaitOutcome as la, WorkerTraceEvidence as li, removeWorktree as ln, ExecutorCancellation as lr, AgentEnvironmentSummary as lt, CliWorktreeSeam as m, SessionMessageLike as ma, workerTraceEnv as mi, LOCAL_HARNESSES as mn, ExecutorMaterialization as mr, AgentTurnResult$1 as mt, AgentTurnInput$1 as n, PendingWait as na, UnconfirmedTeardown as ni, DiffOptions as nn, RetainedRunIntentAdmission as no, AgentExecutionRef as nr, AgentEnvironmentCapabilities$1 as nt, collectAgentTurn as o, WaitRejection as oa, WidenGate as oi, WorktreeHandle as on, RetainedRunTurnInput as oo, ExecutionBindingReceipt as or, AgentEnvironmentProviderRegistry as ot, CliWorktreeBridgeSeam as p, waitUntil as pa, readWorkerTraceContext as pi, DEFAULT_LOCAL_HARNESS as pn, ExecutorFactory as pr, AgentSessionStatus$1 as pt, sumSandboxUsage as q, RetainedRunAdmission as qa, RuntimeStreamEventSummary as qi, peerMailVerbNames as qn, Spend as qr, readCodexRolloutSession as qt, AgentTurnUsage as r, WaitOutcome as ra, UnknownMaterializationReason as ri, DiffResult as rn, RetainedRunReplayPoint as ro, AgentSpec as rr, AgentEnvironmentEvent$1 as rt, streamAgentTurn as s, WaitSpec as sa, WorkerInteractiveSession as si, captureWorktreeDiff as sn, StartRetainedRunInEnvironmentOptions as so, Executor as sr, AgentEnvironmentQuery as st, AgentTurnBackend as t, sanitizeRuntimeStreamEvent as ta, TreeView as ti, CreateWorktreeOptions as tn, RetainedRunHandle as to, Agent as tr, AgentEnvironment$1 as tt, BridgeSeam as u, pollFor as ua, WorkerTraceUnavailableReason as ui, CodexExecutionEvidence as un, ExecutorCancellationRequest as ur, AgentProfileRef$1 as ut, RouterToolsSeam as v, createPushTraceSource as va, createPropagatingTraceEmitter as vi, harnessSupportsReasoningEffort as vn, ExecutorResult as vr, ExecRequest as vt, DEFAULT_SANDBOX_STEERING_MAX_TURNS as w, ExecutorProgress as wa, EvalRunsExportConfig as wi, ToolLoopCallContext as wn, NoWinnerError as wr, ProviderLeafOut as wt, cliWorktreeExecutor as x, ActivityLog as xa, traceContextToEnv as xi, runLocalHarness as xn, Handle as xr, PlacementInfo as xt, SandboxSeam as y, decodeToolPart as ya, mergeTraceEnv as yi, localHarnessExecutable as yn, ExecutorTeardownWarning as yr, ExecResult as yt, assertSandboxServedModel as z, ReconnectRetainedRunOptions as za, generateSpanId as zi, PeerMailRefusal as zn, Runtime as zr, CodexRolloutSession as zt };
4604
+ //# sourceMappingURL=stream-agent-turn-CLOQr497.d.ts.map