@mongodb-js/agent-engine-runner-shared 0.11.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (220) hide show
  1. package/CHANGELOG.md +55 -0
  2. package/LICENSE.md +201 -0
  3. package/README.md +29 -0
  4. package/dist/agent_config.d.ts +167 -0
  5. package/dist/agent_config.d.ts.map +1 -0
  6. package/dist/agent_config.js +544 -0
  7. package/dist/call_interrupted.d.ts +12 -0
  8. package/dist/call_interrupted.d.ts.map +1 -0
  9. package/dist/call_interrupted.js +11 -0
  10. package/dist/checkpoint_workspace.d.ts +25 -0
  11. package/dist/checkpoint_workspace.d.ts.map +1 -0
  12. package/dist/checkpoint_workspace.js +44 -0
  13. package/dist/context.d.ts +235 -0
  14. package/dist/context.d.ts.map +1 -0
  15. package/dist/context.js +322 -0
  16. package/dist/db_config.d.ts +28 -0
  17. package/dist/db_config.d.ts.map +1 -0
  18. package/dist/db_config.js +66 -0
  19. package/dist/db_naming.d.ts +54 -0
  20. package/dist/db_naming.d.ts.map +1 -0
  21. package/dist/db_naming.js +94 -0
  22. package/dist/error_reporting.d.ts +67 -0
  23. package/dist/error_reporting.d.ts.map +1 -0
  24. package/dist/error_reporting.js +311 -0
  25. package/dist/generated/workflow/v1/activity_pb.d.ts +342 -0
  26. package/dist/generated/workflow/v1/activity_pb.d.ts.map +1 -0
  27. package/dist/generated/workflow/v1/activity_pb.js +115 -0
  28. package/dist/generated/workflow/v1/common_pb.d.ts +184 -0
  29. package/dist/generated/workflow/v1/common_pb.d.ts.map +1 -0
  30. package/dist/generated/workflow/v1/common_pb.js +86 -0
  31. package/dist/generated/workflow/v1/runtime_pb.d.ts +200 -0
  32. package/dist/generated/workflow/v1/runtime_pb.d.ts.map +1 -0
  33. package/dist/generated/workflow/v1/runtime_pb.js +40 -0
  34. package/dist/generated/workflow/v1/state_pb.d.ts +254 -0
  35. package/dist/generated/workflow/v1/state_pb.d.ts.map +1 -0
  36. package/dist/generated/workflow/v1/state_pb.js +68 -0
  37. package/dist/guardrails_evaluator/core.d.ts +23 -0
  38. package/dist/guardrails_evaluator/core.d.ts.map +1 -0
  39. package/dist/guardrails_evaluator/core.js +122 -0
  40. package/dist/guardrails_evaluator/index.d.ts +10 -0
  41. package/dist/guardrails_evaluator/index.d.ts.map +1 -0
  42. package/dist/guardrails_evaluator/index.js +11 -0
  43. package/dist/guardrails_evaluator/regex.d.ts +20 -0
  44. package/dist/guardrails_evaluator/regex.d.ts.map +1 -0
  45. package/dist/guardrails_evaluator/regex.js +233 -0
  46. package/dist/hooks.d.ts +109 -0
  47. package/dist/hooks.d.ts.map +1 -0
  48. package/dist/hooks.js +216 -0
  49. package/dist/http_path.d.ts +18 -0
  50. package/dist/http_path.d.ts.map +1 -0
  51. package/dist/http_path.js +53 -0
  52. package/dist/index.d.ts +35 -0
  53. package/dist/index.d.ts.map +1 -0
  54. package/dist/index.js +41 -0
  55. package/dist/launcher.d.ts +130 -0
  56. package/dist/launcher.d.ts.map +1 -0
  57. package/dist/launcher.js +325 -0
  58. package/dist/logger.d.ts +96 -0
  59. package/dist/logger.d.ts.map +1 -0
  60. package/dist/logger.js +204 -0
  61. package/dist/mcp_oauth.d.ts +51 -0
  62. package/dist/mcp_oauth.d.ts.map +1 -0
  63. package/dist/mcp_oauth.js +389 -0
  64. package/dist/mcp_oauth_secret.d.ts +21 -0
  65. package/dist/mcp_oauth_secret.d.ts.map +1 -0
  66. package/dist/mcp_oauth_secret.js +122 -0
  67. package/dist/mcp_tools.d.ts +71 -0
  68. package/dist/mcp_tools.d.ts.map +1 -0
  69. package/dist/mcp_tools.js +301 -0
  70. package/dist/memory_appbound.d.ts +42 -0
  71. package/dist/memory_appbound.d.ts.map +1 -0
  72. package/dist/memory_appbound.js +159 -0
  73. package/dist/memory_writer.d.ts +49 -0
  74. package/dist/memory_writer.d.ts.map +1 -0
  75. package/dist/memory_writer.js +171 -0
  76. package/dist/metrics.d.ts +84 -0
  77. package/dist/metrics.d.ts.map +1 -0
  78. package/dist/metrics.js +205 -0
  79. package/dist/models.d.ts +1458 -0
  80. package/dist/models.d.ts.map +1 -0
  81. package/dist/models.js +1726 -0
  82. package/dist/node_logger.d.ts +43 -0
  83. package/dist/node_logger.d.ts.map +1 -0
  84. package/dist/node_logger.js +158 -0
  85. package/dist/owner_callback.d.ts +16 -0
  86. package/dist/owner_callback.d.ts.map +1 -0
  87. package/dist/owner_callback.js +40 -0
  88. package/dist/progress.d.ts +57 -0
  89. package/dist/progress.d.ts.map +1 -0
  90. package/dist/progress.js +140 -0
  91. package/dist/runtime.d.ts +131 -0
  92. package/dist/runtime.d.ts.map +1 -0
  93. package/dist/runtime.js +351 -0
  94. package/dist/secure_llm_proxy.d.ts +115 -0
  95. package/dist/secure_llm_proxy.d.ts.map +1 -0
  96. package/dist/secure_llm_proxy.js +922 -0
  97. package/dist/secure_wrapper.d.ts +332 -0
  98. package/dist/secure_wrapper.d.ts.map +1 -0
  99. package/dist/secure_wrapper.js +1249 -0
  100. package/dist/server/aer.d.ts +61 -0
  101. package/dist/server/aer.d.ts.map +1 -0
  102. package/dist/server/aer.js +1124 -0
  103. package/dist/server/auth.d.ts +56 -0
  104. package/dist/server/auth.d.ts.map +1 -0
  105. package/dist/server/auth.js +132 -0
  106. package/dist/server/base.d.ts +104 -0
  107. package/dist/server/base.d.ts.map +1 -0
  108. package/dist/server/base.js +150 -0
  109. package/dist/server/callInterrupt.d.ts +49 -0
  110. package/dist/server/callInterrupt.d.ts.map +1 -0
  111. package/dist/server/callInterrupt.js +68 -0
  112. package/dist/server/callback_delivery.d.ts +14 -0
  113. package/dist/server/callback_delivery.d.ts.map +1 -0
  114. package/dist/server/callback_delivery.js +141 -0
  115. package/dist/server/chunk_types.d.ts +50 -0
  116. package/dist/server/chunk_types.d.ts.map +1 -0
  117. package/dist/server/chunk_types.js +62 -0
  118. package/dist/server/cors.d.ts +52 -0
  119. package/dist/server/cors.d.ts.map +1 -0
  120. package/dist/server/cors.js +107 -0
  121. package/dist/server/drain.d.ts +169 -0
  122. package/dist/server/drain.d.ts.map +1 -0
  123. package/dist/server/drain.js +455 -0
  124. package/dist/server/function.d.ts +77 -0
  125. package/dist/server/function.d.ts.map +1 -0
  126. package/dist/server/function.js +337 -0
  127. package/dist/server/http_retry.d.ts +37 -0
  128. package/dist/server/http_retry.d.ts.map +1 -0
  129. package/dist/server/http_retry.js +157 -0
  130. package/dist/server/index.d.ts +7 -0
  131. package/dist/server/index.d.ts.map +1 -0
  132. package/dist/server/index.js +5 -0
  133. package/dist/server/metadata.d.ts +50 -0
  134. package/dist/server/metadata.d.ts.map +1 -0
  135. package/dist/server/metadata.js +193 -0
  136. package/dist/server/oe_url.d.ts +36 -0
  137. package/dist/server/oe_url.d.ts.map +1 -0
  138. package/dist/server/oe_url.js +50 -0
  139. package/dist/server/owner_url.d.ts +35 -0
  140. package/dist/server/owner_url.d.ts.map +1 -0
  141. package/dist/server/owner_url.js +146 -0
  142. package/dist/server/query.d.ts +42 -0
  143. package/dist/server/query.d.ts.map +1 -0
  144. package/dist/server/query.js +28 -0
  145. package/dist/server/tool.d.ts +138 -0
  146. package/dist/server/tool.d.ts.map +1 -0
  147. package/dist/server/tool.js +1017 -0
  148. package/dist/span_names.d.ts +21 -0
  149. package/dist/span_names.d.ts.map +1 -0
  150. package/dist/span_names.js +31 -0
  151. package/dist/structured_logging/constants.d.ts +17 -0
  152. package/dist/structured_logging/constants.d.ts.map +1 -0
  153. package/dist/structured_logging/constants.js +71 -0
  154. package/dist/structured_logging/env.d.ts +18 -0
  155. package/dist/structured_logging/env.d.ts.map +1 -0
  156. package/dist/structured_logging/env.js +39 -0
  157. package/dist/structured_logging/install.d.ts +56 -0
  158. package/dist/structured_logging/install.d.ts.map +1 -0
  159. package/dist/structured_logging/install.js +107 -0
  160. package/dist/structured_logging/layout.d.ts +9 -0
  161. package/dist/structured_logging/layout.d.ts.map +1 -0
  162. package/dist/structured_logging/layout.js +144 -0
  163. package/dist/structured_logging/serialize.d.ts +27 -0
  164. package/dist/structured_logging/serialize.d.ts.map +1 -0
  165. package/dist/structured_logging/serialize.js +61 -0
  166. package/dist/structured_logging/stdio_capture.d.ts +59 -0
  167. package/dist/structured_logging/stdio_capture.d.ts.map +1 -0
  168. package/dist/structured_logging/stdio_capture.js +164 -0
  169. package/dist/structured_logging/uncaught.d.ts +14 -0
  170. package/dist/structured_logging/uncaught.d.ts.map +1 -0
  171. package/dist/structured_logging/uncaught.js +58 -0
  172. package/dist/structured_logging.d.ts +48 -0
  173. package/dist/structured_logging.d.ts.map +1 -0
  174. package/dist/structured_logging.js +47 -0
  175. package/dist/tls_client.d.ts +61 -0
  176. package/dist/tls_client.d.ts.map +1 -0
  177. package/dist/tls_client.js +298 -0
  178. package/dist/tool_api_error.d.ts +62 -0
  179. package/dist/tool_api_error.d.ts.map +1 -0
  180. package/dist/tool_api_error.js +399 -0
  181. package/dist/tool_memory_ownership.d.ts +10 -0
  182. package/dist/tool_memory_ownership.d.ts.map +1 -0
  183. package/dist/tool_memory_ownership.js +36 -0
  184. package/dist/toolpod_handlers.d.ts +126 -0
  185. package/dist/toolpod_handlers.d.ts.map +1 -0
  186. package/dist/toolpod_handlers.js +1016 -0
  187. package/dist/tracing/exporters.d.ts +51 -0
  188. package/dist/tracing/exporters.d.ts.map +1 -0
  189. package/dist/tracing/exporters.js +327 -0
  190. package/dist/tracing/index.d.ts +3 -0
  191. package/dist/tracing/index.d.ts.map +1 -0
  192. package/dist/tracing/index.js +2 -0
  193. package/dist/tracing/setup.d.ts +76 -0
  194. package/dist/tracing/setup.d.ts.map +1 -0
  195. package/dist/tracing/setup.js +436 -0
  196. package/dist/utils.d.ts +204 -0
  197. package/dist/utils.d.ts.map +1 -0
  198. package/dist/utils.js +867 -0
  199. package/dist/workflow/activity.d.ts +71 -0
  200. package/dist/workflow/activity.d.ts.map +1 -0
  201. package/dist/workflow/activity.js +357 -0
  202. package/dist/workflow/attempt.d.ts +12 -0
  203. package/dist/workflow/attempt.d.ts.map +1 -0
  204. package/dist/workflow/attempt.js +96 -0
  205. package/dist/workflow/client.d.ts +46 -0
  206. package/dist/workflow/client.d.ts.map +1 -0
  207. package/dist/workflow/client.js +299 -0
  208. package/dist/workflow/context.d.ts +37 -0
  209. package/dist/workflow/context.d.ts.map +1 -0
  210. package/dist/workflow/context.js +350 -0
  211. package/dist/workflow/heartbeat.d.ts +15 -0
  212. package/dist/workflow/heartbeat.d.ts.map +1 -0
  213. package/dist/workflow/heartbeat.js +78 -0
  214. package/dist/workflow/index.d.ts +14 -0
  215. package/dist/workflow/index.d.ts.map +1 -0
  216. package/dist/workflow/index.js +10 -0
  217. package/dist/workflow/memory.d.ts +17 -0
  218. package/dist/workflow/memory.d.ts.map +1 -0
  219. package/dist/workflow/memory.js +184 -0
  220. package/package.json +73 -0
@@ -0,0 +1,1124 @@
1
+ /**
2
+ * Agent Execution Runtime (AER) server — framework-agnostic agent execution.
3
+ *
4
+ * Responsibilities:
5
+ * - Execute agents via the BaseAgent protocol (framework-neutral)
6
+ * - Use SecureToolWrapper for all tool/LLM calls through OE
7
+ * - Handle HITL via StreamEvent suspend/resume pattern
8
+ * - Report completion/suspension to OE via callback
9
+ *
10
+ * Mirrors Python's `AERServer` in `server/aer.py`.
11
+ */
12
+ import { JsonValueSchema, } from "@mongodb-js/agent-engine-sdk";
13
+ import { explicitFeatures, RuntimeAgentConfig } from "../agent_config.js";
14
+ import { getLogger } from "../logger.js";
15
+ import { withMetrics } from "../metrics.js";
16
+ import { AERExecuteResponseSchema, ExecuteRequestSchema, ExecutorCallbackRequestSchema, SuspendPayloadSchema, } from "../models.js";
17
+ import { NodeExecutionLogger } from "../node_logger.js";
18
+ import { DRAIN_ABORT_REASON } from "./drain.js";
19
+ import { ExternalAPICallError, LLMInvocationError, PolicyDeniedException, SecureToolWrapper, ToolCallTimeoutError, } from "../secure_wrapper.js";
20
+ import { closeSessionFinish, getCurrentExecutionId, getCurrentOeOwnerUrl, isSessionFinishRequested, runWithCustomerOrigin, reportOeOwnerUrlFailure, runWithExecutionContext, } from "../context.js";
21
+ import { AttemptHeartbeat, DurableMemoryState, WorkflowClient, attemptStartRequestFromExecute, newDurabilityOwnerId, runWithAttemptContext, validateDurableMemoryIdentity, } from "../workflow/index.js";
22
+ import { closeAllTLSAgents, fetchPlatform, getFetchOptionsWithTLS, } from "../tls_client.js";
23
+ import { filterThinkingTokens, getEnvFloat, getRequestTimeout, logExecutionCallback, logSection, stripThinking, } from "../utils.js";
24
+ import { getQueryPlugin } from "../hooks.js";
25
+ import { noteCheckpointWireWorkspaceId, resolveCheckpointWorkspaceId, } from "../checkpoint_workspace.js";
26
+ import { BaseServer } from "./base.js";
27
+ import { CallbackDelivery } from "./callback_delivery.js";
28
+ import { DONE, ERROR, LLM_CREDENTIAL_REJECTED_ERROR_CODE, LLM_INVOCATION_ERROR_CODE, LLM_INVOCATION_ERROR_SOURCE, POLICY_DENIED_ERROR_CODE, TEXT, TIMEOUT_ERROR_CODE, TOOL_CREDENTIAL_REJECTED_ERROR_CODE, } from "./chunk_types.js";
29
+ import { resolveOeUrl } from "./oe_url.js";
30
+ import { resolveOwnerUrl } from "./owner_url.js";
31
+ import { discardResponseBody, postWithRetries } from "./http_retry.js";
32
+ import { getTracer } from "../tracing/index.js";
33
+ import { AER_BUILD_AGENT, OPENINFERENCE_SPAN_KIND, OpenInferenceSpanKind, } from "../span_names.js";
34
+ const logger = getLogger("agent_engine_runner_shared.server.aer");
35
+ // Whole-turn execution timeout. Matches the platform's stream deadline so a turn
36
+ // is not cut off short of the limit the caller was promised.
37
+ const EXECUTION_TIMEOUT_MS = getEnvFloat("RUNNER_EXECUTION_TIMEOUT", 600.0) * 1000;
38
+ // Per-request OE callback timeout.
39
+ const REQUEST_TIMEOUT_MS = getRequestTimeout() * 1000;
40
+ // Timeout for the capability advertise POST. Mirrors Python's
41
+ // `_advertise_capabilities` `timeout=10.0`.
42
+ const CAPABILITY_ADVERTISE_TIMEOUT_MS = 10_000;
43
+ // Bounded wait for queued Memory writes before asking OE to release a finished
44
+ // session. Mirrors the Python AER contract.
45
+ const SESSION_FINISH_MEMORY_DRAIN_TIMEOUT_MS = 10_000;
46
+ // `language` fallback when agent.yaml omits it — how the TS SDK identifies
47
+ // itself to OE (mirrors agent.yaml's own `language: typescript` convention
48
+ // for scaffolded TS agents; Python's equivalent fallback is "python").
49
+ const DEFAULT_LANGUAGE = "typescript";
50
+ // Mirrors Python's Query(max_length=200); bounds the $in scan if the OE
51
+ // proxy is ever bypassed (trust model in server/query.ts).
52
+ const MAX_QUERY_SESSION_IDS = 200;
53
+ const QUERY_PLUGIN_UNSUPPORTED_DETAIL = "session queries are not supported by this framework";
54
+ // Plugin errors can echo raw persisted state (user content): bound what is
55
+ // logged and keep the 500 response body generic.
56
+ const QUERY_ERROR_LOG_MAX_CHARS = 200;
57
+ function describeQueryError(exc) {
58
+ const name = exc instanceof Error ? exc.name : typeof exc;
59
+ const message = exc instanceof Error ? exc.message : String(exc);
60
+ return `${name}: ${message.slice(0, QUERY_ERROR_LOG_MAX_CHARS)}`;
61
+ }
62
+ /**
63
+ * Structured metadata for an LLM-provider failure after OE approval.
64
+ * `error_code` is persisted on the execution so unary invokes can classify
65
+ * without the live stream chunk; `code` is the same value so stream
66
+ * consumers that read `metadata.code` (the gateway) match
67
+ * `metadata.error_code` (AER). Mirrors Python's _llm_invocation_metadata.
68
+ *
69
+ * When the tool pod stamped a specific classification (e.g. a credential
70
+ * rejection), it replaces the generic code and `source` is omitted: the
71
+ * gateway maps that bare code to client-owned, while `llm` attribution
72
+ * would read as provider flakiness for a failure the customer can fix.
73
+ *
74
+ * Only the pod's credential stamp is honored. The live relay forwards the
75
+ * pod's frames verbatim and LLMInvocationError is agent-raisable, so the
76
+ * value arriving here is workload-authored: an unrecognized code falls back
77
+ * to the generic classification rather than letting the workload rename its
78
+ * failure for the gateway's allowlist and owner attribution.
79
+ */
80
+ function llmInvocationMetadata(errorCode) {
81
+ if (errorCode === LLM_CREDENTIAL_REJECTED_ERROR_CODE) {
82
+ return { error_code: errorCode, code: errorCode };
83
+ }
84
+ return {
85
+ error_code: LLM_INVOCATION_ERROR_CODE,
86
+ code: LLM_INVOCATION_ERROR_CODE,
87
+ source: LLM_INVOCATION_ERROR_SOURCE,
88
+ };
89
+ }
90
+ /**
91
+ * Structured metadata when the run failed on a rejected credential.
92
+ *
93
+ * Two shapes qualify, and only these: a tool call whose classified
94
+ * external-API failure was an auth rejection (the tool's configured
95
+ * credential), and an LLMInvocationError the framework wrapped before it
96
+ * reached the dedicated catch (the pod's stamped code rides along).
97
+ * Frameworks may nest either under `cause`, so the walk covers the chain;
98
+ * any other failure shape returns undefined and the error stays
99
+ * unclassified. Mirrors Python's _credential_rejection_metadata.
100
+ */
101
+ function credentialRejectionMetadata(err) {
102
+ const seen = new Set();
103
+ const stack = [err];
104
+ while (stack.length > 0) {
105
+ const current = stack.pop();
106
+ if (current === null || current === undefined || seen.has(current)) {
107
+ continue;
108
+ }
109
+ seen.add(current);
110
+ // Literal matches the closed classification the pod stamps
111
+ // (tool_api_error's status map) and the OE allowlists.
112
+ if (current instanceof ExternalAPICallError &&
113
+ current.tool_api_error.classification === "AUTH_FAILED") {
114
+ return {
115
+ error_code: TOOL_CREDENTIAL_REJECTED_ERROR_CODE,
116
+ code: TOOL_CREDENTIAL_REJECTED_ERROR_CODE,
117
+ };
118
+ }
119
+ // The exception type is agent-raisable, so only the exact code the pod
120
+ // can have stamped is honored; anything else falls back to generic.
121
+ if (current instanceof LLMInvocationError &&
122
+ current.error_code === LLM_CREDENTIAL_REJECTED_ERROR_CODE) {
123
+ return { error_code: current.error_code, code: current.error_code };
124
+ }
125
+ if (current instanceof Error) {
126
+ stack.push(current.cause);
127
+ }
128
+ }
129
+ return undefined;
130
+ }
131
+ /**
132
+ * Normalize Fastify's query-string parsing to a string list: a repeated
133
+ * `session_ids` param arrives as an array, a single occurrence as a string,
134
+ * and an absent param as undefined.
135
+ */
136
+ function normalizeSessionIds(raw) {
137
+ if (raw === undefined || raw === null)
138
+ return [];
139
+ if (Array.isArray(raw))
140
+ return raw.map(String);
141
+ return [String(raw)];
142
+ }
143
+ function getOwnerCallbackUrls(server) {
144
+ if (!(server.ownerCallbackUrl instanceof Map)) {
145
+ server.ownerCallbackUrl = new Map();
146
+ }
147
+ return server.ownerCallbackUrl;
148
+ }
149
+ function getPendingSessionFinishes(server) {
150
+ if (!(server.pendingSessionFinish instanceof Set)) {
151
+ server.pendingSessionFinish = new Set();
152
+ }
153
+ return server.pendingSessionFinish;
154
+ }
155
+ /**
156
+ * Build a Promise that rejects when `signal` fires (or immediately if it is
157
+ * already aborted). Used to race against `iter.next()` so a timeout can unwind
158
+ * a pending `await` instead of waiting for the next event.
159
+ *
160
+ * The listener is registered once per execution (the returned promise is
161
+ * reused across every iter.next() race), so this does not accumulate listeners.
162
+ */
163
+ function abortPromise(signal) {
164
+ return new Promise((_, reject) => {
165
+ const err = new Error("Agent execution aborted");
166
+ err.name = "AbortError";
167
+ if (signal.aborted) {
168
+ reject(err);
169
+ return;
170
+ }
171
+ signal.addEventListener("abort", () => reject(err), { once: true });
172
+ });
173
+ }
174
+ function buildExecutorCallback(executionId, status, fields = {}) {
175
+ validateCallbackInterrupts(fields.interrupts);
176
+ const callback = ExecutorCallbackRequestSchema.parse({
177
+ execution_id: executionId,
178
+ status,
179
+ suspend_generation: fields.suspend_generation,
180
+ result: fields.result,
181
+ error: fields.error,
182
+ suspend_reason: fields.suspend_reason,
183
+ suspend_context: fields.suspend_context,
184
+ interrupts: fields.interrupts,
185
+ resume_schema: fields.resume_schema,
186
+ metadata: fields.metadata,
187
+ });
188
+ // Serialize before cleanup so validation and delivery use the same detached
189
+ // payload even if agent-owned values are mutated while resources close.
190
+ return JSON.parse(JSON.stringify(callback));
191
+ }
192
+ /**
193
+ * Resolve the continuation id and message for this execution. Mirrors Python's
194
+ * `_resolve_invocation_params` in `server/aer.py`.
195
+ *
196
+ * `sessionId` falls back to `execution_id` when the caller did not supply
197
+ * one. Framework resume state (e.g. a checkpoint id) is *not* resolved here —
198
+ * it is opaque state the adapter reads from `ctx.metadata`.
199
+ */
200
+ function resolveInvocationParams(request) {
201
+ const reqPayload = request.payload ?? {};
202
+ // Mirror Python's `session_id or execution_id`: an empty-string session_id
203
+ // (e.g. a partially-migrated caller's zero value) must fall through rather
204
+ // than collapse unrelated executions into one session.
205
+ const sessionId = request.session_id || request.execution_id;
206
+ const payloadMessage = reqPayload["message"];
207
+ const message = (typeof payloadMessage === "string" ? payloadMessage : undefined) ??
208
+ request.message;
209
+ return { sessionId, message };
210
+ }
211
+ /**
212
+ * Reject a request that carries `message` both top-level and inside `payload`.
213
+ *
214
+ * The AER is the only component that unpacks the opaque payload, so the
215
+ * collision is caught here and raised as a clean 400. Doing it here (rather
216
+ * than in the model schema) keeps the payload out of the error body and the
217
+ * logs. Mirrors Python's `_reject_ambiguous_message` in `server/aer.py`.
218
+ */
219
+ function rejectAmbiguousMessage(request) {
220
+ const payloadMsg = request.payload ? request.payload["message"] : undefined;
221
+ if (request.message && typeof payloadMsg === "string" && payloadMsg) {
222
+ const err = new Error("Ambiguous request: 'message' was provided both at the top level " +
223
+ "and inside 'payload'. Send the message in exactly one place.");
224
+ err.statusCode = 400;
225
+ throw err;
226
+ }
227
+ }
228
+ async function postJson(url, body) {
229
+ const tlsOptions = getFetchOptionsWithTLS(url);
230
+ const response = await fetchPlatform(url, {
231
+ method: "POST",
232
+ headers: { "Content-Type": "application/json" },
233
+ body: JSON.stringify(body),
234
+ signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS),
235
+ ...tlsOptions,
236
+ });
237
+ if (!response.ok) {
238
+ throw new Error(`HTTP ${response.status} from ${url}`);
239
+ }
240
+ }
241
+ function validateCallbackInterrupts(interrupts) {
242
+ for (const interrupt of interrupts ?? []) {
243
+ try {
244
+ if (JsonValueSchema.safeParse(interrupt.value).success)
245
+ continue;
246
+ }
247
+ catch {
248
+ // Recursive values can overflow the recursive schema before returning a result.
249
+ }
250
+ throw new TypeError(`Interrupt "${interrupt.id}" has a non-JSON-serializable value`);
251
+ }
252
+ }
253
+ export class AERServer extends BaseServer {
254
+ chunkSeq = new Map();
255
+ // Per-execution validated OE owner callback URL, keyed by
256
+ // execution_id like `chunkSeq`. Set in `doHandleExecute` (only when the
257
+ // request's owner URL validates against the resolved service URL) and read by
258
+ // `sendStreamChunk` / `reportCallback` so those two endpoints prefer the
259
+ // owning OE replica. Registered and popped inside the same try/finally as
260
+ // `chunkSeq`, so an entry can never leak when execution setup fails early.
261
+ ownerCallbackUrl = new Map();
262
+ callbackDelivery = new CallbackDelivery();
263
+ // Executions whose agent called finishSession() and completed normally;
264
+ // drained by the /execute route once the response has been flushed.
265
+ pendingSessionFinish = new Set();
266
+ durabilityOwnerId = newDurabilityOwnerId();
267
+ // One-shot capability advertise, mirrors Python's `_capabilities_advertised`.
268
+ // Set only after OE accepts the declaration.
269
+ capabilitiesAdvertised = false;
270
+ get modeName() {
271
+ return "aer";
272
+ }
273
+ async onStartup() {
274
+ if (this.runtime.graphBuilder === null) {
275
+ throw new Error("Agent graph builder not initialized. " +
276
+ "Did you pass a graph_builder to register_and_run()?");
277
+ }
278
+ logger.info("AER graph builder ready");
279
+ const oeUrl = (process.env["OE_URL"] ?? "").trim();
280
+ const workspaceId = (process.env["APP_ID"] ?? "").trim();
281
+ if (!oeUrl || !workspaceId) {
282
+ throw new Error("Capability registration requires OE_URL and APP_ID");
283
+ }
284
+ await this.ensureOeRegistrations(oeUrl, workspaceId);
285
+ }
286
+ /**
287
+ * Run pending one-shot OE registrations (capability advertise).
288
+ *
289
+ * Mirrors Python's `_ensure_oe_registrations`. Called from `onStartup` and
290
+ * defensively from `/execute` so agent work cannot run without a persisted
291
+ * declaration even if startup was bypassed.
292
+ */
293
+ async ensureOeRegistrations(oeUrl, workspaceId) {
294
+ if (!this.capabilitiesAdvertised) {
295
+ await this.advertiseCapabilities(oeUrl, workspaceId);
296
+ }
297
+ }
298
+ /** Push the current declaration to OE, failing startup if it is not accepted. */
299
+ async advertiseCapabilities(oeUrl, workspaceId) {
300
+ const orgId = this.runtime.orgId;
301
+ const projectId = this.runtime.projectId;
302
+ if (!orgId || !projectId) {
303
+ throw new Error("Capability registration requires ORG_ID and PROJECT_ID " +
304
+ `for workspace ${workspaceId}`);
305
+ }
306
+ const agentCfg = typeof this.runtime.getAgentConfig === "function"
307
+ ? this.runtime.getAgentConfig()
308
+ : new RuntimeAgentConfig();
309
+ const language = (agentCfg.language ?? "").trim() || DEFAULT_LANGUAGE;
310
+ const framework = (agentCfg.framework ?? "").trim();
311
+ const features = {
312
+ owner_callback_fallback: true,
313
+ ...explicitFeatures(agentCfg.features),
314
+ };
315
+ const payload = {
316
+ workspace_id: workspaceId,
317
+ org_id: orgId,
318
+ project_id: projectId,
319
+ language,
320
+ framework,
321
+ features,
322
+ };
323
+ const url = `${oeUrl}/agent/capabilities`;
324
+ const tlsOptions = getFetchOptionsWithTLS(url);
325
+ const response = await fetchPlatform(url, {
326
+ method: "POST",
327
+ headers: { "Content-Type": "application/json" },
328
+ body: JSON.stringify(payload),
329
+ signal: AbortSignal.timeout(CAPABILITY_ADVERTISE_TIMEOUT_MS),
330
+ redirect: "manual",
331
+ ...tlsOptions,
332
+ });
333
+ // Only the status is inspected; cancel the body so undici can release the
334
+ // connection instead of retaining an unread response.
335
+ await discardResponseBody(response);
336
+ if (response.status !== 200) {
337
+ throw new Error(`Capability registration was rejected for ${workspaceId} ` +
338
+ `with status ${response.status}`);
339
+ }
340
+ logger.info("Agent capabilities advertised for workspace %s " +
341
+ "(language=%s framework=%s durable_workflow=%s)", workspaceId, language, framework || "-", features["durable_workflow"]);
342
+ this.capabilitiesAdvertised = true;
343
+ }
344
+ async onShutdown() {
345
+ await this.callbackDelivery.shutdown();
346
+ await this.runtime.memoryWriter?.shutdown();
347
+ this.chunkSeq.clear();
348
+ getOwnerCallbackUrls(this).clear();
349
+ // Close all cached TLS agents for graceful shutdown (parity with Python AER)
350
+ closeAllTLSAgents();
351
+ }
352
+ getHealthDetails() {
353
+ return {
354
+ graph_builder_ready: this.runtime.graphBuilder !== null,
355
+ tools_registered: Object.keys(this.runtime.tools),
356
+ };
357
+ }
358
+ registerRoutes(app) {
359
+ app.post("/warm-up", async (_request, reply) => {
360
+ // TypeScript graph builders are synchronous and run on Node's main event
361
+ // loop. Calling one here would make this best-effort optimization block
362
+ // health and every other request, so TypeScript retains lazy first-use
363
+ // graph construction while preserving OE's cross-runtime endpoint.
364
+ return reply.code(204).send();
365
+ });
366
+ app.post("/execute", async (request, reply) => {
367
+ const body = ExecuteRequestSchema.parse(request.body);
368
+ // Throws 409 if the execution was drained after OE dispatched.
369
+ this.drainRegistry.beginWork(body.execution_id);
370
+ try {
371
+ const result = await this.handleExecute(body);
372
+ const responseBody = AERExecuteResponseSchema.parse(result);
373
+ if (getPendingSessionFinishes(this).delete(body.execution_id)) {
374
+ // After the response is flushed, never before: the OE blocks on this
375
+ // call and turns a transport error into a 502 that overwrites the
376
+ // finished run's result. Guard on status too - an error thrown or a
377
+ // client disconnect after this point must not still release.
378
+ reply.raw.once("finish", () => {
379
+ if (reply.raw.statusCode < 400) {
380
+ void this.releaseFinishedSession(body);
381
+ }
382
+ });
383
+ }
384
+ return reply.send(responseBody);
385
+ }
386
+ finally {
387
+ this.drainRegistry.endWork(body.execution_id);
388
+ }
389
+ });
390
+ app.get("/tools", async () => ({
391
+ tools: Object.entries(this.runtime.toolDefinitions).map(([name, defn]) => ({
392
+ name,
393
+ ...defn,
394
+ })),
395
+ }));
396
+ // Session query routes — mirror Python's /query/sessions routes in
397
+ // server/aer.py, including status-code behavior.
398
+ app.get("/query/sessions", async (request, reply) => {
399
+ const query = request.query;
400
+ const sessionIds = normalizeSessionIds(query["session_ids"]);
401
+ if (sessionIds.length > MAX_QUERY_SESSION_IDS) {
402
+ return reply.code(422).send({
403
+ detail: `session_ids accepts at most ${MAX_QUERY_SESSION_IDS} items`,
404
+ });
405
+ }
406
+ if (sessionIds.length === 0) {
407
+ return reply.send({ sessions: [] });
408
+ }
409
+ const plugin = getQueryPlugin();
410
+ if (plugin === null) {
411
+ return reply
412
+ .code(501)
413
+ .send({ detail: QUERY_PLUGIN_UNSUPPORTED_DETAIL });
414
+ }
415
+ try {
416
+ return await reply.send(await plugin.getSummariesForSessions(sessionIds));
417
+ }
418
+ catch (exc) {
419
+ logger.error(`session summary query failed: ${describeQueryError(exc)}`);
420
+ return reply.code(500).send({ detail: "Internal Server Error" });
421
+ }
422
+ });
423
+ app.get("/query/sessions/:session_id/messages", async (request, reply) => {
424
+ const plugin = getQueryPlugin();
425
+ if (plugin === null) {
426
+ return reply
427
+ .code(501)
428
+ .send({ detail: QUERY_PLUGIN_UNSUPPORTED_DETAIL });
429
+ }
430
+ try {
431
+ return await reply.send(await plugin.getMessagesForSession(request.params.session_id));
432
+ }
433
+ catch (exc) {
434
+ logger.error(`session messages query failed: ${describeQueryError(exc)}`);
435
+ return reply.code(500).send({ detail: "Internal Server Error" });
436
+ }
437
+ });
438
+ }
439
+ handleExecute = withMetrics("aer_execute", (request) => this.doHandleExecute(request));
440
+ async doHandleExecute(request) {
441
+ logSection();
442
+ logger.debug("AER: Execute %s... (resume=%s)", request.execution_id.slice(0, 8), request.resume);
443
+ // Pin the callback base to the runner's own configured OE before any of
444
+ // it is used. Every outbound call below derives from this field, and the
445
+ // runner treats the responses as authoritative results and policy
446
+ // decisions — so it must not be chosen by the request. Rebinding here
447
+ // rather than at each use site means a call site added later inherits
448
+ // the trusted value automatically.
449
+ request.platform_api_url = resolveOeUrl(request.platform_api_url);
450
+ logger.info("AER: OE callback URL: %s", request.platform_api_url);
451
+ // Validate the request-supplied owner URL against the trusted
452
+ // service URL just resolved. A forged value validates to null and is
453
+ // ignored; a valid one lets stream chunks, the terminal callback, and
454
+ // in-process tool results prefer the owning OE replica.
455
+ const ownerUrl = resolveOwnerUrl(request.platform_api_owner_url, request.platform_api_url);
456
+ const ownerUrlFailure = { failed: false };
457
+ // Fail fast on a message supplied both top-level and inside payload,
458
+ // before any side effects (execution, callbacks). Mirrors Python.
459
+ rejectAmbiguousMessage(request);
460
+ // Defensively enforce readiness registration if the startup lifecycle was
461
+ // bypassed, using the resolved OE URL. Capability identity comes from
462
+ // trusted deployment configuration, not the caller-controlled payload.
463
+ const registrationWorkspaceId = (process.env["APP_ID"] ?? "").trim();
464
+ if (registrationWorkspaceId) {
465
+ await this.ensureOeRegistrations(request.platform_api_url, registrationWorkspaceId);
466
+ }
467
+ // Resolve message/session_id from payload with top-level fallback.
468
+ // Mirrors Python's _resolve_invocation_params.
469
+ const resolved = resolveInvocationParams(request);
470
+ const message = resolved.message;
471
+ const sessionId = resolved.sessionId;
472
+ noteCheckpointWireWorkspaceId(request.workspace_id ?? null);
473
+ const workspaceId = resolveCheckpointWorkspaceId(request.workspace_id ?? null);
474
+ logger.debug("AER: Message: %s", message.length > 50 ? message.slice(0, 50) + "..." : message);
475
+ const wrapper = new SecureToolWrapper(request.platform_api_url, request.execution_id, request.custom_headers, ownerUrl, this.drainRegistry);
476
+ // Execution-wide cancellation. Python relies on asyncio.wait_for to cancel
477
+ // the inner coroutine on timeout; Node has no implicit cancellation, so a
478
+ // timeout (below) — or an execution drain, via the registry attach below —
479
+ // aborts this controller, which is (1) raced against
480
+ // iter.next() in executeViaAgentStream to stop consuming promptly and (2)
481
+ // threaded through the execution context (`signal`) so requestOeApproval and
482
+ // the SSE stream abort their in-flight OE/LLM fetches — together cancelling
483
+ // active network I/O rather than leaving it to run.
484
+ const abortController = new AbortController();
485
+ // Give the drain receiver the same handle the execution timeout uses, so
486
+ // a drain has exactly the timeout's reach. If the drain landed before this
487
+ // point (between admission and here), the attach aborts immediately.
488
+ this.drainRegistry.attachController(request.execution_id, abortController);
489
+ let pendingCallback;
490
+ const stageCallback = (status, fields = {}) => {
491
+ pendingCallback = buildExecutorCallback(request.execution_id, status, {
492
+ ...fields,
493
+ suspend_generation: request.suspend_generation ?? undefined,
494
+ });
495
+ };
496
+ let cleanupError;
497
+ let cleanupPhase;
498
+ let hasCleanupError = false;
499
+ const rememberCleanupError = (phase, error) => {
500
+ if (!hasCleanupError) {
501
+ cleanupError = error;
502
+ cleanupPhase = phase;
503
+ hasCleanupError = true;
504
+ return;
505
+ }
506
+ logger.error({
507
+ err: error,
508
+ execution_id: request.execution_id,
509
+ cleanup_phase: phase,
510
+ }, "Additional AER cleanup failure");
511
+ };
512
+ const execution = runWithExecutionContext({
513
+ executionId: request.execution_id,
514
+ traceId: request.platform_trace_id,
515
+ wrapper,
516
+ oeUrl: request.platform_api_url,
517
+ // Owner-preferring transports (e.g. emit() from an in-process tool)
518
+ // read this from the execution context.
519
+ oeOwnerUrl: ownerUrl,
520
+ ownerUrlFailure,
521
+ userId: request.user_id,
522
+ sessionId: sessionId,
523
+ workspaceId: workspaceId,
524
+ customHeaders: request.custom_headers,
525
+ // Forward the opaque caller payload so agent tools can read it via
526
+ // getCurrentPayload(). Mirrors Python set_execution_context(payload=...).
527
+ payload: request.payload,
528
+ signal: abortController.signal,
529
+ }, async () => {
530
+ const timeoutHandle = setTimeout(() => abortController.abort(), EXECUTION_TIMEOUT_MS);
531
+ const ctx = {
532
+ executionId: request.execution_id,
533
+ sessionId: sessionId,
534
+ userId: request.user_id,
535
+ workspaceId: workspaceId ?? undefined,
536
+ requestHeaders: request.custom_headers,
537
+ // Framework metadata travels on the context, not the caller payload.
538
+ // AER passes it through without interpreting adapter-owned keys.
539
+ resume: request.resume,
540
+ resumeData: (request.resume_data ?? null),
541
+ metadata: (request.metadata ?? null),
542
+ previousExecutionCancelled: request.previous_execution_cancelled,
543
+ signal: abortController.signal,
544
+ };
545
+ let heartbeat = null;
546
+ try {
547
+ // Register the owner callback URL inside the same try whose finally
548
+ // pops it, so a failure between here and the first
549
+ // sendStreamChunk / reportCallback can never strand an entry. Nothing
550
+ // between context setup and this point delivers a chunk or callback.
551
+ if (ownerUrl) {
552
+ getOwnerCallbackUrls(this).set(request.execution_id, ownerUrl);
553
+ }
554
+ // AgentInput.payload is the opaque caller blob, with the resolved
555
+ // message normalized in. Platform plumbing (identity, resume,
556
+ // checkpoint) travels on ctx, not here.
557
+ const agentInput = {
558
+ payload: {
559
+ ...(request.payload ?? {}),
560
+ message,
561
+ },
562
+ };
563
+ if (request.resume) {
564
+ logger.info("AER: Resume execution for %s", request.execution_id.slice(0, 8));
565
+ }
566
+ logger.info("AER /execute: execution_id=%s, session_id=%s", request.execution_id.slice(0, 8), sessionId.slice(0, 8));
567
+ const nodeLogger = new NodeExecutionLogger({
568
+ oeUrl: request.platform_api_url,
569
+ executionId: request.execution_id,
570
+ sessionId,
571
+ userId: request.user_id,
572
+ orgId: request.org_id,
573
+ projectId: request.project_id,
574
+ });
575
+ logger.debug("Created NodeExecutionLogger for session %s", sessionId.slice(0, 8));
576
+ const agent = getTracer("runner-shared.aer").startActiveSpan(AER_BUILD_AGENT, {
577
+ attributes: {
578
+ [OPENINFERENCE_SPAN_KIND]: OpenInferenceSpanKind.CHAIN,
579
+ },
580
+ }, (span) => {
581
+ try {
582
+ return this.runtime.getAgent({ callbacks: [nodeLogger] });
583
+ }
584
+ finally {
585
+ span.end();
586
+ }
587
+ });
588
+ // Materialize the adapter before asking OE for an attempt. Missing
589
+ // identity, an old OE (bare 404), or an ineligible framework stays
590
+ // native; other start failures fail closed before tenant work.
591
+ const durableAttempt = await this.startDurableAttempt(request, sessionId);
592
+ if (durableAttempt !== null && this.runtime.memoryWriter != null) {
593
+ // != null (not !==): memoryWriter is optional, and an absent
594
+ // writer must behave like a disabled one.
595
+ validateDurableMemoryIdentity(durableAttempt.workflowIdentity, request.user_id);
596
+ wrapper.durableMemory = new DurableMemoryState(message || null);
597
+ }
598
+ if (request.resume_from_step !== undefined &&
599
+ request.resume_from_step !== null &&
600
+ durableAttempt === null) {
601
+ // Raise the shared allocator watermark so the next mint continues
602
+ // past the resumed step (not restarting at 1).
603
+ wrapper.observeOperationalStep(request.resume_from_step);
604
+ }
605
+ heartbeat =
606
+ durableAttempt === null
607
+ ? null
608
+ : new AttemptHeartbeat(durableAttempt, new WorkflowClient(request.platform_api_url));
609
+ if (heartbeat !== null)
610
+ await heartbeat.start();
611
+ let executionOutcome;
612
+ try {
613
+ const execute = () => this.executeViaAgentStream(agent, ctx, agentInput, request.platform_api_url, request.execution_id);
614
+ executionOutcome =
615
+ durableAttempt === null
616
+ ? await execute()
617
+ : await runWithAttemptContext(durableAttempt, execute);
618
+ }
619
+ catch (err) {
620
+ // The controller fires for the execution-timeout setTimeout and
621
+ // for a drain. A drain is a cancellation, not a deadline breach —
622
+ // report it truthfully instead of claiming a phantom timeout.
623
+ if (abortController.signal.aborted) {
624
+ if (abortController.signal.reason === DRAIN_ABORT_REASON) {
625
+ throw new Error("Execution stopped: drained after the execution was cancelled", { cause: err });
626
+ }
627
+ const errorMsg = `Execution timed out after ${EXECUTION_TIMEOUT_MS / 1000} seconds`;
628
+ try {
629
+ await this.sendStreamChunk(request.platform_api_url, request.execution_id, ERROR, "", errorMsg, { error_code: TIMEOUT_ERROR_CODE });
630
+ }
631
+ catch (chunkErr) {
632
+ logger.error({ err: chunkErr }, `Failed to deliver ERROR chunk for execution ${request.execution_id} after retries`);
633
+ }
634
+ stageCallback("ERROR", {
635
+ error: errorMsg,
636
+ metadata: { error_code: TIMEOUT_ERROR_CODE },
637
+ });
638
+ const httpError = new Error(errorMsg);
639
+ httpError.statusCode = 504;
640
+ // Terminal ERROR chunk + callback are already sent above. Mark the
641
+ // error so the outer catch re-raises without reporting a second time
642
+ // — mirrors Python's `except HTTPException: raise`.
643
+ httpError.alreadyReported = true;
644
+ throw httpError;
645
+ }
646
+ throw err;
647
+ }
648
+ if (isInterruptResult(executionOutcome)) {
649
+ const hasEnvelope = executionOutcome.interrupts !== undefined &&
650
+ executionOutcome.resume_schema !== undefined;
651
+ let suspendReason;
652
+ let suspendContext;
653
+ if (hasEnvelope) {
654
+ if (executionOutcome.interrupts?.length === 0) {
655
+ const err = new Error("Agent produced an empty interrupt snapshot; cannot suspend.");
656
+ err.statusCode = 500;
657
+ throw err;
658
+ }
659
+ const parsedPayload = SuspendPayloadSchema.safeParse(executionOutcome.interrupts?.[0]?.value);
660
+ if (parsedPayload.success) {
661
+ suspendReason = parsedPayload.data.suspend_reason;
662
+ suspendContext = parsedPayload.data.suspend_context;
663
+ }
664
+ else {
665
+ suspendReason = "agent_interrupt";
666
+ }
667
+ }
668
+ else {
669
+ const payload = SuspendPayloadSchema.parse(executionOutcome.suspend_payload);
670
+ suspendReason = payload.suspend_reason;
671
+ suspendContext = payload.suspend_context;
672
+ }
673
+ logExecutionCallback(request.execution_id, "SUSPENDED", null, null, suspendReason, "AER");
674
+ if (this.runtime.memoryWriter && durableAttempt === null) {
675
+ try {
676
+ this.runtime.memoryWriter.writeTurnAsync({
677
+ message,
678
+ resultMessages: executionOutcome.messages,
679
+ userId: request.user_id,
680
+ sessionId,
681
+ includeUserTurn: !request.resume,
682
+ });
683
+ }
684
+ catch (error) {
685
+ const kind = error instanceof Error ? error.name : typeof error;
686
+ logger.warn(`Failed to queue pre-suspend memory write (${kind})`);
687
+ }
688
+ }
689
+ stageCallback("SUSPENDED", {
690
+ suspend_reason: suspendReason,
691
+ suspend_context: suspendContext,
692
+ interrupts: executionOutcome.interrupts,
693
+ resume_schema: executionOutcome.resume_schema,
694
+ metadata: executionOutcome.metadata,
695
+ });
696
+ return {
697
+ status: "suspended",
698
+ suspend_reason: suspendReason,
699
+ };
700
+ }
701
+ if (this.runtime.memoryWriter &&
702
+ executionOutcome.messages.length > 0 &&
703
+ durableAttempt === null) {
704
+ try {
705
+ this.runtime.memoryWriter.writeTurnAsync({
706
+ message,
707
+ resultMessages: executionOutcome.messages,
708
+ userId: request.user_id,
709
+ sessionId,
710
+ includeUserTurn: !request.resume,
711
+ });
712
+ }
713
+ catch (error) {
714
+ const kind = error instanceof Error ? error.name : typeof error;
715
+ logger.warn(`Failed to queue memory write (${kind})`);
716
+ }
717
+ }
718
+ try {
719
+ await this.sendStreamChunk(request.platform_api_url, request.execution_id, DONE, executionOutcome.content, "", { status: "completed" });
720
+ }
721
+ catch (chunkErr) {
722
+ logger.error({ err: chunkErr }, `Failed to deliver DONE chunk for execution ${request.execution_id} after retries`);
723
+ }
724
+ stageCallback("COMPLETED", {
725
+ result: executionOutcome.content,
726
+ metadata: executionOutcome.metadata,
727
+ });
728
+ if (isSessionFinishRequested()) {
729
+ // Latched here, acted on after the response: the OE blocks on
730
+ // this /execute call and turns a transport error into a 502 that
731
+ // overwrites the finished run's result, so the session's pods
732
+ // must not be killed before the response is delivered.
733
+ getPendingSessionFinishes(this).add(request.execution_id);
734
+ }
735
+ return { status: "completed", result: executionOutcome.content };
736
+ }
737
+ catch (err) {
738
+ const httpError = err;
739
+ // A path that already emitted its terminal ERROR callback (e.g. the
740
+ // execution-timeout branch) re-raises without reporting again, so OE
741
+ // sees exactly one terminal callback. Mirrors Python's
742
+ // `except HTTPException: raise`.
743
+ if (httpError.alreadyReported) {
744
+ throw httpError;
745
+ }
746
+ // A policy denial is a deliberate, terminal outcome. Attach a
747
+ // machine-readable discriminator (metadata.error_code="policy_denied"
748
+ // plus the bare reason) so consumers can render a "blocked by policy"
749
+ // result instead of string-matching the message.
750
+ //
751
+ // The human-readable `error` keeps the full exception text
752
+ // ("Policy denied: <reason>") so consumers that don't branch on
753
+ // metadata see no change vs. the generic catch-all below; the
754
+ // discriminator and bare reason ride in metadata.
755
+ //
756
+ // The discriminator is guaranteed on the live stream ERROR chunk. It
757
+ // is also attached to the terminal callback, but the OE currently
758
+ // drops terminal error metadata (and the status / SSE-fallback paths
759
+ // don't expose it), so non-streaming consumers can't rely on it until
760
+ // the OE follow-up lands — sending it now is forward-compatible.
761
+ // Mirrors runner-shared (py).
762
+ if (err instanceof PolicyDeniedException) {
763
+ // Guardrail identity is included when the denial came from a
764
+ // guardrail policy (plain tool/model/budget denials carry only the
765
+ // reason). Mirrors runner-shared `_policy_denied_metadata`.
766
+ const policyMeta = {
767
+ error_code: POLICY_DENIED_ERROR_CODE,
768
+ reason: err.reason,
769
+ };
770
+ if (err.guardrailMeta != null) {
771
+ policyMeta.guardrail_id = err.guardrailMeta.guardrail_id;
772
+ policyMeta.guardrail_category =
773
+ err.guardrailMeta.guardrail_category;
774
+ }
775
+ logExecutionCallback(request.execution_id, "POLICY_DENIED", null, err.reason, null, "AER");
776
+ try {
777
+ await this.sendStreamChunk(request.platform_api_url, request.execution_id, ERROR, "", err.message, policyMeta);
778
+ }
779
+ catch (chunkErr) {
780
+ logger.error({ err: chunkErr }, `Failed to deliver policy-denied ERROR chunk for execution ${request.execution_id} after retries`);
781
+ }
782
+ stageCallback("ERROR", {
783
+ error: err.message,
784
+ metadata: policyMeta,
785
+ });
786
+ if (!httpError.statusCode)
787
+ httpError.statusCode = 500;
788
+ throw httpError;
789
+ }
790
+ if (err instanceof ToolCallTimeoutError) {
791
+ // A deadline breach, not a crash: carry the same discriminator as
792
+ // the whole-turn timeout so a consumer handles both the same way,
793
+ // and 504 rather than 500 for the same reason.
794
+ const timeoutMeta = {
795
+ error_code: TIMEOUT_ERROR_CODE,
796
+ tool_name: err.toolName,
797
+ elapsed_seconds: err.elapsedSeconds,
798
+ };
799
+ logExecutionCallback(request.execution_id, "ERROR", null, err.message, null, "AER");
800
+ try {
801
+ await this.sendStreamChunk(request.platform_api_url, request.execution_id, ERROR, "", err.message, timeoutMeta);
802
+ }
803
+ catch (chunkErr) {
804
+ logger.error({ err: chunkErr }, `Failed to deliver timeout ERROR chunk for execution ${request.execution_id} after retries`);
805
+ }
806
+ stageCallback("ERROR", {
807
+ error: err.message,
808
+ metadata: timeoutMeta,
809
+ });
810
+ httpError.statusCode = 504;
811
+ throw httpError;
812
+ }
813
+ if (err instanceof LLMInvocationError) {
814
+ // Stamp source=llm only for provider-owned failures. Relay,
815
+ // truncation, and guardrail plumbing share this exception type
816
+ // without that source and must stay uncoded.
817
+ if (err.source !== LLM_INVOCATION_ERROR_SOURCE) {
818
+ const uncodedMsg = err.message;
819
+ logExecutionCallback(request.execution_id, "ERROR", null, uncodedMsg, null, "AER");
820
+ stageCallback("ERROR", {
821
+ error: uncodedMsg,
822
+ });
823
+ if (!httpError.statusCode)
824
+ httpError.statusCode = 500;
825
+ throw httpError;
826
+ }
827
+ const llmMeta = llmInvocationMetadata(err.error_code);
828
+ logExecutionCallback(request.execution_id, "ERROR", null, err.message, null, "AER");
829
+ try {
830
+ await this.sendStreamChunk(request.platform_api_url, request.execution_id, ERROR, "", err.message, llmMeta);
831
+ }
832
+ catch (chunkErr) {
833
+ logger.error({ err: chunkErr }, `Failed to deliver LLM-invocation ERROR chunk for execution ${request.execution_id} after retries`);
834
+ }
835
+ stageCallback("ERROR", {
836
+ error: err.message,
837
+ metadata: llmMeta,
838
+ });
839
+ if (!httpError.statusCode)
840
+ httpError.statusCode = 500;
841
+ throw httpError;
842
+ }
843
+ const errorMsg = err instanceof Error ? err.message : String(err);
844
+ logExecutionCallback(request.execution_id, "ERROR", null, errorMsg, null, "AER");
845
+ const credentialMeta = credentialRejectionMetadata(err);
846
+ stageCallback("ERROR", {
847
+ error: errorMsg,
848
+ ...(credentialMeta ? { metadata: credentialMeta } : {}),
849
+ });
850
+ if (!httpError.statusCode)
851
+ httpError.statusCode = 500;
852
+ throw httpError;
853
+ }
854
+ finally {
855
+ // Close the session-finish latch before the context that holds it
856
+ // is torn down: covers the success, error,
857
+ // policy-denied, and suspend paths alike, so post-turn work (e.g.
858
+ // a setTimeout or floating promise scheduled during the turn)
859
+ // that calls requestSessionFinish() after this point gets
860
+ // "unavailable" instead of a release promise nothing will act on.
861
+ closeSessionFinish();
862
+ clearTimeout(timeoutHandle);
863
+ this.chunkSeq.delete(request.execution_id);
864
+ try {
865
+ await wrapper.close();
866
+ }
867
+ catch (error) {
868
+ rememberCleanupError("wrapper close", error);
869
+ }
870
+ try {
871
+ if (heartbeat !== null)
872
+ await heartbeat.stop();
873
+ }
874
+ catch (error) {
875
+ rememberCleanupError("heartbeat shutdown", error);
876
+ }
877
+ }
878
+ });
879
+ let response;
880
+ let executionError;
881
+ let hasExecutionError = false;
882
+ try {
883
+ response = await execution;
884
+ }
885
+ catch (error) {
886
+ executionError = error;
887
+ hasExecutionError = true;
888
+ }
889
+ try {
890
+ if (pendingCallback !== undefined) {
891
+ const { execution_id, status, ...fields } = pendingCallback;
892
+ await this.reportCallback(request.platform_api_url, execution_id, status, fields, pendingCallback, ownerUrlFailure, ownerUrl);
893
+ }
894
+ }
895
+ finally {
896
+ getOwnerCallbackUrls(this).delete(request.execution_id);
897
+ }
898
+ if (hasExecutionError) {
899
+ if (hasCleanupError) {
900
+ logger.error({
901
+ err: cleanupError,
902
+ execution_id: request.execution_id,
903
+ cleanup_phase: cleanupPhase,
904
+ }, "Additional AER cleanup failure");
905
+ }
906
+ throw executionError;
907
+ }
908
+ if (hasCleanupError) {
909
+ throw cleanupError;
910
+ }
911
+ if (response === undefined) {
912
+ throw new Error("AER execution completed without a response");
913
+ }
914
+ return response;
915
+ }
916
+ /**
917
+ * Ask OE to activate a fenced attempt, or return null for native routing.
918
+ *
919
+ * Null covers incomplete tenant identity, a framework that never registered
920
+ * a workflow adapter, and an old OE without workflow routes (bare 404). Any
921
+ * other failure propagates so the invocation fails closed rather than
922
+ * running unfenced.
923
+ */
924
+ async startDurableAttempt(request, sessionId) {
925
+ const startRequest = attemptStartRequestFromExecute(request, sessionId, this.durabilityOwnerId, this.runtime.appName, this.runtime.appVersion,
926
+ // != null: memoryWriter is optional, and the declaration must agree
927
+ // with the wrapper setup guard on absent vs disabled writers.
928
+ this.runtime.memoryWriter != null);
929
+ if (startRequest === null)
930
+ return null;
931
+ return new WorkflowClient(request.platform_api_url).startAttempt(startRequest);
932
+ }
933
+ async sendStreamChunk(oeUrl, executionId, chunkType, content = "", error = "",
934
+ // Numbers allowed so numeric discriminator fields (elapsed_seconds) stay
935
+ // numeric on the wire rather than being stringified for TS's benefit.
936
+ metadata = {}) {
937
+ const seq = (this.chunkSeq.get(executionId) ?? 0) + 1;
938
+ this.chunkSeq.set(executionId, seq);
939
+ const chunk = {
940
+ execution_id: executionId,
941
+ chunk_type: chunkType,
942
+ content,
943
+ error,
944
+ metadata,
945
+ seq,
946
+ };
947
+ const terminal = chunkType === DONE || chunkType === ERROR;
948
+ const ownerCallbacks = getOwnerCallbackUrls(this);
949
+ const owner = getCurrentExecutionId() === executionId
950
+ ? getCurrentOeOwnerUrl()
951
+ : (ownerCallbacks.get(executionId) ?? null);
952
+ try {
953
+ await postWithRetries(`${oeUrl}/stream/chunk`, chunk, owner ? `${owner}/stream/chunk` : null, owner
954
+ ? () => {
955
+ ownerCallbacks.delete(executionId);
956
+ if (getCurrentExecutionId() === executionId) {
957
+ reportOeOwnerUrlFailure();
958
+ }
959
+ }
960
+ : null);
961
+ }
962
+ catch (e) {
963
+ if (terminal) {
964
+ // Terminal chunks re-raise on exhaustion so the caller can fall
965
+ // through to reportCallback — OE's exec.Status fallback then
966
+ // synthesizes a terminal SSE chunk from exec.Result / exec.Error.
967
+ // Mirrors Python's `_send_stream_chunk`.
968
+ this.chunkSeq.delete(executionId);
969
+ throw e;
970
+ }
971
+ // Non-terminal chunk (text / subagent boundary): the run continues;
972
+ // the chunk is dropped.
973
+ logger.warn(`Failed to deliver ${chunkType} chunk (seq=${seq}) to OE after retries: ${e}`);
974
+ return;
975
+ }
976
+ if (terminal) {
977
+ this.chunkSeq.delete(executionId);
978
+ }
979
+ }
980
+ async executeViaAgentStream(agent, ctx, agentInput, oeUrl, executionId) {
981
+ // Per-stream <think> filter state keyed by (source, tool_call_id).
982
+ // source alone is insufficient: two parallel dispatches of the same-named
983
+ // subagent share a source string but have distinct tool_call_ids, so their
984
+ // thinking buffers must be isolated to prevent token cross-contamination.
985
+ const thinkingState = new Map();
986
+ // Manual iteration so we can race iter.next() against ctx.signal. for-await
987
+ // would block inside iter.next() with no way to unwind on abort. The race
988
+ // unblocks promptly when the signal fires; iter.return() in `finally` lets
989
+ // the underlying generator run any cleanup it has.
990
+ // Only stream creation and iterator steps run customer code; the
991
+ // rest of the loop is platform-owned.
992
+ const iter = runWithCustomerOrigin(() => agent.execute(ctx, agentInput)[Symbol.asyncIterator]());
993
+ const signal = ctx.signal;
994
+ const abortRacePromise = signal ? abortPromise(signal) : null;
995
+ try {
996
+ while (true) {
997
+ const result = abortRacePromise
998
+ ? await runWithCustomerOrigin(() => Promise.race([iter.next(), abortRacePromise]))
999
+ : await runWithCustomerOrigin(() => iter.next());
1000
+ if (result.done)
1001
+ break;
1002
+ const streamEvent = result.value;
1003
+ if (typeof streamEvent.data !== "object" ||
1004
+ streamEvent.data === null ||
1005
+ Array.isArray(streamEvent.data)) {
1006
+ throw new TypeError(`Expected object event data, got ${typeof streamEvent.data}`);
1007
+ }
1008
+ const eventData = streamEvent.data;
1009
+ if (streamEvent.event === "token") {
1010
+ const tokenContent = eventData["content"] ?? "";
1011
+ const source = String(eventData["source"] ?? "");
1012
+ const toolCallId = String(eventData["tool_call_id"] ?? "");
1013
+ if (tokenContent) {
1014
+ const stateKey = `${source}::${toolCallId}`;
1015
+ const [buf, inside] = thinkingState.get(stateKey) ?? ["", false];
1016
+ const [streamable, newBuf, newInside] = filterThinkingTokens(tokenContent, buf, inside);
1017
+ thinkingState.set(stateKey, [newBuf, newInside]);
1018
+ if (streamable) {
1019
+ await this.sendStreamChunk(oeUrl, executionId, TEXT, streamable, "", {
1020
+ source,
1021
+ tool_call_id: toolCallId,
1022
+ });
1023
+ }
1024
+ }
1025
+ }
1026
+ else if (streamEvent.event === "suspend") {
1027
+ const rawMetadata = eventData["metadata"];
1028
+ const metadata = typeof rawMetadata === "object" &&
1029
+ rawMetadata !== null &&
1030
+ !Array.isArray(rawMetadata)
1031
+ ? rawMetadata
1032
+ : {};
1033
+ return {
1034
+ suspend_payload: eventData["suspend_payload"] ?? {},
1035
+ interrupts: eventData["interrupts"],
1036
+ resume_schema: eventData["resume_schema"],
1037
+ metadata,
1038
+ messages: (eventData["messages"] ??
1039
+ []),
1040
+ };
1041
+ }
1042
+ else if (streamEvent.event === "result") {
1043
+ const content = stripThinking(eventData["response"] ?? "");
1044
+ // Cast through unknown: the agent stream sends raw message objects that match
1045
+ // the Message wire shape but TypeScript cannot verify the deeply nested types.
1046
+ const messages = (eventData["messages"] ??
1047
+ []);
1048
+ return {
1049
+ content,
1050
+ messages,
1051
+ metadata: eventData["metadata"],
1052
+ };
1053
+ }
1054
+ }
1055
+ }
1056
+ finally {
1057
+ // Hint to the underlying generator that we are done so it can run its
1058
+ // own cleanup (e.g. close streams). Fire-and-forget: awaiting
1059
+ // `iter.return()` would block on the generator's current pending
1060
+ // `await`, which defeats the abort race we just performed. On a timeout
1061
+ // the execution signal also aborts the in-flight OE/LLM fetch (see the
1062
+ // execution-context `signal`), so that pending `await` rejects promptly
1063
+ // and the generator unwinds rather than running on; the return is then
1064
+ // processed at its next checkpoint, after which it is eligible for GC.
1065
+ Promise.resolve(
1066
+ // Generator cleanup during return() runs customer code — keep it
1067
+ // inside the customer scope for attribution.
1068
+ runWithCustomerOrigin(() => iter.return?.())).catch(() => {
1069
+ /* ignore */
1070
+ });
1071
+ }
1072
+ throw new Error(`Agent stream ended without result or suspend (execution_id=${executionId})`);
1073
+ }
1074
+ async reportCallback(oeUrl, executionId, status, fields = {}, stagedCallback, ownerUrlFailure, ownerUrlOverride) {
1075
+ const callback = stagedCallback ?? buildExecutorCallback(executionId, status, fields);
1076
+ const ownerCallbacks = getOwnerCallbackUrls(this);
1077
+ const owner = ownerUrlFailure
1078
+ ? ownerUrlFailure.failed
1079
+ ? null
1080
+ : (ownerUrlOverride ?? null)
1081
+ : getCurrentExecutionId() === executionId
1082
+ ? getCurrentOeOwnerUrl()
1083
+ : (ownerCallbacks.get(executionId) ?? null);
1084
+ await this.callbackDelivery.send(oeUrl, callback, owner, owner
1085
+ ? () => {
1086
+ ownerCallbacks.delete(executionId);
1087
+ if (ownerUrlFailure)
1088
+ ownerUrlFailure.failed = true;
1089
+ if (getCurrentExecutionId() === executionId) {
1090
+ reportOeOwnerUrlFailure();
1091
+ }
1092
+ }
1093
+ : null);
1094
+ }
1095
+ /**
1096
+ * Ask the OE to free this session's compute now that the turn is over.
1097
+ *
1098
+ * Called from the /execute route only after its response has been flushed,
1099
+ * because the OE will kill this pod with zero grace. Wait for any retained
1100
+ * terminal callback and queued Memory writes first so the finish request
1101
+ * cannot destroy their only in-memory copies. Best-effort: a failure just
1102
+ * leaves the pods to the idle sweep.
1103
+ */
1104
+ async releaseFinishedSession(request) {
1105
+ if (!(await this.callbackDelivery.waitForTerminal(request.execution_id))) {
1106
+ return;
1107
+ }
1108
+ const writer = this.runtime.memoryWriter;
1109
+ if (writer &&
1110
+ !(await writer.drain(SESSION_FINISH_MEMORY_DRAIN_TIMEOUT_MS))) {
1111
+ logger.warn(`Session finish: ${writer.pendingWrites} memory write(s) still pending after 10s; releasing anyway`);
1112
+ }
1113
+ try {
1114
+ const baseUrl = request.platform_api_url.replace(/\/+$/, "");
1115
+ await postJson(`${baseUrl}/executions/${request.execution_id}/finish`, {});
1116
+ }
1117
+ catch (e) {
1118
+ logger.error({ err: e }, "Session finish request failed");
1119
+ }
1120
+ }
1121
+ }
1122
+ function isInterruptResult(outcome) {
1123
+ return "suspend_payload" in outcome;
1124
+ }