@mongodb-js/agent-engine-runner-shared 0.11.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (220) hide show
  1. package/CHANGELOG.md +55 -0
  2. package/LICENSE.md +201 -0
  3. package/README.md +29 -0
  4. package/dist/agent_config.d.ts +167 -0
  5. package/dist/agent_config.d.ts.map +1 -0
  6. package/dist/agent_config.js +544 -0
  7. package/dist/call_interrupted.d.ts +12 -0
  8. package/dist/call_interrupted.d.ts.map +1 -0
  9. package/dist/call_interrupted.js +11 -0
  10. package/dist/checkpoint_workspace.d.ts +25 -0
  11. package/dist/checkpoint_workspace.d.ts.map +1 -0
  12. package/dist/checkpoint_workspace.js +44 -0
  13. package/dist/context.d.ts +235 -0
  14. package/dist/context.d.ts.map +1 -0
  15. package/dist/context.js +322 -0
  16. package/dist/db_config.d.ts +28 -0
  17. package/dist/db_config.d.ts.map +1 -0
  18. package/dist/db_config.js +66 -0
  19. package/dist/db_naming.d.ts +54 -0
  20. package/dist/db_naming.d.ts.map +1 -0
  21. package/dist/db_naming.js +94 -0
  22. package/dist/error_reporting.d.ts +67 -0
  23. package/dist/error_reporting.d.ts.map +1 -0
  24. package/dist/error_reporting.js +311 -0
  25. package/dist/generated/workflow/v1/activity_pb.d.ts +342 -0
  26. package/dist/generated/workflow/v1/activity_pb.d.ts.map +1 -0
  27. package/dist/generated/workflow/v1/activity_pb.js +115 -0
  28. package/dist/generated/workflow/v1/common_pb.d.ts +184 -0
  29. package/dist/generated/workflow/v1/common_pb.d.ts.map +1 -0
  30. package/dist/generated/workflow/v1/common_pb.js +86 -0
  31. package/dist/generated/workflow/v1/runtime_pb.d.ts +200 -0
  32. package/dist/generated/workflow/v1/runtime_pb.d.ts.map +1 -0
  33. package/dist/generated/workflow/v1/runtime_pb.js +40 -0
  34. package/dist/generated/workflow/v1/state_pb.d.ts +254 -0
  35. package/dist/generated/workflow/v1/state_pb.d.ts.map +1 -0
  36. package/dist/generated/workflow/v1/state_pb.js +68 -0
  37. package/dist/guardrails_evaluator/core.d.ts +23 -0
  38. package/dist/guardrails_evaluator/core.d.ts.map +1 -0
  39. package/dist/guardrails_evaluator/core.js +122 -0
  40. package/dist/guardrails_evaluator/index.d.ts +10 -0
  41. package/dist/guardrails_evaluator/index.d.ts.map +1 -0
  42. package/dist/guardrails_evaluator/index.js +11 -0
  43. package/dist/guardrails_evaluator/regex.d.ts +20 -0
  44. package/dist/guardrails_evaluator/regex.d.ts.map +1 -0
  45. package/dist/guardrails_evaluator/regex.js +233 -0
  46. package/dist/hooks.d.ts +109 -0
  47. package/dist/hooks.d.ts.map +1 -0
  48. package/dist/hooks.js +216 -0
  49. package/dist/http_path.d.ts +18 -0
  50. package/dist/http_path.d.ts.map +1 -0
  51. package/dist/http_path.js +53 -0
  52. package/dist/index.d.ts +35 -0
  53. package/dist/index.d.ts.map +1 -0
  54. package/dist/index.js +41 -0
  55. package/dist/launcher.d.ts +130 -0
  56. package/dist/launcher.d.ts.map +1 -0
  57. package/dist/launcher.js +325 -0
  58. package/dist/logger.d.ts +96 -0
  59. package/dist/logger.d.ts.map +1 -0
  60. package/dist/logger.js +204 -0
  61. package/dist/mcp_oauth.d.ts +51 -0
  62. package/dist/mcp_oauth.d.ts.map +1 -0
  63. package/dist/mcp_oauth.js +389 -0
  64. package/dist/mcp_oauth_secret.d.ts +21 -0
  65. package/dist/mcp_oauth_secret.d.ts.map +1 -0
  66. package/dist/mcp_oauth_secret.js +122 -0
  67. package/dist/mcp_tools.d.ts +71 -0
  68. package/dist/mcp_tools.d.ts.map +1 -0
  69. package/dist/mcp_tools.js +301 -0
  70. package/dist/memory_appbound.d.ts +42 -0
  71. package/dist/memory_appbound.d.ts.map +1 -0
  72. package/dist/memory_appbound.js +159 -0
  73. package/dist/memory_writer.d.ts +49 -0
  74. package/dist/memory_writer.d.ts.map +1 -0
  75. package/dist/memory_writer.js +171 -0
  76. package/dist/metrics.d.ts +84 -0
  77. package/dist/metrics.d.ts.map +1 -0
  78. package/dist/metrics.js +205 -0
  79. package/dist/models.d.ts +1458 -0
  80. package/dist/models.d.ts.map +1 -0
  81. package/dist/models.js +1726 -0
  82. package/dist/node_logger.d.ts +43 -0
  83. package/dist/node_logger.d.ts.map +1 -0
  84. package/dist/node_logger.js +158 -0
  85. package/dist/owner_callback.d.ts +16 -0
  86. package/dist/owner_callback.d.ts.map +1 -0
  87. package/dist/owner_callback.js +40 -0
  88. package/dist/progress.d.ts +57 -0
  89. package/dist/progress.d.ts.map +1 -0
  90. package/dist/progress.js +140 -0
  91. package/dist/runtime.d.ts +131 -0
  92. package/dist/runtime.d.ts.map +1 -0
  93. package/dist/runtime.js +351 -0
  94. package/dist/secure_llm_proxy.d.ts +115 -0
  95. package/dist/secure_llm_proxy.d.ts.map +1 -0
  96. package/dist/secure_llm_proxy.js +922 -0
  97. package/dist/secure_wrapper.d.ts +332 -0
  98. package/dist/secure_wrapper.d.ts.map +1 -0
  99. package/dist/secure_wrapper.js +1249 -0
  100. package/dist/server/aer.d.ts +61 -0
  101. package/dist/server/aer.d.ts.map +1 -0
  102. package/dist/server/aer.js +1124 -0
  103. package/dist/server/auth.d.ts +56 -0
  104. package/dist/server/auth.d.ts.map +1 -0
  105. package/dist/server/auth.js +132 -0
  106. package/dist/server/base.d.ts +104 -0
  107. package/dist/server/base.d.ts.map +1 -0
  108. package/dist/server/base.js +150 -0
  109. package/dist/server/callInterrupt.d.ts +49 -0
  110. package/dist/server/callInterrupt.d.ts.map +1 -0
  111. package/dist/server/callInterrupt.js +68 -0
  112. package/dist/server/callback_delivery.d.ts +14 -0
  113. package/dist/server/callback_delivery.d.ts.map +1 -0
  114. package/dist/server/callback_delivery.js +141 -0
  115. package/dist/server/chunk_types.d.ts +50 -0
  116. package/dist/server/chunk_types.d.ts.map +1 -0
  117. package/dist/server/chunk_types.js +62 -0
  118. package/dist/server/cors.d.ts +52 -0
  119. package/dist/server/cors.d.ts.map +1 -0
  120. package/dist/server/cors.js +107 -0
  121. package/dist/server/drain.d.ts +169 -0
  122. package/dist/server/drain.d.ts.map +1 -0
  123. package/dist/server/drain.js +455 -0
  124. package/dist/server/function.d.ts +77 -0
  125. package/dist/server/function.d.ts.map +1 -0
  126. package/dist/server/function.js +337 -0
  127. package/dist/server/http_retry.d.ts +37 -0
  128. package/dist/server/http_retry.d.ts.map +1 -0
  129. package/dist/server/http_retry.js +157 -0
  130. package/dist/server/index.d.ts +7 -0
  131. package/dist/server/index.d.ts.map +1 -0
  132. package/dist/server/index.js +5 -0
  133. package/dist/server/metadata.d.ts +50 -0
  134. package/dist/server/metadata.d.ts.map +1 -0
  135. package/dist/server/metadata.js +193 -0
  136. package/dist/server/oe_url.d.ts +36 -0
  137. package/dist/server/oe_url.d.ts.map +1 -0
  138. package/dist/server/oe_url.js +50 -0
  139. package/dist/server/owner_url.d.ts +35 -0
  140. package/dist/server/owner_url.d.ts.map +1 -0
  141. package/dist/server/owner_url.js +146 -0
  142. package/dist/server/query.d.ts +42 -0
  143. package/dist/server/query.d.ts.map +1 -0
  144. package/dist/server/query.js +28 -0
  145. package/dist/server/tool.d.ts +138 -0
  146. package/dist/server/tool.d.ts.map +1 -0
  147. package/dist/server/tool.js +1017 -0
  148. package/dist/span_names.d.ts +21 -0
  149. package/dist/span_names.d.ts.map +1 -0
  150. package/dist/span_names.js +31 -0
  151. package/dist/structured_logging/constants.d.ts +17 -0
  152. package/dist/structured_logging/constants.d.ts.map +1 -0
  153. package/dist/structured_logging/constants.js +71 -0
  154. package/dist/structured_logging/env.d.ts +18 -0
  155. package/dist/structured_logging/env.d.ts.map +1 -0
  156. package/dist/structured_logging/env.js +39 -0
  157. package/dist/structured_logging/install.d.ts +56 -0
  158. package/dist/structured_logging/install.d.ts.map +1 -0
  159. package/dist/structured_logging/install.js +107 -0
  160. package/dist/structured_logging/layout.d.ts +9 -0
  161. package/dist/structured_logging/layout.d.ts.map +1 -0
  162. package/dist/structured_logging/layout.js +144 -0
  163. package/dist/structured_logging/serialize.d.ts +27 -0
  164. package/dist/structured_logging/serialize.d.ts.map +1 -0
  165. package/dist/structured_logging/serialize.js +61 -0
  166. package/dist/structured_logging/stdio_capture.d.ts +59 -0
  167. package/dist/structured_logging/stdio_capture.d.ts.map +1 -0
  168. package/dist/structured_logging/stdio_capture.js +164 -0
  169. package/dist/structured_logging/uncaught.d.ts +14 -0
  170. package/dist/structured_logging/uncaught.d.ts.map +1 -0
  171. package/dist/structured_logging/uncaught.js +58 -0
  172. package/dist/structured_logging.d.ts +48 -0
  173. package/dist/structured_logging.d.ts.map +1 -0
  174. package/dist/structured_logging.js +47 -0
  175. package/dist/tls_client.d.ts +61 -0
  176. package/dist/tls_client.d.ts.map +1 -0
  177. package/dist/tls_client.js +298 -0
  178. package/dist/tool_api_error.d.ts +62 -0
  179. package/dist/tool_api_error.d.ts.map +1 -0
  180. package/dist/tool_api_error.js +399 -0
  181. package/dist/tool_memory_ownership.d.ts +10 -0
  182. package/dist/tool_memory_ownership.d.ts.map +1 -0
  183. package/dist/tool_memory_ownership.js +36 -0
  184. package/dist/toolpod_handlers.d.ts +126 -0
  185. package/dist/toolpod_handlers.d.ts.map +1 -0
  186. package/dist/toolpod_handlers.js +1016 -0
  187. package/dist/tracing/exporters.d.ts +51 -0
  188. package/dist/tracing/exporters.d.ts.map +1 -0
  189. package/dist/tracing/exporters.js +327 -0
  190. package/dist/tracing/index.d.ts +3 -0
  191. package/dist/tracing/index.d.ts.map +1 -0
  192. package/dist/tracing/index.js +2 -0
  193. package/dist/tracing/setup.d.ts +76 -0
  194. package/dist/tracing/setup.d.ts.map +1 -0
  195. package/dist/tracing/setup.js +436 -0
  196. package/dist/utils.d.ts +204 -0
  197. package/dist/utils.d.ts.map +1 -0
  198. package/dist/utils.js +867 -0
  199. package/dist/workflow/activity.d.ts +71 -0
  200. package/dist/workflow/activity.d.ts.map +1 -0
  201. package/dist/workflow/activity.js +357 -0
  202. package/dist/workflow/attempt.d.ts +12 -0
  203. package/dist/workflow/attempt.d.ts.map +1 -0
  204. package/dist/workflow/attempt.js +96 -0
  205. package/dist/workflow/client.d.ts +46 -0
  206. package/dist/workflow/client.d.ts.map +1 -0
  207. package/dist/workflow/client.js +299 -0
  208. package/dist/workflow/context.d.ts +37 -0
  209. package/dist/workflow/context.d.ts.map +1 -0
  210. package/dist/workflow/context.js +350 -0
  211. package/dist/workflow/heartbeat.d.ts +15 -0
  212. package/dist/workflow/heartbeat.d.ts.map +1 -0
  213. package/dist/workflow/heartbeat.js +78 -0
  214. package/dist/workflow/index.d.ts +14 -0
  215. package/dist/workflow/index.d.ts.map +1 -0
  216. package/dist/workflow/index.js +10 -0
  217. package/dist/workflow/memory.d.ts +17 -0
  218. package/dist/workflow/memory.d.ts.map +1 -0
  219. package/dist/workflow/memory.js +184 -0
  220. package/package.json +73 -0
@@ -0,0 +1,922 @@
1
+ /**
2
+ * SecureLLMProxy — framework-neutral proxy for secure LLM calls through OE.
3
+ *
4
+ * Packages intercepted invoke_llm requests for the Orchestration Engine and
5
+ * unwraps the streamed OE relay back into agent-engine-sdk models. OE owns approval,
6
+ * routing, live SSE relay, and final audit/result recording.
7
+ */
8
+ import { LLMResponse, LLMToolCall, LLMToolSchemaValidator, JsonValueSchema, } from "@mongodb-js/agent-engine-sdk";
9
+ import { createParser } from "eventsource-parser";
10
+ import { getLogger } from "./logger.js";
11
+ import { getCurrentUserId, withExecutionSignal } from "./context.js";
12
+ import { getSuspendHandler } from "./hooks.js";
13
+ import { LLMPodStreamEventSchema, serializeInvokeLLMRequestArguments, } from "./models.js";
14
+ import { fetchPlatform, getFetchOptionsWithTLS } from "./tls_client.js";
15
+ import { LLM_READ_TIMEOUT, OE_DISPATCH_TAKEOVER_RETRY_DELAY_MS, OE_RETRYABLE_MAX_ATTEMPTS, oeStreamRetry, oeStreamRetryDelayMs, } from "./utils.js";
16
+ import { LLMInvocationError, OperationalStepAllocator, PolicyDeniedException, raiseForOeRejection, requestOeApprovalRetryable, } from "./secure_wrapper.js";
17
+ import { ActivityKind, DurableActivityDeniedError, ReplayedActivityFailedError, WorkflowClient, allocateActivityOrdinal, currentAttemptContext, preallocateActivityOrdinals, runStreamingActivity, toolActivityKey, } from "./workflow/index.js";
18
+ const logger = getLogger("agent_engine_runner_shared.secure_llm_proxy");
19
+ function usageToWireDict(usage) {
20
+ const result = {};
21
+ if (usage.inputTokens != null)
22
+ result["input_tokens"] = usage.inputTokens;
23
+ if (usage.outputTokens != null)
24
+ result["output_tokens"] = usage.outputTokens;
25
+ if (usage.totalTokens != null)
26
+ result["total_tokens"] = usage.totalTokens;
27
+ if (usage.promptTokens != null)
28
+ result["prompt_tokens"] = usage.promptTokens;
29
+ if (usage.completionTokens != null)
30
+ result["completion_tokens"] = usage.completionTokens;
31
+ return result;
32
+ }
33
+ // =============================================================================
34
+ // Internal SSE streaming helper
35
+ // =============================================================================
36
+ /**
37
+ * Async-iterate SSE events from a fetch Response.
38
+ *
39
+ * Uses eventsource-parser to handle multi-line data and edge cases. Cancels
40
+ * the reader on exit (consumer break or throw) to release the HTTP connection.
41
+ *
42
+ * Note: AbortSignal.timeout() starts from the moment the signal is created, not
43
+ * from the last received byte. Python's httpx `read=LLM_READ_TIMEOUT` is a
44
+ * per-read idle timeout. The idle timer below resets after each raw chunk from
45
+ * reader.read(), matching httpx's per-recv semantics exactly.
46
+ */
47
+ async function* parseSseStream(response, options = {}) {
48
+ const { onChunk, signal } = options;
49
+ if (!response.body) {
50
+ // 204/205/304 or a polyfill without a body — surface a legible error that
51
+ // the caller converts into an LLMInvocationError, rather than a cryptic
52
+ // "Cannot read properties of null" TypeError from .getReader().
53
+ throw new Error("SSE response has no body");
54
+ }
55
+ const reader = response.body.getReader();
56
+ const decoder = new TextDecoder();
57
+ const pending = [];
58
+ const parser = createParser({
59
+ onEvent: (event) => {
60
+ if (event.data)
61
+ pending.push(event.data);
62
+ },
63
+ });
64
+ // Belt-and-suspenders: when the caller's signal fires, cancel the reader so
65
+ // reader.read() unblocks immediately. In Node.js/undici the fetch signal
66
+ // already propagates to the body, but this covers edge cases.
67
+ const abortHandler = () => {
68
+ reader.cancel().catch(() => { });
69
+ };
70
+ signal?.addEventListener("abort", abortHandler, { once: true });
71
+ try {
72
+ while (true) {
73
+ const { done, value } = await reader.read();
74
+ if (done)
75
+ break;
76
+ onChunk?.(); // reset idle timer on every raw recv — matches httpx read= semantics
77
+ parser.feed(decoder.decode(value, { stream: true }));
78
+ while (pending.length > 0) {
79
+ const chunk = pending.shift();
80
+ if (chunk !== undefined)
81
+ yield chunk;
82
+ }
83
+ }
84
+ }
85
+ finally {
86
+ signal?.removeEventListener("abort", abortHandler);
87
+ try {
88
+ await reader.cancel();
89
+ }
90
+ catch {
91
+ /* ignore cancel errors */
92
+ }
93
+ }
94
+ }
95
+ function isOeStreamTransportDisconnect(exc) {
96
+ // fetch network failures are TypeError. AbortError is idle timeout,
97
+ // execution cancel, or consumer abort — none of those are owner-death.
98
+ if (exc instanceof DOMException && exc.name === "AbortError")
99
+ return false;
100
+ if (exc instanceof Error && exc.name === "AbortError")
101
+ return false;
102
+ return exc instanceof TypeError;
103
+ }
104
+ // =============================================================================
105
+ // SecureLLMProxy
106
+ // =============================================================================
107
+ export class SecureLLMProxy {
108
+ oeUrl;
109
+ executionId;
110
+ modelName;
111
+ llmId;
112
+ boundTools;
113
+ // Forwarded to the tool pod's bind_tools call so a forced tool choice
114
+ // (e.g. withStructuredOutput) survives the OE round-trip.
115
+ boundToolChoice;
116
+ /** Prefer the wrapper's allocator so tool + LLM share one sequence. */
117
+ operationalSteps;
118
+ durableMemory;
119
+ lastDurationMs;
120
+ lastFromCache;
121
+ lastLatestStepNumber;
122
+ lastPodName;
123
+ constructor(args) {
124
+ this.oeUrl = args.oeUrl.replace(/\/$/, "");
125
+ this.executionId = args.executionId;
126
+ this.modelName = args.modelName ?? "unknown";
127
+ this.llmId = args.llmId ?? "__default__";
128
+ this.boundTools = args.boundTools ?? null;
129
+ this.boundToolChoice = args.boundToolChoice ?? null;
130
+ this.operationalSteps =
131
+ args.operationalSteps ?? new OperationalStepAllocator();
132
+ this.durableMemory = args.durableMemory ?? null;
133
+ this.lastDurationMs = 0.0;
134
+ this.lastFromCache = false;
135
+ this.lastLatestStepNumber = null;
136
+ this.lastPodName = null;
137
+ }
138
+ /** Current operational-step watermark for compatibility readers. */
139
+ get stepCounter() {
140
+ return this.operationalSteps.current();
141
+ }
142
+ allocateStep(step) {
143
+ if (step == null) {
144
+ return this.operationalSteps.next();
145
+ }
146
+ this.operationalSteps.observeAtLeast(step);
147
+ return step;
148
+ }
149
+ // ---------------------------------------------------------------------------
150
+ // Public API
151
+ // ---------------------------------------------------------------------------
152
+ /**
153
+ * Invoke LLM by collecting the stream-oriented execution path.
154
+ * Equivalent to Python's `invoke()` which calls `list(self.stream(...))`.
155
+ */
156
+ async invoke(messages, step, stop, options) {
157
+ const chunks = [];
158
+ for await (const chunk of this.stream(messages, step, stop, options)) {
159
+ chunks.push(chunk);
160
+ }
161
+ return SecureLLMProxy.responseFromStreamChunks(chunks);
162
+ }
163
+ /**
164
+ * Stream invoke_llm chunks through OE approval and OE-owned SSE relay.
165
+ *
166
+ * Step auto-increments when not supplied. If OE returns a cached/sync
167
+ * result, a synthetic LLMStreamChunk is yielded instead of opening an
168
+ * SSE connection.
169
+ */
170
+ async *stream(messages, step, stop, options) {
171
+ const resolvedStep = this.allocateStep(step);
172
+ const invokeRequest = this.buildInvokeRequest(messages, stop ?? null, options ?? null, true);
173
+ if (currentAttemptContext() === null) {
174
+ yield* this.streamNative(invokeRequest, resolvedStep);
175
+ return;
176
+ }
177
+ yield* this.streamDurably(invokeRequest, resolvedStep);
178
+ }
179
+ async *streamDurably(invokeRequest, resolvedStep) {
180
+ const durableMemory = this.durableMemory;
181
+ try {
182
+ yield* runStreamingActivity({
183
+ client: new WorkflowClient(this.oeUrl),
184
+ kind: ActivityKind.LLM,
185
+ name: this.modelName,
186
+ activityOrdinal: allocateActivityOrdinal(`llm:${resolvedStep}`),
187
+ semanticInput: serializeInvokeLLMRequestArguments(invokeRequest),
188
+ execute: () => this.streamActivityEffect(invokeRequest, resolvedStep),
189
+ replay: (result) => this.replayActivityResult(result),
190
+ fold: (chunks) => this.completeActivityStream(chunks),
191
+ onActivityResolved: durableMemory === null
192
+ ? undefined
193
+ : (client, context, result) => durableMemory.synchronizeLlm(client, context, result, getCurrentUserId()),
194
+ // The platform-owned operational step is the LLM call's stable
195
+ // call-order identity. Frameworks may consume tool-call chunks before
196
+ // the provider stream has fully closed.
197
+ exclusive: false,
198
+ });
199
+ }
200
+ catch (error) {
201
+ if (error instanceof DurableActivityDeniedError) {
202
+ if (error.cause instanceof PolicyDeniedException)
203
+ throw error.cause;
204
+ throw new PolicyDeniedException(error.message, null, {
205
+ cause: error,
206
+ });
207
+ }
208
+ if (error instanceof ReplayedActivityFailedError) {
209
+ throw new LLMInvocationError(error.message, { cause: error });
210
+ }
211
+ throw error;
212
+ }
213
+ }
214
+ async *streamActivityEffect(invokeRequest, resolvedStep) {
215
+ try {
216
+ yield* this.streamNative(invokeRequest, resolvedStep, "reject");
217
+ }
218
+ catch (error) {
219
+ // The activity kernel records this error class as DENIED, not FAILED.
220
+ if (error instanceof PolicyDeniedException) {
221
+ throw new DurableActivityDeniedError(error.reason, { cause: error });
222
+ }
223
+ throw error;
224
+ }
225
+ }
226
+ async *streamNative(invokeRequest, resolvedStep, interruptMode = "framework") {
227
+ const response = await this.requestOeExecution(invokeRequest, resolvedStep);
228
+ // Guardrail non-proceed responses: require_review suspends for a human;
229
+ // a genuine block returns the substitute *string* in result, which we yield
230
+ // as a clean chunk so the agent gets a response rather than an exception.
231
+ // Any other non-proceed shape is a hard denial.
232
+ if (!response.proceed) {
233
+ if (response.status === "require_review") {
234
+ if (interruptMode === "reject") {
235
+ throw new LLMInvocationError("durable LLM activities support terminal outcomes only; " +
236
+ "the framework adapter must own review interrupts");
237
+ }
238
+ yield* this.handleRequireReview(response);
239
+ return;
240
+ }
241
+ if (typeof response.result === "string") {
242
+ yield { content: response.result };
243
+ return;
244
+ }
245
+ raiseForOeRejection(response.reason, response.guardrail_meta ?? null);
246
+ }
247
+ const result = response.result ?? response.cached_result;
248
+ const status = response.status ??
249
+ (this.lastFromCache && result != null ? "success" : null);
250
+ if (status === "error") {
251
+ throw new LLMInvocationError(response.error ?? "LLM streaming failed", {
252
+ source: "llm",
253
+ ...(response.error_code ? { error_code: response.error_code } : {}),
254
+ });
255
+ }
256
+ if (status === "interrupted") {
257
+ // OE stopped this call on interrupt (inline or replayed from a durable
258
+ // interrupted step): yield the same clean marker as the live SSE path so
259
+ // the agent continues instead of throwing on an unexpected status.
260
+ yield { responseMetadata: { interrupted: true } };
261
+ return;
262
+ }
263
+ if (response.route_to &&
264
+ response.route_to !== "callback" &&
265
+ result == null) {
266
+ yield* this.streamFromOe(response.route_to, resolvedStep);
267
+ return;
268
+ }
269
+ if (status !== "success" && result == null) {
270
+ throw new LLMInvocationError(`Unexpected OE invoke_llm status: ${status}`);
271
+ }
272
+ yield* SecureLLMProxy.chunksFromResponse(SecureLLMProxy.convertResultToResponse(result));
273
+ }
274
+ *replayActivityResult(result) {
275
+ this.lastFromCache = true;
276
+ if (typeof result === "string") {
277
+ yield { content: result };
278
+ return;
279
+ }
280
+ this.preallocateToolCallsFromResult(result);
281
+ if (typeof result !== "object" ||
282
+ result === null ||
283
+ Array.isArray(result)) {
284
+ throw new LLMInvocationError(`Recorded LLM result has unexpected type: ${typeof result}`);
285
+ }
286
+ yield SecureLLMProxy.completeChunkFromResponse(SecureLLMProxy.convertResultToResponse(result));
287
+ }
288
+ completeActivityStream(chunks) {
289
+ const payload = SecureLLMProxy.responsePayloadFromChunks(chunks);
290
+ this.preallocateToolCallsFromResult(payload);
291
+ return payload;
292
+ }
293
+ preallocateToolCallsFromResult(result) {
294
+ if (typeof result !== "object" ||
295
+ result === null ||
296
+ Array.isArray(result)) {
297
+ return;
298
+ }
299
+ const toolCalls = result["tool_calls"];
300
+ if (!Array.isArray(toolCalls))
301
+ return;
302
+ const keys = toolCalls.flatMap((toolCall) => {
303
+ if (typeof toolCall !== "object" ||
304
+ toolCall === null ||
305
+ Array.isArray(toolCall)) {
306
+ return [];
307
+ }
308
+ const id = toolCall["id"];
309
+ return typeof id === "string" && id ? [toolActivityKey(id)] : [];
310
+ });
311
+ if (keys.length > 0)
312
+ preallocateActivityOrdinals(keys);
313
+ }
314
+ static chunksFromResponse(response) {
315
+ const chunks = [];
316
+ const { usage, ...responseChunk } = SecureLLMProxy.completeChunkFromResponse(response);
317
+ if (responseChunk.content ||
318
+ responseChunk.toolCalls != null ||
319
+ responseChunk.id != null ||
320
+ responseChunk.name != null ||
321
+ responseChunk.responseMetadata != null ||
322
+ responseChunk.additionalKwargs != null) {
323
+ chunks.push(responseChunk);
324
+ }
325
+ if (usage != null)
326
+ chunks.push({ usage });
327
+ return chunks;
328
+ }
329
+ static completeChunkFromResponse(response) {
330
+ return {
331
+ content: response.content || undefined,
332
+ toolCalls: SecureLLMProxy.convertToolCallsToStreamChunks(response.toolCalls ?? null) ?? undefined,
333
+ usage: response.usage ?? undefined,
334
+ id: response.id,
335
+ name: response.name,
336
+ responseMetadata: response.responseMetadata,
337
+ additionalKwargs: response.additionalKwargs,
338
+ };
339
+ }
340
+ static responsePayloadFromChunks(chunks) {
341
+ const response = SecureLLMProxy.responseFromStreamChunks(chunks);
342
+ const result = {
343
+ content: response.content,
344
+ metadata: response.metadata,
345
+ };
346
+ if (response.toolCalls !== undefined) {
347
+ result["tool_calls"] = response.toolCalls.map((toolCall) => {
348
+ const value = {};
349
+ if (toolCall.id !== undefined)
350
+ value["id"] = toolCall.id;
351
+ if (toolCall.name !== undefined)
352
+ value["name"] = toolCall.name;
353
+ if (toolCall.args !== undefined)
354
+ value["args"] = toolCall.args;
355
+ if (toolCall.type !== undefined)
356
+ value["type"] = toolCall.type;
357
+ if (toolCall.index !== undefined)
358
+ value["index"] = toolCall.index;
359
+ return value;
360
+ });
361
+ }
362
+ if (response.usage !== undefined) {
363
+ result["usage"] = response.usage.toJSON();
364
+ }
365
+ if (response.id !== undefined)
366
+ result["id"] = response.id;
367
+ if (response.name !== undefined)
368
+ result["name"] = response.name;
369
+ if (response.responseMetadata !== undefined) {
370
+ result["response_metadata"] = response.responseMetadata;
371
+ }
372
+ if (response.additionalKwargs !== undefined) {
373
+ result["additional_kwargs"] = response.additionalKwargs;
374
+ }
375
+ return result;
376
+ }
377
+ /**
378
+ * Handle a guardrail require_review response by suspending via the framework
379
+ * suspend handler (LangGraph `interrupt`). On first call the handler suspends
380
+ * the node and never returns; on resume it returns the OE-dispatched decision
381
+ * — an object with a top-level `guardrail_review` key. Approve yields the
382
+ * pending LLM content; deny (or an unrecognised decision) denies the call.
383
+ * Mirrors Python's `_handle_require_review`.
384
+ */
385
+ *handleRequireReview(response) {
386
+ const interrupt = getSuspendHandler();
387
+ if (interrupt == null) {
388
+ throw new PolicyDeniedException("Guardrail require_review: no suspend handler registered. " +
389
+ "Ensure the framework SDK calls registerSuspendHandler() before run().", response.guardrail_meta ?? null);
390
+ }
391
+ const suspendPayload = {
392
+ suspend_reason: "guardrail_require_review",
393
+ allowed_decisions: ["approve", "deny"],
394
+ reason: response.reason ?? "Guardrail required human review",
395
+ guardrail_meta: response.guardrail_meta ?? null,
396
+ };
397
+ logger.info({ executionId: this.executionId, step: this.stepCounter }, "Guardrail require_review: suspending for human review");
398
+ // First call suspends via interrupt() and never returns; on resume it
399
+ // returns the stored decision.
400
+ const humanDecision = interrupt(suspendPayload);
401
+ if (humanDecision == null || typeof humanDecision !== "object") {
402
+ throw new LLMInvocationError(`Guardrail require_review: expected object from interrupt, got ${typeof humanDecision}`);
403
+ }
404
+ const guardrailReview = humanDecision
405
+ .guardrail_review;
406
+ if (guardrailReview == null || typeof guardrailReview !== "object") {
407
+ throw new LLMInvocationError("Guardrail require_review: resume data missing or invalid " +
408
+ `'guardrail_review' key (got keys: ${Object.keys(humanDecision).join(", ")})`);
409
+ }
410
+ const review = guardrailReview;
411
+ const decision = review.decision ?? "";
412
+ if (decision === "approve") {
413
+ const pendingContent = review.pending_llm_content;
414
+ if (pendingContent == null) {
415
+ logger.error({ executionId: this.executionId }, "Guardrail require_review: approved but pending_llm_content missing from resume data");
416
+ throw new LLMInvocationError("Guardrail require_review: approved but OE did not include pending LLM content");
417
+ }
418
+ logger.info({ executionId: this.executionId }, "Guardrail require_review: approved — yielding original LLM content");
419
+ yield* SecureLLMProxy.chunksFromResponse(SecureLLMProxy.convertResultToResponse(pendingContent));
420
+ return;
421
+ }
422
+ if (decision !== "deny") {
423
+ logger.warn({ executionId: this.executionId, decision }, "Guardrail require_review: unrecognised decision value — treating as deny");
424
+ }
425
+ else {
426
+ logger.info({ executionId: this.executionId, decision }, "Guardrail require_review: denied by reviewer");
427
+ }
428
+ throw new PolicyDeniedException(`Guardrail require_review: denied by reviewer (decision=${JSON.stringify(decision)})`, response.guardrail_meta ?? null);
429
+ }
430
+ /**
431
+ * Collect streamed sdk-core chunks into a final sdk-core LLMResponse.
432
+ * Static — can be called without a proxy instance (e.g. after parallel streaming).
433
+ */
434
+ static responseFromStreamChunks(chunks) {
435
+ const contentParts = [];
436
+ const streamedToolCalls = [];
437
+ let usage;
438
+ let messageId;
439
+ let messageName;
440
+ let responseMetadata;
441
+ let additionalKwargs;
442
+ for (const chunk of chunks) {
443
+ if (chunk.content)
444
+ contentParts.push(chunk.content);
445
+ if (chunk.toolCalls)
446
+ streamedToolCalls.push(...chunk.toolCalls);
447
+ if (chunk.usage != null) {
448
+ usage = chunk.usage;
449
+ }
450
+ // LangChain JS keeps the first message ID while concatenating streamed
451
+ // AIMessageChunks. Preserve the same ID in the recorded terminal result
452
+ // so a synthetic replay produces the identical checkpoint message.
453
+ if (messageId === undefined && chunk.id != null)
454
+ messageId = chunk.id;
455
+ if (chunk.name != null)
456
+ messageName = chunk.name;
457
+ if (chunk.responseMetadata != null) {
458
+ responseMetadata = {
459
+ ...(responseMetadata ?? {}),
460
+ ...chunk.responseMetadata,
461
+ };
462
+ }
463
+ if (chunk.additionalKwargs != null) {
464
+ additionalKwargs = {
465
+ ...(additionalKwargs ?? {}),
466
+ ...chunk.additionalKwargs,
467
+ };
468
+ }
469
+ }
470
+ const toolCalls = SecureLLMProxy.convertStreamChunksToToolCalls(streamedToolCalls);
471
+ if (toolCalls != null && additionalKwargs?.["tool_calls"] !== undefined) {
472
+ // Provider stream fragments are redundant once semantic tool calls
473
+ // exist; persisting them would smuggle non-portable data into the
474
+ // recorded result and back into replay. Same rule as
475
+ // lcToPlatformMessage.
476
+ const { tool_calls: _providerFragments, ...portableKwargs } = additionalKwargs;
477
+ additionalKwargs =
478
+ Object.keys(portableKwargs).length > 0 ? portableKwargs : undefined;
479
+ }
480
+ const metadata = usage != null ? usageToWireDict(usage) : {};
481
+ return new LLMResponse({
482
+ content: contentParts.join(""),
483
+ toolCalls: toolCalls ?? undefined,
484
+ metadata,
485
+ usage,
486
+ id: messageId,
487
+ name: messageName,
488
+ responseMetadata,
489
+ additionalKwargs,
490
+ });
491
+ }
492
+ // ---------------------------------------------------------------------------
493
+ // Private helpers
494
+ // ---------------------------------------------------------------------------
495
+ /**
496
+ * Serialize bound tools to LLMToolSchema for transmission to OE.
497
+ *
498
+ * TS divergence from Python's `_serialize_bound_tools`: LangChain JS has no
499
+ * `tool_call_schema` attribute on tools (it's a Python-only API). Tools
500
+ * carrying Zod input schemas are normalized to OpenAI-canonical dicts
501
+ * upstream by `_normalizeBoundTool` in agent-engine-sdk-langgraph-ts via
502
+ * `convertToOpenAITool`, so every entry that reaches us here is already
503
+ * either a plain dict or an `LLMToolSchema` instance — both go through
504
+ * `LLMToolSchemaValidator`.
505
+ */
506
+ serializeBoundTools() {
507
+ if (!this.boundTools || this.boundTools.length === 0)
508
+ return null;
509
+ const serialized = [];
510
+ for (const tool of this.boundTools) {
511
+ if (tool === null || typeof tool !== "object" || Array.isArray(tool)) {
512
+ logger.warn({ toolType: typeof tool }, `SecureLLMProxy: Skipping unrecognized tool type ${typeof tool} during serialization`);
513
+ continue;
514
+ }
515
+ const parsed = this.safeParseToolSchema(tool);
516
+ if (parsed)
517
+ serialized.push(parsed);
518
+ }
519
+ return serialized.length > 0 ? serialized : null;
520
+ }
521
+ safeParseToolSchema(value) {
522
+ const result = LLMToolSchemaValidator.safeParse(value);
523
+ if (result.success)
524
+ return result.data;
525
+ logger.warn({ issues: result.error.issues }, "SecureLLMProxy: Skipping tool that failed LLMToolSchema validation");
526
+ return null;
527
+ }
528
+ /**
529
+ * Validate `boundToolChoice` (typed `unknown` at the proxy boundary) as a
530
+ * `JsonValue` for the wire. Rejects non-JSON values rather than letting them
531
+ * break `JSON.stringify` or be silently dropped. `null`/`undefined` → omit
532
+ * the field; `false` is preserved to disable forced tool use.
533
+ */
534
+ resolveToolChoice() {
535
+ if (this.boundToolChoice == null)
536
+ return undefined;
537
+ const result = JsonValueSchema.safeParse(this.boundToolChoice);
538
+ if (!result.success) {
539
+ throw new LLMInvocationError(`Invalid tool_choice: expected a JSON-serializable value, got ${typeof this.boundToolChoice}`);
540
+ }
541
+ return result.data;
542
+ }
543
+ buildInvokeRequest(messages, stop, options, stream) {
544
+ const tools = this.serializeBoundTools();
545
+ // `messages` are already validated internal `Message[]` (built by
546
+ // lcMessagesToPlatform). Do NOT re-run them through
547
+ // InvokeLLMRequestArgumentsSchema.parse — its `messages: z.array(MessageSchema)`
548
+ // applies the snake→camel "parse" transform, which blanks the camelCase
549
+ // fields (toolCallId/toolCalls) of already-internal messages. Parse is the
550
+ // wire→internal boundary; build the typed envelope directly here, and let
551
+ // serializeInvokeLLMRequestArguments do the camel→snake dump on the way out.
552
+ return {
553
+ messages,
554
+ model: this.modelName,
555
+ llm_id: this.llmId,
556
+ stop_sequences: stop ?? undefined,
557
+ tools: tools ?? undefined,
558
+ tool_choice: this.resolveToolChoice(),
559
+ options: options ?? undefined,
560
+ stream,
561
+ };
562
+ }
563
+ async requestOeExecution(invokeRequest, step) {
564
+ // The sole caller (`stream()`) always resolves the step before calling in,
565
+ // so step is required and non-null here. Keep the counter monotonic.
566
+ this.operationalSteps.observeAtLeast(step);
567
+ const response = await requestOeApprovalRetryable({
568
+ oeUrl: this.oeUrl,
569
+ executionId: this.executionId,
570
+ toolName: "invoke_llm",
571
+ // serializeInvokeLLMRequestArguments is the camel→snake "dump" boundary,
572
+ // the equivalent of Python's model_dump(by_alias=True): stop_sequences →
573
+ // stop and each Message dumped to snake_case wire fields.
574
+ arguments: serializeInvokeLLMRequestArguments(invokeRequest),
575
+ step,
576
+ // OE holds /tool/execute open while the (non-streaming) LLM call
577
+ // completes, so use the long LLM read timeout rather than the generic
578
+ // request timeout. Mirrors Python's
579
+ // `httpx.Timeout(get_request_timeout(), read=LLM_READ_TIMEOUT)`.
580
+ timeoutMs: LLM_READ_TIMEOUT * 1000,
581
+ });
582
+ // Let require_review and block-substitute (proceed=false with a result
583
+ // string) responses through — stream() handles them. Only a genuine block
584
+ // with no substitute content is a hard denial here.
585
+ if (!response.proceed &&
586
+ response.result == null &&
587
+ response.status !== "require_review") {
588
+ raiseForOeRejection(response.reason, response.guardrail_meta ?? null);
589
+ }
590
+ this.lastDurationMs = response.duration_ms ?? 0.0;
591
+ this.lastFromCache = response.from_cache ?? response.cached_result != null;
592
+ this.lastLatestStepNumber = response.latest_step_number ?? null;
593
+ this.lastPodName = response.pod_name ?? null;
594
+ if (response.latest_step_number != null) {
595
+ this.operationalSteps.observeAtLeast(response.latest_step_number);
596
+ }
597
+ return response;
598
+ }
599
+ static convertResultToResponse(result) {
600
+ if (result == null)
601
+ throw new LLMInvocationError("OE returned no invoke_llm result");
602
+ let response;
603
+ try {
604
+ response = LLMResponse.fromRaw(result);
605
+ }
606
+ catch (exc) {
607
+ throw new LLMInvocationError(`OE returned unexpected invoke_llm result type: ${typeof result}`, { cause: exc });
608
+ }
609
+ // Python parity: if usage is present but metadata is empty, copy usage fields into metadata.
610
+ // Python's model_validator does this as a post-parse mutation; LLMResponse.fromRaw does not.
611
+ if (response.usage != null && Object.keys(response.metadata).length === 0) {
612
+ const usageDict = usageToWireDict(response.usage);
613
+ return new LLMResponse({
614
+ content: response.content,
615
+ toolCalls: response.toolCalls,
616
+ metadata: usageDict,
617
+ usage: response.usage,
618
+ id: response.id,
619
+ name: response.name,
620
+ additionalKwargs: response.additionalKwargs,
621
+ responseMetadata: response.responseMetadata,
622
+ });
623
+ }
624
+ return response;
625
+ }
626
+ static convertToolCallsToStreamChunks(toolCalls) {
627
+ if (!toolCalls || toolCalls.length === 0)
628
+ return null;
629
+ return toolCalls.map((tc, fallbackIndex) => ({
630
+ id: tc.id,
631
+ name: tc.name,
632
+ args: SecureLLMProxy.normalizeToolCallArgs(tc.args),
633
+ type: tc.type,
634
+ index: tc.index ?? (tc.id == null ? fallbackIndex : undefined),
635
+ }));
636
+ }
637
+ static normalizeToolCallArgs(args) {
638
+ if (args == null)
639
+ return undefined;
640
+ if (typeof args === "string")
641
+ return args;
642
+ try {
643
+ return JSON.stringify(args);
644
+ }
645
+ catch {
646
+ return undefined;
647
+ }
648
+ }
649
+ static parseToolCallArgs(rawArgs) {
650
+ try {
651
+ return JSON.parse(rawArgs);
652
+ }
653
+ catch {
654
+ return rawArgs;
655
+ }
656
+ }
657
+ static convertStreamChunksToToolCalls(chunks) {
658
+ if (chunks.length === 0)
659
+ return null;
660
+ const accumulated = new Map();
661
+ const idToIndex = new Map();
662
+ let currentIndex = null;
663
+ let nextIndex = 0;
664
+ for (const chunk of chunks) {
665
+ let index;
666
+ if (chunk.index != null) {
667
+ index = chunk.index;
668
+ currentIndex = index;
669
+ nextIndex = Math.max(nextIndex, index + 1);
670
+ }
671
+ else if (chunk.id != null) {
672
+ const existing = idToIndex.get(chunk.id);
673
+ if (existing != null) {
674
+ index = existing;
675
+ }
676
+ else {
677
+ index = nextIndex++;
678
+ idToIndex.set(chunk.id, index);
679
+ }
680
+ currentIndex = index;
681
+ }
682
+ else if (currentIndex != null) {
683
+ index = currentIndex;
684
+ }
685
+ else {
686
+ index = nextIndex++;
687
+ currentIndex = index;
688
+ }
689
+ const entry = accumulated.get(index) ?? {};
690
+ accumulated.set(index, entry);
691
+ if (chunk.index != null)
692
+ entry["index"] = chunk.index;
693
+ if (chunk.id != null) {
694
+ const prevIndex = idToIndex.get(chunk.id);
695
+ if (prevIndex != null && prevIndex !== index) {
696
+ logger.warn({ id: chunk.id, prevIndex, index }, `LLM: Tool call id ${chunk.id} changed stream index from ${prevIndex} to ${index}`);
697
+ }
698
+ entry["id"] = chunk.id;
699
+ idToIndex.set(chunk.id, index);
700
+ }
701
+ if (chunk.name != null)
702
+ entry["name"] = chunk.name;
703
+ if (chunk.type != null)
704
+ entry["type"] = chunk.type;
705
+ if (chunk.args != null) {
706
+ const existing = entry["args"];
707
+ entry["args"] =
708
+ typeof existing === "string" ? existing + chunk.args : chunk.args;
709
+ }
710
+ }
711
+ const toolCalls = [];
712
+ for (const index of [...accumulated.keys()].sort((a, b) => a - b)) {
713
+ const accumulatedEntry = accumulated.get(index);
714
+ if (accumulatedEntry === undefined)
715
+ continue;
716
+ const payload = { ...accumulatedEntry };
717
+ const rawArgs = payload["args"];
718
+ if (typeof rawArgs === "string") {
719
+ payload["args"] = SecureLLMProxy.parseToolCallArgs(rawArgs);
720
+ }
721
+ toolCalls.push(new LLMToolCall(payload));
722
+ }
723
+ return toolCalls.length > 0 ? toolCalls : null;
724
+ }
725
+ static streamChunkFromEvent(event) {
726
+ const content = event.content ?? undefined;
727
+ const toolCalls = [];
728
+ if (event.tool_call_chunks)
729
+ toolCalls.push(...event.tool_call_chunks);
730
+ for (const [i, tc] of (event.tool_calls ?? []).entries()) {
731
+ toolCalls.push({
732
+ id: tc["id"],
733
+ name: tc["name"],
734
+ args: SecureLLMProxy.normalizeToolCallArgs(tc["args"]),
735
+ type: tc["type"],
736
+ index: tc["index"] ?? i,
737
+ });
738
+ }
739
+ if (content == null &&
740
+ toolCalls.length === 0 &&
741
+ event.id == null &&
742
+ event.name == null &&
743
+ event.response_metadata == null &&
744
+ event.additional_kwargs == null &&
745
+ event.usage == null) {
746
+ return null;
747
+ }
748
+ return {
749
+ content,
750
+ toolCalls: toolCalls.length > 0 ? toolCalls : undefined,
751
+ usage: event.usage ?? undefined,
752
+ id: event.id ?? undefined,
753
+ name: event.name ?? undefined,
754
+ responseMetadata: event.response_metadata ?? undefined,
755
+ additionalKwargs: event.additional_kwargs ?? undefined,
756
+ };
757
+ }
758
+ /**
759
+ * Stream real-time LLM chunks from OE's audited SSE relay.
760
+ *
761
+ * Timeout: a single idle AbortController whose timer resets on every raw
762
+ * recv (via parseSseStream's onChunk callback). This matches httpx's
763
+ * `read=LLM_READ_TIMEOUT` behavior — the timeout only fires when the
764
+ * connection goes *silent* for LLM_READ_TIMEOUT seconds, not after that
765
+ * many seconds of wall-clock time. Active streams are never cut short.
766
+ * The timer also covers the connect phase (server not responding at all).
767
+ */
768
+ async *streamFromOe(streamUrl, step) {
769
+ let error = null;
770
+ let podName = null;
771
+ let durationMs = 0.0;
772
+ const startTime = Date.now();
773
+ let exposedOutput = false;
774
+ for (let attempt = 0; attempt < OE_RETRYABLE_MAX_ATTEMPTS; attempt++) {
775
+ let status = "success";
776
+ error = null;
777
+ let errorCode;
778
+ let caughtException = null;
779
+ let doneReceived = false;
780
+ let idleTimedOut = false;
781
+ let retrySameUrl = false;
782
+ let retryDelayMs = 0;
783
+ let providerOwned = false;
784
+ const idleController = new AbortController();
785
+ let idleTimer = null;
786
+ const resetIdle = () => {
787
+ if (idleTimer != null)
788
+ clearTimeout(idleTimer);
789
+ idleTimer = setTimeout(() => {
790
+ idleTimedOut = true;
791
+ idleController.abort();
792
+ }, LLM_READ_TIMEOUT * 1000);
793
+ };
794
+ resetIdle(); // starts counting before fetch — covers connect phase
795
+ // Combine the idle-timeout signal with the execution-wide abort signal so
796
+ // an AER execution timeout cancels the in-flight SSE stream (not just the
797
+ // idle window). Mirrors Python's wait_for cancelling the inner coroutine.
798
+ const streamSignal = withExecutionSignal(idleController.signal);
799
+ try {
800
+ const tlsOptions = getFetchOptionsWithTLS(this.oeUrl);
801
+ const resp = await fetchPlatform(streamUrl, {
802
+ method: "POST",
803
+ signal: streamSignal,
804
+ ...tlsOptions,
805
+ });
806
+ resetIdle(); // fresh window after headers received — body read phase starts now
807
+ if (!resp.ok) {
808
+ throw new Error(`SSE stream returned HTTP ${resp.status}: ${resp.statusText}`);
809
+ }
810
+ for await (const data of parseSseStream(resp, {
811
+ onChunk: resetIdle,
812
+ signal: streamSignal,
813
+ })) {
814
+ let streamEvent;
815
+ try {
816
+ streamEvent = LLMPodStreamEventSchema.parse(JSON.parse(data));
817
+ }
818
+ catch {
819
+ logger.warn({ step, data: data.slice(0, 120) }, `LLM: Step ${step} - Skipping malformed SSE payload`);
820
+ continue;
821
+ }
822
+ if (streamEvent.interrupted) {
823
+ // OE stopped this call on interrupt: not error, not truncation.
824
+ // Yield a clean marker so the agent continues instead of raising.
825
+ doneReceived = true;
826
+ podName = streamEvent.pod_name ?? podName;
827
+ if (streamEvent.duration_ms != null)
828
+ durationMs = streamEvent.duration_ms;
829
+ yield { responseMetadata: { interrupted: true } };
830
+ break;
831
+ }
832
+ if (streamEvent.error) {
833
+ error = streamEvent.error;
834
+ if (streamEvent.retryable &&
835
+ attempt < OE_RETRYABLE_MAX_ATTEMPTS - 1 &&
836
+ !exposedOutput) {
837
+ retrySameUrl = true;
838
+ retryDelayMs = oeStreamRetryDelayMs(streamEvent.retry_after_ms);
839
+ }
840
+ else {
841
+ status = "error";
842
+ providerOwned = true;
843
+ errorCode = streamEvent.error_code;
844
+ }
845
+ break;
846
+ }
847
+ if (streamEvent.done) {
848
+ doneReceived = true;
849
+ podName = streamEvent.pod_name ?? null;
850
+ if (streamEvent.duration_ms != null)
851
+ durationMs = streamEvent.duration_ms;
852
+ if (streamEvent.usage != null) {
853
+ yield { usage: streamEvent.usage };
854
+ }
855
+ break;
856
+ }
857
+ const chunk = SecureLLMProxy.streamChunkFromEvent(streamEvent);
858
+ if (chunk != null) {
859
+ exposedOutput = true;
860
+ yield chunk;
861
+ }
862
+ }
863
+ }
864
+ catch (exc) {
865
+ if (status !== "error") {
866
+ status = "error";
867
+ error = exc instanceof Error ? exc.message : String(exc);
868
+ caughtException = exc instanceof Error ? exc : null;
869
+ }
870
+ if (attempt < OE_RETRYABLE_MAX_ATTEMPTS - 1 &&
871
+ !exposedOutput &&
872
+ !idleTimedOut &&
873
+ isOeStreamTransportDisconnect(exc)) {
874
+ retrySameUrl = true;
875
+ retryDelayMs = OE_DISPATCH_TAKEOVER_RETRY_DELAY_MS;
876
+ }
877
+ }
878
+ finally {
879
+ if (idleTimer != null)
880
+ clearTimeout(idleTimer);
881
+ if (durationMs === 0.0)
882
+ durationMs = Date.now() - startTime;
883
+ if (idleTimedOut) {
884
+ // Idle timer fired — override whatever error landed in the catch block
885
+ // (usually "This operation was aborted" from the AbortError) with a
886
+ // human-readable message that matches Python's read-timeout semantics.
887
+ status = "error";
888
+ error = `LLM read idle timeout: no data received for ${LLM_READ_TIMEOUT}s`;
889
+ caughtException = null;
890
+ retrySameUrl = false;
891
+ providerOwned = false;
892
+ }
893
+ else if (!retrySameUrl && status === "success" && !doneReceived) {
894
+ error = "SSE stream ended without done signal (truncated response)";
895
+ if (attempt < OE_RETRYABLE_MAX_ATTEMPTS - 1 && !exposedOutput) {
896
+ retrySameUrl = true;
897
+ retryDelayMs = OE_DISPATCH_TAKEOVER_RETRY_DELAY_MS;
898
+ }
899
+ else {
900
+ status = "error";
901
+ }
902
+ }
903
+ this.lastDurationMs = durationMs;
904
+ this.lastPodName = podName;
905
+ }
906
+ if (retrySameUrl) {
907
+ await oeStreamRetry.sleep(retryDelayMs);
908
+ continue;
909
+ }
910
+ if (status === "error") {
911
+ const msg = error ?? "LLM streaming failed";
912
+ throw new LLMInvocationError(msg, {
913
+ ...(caughtException ? { cause: caughtException } : {}),
914
+ ...(providerOwned ? { source: "llm" } : {}),
915
+ ...(errorCode ? { error_code: errorCode } : {}),
916
+ });
917
+ }
918
+ return;
919
+ }
920
+ throw new LLMInvocationError(error ?? "LLM streaming failed");
921
+ }
922
+ }