@mongodb-js/agent-engine-runner-shared 0.11.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +55 -0
- package/LICENSE.md +201 -0
- package/README.md +29 -0
- package/dist/agent_config.d.ts +167 -0
- package/dist/agent_config.d.ts.map +1 -0
- package/dist/agent_config.js +544 -0
- package/dist/call_interrupted.d.ts +12 -0
- package/dist/call_interrupted.d.ts.map +1 -0
- package/dist/call_interrupted.js +11 -0
- package/dist/checkpoint_workspace.d.ts +25 -0
- package/dist/checkpoint_workspace.d.ts.map +1 -0
- package/dist/checkpoint_workspace.js +44 -0
- package/dist/context.d.ts +235 -0
- package/dist/context.d.ts.map +1 -0
- package/dist/context.js +322 -0
- package/dist/db_config.d.ts +28 -0
- package/dist/db_config.d.ts.map +1 -0
- package/dist/db_config.js +66 -0
- package/dist/db_naming.d.ts +54 -0
- package/dist/db_naming.d.ts.map +1 -0
- package/dist/db_naming.js +94 -0
- package/dist/error_reporting.d.ts +67 -0
- package/dist/error_reporting.d.ts.map +1 -0
- package/dist/error_reporting.js +311 -0
- package/dist/generated/workflow/v1/activity_pb.d.ts +342 -0
- package/dist/generated/workflow/v1/activity_pb.d.ts.map +1 -0
- package/dist/generated/workflow/v1/activity_pb.js +115 -0
- package/dist/generated/workflow/v1/common_pb.d.ts +184 -0
- package/dist/generated/workflow/v1/common_pb.d.ts.map +1 -0
- package/dist/generated/workflow/v1/common_pb.js +86 -0
- package/dist/generated/workflow/v1/runtime_pb.d.ts +200 -0
- package/dist/generated/workflow/v1/runtime_pb.d.ts.map +1 -0
- package/dist/generated/workflow/v1/runtime_pb.js +40 -0
- package/dist/generated/workflow/v1/state_pb.d.ts +254 -0
- package/dist/generated/workflow/v1/state_pb.d.ts.map +1 -0
- package/dist/generated/workflow/v1/state_pb.js +68 -0
- package/dist/guardrails_evaluator/core.d.ts +23 -0
- package/dist/guardrails_evaluator/core.d.ts.map +1 -0
- package/dist/guardrails_evaluator/core.js +122 -0
- package/dist/guardrails_evaluator/index.d.ts +10 -0
- package/dist/guardrails_evaluator/index.d.ts.map +1 -0
- package/dist/guardrails_evaluator/index.js +11 -0
- package/dist/guardrails_evaluator/regex.d.ts +20 -0
- package/dist/guardrails_evaluator/regex.d.ts.map +1 -0
- package/dist/guardrails_evaluator/regex.js +233 -0
- package/dist/hooks.d.ts +109 -0
- package/dist/hooks.d.ts.map +1 -0
- package/dist/hooks.js +216 -0
- package/dist/http_path.d.ts +18 -0
- package/dist/http_path.d.ts.map +1 -0
- package/dist/http_path.js +53 -0
- package/dist/index.d.ts +35 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +41 -0
- package/dist/launcher.d.ts +130 -0
- package/dist/launcher.d.ts.map +1 -0
- package/dist/launcher.js +325 -0
- package/dist/logger.d.ts +96 -0
- package/dist/logger.d.ts.map +1 -0
- package/dist/logger.js +204 -0
- package/dist/mcp_oauth.d.ts +51 -0
- package/dist/mcp_oauth.d.ts.map +1 -0
- package/dist/mcp_oauth.js +389 -0
- package/dist/mcp_oauth_secret.d.ts +21 -0
- package/dist/mcp_oauth_secret.d.ts.map +1 -0
- package/dist/mcp_oauth_secret.js +122 -0
- package/dist/mcp_tools.d.ts +71 -0
- package/dist/mcp_tools.d.ts.map +1 -0
- package/dist/mcp_tools.js +301 -0
- package/dist/memory_appbound.d.ts +42 -0
- package/dist/memory_appbound.d.ts.map +1 -0
- package/dist/memory_appbound.js +159 -0
- package/dist/memory_writer.d.ts +49 -0
- package/dist/memory_writer.d.ts.map +1 -0
- package/dist/memory_writer.js +171 -0
- package/dist/metrics.d.ts +84 -0
- package/dist/metrics.d.ts.map +1 -0
- package/dist/metrics.js +205 -0
- package/dist/models.d.ts +1458 -0
- package/dist/models.d.ts.map +1 -0
- package/dist/models.js +1726 -0
- package/dist/node_logger.d.ts +43 -0
- package/dist/node_logger.d.ts.map +1 -0
- package/dist/node_logger.js +158 -0
- package/dist/owner_callback.d.ts +16 -0
- package/dist/owner_callback.d.ts.map +1 -0
- package/dist/owner_callback.js +40 -0
- package/dist/progress.d.ts +57 -0
- package/dist/progress.d.ts.map +1 -0
- package/dist/progress.js +140 -0
- package/dist/runtime.d.ts +131 -0
- package/dist/runtime.d.ts.map +1 -0
- package/dist/runtime.js +351 -0
- package/dist/secure_llm_proxy.d.ts +115 -0
- package/dist/secure_llm_proxy.d.ts.map +1 -0
- package/dist/secure_llm_proxy.js +922 -0
- package/dist/secure_wrapper.d.ts +332 -0
- package/dist/secure_wrapper.d.ts.map +1 -0
- package/dist/secure_wrapper.js +1249 -0
- package/dist/server/aer.d.ts +61 -0
- package/dist/server/aer.d.ts.map +1 -0
- package/dist/server/aer.js +1124 -0
- package/dist/server/auth.d.ts +56 -0
- package/dist/server/auth.d.ts.map +1 -0
- package/dist/server/auth.js +132 -0
- package/dist/server/base.d.ts +104 -0
- package/dist/server/base.d.ts.map +1 -0
- package/dist/server/base.js +150 -0
- package/dist/server/callInterrupt.d.ts +49 -0
- package/dist/server/callInterrupt.d.ts.map +1 -0
- package/dist/server/callInterrupt.js +68 -0
- package/dist/server/callback_delivery.d.ts +14 -0
- package/dist/server/callback_delivery.d.ts.map +1 -0
- package/dist/server/callback_delivery.js +141 -0
- package/dist/server/chunk_types.d.ts +50 -0
- package/dist/server/chunk_types.d.ts.map +1 -0
- package/dist/server/chunk_types.js +62 -0
- package/dist/server/cors.d.ts +52 -0
- package/dist/server/cors.d.ts.map +1 -0
- package/dist/server/cors.js +107 -0
- package/dist/server/drain.d.ts +169 -0
- package/dist/server/drain.d.ts.map +1 -0
- package/dist/server/drain.js +455 -0
- package/dist/server/function.d.ts +77 -0
- package/dist/server/function.d.ts.map +1 -0
- package/dist/server/function.js +337 -0
- package/dist/server/http_retry.d.ts +37 -0
- package/dist/server/http_retry.d.ts.map +1 -0
- package/dist/server/http_retry.js +157 -0
- package/dist/server/index.d.ts +7 -0
- package/dist/server/index.d.ts.map +1 -0
- package/dist/server/index.js +5 -0
- package/dist/server/metadata.d.ts +50 -0
- package/dist/server/metadata.d.ts.map +1 -0
- package/dist/server/metadata.js +193 -0
- package/dist/server/oe_url.d.ts +36 -0
- package/dist/server/oe_url.d.ts.map +1 -0
- package/dist/server/oe_url.js +50 -0
- package/dist/server/owner_url.d.ts +35 -0
- package/dist/server/owner_url.d.ts.map +1 -0
- package/dist/server/owner_url.js +146 -0
- package/dist/server/query.d.ts +42 -0
- package/dist/server/query.d.ts.map +1 -0
- package/dist/server/query.js +28 -0
- package/dist/server/tool.d.ts +138 -0
- package/dist/server/tool.d.ts.map +1 -0
- package/dist/server/tool.js +1017 -0
- package/dist/span_names.d.ts +21 -0
- package/dist/span_names.d.ts.map +1 -0
- package/dist/span_names.js +31 -0
- package/dist/structured_logging/constants.d.ts +17 -0
- package/dist/structured_logging/constants.d.ts.map +1 -0
- package/dist/structured_logging/constants.js +71 -0
- package/dist/structured_logging/env.d.ts +18 -0
- package/dist/structured_logging/env.d.ts.map +1 -0
- package/dist/structured_logging/env.js +39 -0
- package/dist/structured_logging/install.d.ts +56 -0
- package/dist/structured_logging/install.d.ts.map +1 -0
- package/dist/structured_logging/install.js +107 -0
- package/dist/structured_logging/layout.d.ts +9 -0
- package/dist/structured_logging/layout.d.ts.map +1 -0
- package/dist/structured_logging/layout.js +144 -0
- package/dist/structured_logging/serialize.d.ts +27 -0
- package/dist/structured_logging/serialize.d.ts.map +1 -0
- package/dist/structured_logging/serialize.js +61 -0
- package/dist/structured_logging/stdio_capture.d.ts +59 -0
- package/dist/structured_logging/stdio_capture.d.ts.map +1 -0
- package/dist/structured_logging/stdio_capture.js +164 -0
- package/dist/structured_logging/uncaught.d.ts +14 -0
- package/dist/structured_logging/uncaught.d.ts.map +1 -0
- package/dist/structured_logging/uncaught.js +58 -0
- package/dist/structured_logging.d.ts +48 -0
- package/dist/structured_logging.d.ts.map +1 -0
- package/dist/structured_logging.js +47 -0
- package/dist/tls_client.d.ts +61 -0
- package/dist/tls_client.d.ts.map +1 -0
- package/dist/tls_client.js +298 -0
- package/dist/tool_api_error.d.ts +62 -0
- package/dist/tool_api_error.d.ts.map +1 -0
- package/dist/tool_api_error.js +399 -0
- package/dist/tool_memory_ownership.d.ts +10 -0
- package/dist/tool_memory_ownership.d.ts.map +1 -0
- package/dist/tool_memory_ownership.js +36 -0
- package/dist/toolpod_handlers.d.ts +126 -0
- package/dist/toolpod_handlers.d.ts.map +1 -0
- package/dist/toolpod_handlers.js +1016 -0
- package/dist/tracing/exporters.d.ts +51 -0
- package/dist/tracing/exporters.d.ts.map +1 -0
- package/dist/tracing/exporters.js +327 -0
- package/dist/tracing/index.d.ts +3 -0
- package/dist/tracing/index.d.ts.map +1 -0
- package/dist/tracing/index.js +2 -0
- package/dist/tracing/setup.d.ts +76 -0
- package/dist/tracing/setup.d.ts.map +1 -0
- package/dist/tracing/setup.js +436 -0
- package/dist/utils.d.ts +204 -0
- package/dist/utils.d.ts.map +1 -0
- package/dist/utils.js +867 -0
- package/dist/workflow/activity.d.ts +71 -0
- package/dist/workflow/activity.d.ts.map +1 -0
- package/dist/workflow/activity.js +357 -0
- package/dist/workflow/attempt.d.ts +12 -0
- package/dist/workflow/attempt.d.ts.map +1 -0
- package/dist/workflow/attempt.js +96 -0
- package/dist/workflow/client.d.ts +46 -0
- package/dist/workflow/client.d.ts.map +1 -0
- package/dist/workflow/client.js +299 -0
- package/dist/workflow/context.d.ts +37 -0
- package/dist/workflow/context.d.ts.map +1 -0
- package/dist/workflow/context.js +350 -0
- package/dist/workflow/heartbeat.d.ts +15 -0
- package/dist/workflow/heartbeat.d.ts.map +1 -0
- package/dist/workflow/heartbeat.js +78 -0
- package/dist/workflow/index.d.ts +14 -0
- package/dist/workflow/index.d.ts.map +1 -0
- package/dist/workflow/index.js +10 -0
- package/dist/workflow/memory.d.ts +17 -0
- package/dist/workflow/memory.d.ts.map +1 -0
- package/dist/workflow/memory.js +184 -0
- package/package.json +73 -0
|
@@ -0,0 +1,1017 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tool Executor Pod server — executes tool functions in isolated pods.
|
|
3
|
+
*
|
|
4
|
+
* Responsibilities:
|
|
5
|
+
* - Execute registered tool functions
|
|
6
|
+
* - Invoke LLMs routed from AER via SecureWrappedLLM
|
|
7
|
+
* - Return results to OE
|
|
8
|
+
*
|
|
9
|
+
* Mirrors Python's `ToolServer` in `server/tool.py`.
|
|
10
|
+
*/
|
|
11
|
+
import os from "node:os";
|
|
12
|
+
import { once } from "node:events";
|
|
13
|
+
import { getLogger } from "../logger.js";
|
|
14
|
+
import { withMetrics } from "../metrics.js";
|
|
15
|
+
import { getCurrentExecutionMetadata, getRequestedSuspend, runWithCustomerOrigin, runWithExecutionContext, } from "../context.js";
|
|
16
|
+
import { getLLMAdapterFactory, getNamedLlm, populateLlmRegistryFromEntrypoint, } from "../hooks.js";
|
|
17
|
+
import { isConfiguredMcpSdkToolName, makeMcpToolCallable, MCPConfigError, resolveConfiguredMcpToolBinding, } from "../mcp_tools.js";
|
|
18
|
+
import { GuardrailCheckRequestSchema, LLMPodInvokeRequestSchema, LLMPodInvokeResponseSchema, LLMPodStreamEventSchema, mergeTokenUsage, normalizeLLMPodInvokeResponse, ToolPodExecuteRequestSchema, } from "../models.js";
|
|
19
|
+
import { redactText } from "../error_reporting.js";
|
|
20
|
+
import { evaluateGuardrailCheck } from "../guardrails_evaluator/index.js";
|
|
21
|
+
import { SecureLLMProxy } from "../secure_llm_proxy.js";
|
|
22
|
+
import { closeAllTLSAgents } from "../tls_client.js";
|
|
23
|
+
import { executeErrorResponse, requestCredentialValues, requestNamedCredentialValues, } from "../tool_api_error.js";
|
|
24
|
+
import { withMetadataEnv, withMetadataEnvGen, mergeMetadataEnv, } from "./metadata.js";
|
|
25
|
+
import { LLM_BACKOFF_MULTIPLIER, LLM_INITIAL_BACKOFF, LLM_MAX_BACKOFF, LLM_MAX_RETRIES, formatLlmError, isLlmCredentialRejection, isRetryableError, logToolRequest, normalizeToolCallArgs, toolRedactFields, } from "../utils.js";
|
|
26
|
+
import { BaseServer } from "./base.js";
|
|
27
|
+
import { LLM_CREDENTIAL_REJECTED_ERROR_CODE } from "./chunk_types.js";
|
|
28
|
+
import { BUILTIN_TOOL_NAMES, registerBuiltinTools, } from "../toolpod_handlers.js";
|
|
29
|
+
import { resolveOeUrl } from "./oe_url.js";
|
|
30
|
+
import { resolveOwnerUrl } from "./owner_url.js";
|
|
31
|
+
const logger = getLogger("agent_engine_runner_shared.server.tool");
|
|
32
|
+
/**
|
|
33
|
+
* Max time to wait for a slow SSE client to drain before aborting the stream.
|
|
34
|
+
* Prevents unbounded buffering and long-lived holds on the credential gate
|
|
35
|
+
* when a peer never reads.
|
|
36
|
+
*/
|
|
37
|
+
export const SSE_DRAIN_TIMEOUT_MS = 30_000;
|
|
38
|
+
/**
|
|
39
|
+
* Pipe SSE `data:` lines to a hijacked Node response with disconnect cancel,
|
|
40
|
+
* backpressure, and no process-crashing unhandled `error` events.
|
|
41
|
+
*
|
|
42
|
+
* Fastify's `reply.hijack()` removes framework lifecycle management; this
|
|
43
|
+
* helper owns the raw socket for the rest of the stream.
|
|
44
|
+
*
|
|
45
|
+
* Disconnect is detected via the *response* `close` event when the peer drops
|
|
46
|
+
* before `end()` — not via `IncomingMessage` `close`, which also fires after a
|
|
47
|
+
* normally-consumed POST body and would abort a valid stream.
|
|
48
|
+
*
|
|
49
|
+
* `lines` may be an AsyncIterable or a factory that receives the pipe's
|
|
50
|
+
* AbortSignal so upstream generation (retries/backoff) can stop promptly.
|
|
51
|
+
*
|
|
52
|
+
* Force-close paths (peer drop, drain timeout, socket error) call
|
|
53
|
+
* `res.destroy()` rather than `res.end()`: end only queues FIN behind data
|
|
54
|
+
* already stuck in the writable buffer, which never drains for a never-reading
|
|
55
|
+
* peer and leaks the socket/FD indefinitely.
|
|
56
|
+
*/
|
|
57
|
+
export async function pipeSseLinesToResponse(res, lines, opts = {}) {
|
|
58
|
+
const drainTimeoutMs = opts.drainTimeoutMs ?? SSE_DRAIN_TIMEOUT_MS;
|
|
59
|
+
const ac = new AbortController();
|
|
60
|
+
// `ac` carries the pipe's own lifecycle (peer drop, socket error, drain
|
|
61
|
+
// timeout); an external signal (e.g. an execution drain) combines in so
|
|
62
|
+
// either source unblocks a parked iterator.next() — including when the
|
|
63
|
+
// upstream adapter ignores the signal it was handed.
|
|
64
|
+
const signal = opts.signal
|
|
65
|
+
? combineAbortSignals(ac.signal, opts.signal)
|
|
66
|
+
: ac.signal;
|
|
67
|
+
let iterator;
|
|
68
|
+
try {
|
|
69
|
+
const iterable = typeof lines === "function" ? lines(signal) : lines;
|
|
70
|
+
iterator = iterable[Symbol.asyncIterator]();
|
|
71
|
+
}
|
|
72
|
+
catch (err) {
|
|
73
|
+
// Nothing ever started, so nothing is pending: settle immediately.
|
|
74
|
+
opts.onGeneratorSettled?.();
|
|
75
|
+
throw err;
|
|
76
|
+
}
|
|
77
|
+
const onResError = (err) => {
|
|
78
|
+
logger.warn(`SSE response error: ${err.message}`);
|
|
79
|
+
if (!ac.signal.aborted)
|
|
80
|
+
ac.abort();
|
|
81
|
+
};
|
|
82
|
+
const onResClose = () => {
|
|
83
|
+
// Normal completion also emits `close` after `end()`; only abort when the
|
|
84
|
+
// peer dropped the connection before we finished writing.
|
|
85
|
+
if (!res.writableEnded && !ac.signal.aborted) {
|
|
86
|
+
ac.abort();
|
|
87
|
+
}
|
|
88
|
+
};
|
|
89
|
+
// Without this listener, write/end on a destroyed socket becomes an
|
|
90
|
+
// unhandled stream 'error' and terminates the Node process.
|
|
91
|
+
res.on("error", onResError);
|
|
92
|
+
res.on("close", onResClose);
|
|
93
|
+
// Graceful end only when the iterable finishes without abort/timeout.
|
|
94
|
+
let forceClose = false;
|
|
95
|
+
try {
|
|
96
|
+
while (true) {
|
|
97
|
+
if (signal.aborted || res.destroyed || res.writableEnded) {
|
|
98
|
+
forceClose = true;
|
|
99
|
+
break;
|
|
100
|
+
}
|
|
101
|
+
// Race next() against abort so a hung provider / backoff sleep cannot
|
|
102
|
+
// keep the credential gate locked after the peer is gone.
|
|
103
|
+
const stepped = await nextOrAbort(iterator, signal);
|
|
104
|
+
if (stepped === "aborted") {
|
|
105
|
+
forceClose = true;
|
|
106
|
+
break;
|
|
107
|
+
}
|
|
108
|
+
if (stepped.done)
|
|
109
|
+
break;
|
|
110
|
+
const line = stepped.value;
|
|
111
|
+
if (signal.aborted || res.destroyed || res.writableEnded) {
|
|
112
|
+
forceClose = true;
|
|
113
|
+
break;
|
|
114
|
+
}
|
|
115
|
+
const ok = res.write(line);
|
|
116
|
+
if (!ok) {
|
|
117
|
+
const drained = await waitForDrainOrAbort(res, signal, drainTimeoutMs);
|
|
118
|
+
if (!drained || signal.aborted || res.destroyed || res.writableEnded) {
|
|
119
|
+
forceClose = true;
|
|
120
|
+
if (!ac.signal.aborted)
|
|
121
|
+
ac.abort();
|
|
122
|
+
break;
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
catch (err) {
|
|
128
|
+
forceClose = true;
|
|
129
|
+
if (!ac.signal.aborted) {
|
|
130
|
+
logger.warn(`SSE stream failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
finally {
|
|
134
|
+
// Stop aborting on the post-end `close`; keep `error` attached through
|
|
135
|
+
// close/destroy and the following turn so async EPIPE/hangup is still
|
|
136
|
+
// handled.
|
|
137
|
+
res.off("close", onResClose);
|
|
138
|
+
try {
|
|
139
|
+
if (forceClose) {
|
|
140
|
+
// A generator parked at an unresolved await (e.g. an adapter ignoring
|
|
141
|
+
// the abort signal) never settles return(); teardown must not wait on
|
|
142
|
+
// it. Fire-and-forget — and notify settlement only when the generator
|
|
143
|
+
// actually unwinds, so a drain cannot report completed while provider
|
|
144
|
+
// work is still pending; it times out honestly instead.
|
|
145
|
+
Promise.resolve(iterator.return?.())
|
|
146
|
+
.catch(() => {
|
|
147
|
+
/* ignore */
|
|
148
|
+
})
|
|
149
|
+
.finally(() => opts.onGeneratorSettled?.());
|
|
150
|
+
}
|
|
151
|
+
else {
|
|
152
|
+
await iterator.return?.();
|
|
153
|
+
opts.onGeneratorSettled?.();
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
catch (err) {
|
|
157
|
+
logger.warn(`SSE iterator return failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
158
|
+
// A thrown return() means the generator finished with that error.
|
|
159
|
+
opts.onGeneratorSettled?.();
|
|
160
|
+
}
|
|
161
|
+
if (!res.writableEnded && !res.destroyed) {
|
|
162
|
+
try {
|
|
163
|
+
if (forceClose) {
|
|
164
|
+
res.destroy();
|
|
165
|
+
}
|
|
166
|
+
else {
|
|
167
|
+
res.end();
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
catch (err) {
|
|
171
|
+
logger.warn(`SSE response ${forceClose ? "destroy" : "end"} failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
setImmediate(() => {
|
|
175
|
+
res.off("error", onResError);
|
|
176
|
+
});
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* Pull the next iterator value, or resolve `"aborted"` if `signal` fires
|
|
181
|
+
* first. Does not cancel the in-flight `next()`; callers should still
|
|
182
|
+
* `return()` the iterator in `finally` to unwind generators.
|
|
183
|
+
*/
|
|
184
|
+
function nextOrAbort(iterator, signal) {
|
|
185
|
+
if (signal.aborted)
|
|
186
|
+
return Promise.resolve("aborted");
|
|
187
|
+
return new Promise((resolve, reject) => {
|
|
188
|
+
const onAbort = () => {
|
|
189
|
+
cleanup();
|
|
190
|
+
resolve("aborted");
|
|
191
|
+
};
|
|
192
|
+
const cleanup = () => {
|
|
193
|
+
signal.removeEventListener("abort", onAbort);
|
|
194
|
+
};
|
|
195
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
196
|
+
iterator.next().then((result) => {
|
|
197
|
+
cleanup();
|
|
198
|
+
// Abort may have won the race and already resolved; ignore late next.
|
|
199
|
+
if (signal.aborted)
|
|
200
|
+
return;
|
|
201
|
+
resolve(result);
|
|
202
|
+
}, (err) => {
|
|
203
|
+
cleanup();
|
|
204
|
+
if (signal.aborted)
|
|
205
|
+
return;
|
|
206
|
+
reject(err);
|
|
207
|
+
});
|
|
208
|
+
});
|
|
209
|
+
}
|
|
210
|
+
/**
|
|
211
|
+
* Abort the returned controller when either source fires. Local
|
|
212
|
+
* `AbortSignal.any` — both signals stay independent of each other otherwise.
|
|
213
|
+
*/
|
|
214
|
+
function combineAbortSignals(a, b) {
|
|
215
|
+
const combined = new AbortController();
|
|
216
|
+
if (a.aborted || b.aborted) {
|
|
217
|
+
combined.abort();
|
|
218
|
+
return combined.signal;
|
|
219
|
+
}
|
|
220
|
+
const abort = () => combined.abort();
|
|
221
|
+
a.addEventListener("abort", abort, { once: true });
|
|
222
|
+
b.addEventListener("abort", abort, { once: true });
|
|
223
|
+
return combined.signal;
|
|
224
|
+
}
|
|
225
|
+
/**
|
|
226
|
+
* Wait until the response drains, the abort signal fires, or the drain
|
|
227
|
+
* timeout elapses. Returns true only when drain won.
|
|
228
|
+
*/
|
|
229
|
+
function waitForDrainOrAbort(res, signal, timeoutMs) {
|
|
230
|
+
if (signal.aborted || res.destroyed || res.writableEnded) {
|
|
231
|
+
return Promise.resolve(false);
|
|
232
|
+
}
|
|
233
|
+
return once(res, "drain", {
|
|
234
|
+
signal: AbortSignal.any([signal, AbortSignal.timeout(timeoutMs)]),
|
|
235
|
+
}).then(() => true, () => false);
|
|
236
|
+
}
|
|
237
|
+
function isAbortError(err) {
|
|
238
|
+
return ((err instanceof Error && err.name === "AbortError") ||
|
|
239
|
+
(typeof DOMException !== "undefined" &&
|
|
240
|
+
err instanceof DOMException &&
|
|
241
|
+
err.name === "AbortError"));
|
|
242
|
+
}
|
|
243
|
+
/** Abortable sleep used by LLM stream retry backoff. */
|
|
244
|
+
function sleep(ms, signal) {
|
|
245
|
+
return new Promise((resolve, reject) => {
|
|
246
|
+
if (signal?.aborted) {
|
|
247
|
+
reject(new DOMException("This operation was aborted", "AbortError"));
|
|
248
|
+
return;
|
|
249
|
+
}
|
|
250
|
+
const timer = setTimeout(() => {
|
|
251
|
+
signal?.removeEventListener("abort", onAbort);
|
|
252
|
+
resolve();
|
|
253
|
+
}, ms);
|
|
254
|
+
const onAbort = () => {
|
|
255
|
+
clearTimeout(timer);
|
|
256
|
+
reject(new DOMException("This operation was aborted", "AbortError"));
|
|
257
|
+
};
|
|
258
|
+
signal?.addEventListener("abort", onAbort, { once: true });
|
|
259
|
+
});
|
|
260
|
+
}
|
|
261
|
+
/**
|
|
262
|
+
* Promise-chain mutex serializing async work within one ToolServer instance.
|
|
263
|
+
* Mirrors Python's asyncio.Lock on ToolServer._execute_gate.
|
|
264
|
+
* Exported for tests that build a ToolServer without its constructor.
|
|
265
|
+
*/
|
|
266
|
+
export class AsyncMutex {
|
|
267
|
+
locked = Promise.resolve();
|
|
268
|
+
async acquire() {
|
|
269
|
+
const previous = this.locked;
|
|
270
|
+
let release;
|
|
271
|
+
this.locked = new Promise((resolve) => {
|
|
272
|
+
release = resolve;
|
|
273
|
+
});
|
|
274
|
+
await previous;
|
|
275
|
+
return release;
|
|
276
|
+
}
|
|
277
|
+
async runExclusive(fn) {
|
|
278
|
+
const release = await this.acquire();
|
|
279
|
+
try {
|
|
280
|
+
return await fn();
|
|
281
|
+
}
|
|
282
|
+
finally {
|
|
283
|
+
release();
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
/**
|
|
287
|
+
* Generator form of runExclusive: the lock is held for the lifetime of the
|
|
288
|
+
* iteration and released when it completes, throws, or is closed early.
|
|
289
|
+
*/
|
|
290
|
+
async *runExclusiveGen(genFactory) {
|
|
291
|
+
const release = await this.acquire();
|
|
292
|
+
try {
|
|
293
|
+
yield* genFactory();
|
|
294
|
+
}
|
|
295
|
+
finally {
|
|
296
|
+
release();
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
/**
|
|
301
|
+
* Normalize adapter stream output into an LLMStreamChunk shape.
|
|
302
|
+
*
|
|
303
|
+
* Mirrors Python's `_coerce_stream_chunk`. The framework adapter
|
|
304
|
+
* (`agent-engine-sdk-langgraph-ts`) translates raw LangChain chunks into the
|
|
305
|
+
* camelCase sdk-core `LLMStreamChunk` contract before yielding (see its
|
|
306
|
+
* `chunkToLlmStreamChunk`), so this reads camelCase keys only. Python reads
|
|
307
|
+
* snake_case because it inspects raw LangChain objects directly; the TS
|
|
308
|
+
* adapter pre-translates, so no snake_case ever reaches here. Always emits
|
|
309
|
+
* camelCase.
|
|
310
|
+
*/
|
|
311
|
+
function coerceStreamChunk(chunk) {
|
|
312
|
+
if (chunk !== null && typeof chunk === "object" && !Array.isArray(chunk)) {
|
|
313
|
+
const obj = chunk;
|
|
314
|
+
// Prefer streaming tool-call chunks, then fall back to merged tool calls.
|
|
315
|
+
const rawChunks = obj["toolCallChunks"] ?? [];
|
|
316
|
+
let toolCalls = rawChunks.length > 0
|
|
317
|
+
? rawChunks.flatMap((tc, index) => {
|
|
318
|
+
if (typeof tc !== "object" || tc === null)
|
|
319
|
+
return [];
|
|
320
|
+
const t = tc;
|
|
321
|
+
return [
|
|
322
|
+
{
|
|
323
|
+
id: t["id"],
|
|
324
|
+
name: t["name"],
|
|
325
|
+
args: normalizeToolCallArgs(t["args"]) ?? undefined,
|
|
326
|
+
type: t["type"],
|
|
327
|
+
index: t["index"] ?? index,
|
|
328
|
+
},
|
|
329
|
+
];
|
|
330
|
+
})
|
|
331
|
+
: undefined;
|
|
332
|
+
if (!toolCalls) {
|
|
333
|
+
const rawCalls = obj["toolCalls"] ?? [];
|
|
334
|
+
const built = rawCalls.flatMap((tc, index) => {
|
|
335
|
+
if (typeof tc !== "object" || tc === null)
|
|
336
|
+
return [];
|
|
337
|
+
const t = tc;
|
|
338
|
+
return [
|
|
339
|
+
{
|
|
340
|
+
id: t["id"],
|
|
341
|
+
name: t["name"],
|
|
342
|
+
args: normalizeToolCallArgs(t["args"] ?? t["arguments"]) ?? undefined,
|
|
343
|
+
type: t["type"],
|
|
344
|
+
index: t["index"] ?? index,
|
|
345
|
+
},
|
|
346
|
+
];
|
|
347
|
+
});
|
|
348
|
+
toolCalls = built.length > 0 ? built : undefined;
|
|
349
|
+
}
|
|
350
|
+
return {
|
|
351
|
+
content: typeof obj["content"] === "string" ? obj["content"] : undefined,
|
|
352
|
+
toolCalls,
|
|
353
|
+
usage: obj["usage"],
|
|
354
|
+
id: obj["id"],
|
|
355
|
+
name: obj["name"],
|
|
356
|
+
responseMetadata: obj["responseMetadata"],
|
|
357
|
+
additionalKwargs: obj["additionalKwargs"],
|
|
358
|
+
};
|
|
359
|
+
}
|
|
360
|
+
return {};
|
|
361
|
+
}
|
|
362
|
+
function streamChunkHasPayload(streamChunk) {
|
|
363
|
+
return Boolean(streamChunk.content ||
|
|
364
|
+
streamChunk.toolCalls ||
|
|
365
|
+
streamChunk.id !== undefined ||
|
|
366
|
+
streamChunk.name !== undefined ||
|
|
367
|
+
streamChunk.responseMetadata !== undefined ||
|
|
368
|
+
streamChunk.additionalKwargs !== undefined);
|
|
369
|
+
}
|
|
370
|
+
// Bound on the entrypoint-derived text echoed to the caller. The text comes
|
|
371
|
+
// from the agent's own code, which makes it useful, but it is unbounded
|
|
372
|
+
// user-authored input on a caller-facing path.
|
|
373
|
+
const MAX_ENTRYPOINT_ERROR_TEXT = 512;
|
|
374
|
+
/**
|
|
375
|
+
* Stands for "the entrypoint load has not failed" in `llmRegistryLoadError`.
|
|
376
|
+
*
|
|
377
|
+
* A unique symbol rather than `null`: JavaScript permits `throw null`, and a
|
|
378
|
+
* null initializer would make that failure indistinguishable from a successful
|
|
379
|
+
* load, so its bare registration error would be reported instead of the real
|
|
380
|
+
* cause. No user throwable can equal this value.
|
|
381
|
+
*/
|
|
382
|
+
const NO_RECORDED_LOAD_ERROR = Symbol("noRecordedLlmRegistryLoadError");
|
|
383
|
+
/**
|
|
384
|
+
* Error class name, for parity with Python's `type(error).__name__`.
|
|
385
|
+
*
|
|
386
|
+
* `error.name` is inherited as "Error" unless a subclass assigns `this.name`,
|
|
387
|
+
* so an unmarked `class ProviderError extends Error {}` would render as
|
|
388
|
+
* "Error" and lose the specificity the Python implementation reports for the
|
|
389
|
+
* same agent code. Prefer an explicit `this.name` override when one is set,
|
|
390
|
+
* otherwise fall back to the constructor's class name.
|
|
391
|
+
*/
|
|
392
|
+
function thrownName(error) {
|
|
393
|
+
const explicit = error.name;
|
|
394
|
+
if (explicit && explicit !== "Error")
|
|
395
|
+
return explicit;
|
|
396
|
+
return error.constructor?.name || explicit || "Error";
|
|
397
|
+
}
|
|
398
|
+
/**
|
|
399
|
+
* Render a thrown value for display. JavaScript permits throwing anything, so
|
|
400
|
+
* this must not assume an Error: `null` becomes "null" rather than the
|
|
401
|
+
* misleading "object: null" that `typeof` would produce.
|
|
402
|
+
*/
|
|
403
|
+
function describeThrown(error) {
|
|
404
|
+
if (error instanceof Error)
|
|
405
|
+
return `${thrownName(error)}: ${error.message}`;
|
|
406
|
+
return String(error);
|
|
407
|
+
}
|
|
408
|
+
/**
|
|
409
|
+
* Render an entrypoint failure for a caller-facing error, redacted.
|
|
410
|
+
*
|
|
411
|
+
* The cause is the agent's own code failing, which is what makes it worth
|
|
412
|
+
* surfacing at all; but its text can echo a connection string, an API key, or a
|
|
413
|
+
* credential-bearing URL. Two layers, both used for caller-facing text in this
|
|
414
|
+
* package: the shared pattern redactor (URL userinfo, bearer tokens, inline
|
|
415
|
+
* secrets), then exact replacement of the credential values a name marks as
|
|
416
|
+
* secrets (`requestNamedCredentialValues`).
|
|
417
|
+
*
|
|
418
|
+
* Both choices are load-bearing:
|
|
419
|
+
*
|
|
420
|
+
* - Patterns run on the pristine text. Replacing values verbatim first can
|
|
421
|
+
* destroy the `://` the userinfo patterns anchor on, leaving the credential
|
|
422
|
+
* in place.
|
|
423
|
+
* - Candidates are selected by variable *name*, not by value length. Tenant env
|
|
424
|
+
* is mostly not secret and includes single-character values ("1", "/"), and a
|
|
425
|
+
* length threshold cannot tell those from a short token -- it drops the token
|
|
426
|
+
* too. Selecting by name keeps "/" out (never redacted) while redacting a
|
|
427
|
+
* 7-character secret (always redacted).
|
|
428
|
+
*
|
|
429
|
+
* Deliberately renders `name: message` rather than `formatLlmError`: that
|
|
430
|
+
* helper unpacks a provider error's `.details`/`.body` wholesale, which is the
|
|
431
|
+
* highest-risk payload to echo, and the error name plus message already
|
|
432
|
+
* carries the actionable part. Mirrors Python's _safe_entrypoint_error_text.
|
|
433
|
+
*/
|
|
434
|
+
function safeEntrypointErrorText(error) {
|
|
435
|
+
let text = redactText(describeThrown(error));
|
|
436
|
+
const values = requestNamedCredentialValues().sort((a, b) => b.length - a.length);
|
|
437
|
+
for (const value of values) {
|
|
438
|
+
text = text.split(value).join("<redacted>");
|
|
439
|
+
}
|
|
440
|
+
if (text.length > MAX_ENTRYPOINT_ERROR_TEXT) {
|
|
441
|
+
text = text.slice(0, MAX_ENTRYPOINT_ERROR_TEXT - 1) + "…";
|
|
442
|
+
}
|
|
443
|
+
return text;
|
|
444
|
+
}
|
|
445
|
+
/**
|
|
446
|
+
* Wire code for a provider credential rejection, else undefined.
|
|
447
|
+
*
|
|
448
|
+
* Deliberately narrow: only a rejection authenticated by the provider's own
|
|
449
|
+
* HTTP status earns the code, so a rate limit, outage, or agent-side bug
|
|
450
|
+
* keeps its existing generic classification instead of being relabeled as a
|
|
451
|
+
* customer-secret problem. Mirrors Python's _llm_failure_error_code.
|
|
452
|
+
*/
|
|
453
|
+
function llmFailureErrorCode(error) {
|
|
454
|
+
return isLlmCredentialRejection(error)
|
|
455
|
+
? LLM_CREDENTIAL_REJECTED_ERROR_CODE
|
|
456
|
+
: undefined;
|
|
457
|
+
}
|
|
458
|
+
/**
|
|
459
|
+
* Thrown when an LLM lookup fails because the entrypoint registry load failed.
|
|
460
|
+
*
|
|
461
|
+
* The named-LLM registry is populated by running the user entrypoint. When that
|
|
462
|
+
* run throws, the registry is left empty (or holding only import-time
|
|
463
|
+
* registrations) and every lookup of an LLM the entrypoint would have
|
|
464
|
+
* registered fails with a bare "not registered" error -- which points the agent
|
|
465
|
+
* developer at their `app.llm()` calls instead of at the transient cause that
|
|
466
|
+
* actually broke the load. The real cause is named in the message only; it is
|
|
467
|
+
* deliberately not attached as `cause`, because the caller-facing response is
|
|
468
|
+
* rendered by `formatLlmError`, which walks `cause` and would dump a raw
|
|
469
|
+
* provider `.details`/`.body` from there, bypassing the message redaction.
|
|
470
|
+
* Mirrors Python's LLMRegistryLoadError.
|
|
471
|
+
*/
|
|
472
|
+
export class LLMRegistryLoadError extends Error {
|
|
473
|
+
constructor(message, options) {
|
|
474
|
+
super(message, options);
|
|
475
|
+
this.name = "LLMRegistryLoadError";
|
|
476
|
+
}
|
|
477
|
+
}
|
|
478
|
+
export class ToolServer extends BaseServer {
|
|
479
|
+
// Serialize the credential-bearing endpoints (/execute, /invoke_llm,
|
|
480
|
+
// /invoke_llm/stream): applyMetadataEnv mutates process-global process.env,
|
|
481
|
+
// so per-call delegated credentials and request context must never
|
|
482
|
+
// overlap. Restriction is off, so the gate is skipped.
|
|
483
|
+
executeGate = new AsyncMutex();
|
|
484
|
+
restrictionDisabled;
|
|
485
|
+
// Guards one-time registry loading across startup and request paths.
|
|
486
|
+
llmRegistryLock = new AsyncMutex();
|
|
487
|
+
llmRegistryLoaded = false;
|
|
488
|
+
// Why the last entrypoint run failed, or NO_RECORDED_LOAD_ERROR after a
|
|
489
|
+
// successful load. Read by createLlmForPod to replace a downstream "not
|
|
490
|
+
// registered" throw with the real cause. Mirrors Python's
|
|
491
|
+
// _llm_registry_load_error, except that the sentinel is distinct from every
|
|
492
|
+
// throwable value: a JavaScript entrypoint may legally `throw null`, and a
|
|
493
|
+
// null initializer would make that failure indistinguishable from success.
|
|
494
|
+
llmRegistryLoadError = NO_RECORDED_LOAD_ERROR;
|
|
495
|
+
constructor(runtime) {
|
|
496
|
+
super(runtime);
|
|
497
|
+
this.restrictionDisabled = true;
|
|
498
|
+
}
|
|
499
|
+
/**
|
|
500
|
+
* Run `fn` under restricted or unrestricted secret handling.
|
|
501
|
+
*
|
|
502
|
+
* Restricted (default): take `executeGate` and apply/restore metadata env.
|
|
503
|
+
* Unrestricted: merge metadata without restore and skip the gate.
|
|
504
|
+
*/
|
|
505
|
+
async withCredentialIsolation(fn) {
|
|
506
|
+
if (this.restrictionDisabled) {
|
|
507
|
+
mergeMetadataEnv();
|
|
508
|
+
return fn();
|
|
509
|
+
}
|
|
510
|
+
return this.executeGate.runExclusive(() => withMetadataEnv(fn));
|
|
511
|
+
}
|
|
512
|
+
/**
|
|
513
|
+
* Streaming variant of `withCredentialIsolation` for `/invoke_llm/stream`.
|
|
514
|
+
*/
|
|
515
|
+
async *withCredentialIsolationGen(genFactory) {
|
|
516
|
+
if (this.restrictionDisabled) {
|
|
517
|
+
mergeMetadataEnv();
|
|
518
|
+
yield* genFactory();
|
|
519
|
+
return;
|
|
520
|
+
}
|
|
521
|
+
yield* this.executeGate.runExclusiveGen(() => withMetadataEnvGen(genFactory));
|
|
522
|
+
}
|
|
523
|
+
get modeName() {
|
|
524
|
+
return "tool";
|
|
525
|
+
}
|
|
526
|
+
async onStartup() {
|
|
527
|
+
// `features.deep_agent` gates the built-in filesystem + shell handlers.
|
|
528
|
+
// Tenants that don't opt in never get a file/shell attack surface
|
|
529
|
+
// registered on their Tool Pod, so the platform stays agent-type-agnostic
|
|
530
|
+
// and tenants pay only for what they use.
|
|
531
|
+
const deepAgentEnabled = this.runtime
|
|
532
|
+
.getAgentConfig()
|
|
533
|
+
.featureEnabled("deep_agent", false);
|
|
534
|
+
if (deepAgentEnabled) {
|
|
535
|
+
registerBuiltinTools(this.runtime);
|
|
536
|
+
const missing = [...BUILTIN_TOOL_NAMES].filter((name) => this.runtime.tools[name] === undefined);
|
|
537
|
+
if (missing.length > 0) {
|
|
538
|
+
throw new Error(`ToolServer startup: missing built-in tool handlers: ${JSON.stringify(missing.sort())}. ` +
|
|
539
|
+
`Expected all of ${JSON.stringify([...BUILTIN_TOOL_NAMES].sort())} to be registered by registerBuiltinTools().`);
|
|
540
|
+
}
|
|
541
|
+
logger.info(`Deep-agent built-in tools registered (features.deep_agent=true): ${JSON.stringify([...BUILTIN_TOOL_NAMES].sort())}`);
|
|
542
|
+
}
|
|
543
|
+
else {
|
|
544
|
+
logger.info("Deep-agent built-in tools NOT registered (features.deep_agent is off in agent.yaml).");
|
|
545
|
+
}
|
|
546
|
+
const tools = Object.keys(this.runtime.tools);
|
|
547
|
+
logger.info(`Registered tools: ${JSON.stringify(tools)}`);
|
|
548
|
+
// Populate the registry while the guest warms, before the first model request.
|
|
549
|
+
await this.ensureLlmRegistryLoaded();
|
|
550
|
+
}
|
|
551
|
+
/**
|
|
552
|
+
* Populate the registry at startup, retrying failed construction on requests.
|
|
553
|
+
* The lock ensures subsequent calls reuse successful preparation.
|
|
554
|
+
*/
|
|
555
|
+
async ensureLlmRegistryLoaded() {
|
|
556
|
+
if (this.llmRegistryLoaded)
|
|
557
|
+
return;
|
|
558
|
+
await this.llmRegistryLock.runExclusive(async () => {
|
|
559
|
+
if (this.llmRegistryLoaded)
|
|
560
|
+
return;
|
|
561
|
+
let loadError = NO_RECORDED_LOAD_ERROR;
|
|
562
|
+
this.llmRegistryLoaded = populateLlmRegistryFromEntrypoint(this.runtime.graphBuilder, logger, {
|
|
563
|
+
entrypointFailed: "Failed to run entrypoint for LLM registration; " +
|
|
564
|
+
"/invoke_llm will fail if no LLM was registered",
|
|
565
|
+
noLlmRegistered: "Entrypoint ran but did not call app.llm(); " +
|
|
566
|
+
"/invoke_llm will fail until an LLM is registered",
|
|
567
|
+
}, (error) => {
|
|
568
|
+
loadError = error;
|
|
569
|
+
});
|
|
570
|
+
// A successful load clears any cause recorded by an earlier failure, so
|
|
571
|
+
// a later lookup miss is reported as a genuine registration error. The
|
|
572
|
+
// sink only fires on failure, so a success leaves the sentinel in place.
|
|
573
|
+
this.llmRegistryLoadError = loadError;
|
|
574
|
+
});
|
|
575
|
+
}
|
|
576
|
+
async onShutdown() {
|
|
577
|
+
// Close all cached TLS agents for graceful shutdown (parity with Python)
|
|
578
|
+
closeAllTLSAgents();
|
|
579
|
+
}
|
|
580
|
+
getHealthDetails() {
|
|
581
|
+
return {
|
|
582
|
+
tools: Object.keys(this.runtime.tools),
|
|
583
|
+
tool_count: Object.keys(this.runtime.tools).length,
|
|
584
|
+
};
|
|
585
|
+
}
|
|
586
|
+
registerRoutes(app) {
|
|
587
|
+
app.post("/execute", async (request, reply) => {
|
|
588
|
+
const body = ToolPodExecuteRequestSchema.parse(request.body);
|
|
589
|
+
// Tool functions receive no AbortSignal (ServerToolFn takes only args),
|
|
590
|
+
// so a drain can only wait this work out: count-only registration, and
|
|
591
|
+
// outliving the deadline is an honest timed_out. Throws 409 if the
|
|
592
|
+
// execution was drained after OE dispatched. The step number makes the
|
|
593
|
+
// call addressable by the per-call interrupt even though tool functions
|
|
594
|
+
// have no signal channel (it reports not_cancellable honestly).
|
|
595
|
+
this.drainRegistry.beginWork(body.execution_id, undefined, body.step_number);
|
|
596
|
+
try {
|
|
597
|
+
const result = await this.handleExecute(body);
|
|
598
|
+
return reply.send(result);
|
|
599
|
+
}
|
|
600
|
+
finally {
|
|
601
|
+
this.drainRegistry.endWork(body.execution_id, undefined, body.step_number);
|
|
602
|
+
}
|
|
603
|
+
});
|
|
604
|
+
app.post("/invoke_llm", async (request, reply) => {
|
|
605
|
+
const body = LLMPodInvokeRequestSchema.parse(request.body);
|
|
606
|
+
const drainController = new AbortController();
|
|
607
|
+
this.drainRegistry.beginWork(body.execution_id, drainController, body.step_number);
|
|
608
|
+
try {
|
|
609
|
+
const result = await this.handleInvokeLlm(body, drainController.signal);
|
|
610
|
+
// Mirror Python's `_sync_usage_with_result` model validator: ensure
|
|
611
|
+
// `result.usage` is populated (OE reads `result["usage"]`), not just the
|
|
612
|
+
// top-level `usage`.
|
|
613
|
+
return reply.send(normalizeLLMPodInvokeResponse(LLMPodInvokeResponseSchema.parse(result)));
|
|
614
|
+
}
|
|
615
|
+
finally {
|
|
616
|
+
this.drainRegistry.endWork(body.execution_id, drainController, body.step_number);
|
|
617
|
+
}
|
|
618
|
+
});
|
|
619
|
+
app.post("/invoke_llm/stream", async (request, reply) => {
|
|
620
|
+
const body = LLMPodInvokeRequestSchema.parse(request.body);
|
|
621
|
+
// Checked here so a drained execution gets a 409 rather than an SSE
|
|
622
|
+
// error event after the stream has already started.
|
|
623
|
+
this.drainRegistry.checkAdmission(body.execution_id);
|
|
624
|
+
reply.hijack();
|
|
625
|
+
const res = reply.raw;
|
|
626
|
+
res.setHeader("Content-Type", "text/event-stream");
|
|
627
|
+
res.setHeader("Cache-Control", "no-cache");
|
|
628
|
+
res.setHeader("Connection", "keep-alive");
|
|
629
|
+
res.flushHeaders();
|
|
630
|
+
const drainController = new AbortController();
|
|
631
|
+
this.drainRegistry.beginWork(body.execution_id, drainController, body.step_number);
|
|
632
|
+
// The registration is released when the generator has truly settled,
|
|
633
|
+
// not when the socket closes: a signal-ignoring generator can outlive
|
|
634
|
+
// force-close, and a drain must never report completed while provider
|
|
635
|
+
// work is still pending — it times out honestly instead (mirrors the
|
|
636
|
+
// Python twin, whose end_work runs only when the task unwinds).
|
|
637
|
+
//
|
|
638
|
+
// Hijack removes Fastify lifecycle management; pipe owns disconnect
|
|
639
|
+
// cancel, backpressure, and safe end so a dropped client cannot crash
|
|
640
|
+
// the shared tool pod. The drain signal rides
|
|
641
|
+
// the pipe's combined signal, so a drain unblocks the iterator even
|
|
642
|
+
// when the LLM adapter ignores the signal it was handed.
|
|
643
|
+
await pipeSseLinesToResponse(res, (signal) => this.handleInvokeLlmStream(body, signal), {
|
|
644
|
+
signal: drainController.signal,
|
|
645
|
+
onGeneratorSettled: () => this.drainRegistry.endWork(body.execution_id, drainController, body.step_number),
|
|
646
|
+
});
|
|
647
|
+
});
|
|
648
|
+
app.get("/tools", async () => ({
|
|
649
|
+
tools: Object.entries(this.runtime.toolDefinitions).map(([name, defn]) => ({ name, ...defn })),
|
|
650
|
+
count: Object.keys(this.runtime.toolDefinitions).length,
|
|
651
|
+
}));
|
|
652
|
+
app.post("/guardrails/check", async (request, reply) => {
|
|
653
|
+
const body = GuardrailCheckRequestSchema.parse(request.body);
|
|
654
|
+
// Count-only registration: the evaluation is synchronous and fast, so
|
|
655
|
+
// there is no abort channel to signal; a drain waits it out.
|
|
656
|
+
this.drainRegistry.beginWork(body.execution_id);
|
|
657
|
+
try {
|
|
658
|
+
return reply.send(await this.handleGuardrailsCheck(body));
|
|
659
|
+
}
|
|
660
|
+
finally {
|
|
661
|
+
this.drainRegistry.endWork(body.execution_id);
|
|
662
|
+
}
|
|
663
|
+
});
|
|
664
|
+
}
|
|
665
|
+
handleGuardrailsCheck = withMetrics("tool_pod_guardrails_check", (request) => evaluateGuardrailCheck(request), { recordErrors: false });
|
|
666
|
+
handleExecute = withMetrics("tool_pod_execute", (request) => this.doHandleExecute(request), { recordErrors: false });
|
|
667
|
+
async doHandleExecute(request) {
|
|
668
|
+
const podName = os.hostname();
|
|
669
|
+
logToolRequest(request.tool_name ?? "(empty)", request.arguments, 0, "TOOL_POD", toolRedactFields(this.runtime.toolDefinitions, request.tool_name));
|
|
670
|
+
if (!request.tool_name) {
|
|
671
|
+
return {
|
|
672
|
+
status: "error",
|
|
673
|
+
error: "Tool name cannot be empty",
|
|
674
|
+
pod_name: podName,
|
|
675
|
+
metadata: {},
|
|
676
|
+
};
|
|
677
|
+
}
|
|
678
|
+
const resolved = this.resolveTool(request.tool_name, request.metadata);
|
|
679
|
+
if ("error" in resolved) {
|
|
680
|
+
return {
|
|
681
|
+
status: "error",
|
|
682
|
+
error: resolved.error,
|
|
683
|
+
pod_name: podName,
|
|
684
|
+
metadata: {},
|
|
685
|
+
};
|
|
686
|
+
}
|
|
687
|
+
const func = resolved.func;
|
|
688
|
+
// Pin the callback base to the runner's own configured OE (deploy-time
|
|
689
|
+
// OE_URL) before it is used as the callback base or the owner-URL trust
|
|
690
|
+
// anchor. The request field is only honoured when no OE_URL is stamped
|
|
691
|
+
// (local dev / tests). Resolving once here means the owner URL is validated
|
|
692
|
+
// against a trusted anchor, not the raw request value.
|
|
693
|
+
const oeUrl = resolveOeUrl(request.oe_url ?? "");
|
|
694
|
+
return this.withCredentialIsolation(() => runWithExecutionContext({
|
|
695
|
+
executionId: request.execution_id,
|
|
696
|
+
traceId: request.platform_trace_id,
|
|
697
|
+
wrapper: null,
|
|
698
|
+
oeUrl,
|
|
699
|
+
// Emit() from the tool prefers this owner replica, validated against
|
|
700
|
+
// the trusted resolved oeUrl above.
|
|
701
|
+
oeOwnerUrl: resolveOwnerUrl(request.oe_owner_url, oeUrl),
|
|
702
|
+
userId: request.user_id,
|
|
703
|
+
sessionId: request.session_id,
|
|
704
|
+
authorization: request.authorization,
|
|
705
|
+
customHeaders: request.custom_headers,
|
|
706
|
+
payload: request.payload,
|
|
707
|
+
}, async () => {
|
|
708
|
+
// Sync tools run on the event loop (no asyncio.to_thread equivalent
|
|
709
|
+
// in TS). This diverges from Python which off-loads sync functions to
|
|
710
|
+
// a thread pool.
|
|
711
|
+
try {
|
|
712
|
+
const maybePromise = runWithCustomerOrigin(() => func(request.arguments));
|
|
713
|
+
const result = maybePromise instanceof Promise
|
|
714
|
+
? await maybePromise
|
|
715
|
+
: maybePromise;
|
|
716
|
+
const metadata = getCurrentExecutionMetadata();
|
|
717
|
+
// Author-intended suspend is signaled out of band via
|
|
718
|
+
// suspendPayloadToJson, never inferred from result content — so
|
|
719
|
+
// relayed untrusted data cannot forge a HITL suspend.
|
|
720
|
+
const suspendMarker = getRequestedSuspend();
|
|
721
|
+
if (suspendMarker !== null) {
|
|
722
|
+
return {
|
|
723
|
+
status: "suspend",
|
|
724
|
+
result: JSON.stringify(suspendMarker),
|
|
725
|
+
pod_name: podName,
|
|
726
|
+
kind: metadata["memory"] ? "memory" : undefined,
|
|
727
|
+
metadata,
|
|
728
|
+
oob_suspend_supported: true,
|
|
729
|
+
};
|
|
730
|
+
}
|
|
731
|
+
return {
|
|
732
|
+
status: "success",
|
|
733
|
+
result,
|
|
734
|
+
pod_name: podName,
|
|
735
|
+
kind: metadata["memory"] ? "memory" : undefined,
|
|
736
|
+
metadata,
|
|
737
|
+
oob_suspend_supported: true,
|
|
738
|
+
};
|
|
739
|
+
}
|
|
740
|
+
catch (e) {
|
|
741
|
+
return executeErrorResponse(e, this.runtime.toolDefinitions[request.tool_name], podName, getCurrentExecutionMetadata(), requestCredentialValues());
|
|
742
|
+
}
|
|
743
|
+
}));
|
|
744
|
+
}
|
|
745
|
+
/**
|
|
746
|
+
* Resolve a tool name to a callable, falling back to a configured MCP tool.
|
|
747
|
+
*
|
|
748
|
+
* Mirrors Python's `ToolServer._resolve_tool()`. AER discovers remote MCP
|
|
749
|
+
* schemas at startup and forwards `mcp_server`/`mcp_tool` call metadata
|
|
750
|
+
* when invoking a Tool-Pod-dispatched MCP tool; the direct registry lookup
|
|
751
|
+
* misses for those names because MCP tools aren't pre-registered on the
|
|
752
|
+
* Tool Pod side (it skips `tools/list` startup discovery).
|
|
753
|
+
*/
|
|
754
|
+
resolveTool(toolName, metadata) {
|
|
755
|
+
const direct = this.runtime.tools[toolName];
|
|
756
|
+
if (direct)
|
|
757
|
+
return { func: direct };
|
|
758
|
+
const mcpConfig = this.runtime.getAgentConfig().mcp;
|
|
759
|
+
const mcpServerName = metadata["mcp_server"];
|
|
760
|
+
const mcpToolName = metadata["mcp_tool"];
|
|
761
|
+
if (typeof mcpServerName === "string" && typeof mcpToolName === "string") {
|
|
762
|
+
try {
|
|
763
|
+
const binding = resolveConfiguredMcpToolBinding(mcpConfig, toolName, mcpServerName, mcpToolName);
|
|
764
|
+
if (binding === null) {
|
|
765
|
+
return { error: `Unknown tool: ${toolName}` };
|
|
766
|
+
}
|
|
767
|
+
const callable = makeMcpToolCallable(binding);
|
|
768
|
+
return { func: (args) => callable(args) };
|
|
769
|
+
}
|
|
770
|
+
catch (err) {
|
|
771
|
+
if (err instanceof MCPConfigError)
|
|
772
|
+
return { error: err.message };
|
|
773
|
+
throw err;
|
|
774
|
+
}
|
|
775
|
+
}
|
|
776
|
+
if (isConfiguredMcpSdkToolName(mcpConfig, toolName)) {
|
|
777
|
+
return {
|
|
778
|
+
error: `Tool '${toolName}' is a configured MCP tool but the request did ` +
|
|
779
|
+
"not include mcp_server/mcp_tool call metadata",
|
|
780
|
+
};
|
|
781
|
+
}
|
|
782
|
+
return { error: `Unknown tool: ${toolName}` };
|
|
783
|
+
}
|
|
784
|
+
/**
|
|
785
|
+
* Create a BaseLLM-compatible instance for tool pod execution.
|
|
786
|
+
*
|
|
787
|
+
* Retrieves the named LLM from the registry (registered via
|
|
788
|
+
* `app.llm(llm, { llmId })`) and wraps it through the registered adapter
|
|
789
|
+
* factory. No SecureWrappedLLM needed — OE approval already happened in AER.
|
|
790
|
+
*/
|
|
791
|
+
createLlmForPod(llmId, tools, toolChoice) {
|
|
792
|
+
let llm;
|
|
793
|
+
try {
|
|
794
|
+
llm = getNamedLlm(llmId);
|
|
795
|
+
}
|
|
796
|
+
catch (error) {
|
|
797
|
+
const loadError = this.llmRegistryLoadError;
|
|
798
|
+
if (loadError === NO_RECORDED_LOAD_ERROR)
|
|
799
|
+
throw error;
|
|
800
|
+
// The lookup miss is a symptom of a failed entrypoint load, so reporting
|
|
801
|
+
// it as a missing app.llm() call would misdirect the agent developer.
|
|
802
|
+
// Surface the real cause instead -- redacted, since this text is returned
|
|
803
|
+
// to the caller. This branch is reachable only for a lookup that already
|
|
804
|
+
// failed, so it can never turn a working call into a failure -- in
|
|
805
|
+
// particular the import-time-snapshot path still resolves an llmId the
|
|
806
|
+
// snapshot holds even when the entrypoint run threw.
|
|
807
|
+
//
|
|
808
|
+
// No `cause`: the caller-facing error is rendered by formatLlmError,
|
|
809
|
+
// which walks `error.cause` and would dump a provider error's raw
|
|
810
|
+
// `.details`/`.body` from there, bypassing the redaction in the message.
|
|
811
|
+
throw new LLMRegistryLoadError(`llm_id ${JSON.stringify(llmId)} is not registered because the agent ` +
|
|
812
|
+
`entrypoint failed when the LLM registry was loaded. The next LLM ` +
|
|
813
|
+
`call re-runs the entrypoint, so this recovers once the underlying ` +
|
|
814
|
+
`cause clears. Underlying failure: ${safeEntrypointErrorText(loadError)}`);
|
|
815
|
+
}
|
|
816
|
+
// tool_choice forces a specific bound tool (e.g. the schema bound by
|
|
817
|
+
// withStructuredOutput); forward it so the adapter's bindTools call can
|
|
818
|
+
// translate it. OE approval already happened in AER.
|
|
819
|
+
return getLLMAdapterFactory()(llm, {
|
|
820
|
+
tools,
|
|
821
|
+
tool_choice: toolChoice,
|
|
822
|
+
});
|
|
823
|
+
}
|
|
824
|
+
handleInvokeLlm = withMetrics("llm_pod_invoke", (request, signal) => this.doHandleInvokeLlm(request, signal), { recordErrors: false });
|
|
825
|
+
async doHandleInvokeLlm(request, signal) {
|
|
826
|
+
const podName = os.hostname();
|
|
827
|
+
const startTime = performance.now();
|
|
828
|
+
try {
|
|
829
|
+
// Usually a no-op after startup; keep any registry load inside the
|
|
830
|
+
// request's credential isolation, as with model execution below.
|
|
831
|
+
const chunks = await runWithExecutionContext({
|
|
832
|
+
executionId: request.execution_id ?? "",
|
|
833
|
+
wrapper: null,
|
|
834
|
+
oeUrl: "",
|
|
835
|
+
traceId: request.platform_trace_id,
|
|
836
|
+
}, () => this.withCredentialIsolation(async () => {
|
|
837
|
+
await this.ensureLlmRegistryLoaded();
|
|
838
|
+
const collected = [];
|
|
839
|
+
for await (const chunk of this.streamLlmChunks(request, signal)) {
|
|
840
|
+
collected.push(chunk);
|
|
841
|
+
}
|
|
842
|
+
return collected;
|
|
843
|
+
}));
|
|
844
|
+
const llmResponse = SecureLLMProxy.responseFromStreamChunks(chunks);
|
|
845
|
+
const durationMs = performance.now() - startTime;
|
|
846
|
+
return {
|
|
847
|
+
status: "success",
|
|
848
|
+
result: {
|
|
849
|
+
content: llmResponse.content,
|
|
850
|
+
tool_calls: (llmResponse.toolCalls?.length ?? 0) > 0
|
|
851
|
+
? llmResponse.toolCalls
|
|
852
|
+
: undefined,
|
|
853
|
+
metadata: llmResponse.metadata,
|
|
854
|
+
id: llmResponse.id,
|
|
855
|
+
name: llmResponse.name,
|
|
856
|
+
additional_kwargs: llmResponse.additionalKwargs,
|
|
857
|
+
response_metadata: llmResponse.responseMetadata,
|
|
858
|
+
},
|
|
859
|
+
pod_name: podName,
|
|
860
|
+
duration_ms: durationMs,
|
|
861
|
+
usage: llmResponse.usage,
|
|
862
|
+
};
|
|
863
|
+
}
|
|
864
|
+
catch (e) {
|
|
865
|
+
const durationMs = performance.now() - startTime;
|
|
866
|
+
return {
|
|
867
|
+
status: "error",
|
|
868
|
+
error: formatLlmError(e),
|
|
869
|
+
error_code: llmFailureErrorCode(e),
|
|
870
|
+
pod_name: podName,
|
|
871
|
+
duration_ms: durationMs,
|
|
872
|
+
};
|
|
873
|
+
}
|
|
874
|
+
}
|
|
875
|
+
/**
|
|
876
|
+
* ensureLlmRegistryLoaded() then streamLlmChunks(), both inside the
|
|
877
|
+
* request's credential isolation window. Registry loading is normally a
|
|
878
|
+
* no-op after startup; the gate still covers the complete stream.
|
|
879
|
+
*/
|
|
880
|
+
async *loadRegistryThenStreamLlmChunks(request, signal) {
|
|
881
|
+
yield* this.registryLoadedLlmChunks(request, signal);
|
|
882
|
+
}
|
|
883
|
+
async *registryLoadedLlmChunks(request, signal) {
|
|
884
|
+
await this.ensureLlmRegistryLoaded();
|
|
885
|
+
yield* this.streamLlmChunks(request, signal);
|
|
886
|
+
}
|
|
887
|
+
async *streamLlmChunks(request, signal) {
|
|
888
|
+
const invokeArgs = request.arguments;
|
|
889
|
+
const llm = this.createLlmForPod(invokeArgs.llm_id, invokeArgs.tools, invokeArgs.tool_choice);
|
|
890
|
+
const messages = invokeArgs.messages;
|
|
891
|
+
const extraKwargs = invokeArgs.options?.toModelKwargs() ?? {};
|
|
892
|
+
// `stop` is the wire key the LLM adapter expects (Python Pydantic serialization alias).
|
|
893
|
+
extraKwargs["stop"] = invokeArgs.stop_sequences;
|
|
894
|
+
// Forward abort so LangChain adapters that honor `signal` can cancel the
|
|
895
|
+
// in-flight provider request (best-effort; not all adapters support it).
|
|
896
|
+
if (signal !== undefined) {
|
|
897
|
+
extraKwargs["signal"] = signal;
|
|
898
|
+
}
|
|
899
|
+
let lastError = null;
|
|
900
|
+
let streamStarted = false;
|
|
901
|
+
for (let attempt = 0; attempt < LLM_MAX_RETRIES; attempt++) {
|
|
902
|
+
let usage;
|
|
903
|
+
if (signal?.aborted) {
|
|
904
|
+
throw new DOMException("This operation was aborted", "AbortError");
|
|
905
|
+
}
|
|
906
|
+
try {
|
|
907
|
+
const stream = llm.stream(messages, extraKwargs);
|
|
908
|
+
for await (const chunk of stream) {
|
|
909
|
+
if (signal?.aborted) {
|
|
910
|
+
throw new DOMException("This operation was aborted", "AbortError");
|
|
911
|
+
}
|
|
912
|
+
const streamChunk = coerceStreamChunk(chunk);
|
|
913
|
+
if (streamChunkHasPayload(streamChunk)) {
|
|
914
|
+
streamStarted = true;
|
|
915
|
+
yield streamChunk;
|
|
916
|
+
}
|
|
917
|
+
if (streamChunk.usage !== undefined) {
|
|
918
|
+
usage = mergeTokenUsage(usage, streamChunk.usage) ?? usage;
|
|
919
|
+
}
|
|
920
|
+
}
|
|
921
|
+
if (usage !== undefined) {
|
|
922
|
+
yield { usage };
|
|
923
|
+
}
|
|
924
|
+
return;
|
|
925
|
+
}
|
|
926
|
+
catch (e) {
|
|
927
|
+
if (isAbortError(e) || signal?.aborted) {
|
|
928
|
+
throw isAbortError(e)
|
|
929
|
+
? e
|
|
930
|
+
: new DOMException("This operation was aborted", "AbortError");
|
|
931
|
+
}
|
|
932
|
+
lastError = e;
|
|
933
|
+
if (!streamStarted &&
|
|
934
|
+
isRetryableError(e) &&
|
|
935
|
+
attempt < LLM_MAX_RETRIES - 1) {
|
|
936
|
+
const backoff = Math.min(LLM_INITIAL_BACKOFF * Math.pow(LLM_BACKOFF_MULTIPLIER, attempt), LLM_MAX_BACKOFF);
|
|
937
|
+
logger.warn(`LLM Pod Stream: Transient provider failure (attempt ${attempt + 1}/${LLM_MAX_RETRIES}). ` +
|
|
938
|
+
`Retrying in ${backoff.toFixed(1)}s...`);
|
|
939
|
+
await sleep(backoff * 1000, signal);
|
|
940
|
+
continue;
|
|
941
|
+
}
|
|
942
|
+
break;
|
|
943
|
+
}
|
|
944
|
+
}
|
|
945
|
+
if (lastError !== null)
|
|
946
|
+
throw lastError;
|
|
947
|
+
throw new Error("Failed to create LLM stream");
|
|
948
|
+
}
|
|
949
|
+
async *handleInvokeLlmStream(request, signal) {
|
|
950
|
+
const podName = os.hostname();
|
|
951
|
+
const startTime = performance.now();
|
|
952
|
+
const ctx = {
|
|
953
|
+
executionId: request.execution_id ?? "",
|
|
954
|
+
wrapper: null,
|
|
955
|
+
oeUrl: "",
|
|
956
|
+
traceId: request.platform_trace_id,
|
|
957
|
+
};
|
|
958
|
+
const gen = this.invokeLlmStreamEvents(request, signal, podName, startTime);
|
|
959
|
+
try {
|
|
960
|
+
while (true) {
|
|
961
|
+
const step = await runWithExecutionContext(ctx, () => gen.next());
|
|
962
|
+
if (step.done) {
|
|
963
|
+
return;
|
|
964
|
+
}
|
|
965
|
+
yield step.value;
|
|
966
|
+
}
|
|
967
|
+
}
|
|
968
|
+
finally {
|
|
969
|
+
await gen.return(undefined);
|
|
970
|
+
}
|
|
971
|
+
}
|
|
972
|
+
async *invokeLlmStreamEvents(request, signal, podName, startTime) {
|
|
973
|
+
try {
|
|
974
|
+
let usage;
|
|
975
|
+
// See the same-gate rationale in doHandleInvokeLlm above.
|
|
976
|
+
for await (const streamChunk of this.withCredentialIsolationGen(() => this.loadRegistryThenStreamLlmChunks(request, signal))) {
|
|
977
|
+
if (!streamChunkHasPayload(streamChunk)) {
|
|
978
|
+
if (streamChunk.usage !== undefined) {
|
|
979
|
+
usage = streamChunk.usage;
|
|
980
|
+
}
|
|
981
|
+
continue;
|
|
982
|
+
}
|
|
983
|
+
const chunkEvent = {
|
|
984
|
+
content: streamChunk.content ?? undefined,
|
|
985
|
+
tool_call_chunks: streamChunk.toolCalls ?? undefined,
|
|
986
|
+
id: streamChunk.id,
|
|
987
|
+
name: streamChunk.name,
|
|
988
|
+
response_metadata: streamChunk.responseMetadata,
|
|
989
|
+
additional_kwargs: streamChunk.additionalKwargs,
|
|
990
|
+
};
|
|
991
|
+
yield `data: ${JSON.stringify(LLMPodStreamEventSchema.parse(chunkEvent))}\n\n`;
|
|
992
|
+
}
|
|
993
|
+
const durationMs = performance.now() - startTime;
|
|
994
|
+
const summary = {
|
|
995
|
+
done: true,
|
|
996
|
+
pod_name: podName,
|
|
997
|
+
duration_ms: durationMs,
|
|
998
|
+
usage,
|
|
999
|
+
};
|
|
1000
|
+
yield `data: ${JSON.stringify(LLMPodStreamEventSchema.parse(summary))}\n\n`;
|
|
1001
|
+
}
|
|
1002
|
+
catch (e) {
|
|
1003
|
+
// Abort is expected on peer drop / drain timeout — no error frame.
|
|
1004
|
+
if (isAbortError(e) || signal?.aborted) {
|
|
1005
|
+
return;
|
|
1006
|
+
}
|
|
1007
|
+
// Log even when the SSE consumer is already gone: otherwise adapter
|
|
1008
|
+
// teardown failures after disconnect are invisible server-side.
|
|
1009
|
+
logger.warn(`LLM stream failed: ${e instanceof Error ? e.message : String(e)}`);
|
|
1010
|
+
const errorEvent = {
|
|
1011
|
+
error: formatLlmError(e),
|
|
1012
|
+
error_code: llmFailureErrorCode(e),
|
|
1013
|
+
};
|
|
1014
|
+
yield `data: ${JSON.stringify(LLMPodStreamEventSchema.parse(errorEvent))}\n\n`;
|
|
1015
|
+
}
|
|
1016
|
+
}
|
|
1017
|
+
}
|