@velum-labs/routekit-gateway 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +28 -0
  3. package/dist/acp-agent.d.ts +38 -0
  4. package/dist/acp-agent.js +142 -0
  5. package/dist/acp-registry.d.ts +36 -0
  6. package/dist/acp-registry.js +85 -0
  7. package/dist/adapters/anthropic.d.ts +131 -0
  8. package/dist/adapters/anthropic.js +1195 -0
  9. package/dist/adapters/chat.d.ts +14 -0
  10. package/dist/adapters/chat.js +34 -0
  11. package/dist/adapters/cursor.d.ts +34 -0
  12. package/dist/adapters/cursor.js +305 -0
  13. package/dist/adapters/dropped.d.ts +10 -0
  14. package/dist/adapters/dropped.js +24 -0
  15. package/dist/adapters/openai-chat-wire.d.ts +93 -0
  16. package/dist/adapters/openai-chat-wire.js +143 -0
  17. package/dist/adapters/responses-stream.d.ts +7 -0
  18. package/dist/adapters/responses-stream.js +597 -0
  19. package/dist/adapters/responses.d.ts +174 -0
  20. package/dist/adapters/responses.js +778 -0
  21. package/dist/adapters/server-tool-loop.d.ts +94 -0
  22. package/dist/adapters/server-tool-loop.js +477 -0
  23. package/dist/adapters/upstream-error.d.ts +14 -0
  24. package/dist/adapters/upstream-error.js +25 -0
  25. package/dist/adapters/validate.d.ts +27 -0
  26. package/dist/adapters/validate.js +180 -0
  27. package/dist/adapters/web-search.d.ts +46 -0
  28. package/dist/adapters/web-search.js +151 -0
  29. package/dist/auth.d.ts +10 -0
  30. package/dist/auth.js +28 -0
  31. package/dist/backend.d.ts +151 -0
  32. package/dist/backend.js +143 -0
  33. package/dist/capacity-pool.d.ts +31 -0
  34. package/dist/capacity-pool.js +99 -0
  35. package/dist/cost.d.ts +49 -0
  36. package/dist/cost.js +112 -0
  37. package/dist/endpoint-health.d.ts +54 -0
  38. package/dist/endpoint-health.js +123 -0
  39. package/dist/index.d.ts +40 -0
  40. package/dist/index.js +23 -0
  41. package/dist/provenance.d.ts +31 -0
  42. package/dist/provenance.js +191 -0
  43. package/dist/provider-backends.d.ts +40 -0
  44. package/dist/provider-backends.js +1050 -0
  45. package/dist/provider-source.d.ts +40 -0
  46. package/dist/provider-source.js +293 -0
  47. package/dist/router.d.ts +168 -0
  48. package/dist/router.js +474 -0
  49. package/dist/server.d.ts +67 -0
  50. package/dist/server.js +930 -0
  51. package/dist/sse/chat-assembler.d.ts +45 -0
  52. package/dist/sse/chat-assembler.js +190 -0
  53. package/dist/sse/parse.d.ts +50 -0
  54. package/dist/sse/parse.js +149 -0
  55. package/dist/sse-wire.d.ts +10 -0
  56. package/dist/sse-wire.js +31 -0
  57. package/dist/switching-proxy.d.ts +15 -0
  58. package/dist/switching-proxy.js +232 -0
  59. package/dist/test/acp-agent.test.d.ts +1 -0
  60. package/dist/test/acp-agent.test.js +66 -0
  61. package/dist/test/acp-registry.test.d.ts +1 -0
  62. package/dist/test/acp-registry.test.js +70 -0
  63. package/dist/test/anthropic.test.d.ts +1 -0
  64. package/dist/test/anthropic.test.js +793 -0
  65. package/dist/test/auth.test.d.ts +1 -0
  66. package/dist/test/auth.test.js +25 -0
  67. package/dist/test/boundary.test.d.ts +1 -0
  68. package/dist/test/boundary.test.js +32 -0
  69. package/dist/test/chat.test.d.ts +1 -0
  70. package/dist/test/chat.test.js +418 -0
  71. package/dist/test/cost.test.d.ts +1 -0
  72. package/dist/test/cost.test.js +60 -0
  73. package/dist/test/cursor.test.d.ts +1 -0
  74. package/dist/test/cursor.test.js +100 -0
  75. package/dist/test/drain.test.d.ts +1 -0
  76. package/dist/test/drain.test.js +116 -0
  77. package/dist/test/dropped.test.d.ts +1 -0
  78. package/dist/test/dropped.test.js +80 -0
  79. package/dist/test/endpoint-health.test.d.ts +1 -0
  80. package/dist/test/endpoint-health.test.js +73 -0
  81. package/dist/test/provenance.test.d.ts +1 -0
  82. package/dist/test/provenance.test.js +176 -0
  83. package/dist/test/provider-backends.test.d.ts +1 -0
  84. package/dist/test/provider-backends.test.js +699 -0
  85. package/dist/test/responses.test.d.ts +1 -0
  86. package/dist/test/responses.test.js +813 -0
  87. package/dist/test/routed-backend.test.d.ts +1 -0
  88. package/dist/test/routed-backend.test.js +39 -0
  89. package/dist/test/router.test.d.ts +1 -0
  90. package/dist/test/router.test.js +297 -0
  91. package/dist/test/server-resilience.test.d.ts +1 -0
  92. package/dist/test/server-resilience.test.js +169 -0
  93. package/dist/test/sse-codec.test.d.ts +1 -0
  94. package/dist/test/sse-codec.test.js +186 -0
  95. package/dist/test/web-search-loop.test.d.ts +1 -0
  96. package/dist/test/web-search-loop.test.js +469 -0
  97. package/dist/test/wire-validation.test.d.ts +1 -0
  98. package/dist/test/wire-validation.test.js +140 -0
  99. package/package.json +48 -0
@@ -0,0 +1,94 @@
1
+ /**
2
+ * The server-tool inner loop (gateway-executed web search).
3
+ *
4
+ * When the upstream model calls a *server-executed* tool (today: `web_search`),
5
+ * nobody on the caller's side can answer it — the caller declared the tool
6
+ * expecting the "server" to run it. This loop makes the gateway that server:
7
+ * it intercepts server-tool calls from a model step, executes them via a
8
+ * {@link WebSearchExecutor}, appends the exchange to the chat transcript, and
9
+ * runs another model step — repeating until a step commits to something the
10
+ * caller can actually handle (text, client tool calls, or a clean stop).
11
+ *
12
+ * The loop operates at the chat-completions layer, around `backend.chat`:
13
+ * each inner step is an ordinary backend turn, exactly as
14
+ * if the caller had executed a client tool and come back. The dialect egress
15
+ * translators stay single-stream: in streaming mode the loop composes the
16
+ * steps' chat SSE into one continuous stream, suppressing the server-tool
17
+ * fragments and injecting {@link ServerToolMarker} chunks (an in-process
18
+ * convention) that the translators render as their dialect's native search
19
+ * items (`web_search_call` / `server_tool_use` + `web_search_tool_result`).
20
+ *
21
+ * Mixed batches (server + client calls in one step) terminate the turn: the
22
+ * client calls surface and the server calls are dropped un-executed — results
23
+ * could not be fed back into a turn that just ended, and the upstream model can
24
+ * simply re-issue the search next turn.
25
+ */
26
+ import { type AnthropicReasoningDetail } from "./openai-chat-wire.js";
27
+ import type { WebSearchExecutor, WebSearchOutcome } from "./web-search.js";
28
+ /** In-process marker chunk field the loop injects between composed steps. */
29
+ export declare const SERVER_TOOL_MARKER_FIELD = "routekit_server_tool";
30
+ export type ServerToolMarker = {
31
+ kind: "web_search";
32
+ phase: "start" | "done";
33
+ item_id: string;
34
+ query: string;
35
+ status?: "completed" | "failed";
36
+ /** Anthropic-native result blocks for the Anthropic egress (done phase). */
37
+ result_blocks?: unknown[];
38
+ };
39
+ /** The marker on a parsed chat chunk, if present. */
40
+ export declare function serverToolMarkerOf(chunk: unknown): ServerToolMarker | undefined;
41
+ export type ExecutedSearch = {
42
+ itemId: string;
43
+ query: string;
44
+ status: "completed" | "failed";
45
+ outcome?: WebSearchOutcome;
46
+ };
47
+ export type ServerToolLoopEvent = {
48
+ kind: "reasoning";
49
+ details: AnthropicReasoningDetail[];
50
+ } | {
51
+ kind: "search";
52
+ search: ExecutedSearch;
53
+ };
54
+ export type ServerToolLoopOptions = {
55
+ /** The translated chat body; the loop appends search exchanges to `messages`. */
56
+ chat: Record<string, unknown>;
57
+ runStep: (chat: Record<string, unknown>) => Promise<Response>;
58
+ serverToolNames: ReadonlySet<string>;
59
+ executor: WebSearchExecutor;
60
+ maxSearches?: number;
61
+ signal?: AbortSignal;
62
+ };
63
+ export type BufferedLoopOutcome = {
64
+ kind: "openai";
65
+ openai: Record<string, unknown>;
66
+ searches: ExecutedSearch[];
67
+ events: ServerToolLoopEvent[];
68
+ } | {
69
+ kind: "upstream_error";
70
+ response: Response;
71
+ };
72
+ /**
73
+ * Run the loop over buffered (non-streaming) model steps. `firstStep` is the
74
+ * already-awaited first model step (the handler surfaces its HTTP errors
75
+ * before entering the loop). Returns the terminal step's OpenAI payload (with
76
+ * any un-executable mixed-batch server calls stripped) plus the searches
77
+ * executed along the way, for the dialect egress to render as native items.
78
+ */
79
+ export declare function runBufferedServerToolLoop(options: ServerToolLoopOptions & {
80
+ firstStep: Response;
81
+ }): Promise<BufferedLoopOutcome>;
82
+ /**
83
+ * Compose the loop's model steps into one continuous chat SSE stream.
84
+ *
85
+ * `firstStep` is the already-awaited first model step (the handler surfaces
86
+ * its HTTP errors exactly as the single-step path does). Server-tool call
87
+ * fragments are suppressed from the forwarded stream; each executed search is
88
+ * injected as a pair of {@link ServerToolMarker} chunks for the dialect
89
+ * translator. Per-step usage is withheld and re-emitted summed before the
90
+ * terminal finish chunk, so the client-visible usage covers the whole loop.
91
+ */
92
+ export declare function composeServerToolStream(options: ServerToolLoopOptions & {
93
+ firstStep: Response;
94
+ }): ReadableStream<Uint8Array>;
@@ -0,0 +1,477 @@
1
+ /**
2
+ * The server-tool inner loop (gateway-executed web search).
3
+ *
4
+ * When the upstream model calls a *server-executed* tool (today: `web_search`),
5
+ * nobody on the caller's side can answer it — the caller declared the tool
6
+ * expecting the "server" to run it. This loop makes the gateway that server:
7
+ * it intercepts server-tool calls from a model step, executes them via a
8
+ * {@link WebSearchExecutor}, appends the exchange to the chat transcript, and
9
+ * runs another model step — repeating until a step commits to something the
10
+ * caller can actually handle (text, client tool calls, or a clean stop).
11
+ *
12
+ * The loop operates at the chat-completions layer, around `backend.chat`:
13
+ * each inner step is an ordinary backend turn, exactly as
14
+ * if the caller had executed a client tool and come back. The dialect egress
15
+ * translators stay single-stream: in streaming mode the loop composes the
16
+ * steps' chat SSE into one continuous stream, suppressing the server-tool
17
+ * fragments and injecting {@link ServerToolMarker} chunks (an in-process
18
+ * convention) that the translators render as their dialect's native search
19
+ * items (`web_search_call` / `server_tool_use` + `web_search_tool_result`).
20
+ *
21
+ * Mixed batches (server + client calls in one step) terminate the turn: the
22
+ * client calls surface and the server calls are dropped un-executed — results
23
+ * could not be fed back into a turn that just ended, and the upstream model can
24
+ * simply re-issue the search next turn.
25
+ */
26
+ import { randomId } from "@velum-labs/routekit-runtime";
27
+ import { SseDecoder } from "../sse/parse.js";
28
+ import { ChatStreamAssembler } from "../sse/chat-assembler.js";
29
+ import { ANTHROPIC_MESSAGE_CONTENT, anthropicReasoningDetailsOf } from "./openai-chat-wire.js";
30
+ import { MAX_WEB_SEARCHES_PER_TURN } from "./web-search.js";
31
+ const ENCODER = new TextEncoder();
32
+ /** Absolute bound on model steps per caller turn (defense against a model that
33
+ * keeps searching after being told the search budget is exhausted). */
34
+ const MAX_LOOP_STEPS = 16;
35
+ /** In-process marker chunk field the loop injects between composed steps. */
36
+ export const SERVER_TOOL_MARKER_FIELD = "routekit_server_tool";
37
+ /** The marker on a parsed chat chunk, if present. */
38
+ export function serverToolMarkerOf(chunk) {
39
+ if (chunk === null || typeof chunk !== "object")
40
+ return undefined;
41
+ const marker = chunk[SERVER_TOOL_MARKER_FIELD];
42
+ return marker !== null && typeof marker === "object" ? marker : undefined;
43
+ }
44
+ function callName(call) {
45
+ return "function" in call && call.function !== undefined ? (call.function.name ?? "") : (call.name ?? "");
46
+ }
47
+ function queryOf(args) {
48
+ if (args === undefined || args.trim().length === 0)
49
+ return "";
50
+ try {
51
+ const parsed = JSON.parse(args);
52
+ return typeof parsed.query === "string" ? parsed.query : args;
53
+ }
54
+ catch {
55
+ return args;
56
+ }
57
+ }
58
+ function renderSearchResult(search) {
59
+ if (search.status === "failed" || search.outcome === undefined) {
60
+ return `[web_search_error] the search could not be executed${search.outcome?.text !== undefined && search.outcome.text.length > 0 ? `: ${search.outcome.text}` : ""}. Answer from what you already know, or try a different query.`;
61
+ }
62
+ const sources = search.outcome.citations.map((citation) => `- ${citation.url}${citation.title !== undefined ? ` (${citation.title})` : ""}`);
63
+ return sources.length > 0 ? `${search.outcome.text}\n\nSources:\n${sources.join("\n")}` : search.outcome.text;
64
+ }
65
+ const LIMIT_MESSAGE = "[web_search_limit] the web search budget for this turn is exhausted; answer with the information you already have.";
66
+ function chatMessages(chat) {
67
+ if (!Array.isArray(chat.messages))
68
+ chat.messages = [];
69
+ return chat.messages;
70
+ }
71
+ /**
72
+ * Execute one pure-server step's calls (respecting the remaining budget) and
73
+ * append the assistant tool-call message + tool results to the transcript.
74
+ * Emits `onSearch` start/done callbacks around each execution so the streaming
75
+ * composer can inject markers live.
76
+ */
77
+ async function executeServerCalls(input) {
78
+ const { options, calls, searches } = input;
79
+ const max = options.maxSearches ?? MAX_WEB_SEARCHES_PER_TURN;
80
+ const messages = chatMessages(options.chat);
81
+ const toolCalls = calls.map((call) => ({
82
+ id: call.id ?? `call_${randomId()}`,
83
+ type: "function",
84
+ function: { name: call.name ?? "web_search", arguments: call.arguments ?? "" }
85
+ }));
86
+ const assistant = {
87
+ role: "assistant",
88
+ content: typeof input.stepContent === "string" && input.stepContent.length > 0 ? input.stepContent : null,
89
+ tool_calls: toolCalls
90
+ };
91
+ const nativeReasoning = anthropicReasoningDetailsOf(input.reasoningDetails, "message")
92
+ .filter((detail) => detail.type === "redacted_thinking" ||
93
+ (detail.type === "thinking" &&
94
+ typeof detail.signature === "string" &&
95
+ detail.signature.length > 0))
96
+ .sort((a, b) => a.index - b.index);
97
+ if (nativeReasoning.length > 0) {
98
+ const nativeContent = nativeReasoning.map((detail) => detail.type === "redacted_thinking"
99
+ ? { type: "redacted_thinking", data: detail.data }
100
+ : {
101
+ type: "thinking",
102
+ thinking: detail.thinking ?? "",
103
+ signature: detail.signature ?? ""
104
+ });
105
+ if (typeof input.stepContent === "string" && input.stepContent.length > 0) {
106
+ nativeContent.push({ type: "text", text: input.stepContent });
107
+ }
108
+ for (const call of toolCalls) {
109
+ let toolInput = {};
110
+ try {
111
+ toolInput = JSON.parse(call.function.arguments);
112
+ }
113
+ catch {
114
+ toolInput = { raw: call.function.arguments };
115
+ }
116
+ nativeContent.push({
117
+ type: "tool_use",
118
+ id: call.id,
119
+ name: call.function.name,
120
+ input: toolInput
121
+ });
122
+ }
123
+ Object.defineProperty(assistant, ANTHROPIC_MESSAGE_CONTENT, {
124
+ value: nativeContent,
125
+ enumerable: true
126
+ });
127
+ }
128
+ messages.push(assistant);
129
+ for (let i = 0; i < calls.length; i += 1) {
130
+ const call = calls[i];
131
+ if (call === undefined)
132
+ continue;
133
+ const query = queryOf(call.arguments);
134
+ const callId = toolCalls[i]?.id ?? `call_${randomId()}`;
135
+ if (searches.length >= max) {
136
+ messages.push({ role: "tool", tool_call_id: callId, content: LIMIT_MESSAGE });
137
+ continue;
138
+ }
139
+ const itemId = `ws_${randomId()}`;
140
+ input.onSearchStart?.({ itemId, query });
141
+ let search;
142
+ try {
143
+ const outcome = await options.executor.search(query, options.signal);
144
+ search = { itemId, query, status: "completed", outcome };
145
+ }
146
+ catch (error) {
147
+ search = {
148
+ itemId,
149
+ query,
150
+ status: "failed",
151
+ outcome: { text: error instanceof Error ? error.message : String(error), citations: [] }
152
+ };
153
+ }
154
+ searches.push(search);
155
+ input.onSearchDone?.(search);
156
+ messages.push({ role: "tool", tool_call_id: callId, content: renderSearchResult(search) });
157
+ }
158
+ }
159
+ /**
160
+ * Run the loop over buffered (non-streaming) model steps. `firstStep` is the
161
+ * already-awaited first model step (the handler surfaces its HTTP errors
162
+ * before entering the loop). Returns the terminal step's OpenAI payload (with
163
+ * any un-executable mixed-batch server calls stripped) plus the searches
164
+ * executed along the way, for the dialect egress to render as native items.
165
+ */
166
+ export async function runBufferedServerToolLoop(options) {
167
+ const searches = [];
168
+ const events = [];
169
+ const totals = { prompt: 0, completion: 0, seen: false };
170
+ for (let step = 0; step < MAX_LOOP_STEPS; step += 1) {
171
+ const upstream = step === 0 ? options.firstStep : await options.runStep(options.chat);
172
+ if (!upstream.ok)
173
+ return { kind: "upstream_error", response: upstream };
174
+ const openai = (await upstream.json());
175
+ accumulateUsage(totals, openai.usage);
176
+ const choice = (Array.isArray(openai.choices) ? openai.choices[0] : undefined);
177
+ const message = choice?.message;
178
+ const calls = (Array.isArray(message?.tool_calls) ? message.tool_calls : []);
179
+ const server = calls.filter((call) => options.serverToolNames.has(callName(call)));
180
+ const client = calls.filter((call) => !options.serverToolNames.has(callName(call)));
181
+ if (server.length === 0) {
182
+ return {
183
+ kind: "openai",
184
+ openai: withAccumulatedUsage(openai, totals),
185
+ searches,
186
+ events
187
+ };
188
+ }
189
+ if (client.length > 0 || typeof choice?.finish_reason !== "string") {
190
+ // Mixed batch (or truncated step): surface what the caller can handle;
191
+ // the un-executable server calls are dropped, the model can re-search
192
+ // next turn.
193
+ if (message !== undefined)
194
+ message.tool_calls = client;
195
+ return {
196
+ kind: "openai",
197
+ openai: withAccumulatedUsage(openai, totals),
198
+ searches,
199
+ events
200
+ };
201
+ }
202
+ // Past the search budget, executeServerCalls answers each call with a
203
+ // limit notice instead of executing — the model gets one more step to
204
+ // answer from what it has (MAX_LOOP_STEPS bounds a model that will not).
205
+ const stepReasoning = anthropicReasoningDetailsOf(message?.reasoning_details, "message").filter((detail) => detail.type === "redacted_thinking" ||
206
+ (detail.type === "thinking" &&
207
+ typeof detail.signature === "string" &&
208
+ detail.signature.length > 0));
209
+ if (stepReasoning.length > 0) {
210
+ events.push({ kind: "reasoning", details: stepReasoning });
211
+ }
212
+ await executeServerCalls({
213
+ options,
214
+ calls: server.map((call) => ({ id: call.id, name: call.function?.name, arguments: call.function?.arguments })),
215
+ stepContent: typeof message?.content === "string" ? message.content : undefined,
216
+ reasoningDetails: message?.reasoning_details,
217
+ searches,
218
+ onSearchDone: (search) => events.push({ kind: "search", search })
219
+ });
220
+ }
221
+ return {
222
+ kind: "openai",
223
+ openai: withAccumulatedUsage({
224
+ choices: [
225
+ { index: 0, message: { role: "assistant", content: LIMIT_MESSAGE }, finish_reason: "stop" }
226
+ ]
227
+ }, totals),
228
+ searches,
229
+ events
230
+ };
231
+ }
232
+ function isFragmentSuppressed(state, call, serverToolNames) {
233
+ const index = typeof call.index === "number" ? call.index : undefined;
234
+ const id = typeof call.id === "string" && call.id.length > 0 ? call.id : undefined;
235
+ const name = call.function?.name;
236
+ if (typeof name === "string" && name.length > 0) {
237
+ const suppressed = serverToolNames.has(name);
238
+ if (suppressed) {
239
+ if (index !== undefined)
240
+ state.suppressedIndexes.add(index);
241
+ if (id !== undefined)
242
+ state.suppressedIds.add(id);
243
+ }
244
+ state.lastFragmentSuppressed = suppressed;
245
+ return suppressed;
246
+ }
247
+ const suppressed = index !== undefined
248
+ ? state.suppressedIndexes.has(index)
249
+ : id !== undefined
250
+ ? state.suppressedIds.has(id)
251
+ : state.lastFragmentSuppressed;
252
+ state.lastFragmentSuppressed = suppressed;
253
+ return suppressed;
254
+ }
255
+ function encodeChunk(chunk) {
256
+ return ENCODER.encode(`data: ${JSON.stringify(chunk)}\n\n`);
257
+ }
258
+ function markerChunk(marker) {
259
+ return encodeChunk({ [SERVER_TOOL_MARKER_FIELD]: marker });
260
+ }
261
+ function accumulateUsage(totals, usage) {
262
+ if (usage === null || typeof usage !== "object")
263
+ return;
264
+ const source = usage;
265
+ if (typeof source.prompt_tokens === "number")
266
+ totals.prompt += source.prompt_tokens;
267
+ if (typeof source.completion_tokens === "number")
268
+ totals.completion += source.completion_tokens;
269
+ totals.seen = true;
270
+ }
271
+ function withAccumulatedUsage(openai, totals) {
272
+ if (!totals.seen)
273
+ return openai;
274
+ const existing = openai.usage !== null &&
275
+ typeof openai.usage === "object" &&
276
+ !Array.isArray(openai.usage)
277
+ ? openai.usage
278
+ : {};
279
+ return {
280
+ ...openai,
281
+ usage: {
282
+ ...existing,
283
+ prompt_tokens: totals.prompt,
284
+ completion_tokens: totals.completion,
285
+ total_tokens: totals.prompt + totals.completion
286
+ }
287
+ };
288
+ }
289
+ /**
290
+ * Compose the loop's model steps into one continuous chat SSE stream.
291
+ *
292
+ * `firstStep` is the already-awaited first model step (the handler surfaces
293
+ * its HTTP errors exactly as the single-step path does). Server-tool call
294
+ * fragments are suppressed from the forwarded stream; each executed search is
295
+ * injected as a pair of {@link ServerToolMarker} chunks for the dialect
296
+ * translator. Per-step usage is withheld and re-emitted summed before the
297
+ * terminal finish chunk, so the client-visible usage covers the whole loop.
298
+ */
299
+ export function composeServerToolStream(options) {
300
+ const searches = [];
301
+ const totals = { prompt: 0, completion: 0, seen: false };
302
+ return new ReadableStream({
303
+ start(controller) {
304
+ void (async () => {
305
+ try {
306
+ let upstream = options.firstStep;
307
+ for (let step = 0; step < MAX_LOOP_STEPS; step += 1) {
308
+ if (step > 0) {
309
+ upstream = await options.runStep(options.chat);
310
+ if (!upstream.ok) {
311
+ throw new Error(`model step failed mid web-search loop (${upstream.status}): ${(await upstream.text()).slice(0, 500)}`);
312
+ }
313
+ }
314
+ const source = upstream.body;
315
+ if (source === null)
316
+ throw new Error("model step produced no stream mid web-search loop");
317
+ const terminal = await forwardStep(controller, source);
318
+ if (terminal) {
319
+ controller.close();
320
+ return;
321
+ }
322
+ }
323
+ // Step bound exhausted: close the turn rather than looping forever.
324
+ finalize(controller, "stop");
325
+ }
326
+ catch (error) {
327
+ controller.error(error);
328
+ return;
329
+ }
330
+ controller.close();
331
+ })();
332
+ }
333
+ });
334
+ /** Emit summed usage + a finish chunk + [DONE], ending the composed stream. */
335
+ function finalize(controller, finishReason, heldFinishChunk) {
336
+ if (totals.seen) {
337
+ controller.enqueue(encodeChunk({
338
+ choices: [],
339
+ usage: {
340
+ prompt_tokens: totals.prompt,
341
+ completion_tokens: totals.completion,
342
+ total_tokens: totals.prompt + totals.completion
343
+ }
344
+ }));
345
+ }
346
+ controller.enqueue(encodeChunk(heldFinishChunk ?? { choices: [{ index: 0, delta: {}, finish_reason: finishReason }] }));
347
+ controller.enqueue(ENCODER.encode("data: [DONE]\n\n"));
348
+ }
349
+ /**
350
+ * Forward one step's SSE into the composed stream. Returns true when the
351
+ * step was terminal (stream finished); false when the loop must run another
352
+ * model step (a pure server-tool step whose searches were executed).
353
+ */
354
+ async function forwardStep(controller, source) {
355
+ const reader = source.getReader();
356
+ const decoder = new SseDecoder();
357
+ const assembler = new ChatStreamAssembler();
358
+ const state = {
359
+ suppressedIndexes: new Set(),
360
+ suppressedIds: new Set(),
361
+ lastFragmentSuppressed: false,
362
+ heldFinishChunk: undefined
363
+ };
364
+ let stepContent = "";
365
+ const handleData = (data) => {
366
+ if (data.length === 0 || data === "[DONE]")
367
+ return;
368
+ let chunk;
369
+ try {
370
+ chunk = JSON.parse(data);
371
+ }
372
+ catch {
373
+ // Forward unparseable payloads untouched; the dialect translator owns
374
+ // strictness (it raises SseParseError on malformed chunks).
375
+ controller.enqueue(ENCODER.encode(`data: ${data}\n\n`));
376
+ return;
377
+ }
378
+ assembler.pushParsed(chunk);
379
+ let rewritten = chunk;
380
+ if (chunk.usage !== undefined && chunk.usage !== null) {
381
+ accumulateUsage(totals, chunk.usage);
382
+ rewritten = { ...rewritten };
383
+ delete rewritten.usage;
384
+ }
385
+ const choice = (Array.isArray(rewritten.choices) ? rewritten.choices[0] : undefined);
386
+ const delta = choice?.delta;
387
+ if (typeof delta?.content === "string")
388
+ stepContent += delta.content;
389
+ if (choice !== undefined && Array.isArray(delta?.tool_calls)) {
390
+ const kept = delta.tool_calls.filter((call) => !isFragmentSuppressed(state, call, options.serverToolNames));
391
+ if (kept.length !== delta.tool_calls.length) {
392
+ rewritten = {
393
+ ...rewritten,
394
+ choices: [{ ...choice, delta: { ...delta, tool_calls: kept } }]
395
+ };
396
+ const rewrittenChoice = rewritten.choices[0];
397
+ if (kept.length === 0 && rewrittenChoice !== undefined)
398
+ delete rewrittenChoice.delta.tool_calls;
399
+ }
400
+ }
401
+ const finishReason = choice?.finish_reason;
402
+ if (typeof finishReason === "string") {
403
+ // Hold the finish: whether it surfaces depends on what the step
404
+ // committed to (decided once the step's stream ends).
405
+ state.heldFinishChunk = rewritten;
406
+ return;
407
+ }
408
+ const survivingChoice = (Array.isArray(rewritten.choices) ? rewritten.choices[0] : undefined);
409
+ const emptyDelta = survivingChoice?.delta !== undefined && Object.keys(survivingChoice.delta).length === 0;
410
+ const bareUsageChunk = chunk.usage !== undefined && (rewritten.choices === undefined || rewritten.choices.length === 0);
411
+ if (emptyDelta || bareUsageChunk)
412
+ return;
413
+ controller.enqueue(encodeChunk(rewritten));
414
+ };
415
+ for (;;) {
416
+ const { done, value } = await reader.read();
417
+ if (done) {
418
+ for (const event of decoder.flush())
419
+ handleData(event.data);
420
+ break;
421
+ }
422
+ if (value !== undefined) {
423
+ for (const event of decoder.feed(value))
424
+ handleData(event.data);
425
+ }
426
+ }
427
+ const turn = assembler.result();
428
+ const server = turn.toolCalls.filter((call) => options.serverToolNames.has(call.name ?? ""));
429
+ const client = turn.toolCalls.filter((call) => !options.serverToolNames.has(call.name ?? ""));
430
+ const pureServerStep = server.length > 0 && client.length === 0 && turn.finishReason !== undefined;
431
+ if (!pureServerStep) {
432
+ if (state.heldFinishChunk !== undefined) {
433
+ finalize(controller, "stop", state.heldFinishChunk);
434
+ }
435
+ else if (turn.finishReason === undefined) {
436
+ // Truncated upstream: end without a finish chunk so the translator
437
+ // reports the turn as incomplete rather than fabricating completion.
438
+ controller.enqueue(ENCODER.encode("data: [DONE]\n\n"));
439
+ }
440
+ else {
441
+ finalize(controller, turn.finishReason);
442
+ }
443
+ return true;
444
+ }
445
+ await executeServerCalls({
446
+ options,
447
+ calls: server.map((call) => ({ id: call.id, name: call.name, arguments: call.arguments })),
448
+ stepContent: stepContent.length > 0 ? stepContent : undefined,
449
+ reasoningDetails: turn.reasoningDetails,
450
+ searches,
451
+ onSearchStart: (search) => {
452
+ controller.enqueue(markerChunk({ kind: "web_search", phase: "start", item_id: search.itemId, query: search.query }));
453
+ },
454
+ onSearchDone: (search) => {
455
+ controller.enqueue(markerChunk({
456
+ kind: "web_search",
457
+ phase: "done",
458
+ item_id: search.itemId,
459
+ query: search.query,
460
+ status: search.status,
461
+ ...(search.outcome?.anthropicResultBlocks !== undefined
462
+ ? { result_blocks: search.outcome.anthropicResultBlocks }
463
+ : search.outcome !== undefined && search.status === "completed"
464
+ ? {
465
+ result_blocks: search.outcome.citations.map((citation) => ({
466
+ type: "web_search_result",
467
+ url: citation.url,
468
+ ...(citation.title !== undefined ? { title: citation.title } : {})
469
+ }))
470
+ }
471
+ : {})
472
+ }));
473
+ }
474
+ });
475
+ return false;
476
+ }
477
+ }
@@ -0,0 +1,14 @@
1
+ /**
2
+ * Unwrap a backend error body for re-emission on a translated door.
3
+ *
4
+ * The Anthropic/Responses adapters run over the OpenAI-chat backend; when it
5
+ * fails, its body is already an OpenAI error envelope. Re-wrapping that JSON
6
+ * string as the `message` of a fresh `api_error` (the old behavior) both
7
+ * double-encodes the message and misclassifies caller errors — a 400
8
+ * `invalid_request_error` from the backend must stay an
9
+ * `invalid_request_error` on the door.
10
+ */
11
+ export declare function unwrapUpstreamError(detail: string): {
12
+ type: string;
13
+ message: string;
14
+ };
@@ -0,0 +1,25 @@
1
+ /**
2
+ * Unwrap a backend error body for re-emission on a translated door.
3
+ *
4
+ * The Anthropic/Responses adapters run over the OpenAI-chat backend; when it
5
+ * fails, its body is already an OpenAI error envelope. Re-wrapping that JSON
6
+ * string as the `message` of a fresh `api_error` (the old behavior) both
7
+ * double-encodes the message and misclassifies caller errors — a 400
8
+ * `invalid_request_error` from the backend must stay an
9
+ * `invalid_request_error` on the door.
10
+ */
11
+ export function unwrapUpstreamError(detail) {
12
+ try {
13
+ const parsed = JSON.parse(detail);
14
+ if (typeof parsed.error?.message === "string") {
15
+ return {
16
+ type: typeof parsed.error.type === "string" ? parsed.error.type : "api_error",
17
+ message: parsed.error.message
18
+ };
19
+ }
20
+ }
21
+ catch {
22
+ // not JSON — fall through to the raw detail
23
+ }
24
+ return { type: "api_error", message: detail.slice(0, 2000) };
25
+ }
@@ -0,0 +1,27 @@
1
+ /**
2
+ * Structural request validation at the gateway's wire doors.
3
+ *
4
+ * Hostile-input fuzzing showed that malformed bodies (a `model` array, a
5
+ * `messages` string, an empty body) sailed past the doors, hit deep code, and
6
+ * surfaced as 502 `upstream_error`s carrying raw TypeError text
7
+ * ("requested.startsWith is not a function", "body.messages is not iterable")
8
+ * or internal implementation details. A caller error must be a 400 in the
9
+ * door's native error envelope, and it must never reach a provider.
10
+ *
11
+ * Validation here is *structural only* — field presence and JSON types
12
+ * matching what the real providers enforce. Semantic rules (role enums,
13
+ * schema shapes) stay with the providers, so the doors never reject a shape a
14
+ * real provider would accept.
15
+ */
16
+ export type WireRejection = {
17
+ status: number;
18
+ body: unknown;
19
+ };
20
+ /** OpenAI Chat Completions door (`/v1/chat/completions`). */
21
+ export declare function validateChatRequest(body: unknown): WireRejection | undefined;
22
+ /** Anthropic Messages door (`/v1/messages`). */
23
+ export declare function validateAnthropicRequest(body: unknown): WireRejection | undefined;
24
+ /** Anthropic `count_tokens` door: same message-shape contract, no minimum length. */
25
+ export declare function validateCountTokensRequest(body: unknown): WireRejection | undefined;
26
+ /** OpenAI Responses door (`/v1/responses`). */
27
+ export declare function validateResponsesRequest(body: unknown): WireRejection | undefined;