@velum-labs/routekit-gateway 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +28 -0
- package/dist/acp-agent.d.ts +38 -0
- package/dist/acp-agent.js +142 -0
- package/dist/acp-registry.d.ts +36 -0
- package/dist/acp-registry.js +85 -0
- package/dist/adapters/anthropic.d.ts +131 -0
- package/dist/adapters/anthropic.js +1195 -0
- package/dist/adapters/chat.d.ts +14 -0
- package/dist/adapters/chat.js +34 -0
- package/dist/adapters/cursor.d.ts +34 -0
- package/dist/adapters/cursor.js +305 -0
- package/dist/adapters/dropped.d.ts +10 -0
- package/dist/adapters/dropped.js +24 -0
- package/dist/adapters/openai-chat-wire.d.ts +93 -0
- package/dist/adapters/openai-chat-wire.js +143 -0
- package/dist/adapters/responses-stream.d.ts +7 -0
- package/dist/adapters/responses-stream.js +597 -0
- package/dist/adapters/responses.d.ts +174 -0
- package/dist/adapters/responses.js +778 -0
- package/dist/adapters/server-tool-loop.d.ts +94 -0
- package/dist/adapters/server-tool-loop.js +477 -0
- package/dist/adapters/upstream-error.d.ts +14 -0
- package/dist/adapters/upstream-error.js +25 -0
- package/dist/adapters/validate.d.ts +27 -0
- package/dist/adapters/validate.js +180 -0
- package/dist/adapters/web-search.d.ts +46 -0
- package/dist/adapters/web-search.js +151 -0
- package/dist/auth.d.ts +10 -0
- package/dist/auth.js +28 -0
- package/dist/backend.d.ts +151 -0
- package/dist/backend.js +143 -0
- package/dist/capacity-pool.d.ts +31 -0
- package/dist/capacity-pool.js +99 -0
- package/dist/cost.d.ts +49 -0
- package/dist/cost.js +112 -0
- package/dist/endpoint-health.d.ts +54 -0
- package/dist/endpoint-health.js +123 -0
- package/dist/index.d.ts +40 -0
- package/dist/index.js +23 -0
- package/dist/provenance.d.ts +31 -0
- package/dist/provenance.js +191 -0
- package/dist/provider-backends.d.ts +40 -0
- package/dist/provider-backends.js +1050 -0
- package/dist/provider-source.d.ts +40 -0
- package/dist/provider-source.js +293 -0
- package/dist/router.d.ts +168 -0
- package/dist/router.js +474 -0
- package/dist/server.d.ts +67 -0
- package/dist/server.js +930 -0
- package/dist/sse/chat-assembler.d.ts +45 -0
- package/dist/sse/chat-assembler.js +190 -0
- package/dist/sse/parse.d.ts +50 -0
- package/dist/sse/parse.js +149 -0
- package/dist/sse-wire.d.ts +10 -0
- package/dist/sse-wire.js +31 -0
- package/dist/switching-proxy.d.ts +15 -0
- package/dist/switching-proxy.js +232 -0
- package/dist/test/acp-agent.test.d.ts +1 -0
- package/dist/test/acp-agent.test.js +66 -0
- package/dist/test/acp-registry.test.d.ts +1 -0
- package/dist/test/acp-registry.test.js +70 -0
- package/dist/test/anthropic.test.d.ts +1 -0
- package/dist/test/anthropic.test.js +793 -0
- package/dist/test/auth.test.d.ts +1 -0
- package/dist/test/auth.test.js +25 -0
- package/dist/test/boundary.test.d.ts +1 -0
- package/dist/test/boundary.test.js +32 -0
- package/dist/test/chat.test.d.ts +1 -0
- package/dist/test/chat.test.js +418 -0
- package/dist/test/cost.test.d.ts +1 -0
- package/dist/test/cost.test.js +60 -0
- package/dist/test/cursor.test.d.ts +1 -0
- package/dist/test/cursor.test.js +100 -0
- package/dist/test/drain.test.d.ts +1 -0
- package/dist/test/drain.test.js +116 -0
- package/dist/test/dropped.test.d.ts +1 -0
- package/dist/test/dropped.test.js +80 -0
- package/dist/test/endpoint-health.test.d.ts +1 -0
- package/dist/test/endpoint-health.test.js +73 -0
- package/dist/test/provenance.test.d.ts +1 -0
- package/dist/test/provenance.test.js +176 -0
- package/dist/test/provider-backends.test.d.ts +1 -0
- package/dist/test/provider-backends.test.js +699 -0
- package/dist/test/responses.test.d.ts +1 -0
- package/dist/test/responses.test.js +813 -0
- package/dist/test/routed-backend.test.d.ts +1 -0
- package/dist/test/routed-backend.test.js +39 -0
- package/dist/test/router.test.d.ts +1 -0
- package/dist/test/router.test.js +297 -0
- package/dist/test/server-resilience.test.d.ts +1 -0
- package/dist/test/server-resilience.test.js +169 -0
- package/dist/test/sse-codec.test.d.ts +1 -0
- package/dist/test/sse-codec.test.js +186 -0
- package/dist/test/web-search-loop.test.d.ts +1 -0
- package/dist/test/web-search-loop.test.js +469 -0
- package/dist/test/wire-validation.test.d.ts +1 -0
- package/dist/test/wire-validation.test.js +140 -0
- package/package.json +48 -0
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The server-tool inner loop (gateway-executed web search).
|
|
3
|
+
*
|
|
4
|
+
* When the upstream model calls a *server-executed* tool (today: `web_search`),
|
|
5
|
+
* nobody on the caller's side can answer it — the caller declared the tool
|
|
6
|
+
* expecting the "server" to run it. This loop makes the gateway that server:
|
|
7
|
+
* it intercepts server-tool calls from a model step, executes them via a
|
|
8
|
+
* {@link WebSearchExecutor}, appends the exchange to the chat transcript, and
|
|
9
|
+
* runs another model step — repeating until a step commits to something the
|
|
10
|
+
* caller can actually handle (text, client tool calls, or a clean stop).
|
|
11
|
+
*
|
|
12
|
+
* The loop operates at the chat-completions layer, around `backend.chat`:
|
|
13
|
+
* each inner step is an ordinary backend turn, exactly as
|
|
14
|
+
* if the caller had executed a client tool and come back. The dialect egress
|
|
15
|
+
* translators stay single-stream: in streaming mode the loop composes the
|
|
16
|
+
* steps' chat SSE into one continuous stream, suppressing the server-tool
|
|
17
|
+
* fragments and injecting {@link ServerToolMarker} chunks (an in-process
|
|
18
|
+
* convention) that the translators render as their dialect's native search
|
|
19
|
+
* items (`web_search_call` / `server_tool_use` + `web_search_tool_result`).
|
|
20
|
+
*
|
|
21
|
+
* Mixed batches (server + client calls in one step) terminate the turn: the
|
|
22
|
+
* client calls surface and the server calls are dropped un-executed — results
|
|
23
|
+
* could not be fed back into a turn that just ended, and the upstream model can
|
|
24
|
+
* simply re-issue the search next turn.
|
|
25
|
+
*/
|
|
26
|
+
import { type AnthropicReasoningDetail } from "./openai-chat-wire.js";
|
|
27
|
+
import type { WebSearchExecutor, WebSearchOutcome } from "./web-search.js";
|
|
28
|
+
/** In-process marker chunk field the loop injects between composed steps. */
|
|
29
|
+
export declare const SERVER_TOOL_MARKER_FIELD = "routekit_server_tool";
|
|
30
|
+
export type ServerToolMarker = {
|
|
31
|
+
kind: "web_search";
|
|
32
|
+
phase: "start" | "done";
|
|
33
|
+
item_id: string;
|
|
34
|
+
query: string;
|
|
35
|
+
status?: "completed" | "failed";
|
|
36
|
+
/** Anthropic-native result blocks for the Anthropic egress (done phase). */
|
|
37
|
+
result_blocks?: unknown[];
|
|
38
|
+
};
|
|
39
|
+
/** The marker on a parsed chat chunk, if present. */
|
|
40
|
+
export declare function serverToolMarkerOf(chunk: unknown): ServerToolMarker | undefined;
|
|
41
|
+
export type ExecutedSearch = {
|
|
42
|
+
itemId: string;
|
|
43
|
+
query: string;
|
|
44
|
+
status: "completed" | "failed";
|
|
45
|
+
outcome?: WebSearchOutcome;
|
|
46
|
+
};
|
|
47
|
+
export type ServerToolLoopEvent = {
|
|
48
|
+
kind: "reasoning";
|
|
49
|
+
details: AnthropicReasoningDetail[];
|
|
50
|
+
} | {
|
|
51
|
+
kind: "search";
|
|
52
|
+
search: ExecutedSearch;
|
|
53
|
+
};
|
|
54
|
+
export type ServerToolLoopOptions = {
|
|
55
|
+
/** The translated chat body; the loop appends search exchanges to `messages`. */
|
|
56
|
+
chat: Record<string, unknown>;
|
|
57
|
+
runStep: (chat: Record<string, unknown>) => Promise<Response>;
|
|
58
|
+
serverToolNames: ReadonlySet<string>;
|
|
59
|
+
executor: WebSearchExecutor;
|
|
60
|
+
maxSearches?: number;
|
|
61
|
+
signal?: AbortSignal;
|
|
62
|
+
};
|
|
63
|
+
export type BufferedLoopOutcome = {
|
|
64
|
+
kind: "openai";
|
|
65
|
+
openai: Record<string, unknown>;
|
|
66
|
+
searches: ExecutedSearch[];
|
|
67
|
+
events: ServerToolLoopEvent[];
|
|
68
|
+
} | {
|
|
69
|
+
kind: "upstream_error";
|
|
70
|
+
response: Response;
|
|
71
|
+
};
|
|
72
|
+
/**
|
|
73
|
+
* Run the loop over buffered (non-streaming) model steps. `firstStep` is the
|
|
74
|
+
* already-awaited first model step (the handler surfaces its HTTP errors
|
|
75
|
+
* before entering the loop). Returns the terminal step's OpenAI payload (with
|
|
76
|
+
* any un-executable mixed-batch server calls stripped) plus the searches
|
|
77
|
+
* executed along the way, for the dialect egress to render as native items.
|
|
78
|
+
*/
|
|
79
|
+
export declare function runBufferedServerToolLoop(options: ServerToolLoopOptions & {
|
|
80
|
+
firstStep: Response;
|
|
81
|
+
}): Promise<BufferedLoopOutcome>;
|
|
82
|
+
/**
|
|
83
|
+
* Compose the loop's model steps into one continuous chat SSE stream.
|
|
84
|
+
*
|
|
85
|
+
* `firstStep` is the already-awaited first model step (the handler surfaces
|
|
86
|
+
* its HTTP errors exactly as the single-step path does). Server-tool call
|
|
87
|
+
* fragments are suppressed from the forwarded stream; each executed search is
|
|
88
|
+
* injected as a pair of {@link ServerToolMarker} chunks for the dialect
|
|
89
|
+
* translator. Per-step usage is withheld and re-emitted summed before the
|
|
90
|
+
* terminal finish chunk, so the client-visible usage covers the whole loop.
|
|
91
|
+
*/
|
|
92
|
+
export declare function composeServerToolStream(options: ServerToolLoopOptions & {
|
|
93
|
+
firstStep: Response;
|
|
94
|
+
}): ReadableStream<Uint8Array>;
|
|
@@ -0,0 +1,477 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The server-tool inner loop (gateway-executed web search).
|
|
3
|
+
*
|
|
4
|
+
* When the upstream model calls a *server-executed* tool (today: `web_search`),
|
|
5
|
+
* nobody on the caller's side can answer it — the caller declared the tool
|
|
6
|
+
* expecting the "server" to run it. This loop makes the gateway that server:
|
|
7
|
+
* it intercepts server-tool calls from a model step, executes them via a
|
|
8
|
+
* {@link WebSearchExecutor}, appends the exchange to the chat transcript, and
|
|
9
|
+
* runs another model step — repeating until a step commits to something the
|
|
10
|
+
* caller can actually handle (text, client tool calls, or a clean stop).
|
|
11
|
+
*
|
|
12
|
+
* The loop operates at the chat-completions layer, around `backend.chat`:
|
|
13
|
+
* each inner step is an ordinary backend turn, exactly as
|
|
14
|
+
* if the caller had executed a client tool and come back. The dialect egress
|
|
15
|
+
* translators stay single-stream: in streaming mode the loop composes the
|
|
16
|
+
* steps' chat SSE into one continuous stream, suppressing the server-tool
|
|
17
|
+
* fragments and injecting {@link ServerToolMarker} chunks (an in-process
|
|
18
|
+
* convention) that the translators render as their dialect's native search
|
|
19
|
+
* items (`web_search_call` / `server_tool_use` + `web_search_tool_result`).
|
|
20
|
+
*
|
|
21
|
+
* Mixed batches (server + client calls in one step) terminate the turn: the
|
|
22
|
+
* client calls surface and the server calls are dropped un-executed — results
|
|
23
|
+
* could not be fed back into a turn that just ended, and the upstream model can
|
|
24
|
+
* simply re-issue the search next turn.
|
|
25
|
+
*/
|
|
26
|
+
import { randomId } from "@velum-labs/routekit-runtime";
|
|
27
|
+
import { SseDecoder } from "../sse/parse.js";
|
|
28
|
+
import { ChatStreamAssembler } from "../sse/chat-assembler.js";
|
|
29
|
+
import { ANTHROPIC_MESSAGE_CONTENT, anthropicReasoningDetailsOf } from "./openai-chat-wire.js";
|
|
30
|
+
import { MAX_WEB_SEARCHES_PER_TURN } from "./web-search.js";
|
|
31
|
+
const ENCODER = new TextEncoder();
|
|
32
|
+
/** Absolute bound on model steps per caller turn (defense against a model that
|
|
33
|
+
* keeps searching after being told the search budget is exhausted). */
|
|
34
|
+
const MAX_LOOP_STEPS = 16;
|
|
35
|
+
/** In-process marker chunk field the loop injects between composed steps. */
|
|
36
|
+
export const SERVER_TOOL_MARKER_FIELD = "routekit_server_tool";
|
|
37
|
+
/** The marker on a parsed chat chunk, if present. */
|
|
38
|
+
export function serverToolMarkerOf(chunk) {
|
|
39
|
+
if (chunk === null || typeof chunk !== "object")
|
|
40
|
+
return undefined;
|
|
41
|
+
const marker = chunk[SERVER_TOOL_MARKER_FIELD];
|
|
42
|
+
return marker !== null && typeof marker === "object" ? marker : undefined;
|
|
43
|
+
}
|
|
44
|
+
function callName(call) {
|
|
45
|
+
return "function" in call && call.function !== undefined ? (call.function.name ?? "") : (call.name ?? "");
|
|
46
|
+
}
|
|
47
|
+
function queryOf(args) {
|
|
48
|
+
if (args === undefined || args.trim().length === 0)
|
|
49
|
+
return "";
|
|
50
|
+
try {
|
|
51
|
+
const parsed = JSON.parse(args);
|
|
52
|
+
return typeof parsed.query === "string" ? parsed.query : args;
|
|
53
|
+
}
|
|
54
|
+
catch {
|
|
55
|
+
return args;
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
function renderSearchResult(search) {
|
|
59
|
+
if (search.status === "failed" || search.outcome === undefined) {
|
|
60
|
+
return `[web_search_error] the search could not be executed${search.outcome?.text !== undefined && search.outcome.text.length > 0 ? `: ${search.outcome.text}` : ""}. Answer from what you already know, or try a different query.`;
|
|
61
|
+
}
|
|
62
|
+
const sources = search.outcome.citations.map((citation) => `- ${citation.url}${citation.title !== undefined ? ` (${citation.title})` : ""}`);
|
|
63
|
+
return sources.length > 0 ? `${search.outcome.text}\n\nSources:\n${sources.join("\n")}` : search.outcome.text;
|
|
64
|
+
}
|
|
65
|
+
const LIMIT_MESSAGE = "[web_search_limit] the web search budget for this turn is exhausted; answer with the information you already have.";
|
|
66
|
+
function chatMessages(chat) {
|
|
67
|
+
if (!Array.isArray(chat.messages))
|
|
68
|
+
chat.messages = [];
|
|
69
|
+
return chat.messages;
|
|
70
|
+
}
|
|
71
|
+
/**
|
|
72
|
+
* Execute one pure-server step's calls (respecting the remaining budget) and
|
|
73
|
+
* append the assistant tool-call message + tool results to the transcript.
|
|
74
|
+
* Emits `onSearch` start/done callbacks around each execution so the streaming
|
|
75
|
+
* composer can inject markers live.
|
|
76
|
+
*/
|
|
77
|
+
async function executeServerCalls(input) {
|
|
78
|
+
const { options, calls, searches } = input;
|
|
79
|
+
const max = options.maxSearches ?? MAX_WEB_SEARCHES_PER_TURN;
|
|
80
|
+
const messages = chatMessages(options.chat);
|
|
81
|
+
const toolCalls = calls.map((call) => ({
|
|
82
|
+
id: call.id ?? `call_${randomId()}`,
|
|
83
|
+
type: "function",
|
|
84
|
+
function: { name: call.name ?? "web_search", arguments: call.arguments ?? "" }
|
|
85
|
+
}));
|
|
86
|
+
const assistant = {
|
|
87
|
+
role: "assistant",
|
|
88
|
+
content: typeof input.stepContent === "string" && input.stepContent.length > 0 ? input.stepContent : null,
|
|
89
|
+
tool_calls: toolCalls
|
|
90
|
+
};
|
|
91
|
+
const nativeReasoning = anthropicReasoningDetailsOf(input.reasoningDetails, "message")
|
|
92
|
+
.filter((detail) => detail.type === "redacted_thinking" ||
|
|
93
|
+
(detail.type === "thinking" &&
|
|
94
|
+
typeof detail.signature === "string" &&
|
|
95
|
+
detail.signature.length > 0))
|
|
96
|
+
.sort((a, b) => a.index - b.index);
|
|
97
|
+
if (nativeReasoning.length > 0) {
|
|
98
|
+
const nativeContent = nativeReasoning.map((detail) => detail.type === "redacted_thinking"
|
|
99
|
+
? { type: "redacted_thinking", data: detail.data }
|
|
100
|
+
: {
|
|
101
|
+
type: "thinking",
|
|
102
|
+
thinking: detail.thinking ?? "",
|
|
103
|
+
signature: detail.signature ?? ""
|
|
104
|
+
});
|
|
105
|
+
if (typeof input.stepContent === "string" && input.stepContent.length > 0) {
|
|
106
|
+
nativeContent.push({ type: "text", text: input.stepContent });
|
|
107
|
+
}
|
|
108
|
+
for (const call of toolCalls) {
|
|
109
|
+
let toolInput = {};
|
|
110
|
+
try {
|
|
111
|
+
toolInput = JSON.parse(call.function.arguments);
|
|
112
|
+
}
|
|
113
|
+
catch {
|
|
114
|
+
toolInput = { raw: call.function.arguments };
|
|
115
|
+
}
|
|
116
|
+
nativeContent.push({
|
|
117
|
+
type: "tool_use",
|
|
118
|
+
id: call.id,
|
|
119
|
+
name: call.function.name,
|
|
120
|
+
input: toolInput
|
|
121
|
+
});
|
|
122
|
+
}
|
|
123
|
+
Object.defineProperty(assistant, ANTHROPIC_MESSAGE_CONTENT, {
|
|
124
|
+
value: nativeContent,
|
|
125
|
+
enumerable: true
|
|
126
|
+
});
|
|
127
|
+
}
|
|
128
|
+
messages.push(assistant);
|
|
129
|
+
for (let i = 0; i < calls.length; i += 1) {
|
|
130
|
+
const call = calls[i];
|
|
131
|
+
if (call === undefined)
|
|
132
|
+
continue;
|
|
133
|
+
const query = queryOf(call.arguments);
|
|
134
|
+
const callId = toolCalls[i]?.id ?? `call_${randomId()}`;
|
|
135
|
+
if (searches.length >= max) {
|
|
136
|
+
messages.push({ role: "tool", tool_call_id: callId, content: LIMIT_MESSAGE });
|
|
137
|
+
continue;
|
|
138
|
+
}
|
|
139
|
+
const itemId = `ws_${randomId()}`;
|
|
140
|
+
input.onSearchStart?.({ itemId, query });
|
|
141
|
+
let search;
|
|
142
|
+
try {
|
|
143
|
+
const outcome = await options.executor.search(query, options.signal);
|
|
144
|
+
search = { itemId, query, status: "completed", outcome };
|
|
145
|
+
}
|
|
146
|
+
catch (error) {
|
|
147
|
+
search = {
|
|
148
|
+
itemId,
|
|
149
|
+
query,
|
|
150
|
+
status: "failed",
|
|
151
|
+
outcome: { text: error instanceof Error ? error.message : String(error), citations: [] }
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
searches.push(search);
|
|
155
|
+
input.onSearchDone?.(search);
|
|
156
|
+
messages.push({ role: "tool", tool_call_id: callId, content: renderSearchResult(search) });
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
/**
|
|
160
|
+
* Run the loop over buffered (non-streaming) model steps. `firstStep` is the
|
|
161
|
+
* already-awaited first model step (the handler surfaces its HTTP errors
|
|
162
|
+
* before entering the loop). Returns the terminal step's OpenAI payload (with
|
|
163
|
+
* any un-executable mixed-batch server calls stripped) plus the searches
|
|
164
|
+
* executed along the way, for the dialect egress to render as native items.
|
|
165
|
+
*/
|
|
166
|
+
export async function runBufferedServerToolLoop(options) {
|
|
167
|
+
const searches = [];
|
|
168
|
+
const events = [];
|
|
169
|
+
const totals = { prompt: 0, completion: 0, seen: false };
|
|
170
|
+
for (let step = 0; step < MAX_LOOP_STEPS; step += 1) {
|
|
171
|
+
const upstream = step === 0 ? options.firstStep : await options.runStep(options.chat);
|
|
172
|
+
if (!upstream.ok)
|
|
173
|
+
return { kind: "upstream_error", response: upstream };
|
|
174
|
+
const openai = (await upstream.json());
|
|
175
|
+
accumulateUsage(totals, openai.usage);
|
|
176
|
+
const choice = (Array.isArray(openai.choices) ? openai.choices[0] : undefined);
|
|
177
|
+
const message = choice?.message;
|
|
178
|
+
const calls = (Array.isArray(message?.tool_calls) ? message.tool_calls : []);
|
|
179
|
+
const server = calls.filter((call) => options.serverToolNames.has(callName(call)));
|
|
180
|
+
const client = calls.filter((call) => !options.serverToolNames.has(callName(call)));
|
|
181
|
+
if (server.length === 0) {
|
|
182
|
+
return {
|
|
183
|
+
kind: "openai",
|
|
184
|
+
openai: withAccumulatedUsage(openai, totals),
|
|
185
|
+
searches,
|
|
186
|
+
events
|
|
187
|
+
};
|
|
188
|
+
}
|
|
189
|
+
if (client.length > 0 || typeof choice?.finish_reason !== "string") {
|
|
190
|
+
// Mixed batch (or truncated step): surface what the caller can handle;
|
|
191
|
+
// the un-executable server calls are dropped, the model can re-search
|
|
192
|
+
// next turn.
|
|
193
|
+
if (message !== undefined)
|
|
194
|
+
message.tool_calls = client;
|
|
195
|
+
return {
|
|
196
|
+
kind: "openai",
|
|
197
|
+
openai: withAccumulatedUsage(openai, totals),
|
|
198
|
+
searches,
|
|
199
|
+
events
|
|
200
|
+
};
|
|
201
|
+
}
|
|
202
|
+
// Past the search budget, executeServerCalls answers each call with a
|
|
203
|
+
// limit notice instead of executing — the model gets one more step to
|
|
204
|
+
// answer from what it has (MAX_LOOP_STEPS bounds a model that will not).
|
|
205
|
+
const stepReasoning = anthropicReasoningDetailsOf(message?.reasoning_details, "message").filter((detail) => detail.type === "redacted_thinking" ||
|
|
206
|
+
(detail.type === "thinking" &&
|
|
207
|
+
typeof detail.signature === "string" &&
|
|
208
|
+
detail.signature.length > 0));
|
|
209
|
+
if (stepReasoning.length > 0) {
|
|
210
|
+
events.push({ kind: "reasoning", details: stepReasoning });
|
|
211
|
+
}
|
|
212
|
+
await executeServerCalls({
|
|
213
|
+
options,
|
|
214
|
+
calls: server.map((call) => ({ id: call.id, name: call.function?.name, arguments: call.function?.arguments })),
|
|
215
|
+
stepContent: typeof message?.content === "string" ? message.content : undefined,
|
|
216
|
+
reasoningDetails: message?.reasoning_details,
|
|
217
|
+
searches,
|
|
218
|
+
onSearchDone: (search) => events.push({ kind: "search", search })
|
|
219
|
+
});
|
|
220
|
+
}
|
|
221
|
+
return {
|
|
222
|
+
kind: "openai",
|
|
223
|
+
openai: withAccumulatedUsage({
|
|
224
|
+
choices: [
|
|
225
|
+
{ index: 0, message: { role: "assistant", content: LIMIT_MESSAGE }, finish_reason: "stop" }
|
|
226
|
+
]
|
|
227
|
+
}, totals),
|
|
228
|
+
searches,
|
|
229
|
+
events
|
|
230
|
+
};
|
|
231
|
+
}
|
|
232
|
+
function isFragmentSuppressed(state, call, serverToolNames) {
|
|
233
|
+
const index = typeof call.index === "number" ? call.index : undefined;
|
|
234
|
+
const id = typeof call.id === "string" && call.id.length > 0 ? call.id : undefined;
|
|
235
|
+
const name = call.function?.name;
|
|
236
|
+
if (typeof name === "string" && name.length > 0) {
|
|
237
|
+
const suppressed = serverToolNames.has(name);
|
|
238
|
+
if (suppressed) {
|
|
239
|
+
if (index !== undefined)
|
|
240
|
+
state.suppressedIndexes.add(index);
|
|
241
|
+
if (id !== undefined)
|
|
242
|
+
state.suppressedIds.add(id);
|
|
243
|
+
}
|
|
244
|
+
state.lastFragmentSuppressed = suppressed;
|
|
245
|
+
return suppressed;
|
|
246
|
+
}
|
|
247
|
+
const suppressed = index !== undefined
|
|
248
|
+
? state.suppressedIndexes.has(index)
|
|
249
|
+
: id !== undefined
|
|
250
|
+
? state.suppressedIds.has(id)
|
|
251
|
+
: state.lastFragmentSuppressed;
|
|
252
|
+
state.lastFragmentSuppressed = suppressed;
|
|
253
|
+
return suppressed;
|
|
254
|
+
}
|
|
255
|
+
function encodeChunk(chunk) {
|
|
256
|
+
return ENCODER.encode(`data: ${JSON.stringify(chunk)}\n\n`);
|
|
257
|
+
}
|
|
258
|
+
function markerChunk(marker) {
|
|
259
|
+
return encodeChunk({ [SERVER_TOOL_MARKER_FIELD]: marker });
|
|
260
|
+
}
|
|
261
|
+
function accumulateUsage(totals, usage) {
|
|
262
|
+
if (usage === null || typeof usage !== "object")
|
|
263
|
+
return;
|
|
264
|
+
const source = usage;
|
|
265
|
+
if (typeof source.prompt_tokens === "number")
|
|
266
|
+
totals.prompt += source.prompt_tokens;
|
|
267
|
+
if (typeof source.completion_tokens === "number")
|
|
268
|
+
totals.completion += source.completion_tokens;
|
|
269
|
+
totals.seen = true;
|
|
270
|
+
}
|
|
271
|
+
function withAccumulatedUsage(openai, totals) {
|
|
272
|
+
if (!totals.seen)
|
|
273
|
+
return openai;
|
|
274
|
+
const existing = openai.usage !== null &&
|
|
275
|
+
typeof openai.usage === "object" &&
|
|
276
|
+
!Array.isArray(openai.usage)
|
|
277
|
+
? openai.usage
|
|
278
|
+
: {};
|
|
279
|
+
return {
|
|
280
|
+
...openai,
|
|
281
|
+
usage: {
|
|
282
|
+
...existing,
|
|
283
|
+
prompt_tokens: totals.prompt,
|
|
284
|
+
completion_tokens: totals.completion,
|
|
285
|
+
total_tokens: totals.prompt + totals.completion
|
|
286
|
+
}
|
|
287
|
+
};
|
|
288
|
+
}
|
|
289
|
+
/**
|
|
290
|
+
* Compose the loop's model steps into one continuous chat SSE stream.
|
|
291
|
+
*
|
|
292
|
+
* `firstStep` is the already-awaited first model step (the handler surfaces
|
|
293
|
+
* its HTTP errors exactly as the single-step path does). Server-tool call
|
|
294
|
+
* fragments are suppressed from the forwarded stream; each executed search is
|
|
295
|
+
* injected as a pair of {@link ServerToolMarker} chunks for the dialect
|
|
296
|
+
* translator. Per-step usage is withheld and re-emitted summed before the
|
|
297
|
+
* terminal finish chunk, so the client-visible usage covers the whole loop.
|
|
298
|
+
*/
|
|
299
|
+
export function composeServerToolStream(options) {
|
|
300
|
+
const searches = [];
|
|
301
|
+
const totals = { prompt: 0, completion: 0, seen: false };
|
|
302
|
+
return new ReadableStream({
|
|
303
|
+
start(controller) {
|
|
304
|
+
void (async () => {
|
|
305
|
+
try {
|
|
306
|
+
let upstream = options.firstStep;
|
|
307
|
+
for (let step = 0; step < MAX_LOOP_STEPS; step += 1) {
|
|
308
|
+
if (step > 0) {
|
|
309
|
+
upstream = await options.runStep(options.chat);
|
|
310
|
+
if (!upstream.ok) {
|
|
311
|
+
throw new Error(`model step failed mid web-search loop (${upstream.status}): ${(await upstream.text()).slice(0, 500)}`);
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
const source = upstream.body;
|
|
315
|
+
if (source === null)
|
|
316
|
+
throw new Error("model step produced no stream mid web-search loop");
|
|
317
|
+
const terminal = await forwardStep(controller, source);
|
|
318
|
+
if (terminal) {
|
|
319
|
+
controller.close();
|
|
320
|
+
return;
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
// Step bound exhausted: close the turn rather than looping forever.
|
|
324
|
+
finalize(controller, "stop");
|
|
325
|
+
}
|
|
326
|
+
catch (error) {
|
|
327
|
+
controller.error(error);
|
|
328
|
+
return;
|
|
329
|
+
}
|
|
330
|
+
controller.close();
|
|
331
|
+
})();
|
|
332
|
+
}
|
|
333
|
+
});
|
|
334
|
+
/** Emit summed usage + a finish chunk + [DONE], ending the composed stream. */
|
|
335
|
+
function finalize(controller, finishReason, heldFinishChunk) {
|
|
336
|
+
if (totals.seen) {
|
|
337
|
+
controller.enqueue(encodeChunk({
|
|
338
|
+
choices: [],
|
|
339
|
+
usage: {
|
|
340
|
+
prompt_tokens: totals.prompt,
|
|
341
|
+
completion_tokens: totals.completion,
|
|
342
|
+
total_tokens: totals.prompt + totals.completion
|
|
343
|
+
}
|
|
344
|
+
}));
|
|
345
|
+
}
|
|
346
|
+
controller.enqueue(encodeChunk(heldFinishChunk ?? { choices: [{ index: 0, delta: {}, finish_reason: finishReason }] }));
|
|
347
|
+
controller.enqueue(ENCODER.encode("data: [DONE]\n\n"));
|
|
348
|
+
}
|
|
349
|
+
/**
|
|
350
|
+
* Forward one step's SSE into the composed stream. Returns true when the
|
|
351
|
+
* step was terminal (stream finished); false when the loop must run another
|
|
352
|
+
* model step (a pure server-tool step whose searches were executed).
|
|
353
|
+
*/
|
|
354
|
+
async function forwardStep(controller, source) {
|
|
355
|
+
const reader = source.getReader();
|
|
356
|
+
const decoder = new SseDecoder();
|
|
357
|
+
const assembler = new ChatStreamAssembler();
|
|
358
|
+
const state = {
|
|
359
|
+
suppressedIndexes: new Set(),
|
|
360
|
+
suppressedIds: new Set(),
|
|
361
|
+
lastFragmentSuppressed: false,
|
|
362
|
+
heldFinishChunk: undefined
|
|
363
|
+
};
|
|
364
|
+
let stepContent = "";
|
|
365
|
+
const handleData = (data) => {
|
|
366
|
+
if (data.length === 0 || data === "[DONE]")
|
|
367
|
+
return;
|
|
368
|
+
let chunk;
|
|
369
|
+
try {
|
|
370
|
+
chunk = JSON.parse(data);
|
|
371
|
+
}
|
|
372
|
+
catch {
|
|
373
|
+
// Forward unparseable payloads untouched; the dialect translator owns
|
|
374
|
+
// strictness (it raises SseParseError on malformed chunks).
|
|
375
|
+
controller.enqueue(ENCODER.encode(`data: ${data}\n\n`));
|
|
376
|
+
return;
|
|
377
|
+
}
|
|
378
|
+
assembler.pushParsed(chunk);
|
|
379
|
+
let rewritten = chunk;
|
|
380
|
+
if (chunk.usage !== undefined && chunk.usage !== null) {
|
|
381
|
+
accumulateUsage(totals, chunk.usage);
|
|
382
|
+
rewritten = { ...rewritten };
|
|
383
|
+
delete rewritten.usage;
|
|
384
|
+
}
|
|
385
|
+
const choice = (Array.isArray(rewritten.choices) ? rewritten.choices[0] : undefined);
|
|
386
|
+
const delta = choice?.delta;
|
|
387
|
+
if (typeof delta?.content === "string")
|
|
388
|
+
stepContent += delta.content;
|
|
389
|
+
if (choice !== undefined && Array.isArray(delta?.tool_calls)) {
|
|
390
|
+
const kept = delta.tool_calls.filter((call) => !isFragmentSuppressed(state, call, options.serverToolNames));
|
|
391
|
+
if (kept.length !== delta.tool_calls.length) {
|
|
392
|
+
rewritten = {
|
|
393
|
+
...rewritten,
|
|
394
|
+
choices: [{ ...choice, delta: { ...delta, tool_calls: kept } }]
|
|
395
|
+
};
|
|
396
|
+
const rewrittenChoice = rewritten.choices[0];
|
|
397
|
+
if (kept.length === 0 && rewrittenChoice !== undefined)
|
|
398
|
+
delete rewrittenChoice.delta.tool_calls;
|
|
399
|
+
}
|
|
400
|
+
}
|
|
401
|
+
const finishReason = choice?.finish_reason;
|
|
402
|
+
if (typeof finishReason === "string") {
|
|
403
|
+
// Hold the finish: whether it surfaces depends on what the step
|
|
404
|
+
// committed to (decided once the step's stream ends).
|
|
405
|
+
state.heldFinishChunk = rewritten;
|
|
406
|
+
return;
|
|
407
|
+
}
|
|
408
|
+
const survivingChoice = (Array.isArray(rewritten.choices) ? rewritten.choices[0] : undefined);
|
|
409
|
+
const emptyDelta = survivingChoice?.delta !== undefined && Object.keys(survivingChoice.delta).length === 0;
|
|
410
|
+
const bareUsageChunk = chunk.usage !== undefined && (rewritten.choices === undefined || rewritten.choices.length === 0);
|
|
411
|
+
if (emptyDelta || bareUsageChunk)
|
|
412
|
+
return;
|
|
413
|
+
controller.enqueue(encodeChunk(rewritten));
|
|
414
|
+
};
|
|
415
|
+
for (;;) {
|
|
416
|
+
const { done, value } = await reader.read();
|
|
417
|
+
if (done) {
|
|
418
|
+
for (const event of decoder.flush())
|
|
419
|
+
handleData(event.data);
|
|
420
|
+
break;
|
|
421
|
+
}
|
|
422
|
+
if (value !== undefined) {
|
|
423
|
+
for (const event of decoder.feed(value))
|
|
424
|
+
handleData(event.data);
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
const turn = assembler.result();
|
|
428
|
+
const server = turn.toolCalls.filter((call) => options.serverToolNames.has(call.name ?? ""));
|
|
429
|
+
const client = turn.toolCalls.filter((call) => !options.serverToolNames.has(call.name ?? ""));
|
|
430
|
+
const pureServerStep = server.length > 0 && client.length === 0 && turn.finishReason !== undefined;
|
|
431
|
+
if (!pureServerStep) {
|
|
432
|
+
if (state.heldFinishChunk !== undefined) {
|
|
433
|
+
finalize(controller, "stop", state.heldFinishChunk);
|
|
434
|
+
}
|
|
435
|
+
else if (turn.finishReason === undefined) {
|
|
436
|
+
// Truncated upstream: end without a finish chunk so the translator
|
|
437
|
+
// reports the turn as incomplete rather than fabricating completion.
|
|
438
|
+
controller.enqueue(ENCODER.encode("data: [DONE]\n\n"));
|
|
439
|
+
}
|
|
440
|
+
else {
|
|
441
|
+
finalize(controller, turn.finishReason);
|
|
442
|
+
}
|
|
443
|
+
return true;
|
|
444
|
+
}
|
|
445
|
+
await executeServerCalls({
|
|
446
|
+
options,
|
|
447
|
+
calls: server.map((call) => ({ id: call.id, name: call.name, arguments: call.arguments })),
|
|
448
|
+
stepContent: stepContent.length > 0 ? stepContent : undefined,
|
|
449
|
+
reasoningDetails: turn.reasoningDetails,
|
|
450
|
+
searches,
|
|
451
|
+
onSearchStart: (search) => {
|
|
452
|
+
controller.enqueue(markerChunk({ kind: "web_search", phase: "start", item_id: search.itemId, query: search.query }));
|
|
453
|
+
},
|
|
454
|
+
onSearchDone: (search) => {
|
|
455
|
+
controller.enqueue(markerChunk({
|
|
456
|
+
kind: "web_search",
|
|
457
|
+
phase: "done",
|
|
458
|
+
item_id: search.itemId,
|
|
459
|
+
query: search.query,
|
|
460
|
+
status: search.status,
|
|
461
|
+
...(search.outcome?.anthropicResultBlocks !== undefined
|
|
462
|
+
? { result_blocks: search.outcome.anthropicResultBlocks }
|
|
463
|
+
: search.outcome !== undefined && search.status === "completed"
|
|
464
|
+
? {
|
|
465
|
+
result_blocks: search.outcome.citations.map((citation) => ({
|
|
466
|
+
type: "web_search_result",
|
|
467
|
+
url: citation.url,
|
|
468
|
+
...(citation.title !== undefined ? { title: citation.title } : {})
|
|
469
|
+
}))
|
|
470
|
+
}
|
|
471
|
+
: {})
|
|
472
|
+
}));
|
|
473
|
+
}
|
|
474
|
+
});
|
|
475
|
+
return false;
|
|
476
|
+
}
|
|
477
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Unwrap a backend error body for re-emission on a translated door.
|
|
3
|
+
*
|
|
4
|
+
* The Anthropic/Responses adapters run over the OpenAI-chat backend; when it
|
|
5
|
+
* fails, its body is already an OpenAI error envelope. Re-wrapping that JSON
|
|
6
|
+
* string as the `message` of a fresh `api_error` (the old behavior) both
|
|
7
|
+
* double-encodes the message and misclassifies caller errors — a 400
|
|
8
|
+
* `invalid_request_error` from the backend must stay an
|
|
9
|
+
* `invalid_request_error` on the door.
|
|
10
|
+
*/
|
|
11
|
+
export declare function unwrapUpstreamError(detail: string): {
|
|
12
|
+
type: string;
|
|
13
|
+
message: string;
|
|
14
|
+
};
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Unwrap a backend error body for re-emission on a translated door.
|
|
3
|
+
*
|
|
4
|
+
* The Anthropic/Responses adapters run over the OpenAI-chat backend; when it
|
|
5
|
+
* fails, its body is already an OpenAI error envelope. Re-wrapping that JSON
|
|
6
|
+
* string as the `message` of a fresh `api_error` (the old behavior) both
|
|
7
|
+
* double-encodes the message and misclassifies caller errors — a 400
|
|
8
|
+
* `invalid_request_error` from the backend must stay an
|
|
9
|
+
* `invalid_request_error` on the door.
|
|
10
|
+
*/
|
|
11
|
+
export function unwrapUpstreamError(detail) {
|
|
12
|
+
try {
|
|
13
|
+
const parsed = JSON.parse(detail);
|
|
14
|
+
if (typeof parsed.error?.message === "string") {
|
|
15
|
+
return {
|
|
16
|
+
type: typeof parsed.error.type === "string" ? parsed.error.type : "api_error",
|
|
17
|
+
message: parsed.error.message
|
|
18
|
+
};
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
catch {
|
|
22
|
+
// not JSON — fall through to the raw detail
|
|
23
|
+
}
|
|
24
|
+
return { type: "api_error", message: detail.slice(0, 2000) };
|
|
25
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Structural request validation at the gateway's wire doors.
|
|
3
|
+
*
|
|
4
|
+
* Hostile-input fuzzing showed that malformed bodies (a `model` array, a
|
|
5
|
+
* `messages` string, an empty body) sailed past the doors, hit deep code, and
|
|
6
|
+
* surfaced as 502 `upstream_error`s carrying raw TypeError text
|
|
7
|
+
* ("requested.startsWith is not a function", "body.messages is not iterable")
|
|
8
|
+
* or internal implementation details. A caller error must be a 400 in the
|
|
9
|
+
* door's native error envelope, and it must never reach a provider.
|
|
10
|
+
*
|
|
11
|
+
* Validation here is *structural only* — field presence and JSON types
|
|
12
|
+
* matching what the real providers enforce. Semantic rules (role enums,
|
|
13
|
+
* schema shapes) stay with the providers, so the doors never reject a shape a
|
|
14
|
+
* real provider would accept.
|
|
15
|
+
*/
|
|
16
|
+
export type WireRejection = {
|
|
17
|
+
status: number;
|
|
18
|
+
body: unknown;
|
|
19
|
+
};
|
|
20
|
+
/** OpenAI Chat Completions door (`/v1/chat/completions`). */
|
|
21
|
+
export declare function validateChatRequest(body: unknown): WireRejection | undefined;
|
|
22
|
+
/** Anthropic Messages door (`/v1/messages`). */
|
|
23
|
+
export declare function validateAnthropicRequest(body: unknown): WireRejection | undefined;
|
|
24
|
+
/** Anthropic `count_tokens` door: same message-shape contract, no minimum length. */
|
|
25
|
+
export declare function validateCountTokensRequest(body: unknown): WireRejection | undefined;
|
|
26
|
+
/** OpenAI Responses door (`/v1/responses`). */
|
|
27
|
+
export declare function validateResponsesRequest(body: unknown): WireRejection | undefined;
|