@velum-labs/routekit-gateway 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +28 -0
- package/dist/acp-agent.d.ts +38 -0
- package/dist/acp-agent.js +142 -0
- package/dist/acp-registry.d.ts +36 -0
- package/dist/acp-registry.js +85 -0
- package/dist/adapters/anthropic.d.ts +131 -0
- package/dist/adapters/anthropic.js +1195 -0
- package/dist/adapters/chat.d.ts +14 -0
- package/dist/adapters/chat.js +34 -0
- package/dist/adapters/cursor.d.ts +34 -0
- package/dist/adapters/cursor.js +305 -0
- package/dist/adapters/dropped.d.ts +10 -0
- package/dist/adapters/dropped.js +24 -0
- package/dist/adapters/openai-chat-wire.d.ts +93 -0
- package/dist/adapters/openai-chat-wire.js +143 -0
- package/dist/adapters/responses-stream.d.ts +7 -0
- package/dist/adapters/responses-stream.js +597 -0
- package/dist/adapters/responses.d.ts +174 -0
- package/dist/adapters/responses.js +778 -0
- package/dist/adapters/server-tool-loop.d.ts +94 -0
- package/dist/adapters/server-tool-loop.js +477 -0
- package/dist/adapters/upstream-error.d.ts +14 -0
- package/dist/adapters/upstream-error.js +25 -0
- package/dist/adapters/validate.d.ts +27 -0
- package/dist/adapters/validate.js +180 -0
- package/dist/adapters/web-search.d.ts +46 -0
- package/dist/adapters/web-search.js +151 -0
- package/dist/auth.d.ts +10 -0
- package/dist/auth.js +28 -0
- package/dist/backend.d.ts +151 -0
- package/dist/backend.js +143 -0
- package/dist/capacity-pool.d.ts +31 -0
- package/dist/capacity-pool.js +99 -0
- package/dist/cost.d.ts +49 -0
- package/dist/cost.js +112 -0
- package/dist/endpoint-health.d.ts +54 -0
- package/dist/endpoint-health.js +123 -0
- package/dist/index.d.ts +40 -0
- package/dist/index.js +23 -0
- package/dist/provenance.d.ts +31 -0
- package/dist/provenance.js +191 -0
- package/dist/provider-backends.d.ts +40 -0
- package/dist/provider-backends.js +1050 -0
- package/dist/provider-source.d.ts +40 -0
- package/dist/provider-source.js +293 -0
- package/dist/router.d.ts +168 -0
- package/dist/router.js +474 -0
- package/dist/server.d.ts +67 -0
- package/dist/server.js +930 -0
- package/dist/sse/chat-assembler.d.ts +45 -0
- package/dist/sse/chat-assembler.js +190 -0
- package/dist/sse/parse.d.ts +50 -0
- package/dist/sse/parse.js +149 -0
- package/dist/sse-wire.d.ts +10 -0
- package/dist/sse-wire.js +31 -0
- package/dist/switching-proxy.d.ts +15 -0
- package/dist/switching-proxy.js +232 -0
- package/dist/test/acp-agent.test.d.ts +1 -0
- package/dist/test/acp-agent.test.js +66 -0
- package/dist/test/acp-registry.test.d.ts +1 -0
- package/dist/test/acp-registry.test.js +70 -0
- package/dist/test/anthropic.test.d.ts +1 -0
- package/dist/test/anthropic.test.js +793 -0
- package/dist/test/auth.test.d.ts +1 -0
- package/dist/test/auth.test.js +25 -0
- package/dist/test/boundary.test.d.ts +1 -0
- package/dist/test/boundary.test.js +32 -0
- package/dist/test/chat.test.d.ts +1 -0
- package/dist/test/chat.test.js +418 -0
- package/dist/test/cost.test.d.ts +1 -0
- package/dist/test/cost.test.js +60 -0
- package/dist/test/cursor.test.d.ts +1 -0
- package/dist/test/cursor.test.js +100 -0
- package/dist/test/drain.test.d.ts +1 -0
- package/dist/test/drain.test.js +116 -0
- package/dist/test/dropped.test.d.ts +1 -0
- package/dist/test/dropped.test.js +80 -0
- package/dist/test/endpoint-health.test.d.ts +1 -0
- package/dist/test/endpoint-health.test.js +73 -0
- package/dist/test/provenance.test.d.ts +1 -0
- package/dist/test/provenance.test.js +176 -0
- package/dist/test/provider-backends.test.d.ts +1 -0
- package/dist/test/provider-backends.test.js +699 -0
- package/dist/test/responses.test.d.ts +1 -0
- package/dist/test/responses.test.js +813 -0
- package/dist/test/routed-backend.test.d.ts +1 -0
- package/dist/test/routed-backend.test.js +39 -0
- package/dist/test/router.test.d.ts +1 -0
- package/dist/test/router.test.js +297 -0
- package/dist/test/server-resilience.test.d.ts +1 -0
- package/dist/test/server-resilience.test.js +169 -0
- package/dist/test/sse-codec.test.d.ts +1 -0
- package/dist/test/sse-codec.test.js +186 -0
- package/dist/test/web-search-loop.test.d.ts +1 -0
- package/dist/test/web-search-loop.test.js +469 -0
- package/dist/test/wire-validation.test.d.ts +1 -0
- package/dist/test/wire-validation.test.js +140 -0
- package/package.json +48 -0
|
@@ -0,0 +1,597 @@
|
|
|
1
|
+
import { randomId } from "@velum-labs/routekit-runtime";
|
|
2
|
+
import { SseDecoder, SseParseError } from "../sse/parse.js";
|
|
3
|
+
import { serverToolMarkerOf } from "./server-tool-loop.js";
|
|
4
|
+
const ENCODER = new TextEncoder();
|
|
5
|
+
function typedToolArguments(args) {
|
|
6
|
+
if (args.trim().length === 0)
|
|
7
|
+
return {};
|
|
8
|
+
try {
|
|
9
|
+
return JSON.parse(args);
|
|
10
|
+
}
|
|
11
|
+
catch {
|
|
12
|
+
return {};
|
|
13
|
+
}
|
|
14
|
+
}
|
|
15
|
+
function typedToolCallItem(input) {
|
|
16
|
+
return {
|
|
17
|
+
type: `${input.name}_call`,
|
|
18
|
+
id: input.itemId,
|
|
19
|
+
call_id: input.callId,
|
|
20
|
+
status: "completed",
|
|
21
|
+
execution: "client",
|
|
22
|
+
arguments: typedToolArguments(input.args)
|
|
23
|
+
};
|
|
24
|
+
}
|
|
25
|
+
function customToolInput(args) {
|
|
26
|
+
if (args.trim().length === 0)
|
|
27
|
+
return "";
|
|
28
|
+
try {
|
|
29
|
+
const parsed = JSON.parse(args);
|
|
30
|
+
return typeof parsed.input === "string" ? parsed.input : args;
|
|
31
|
+
}
|
|
32
|
+
catch {
|
|
33
|
+
return args;
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
function sse(type, data, sequenceNumber) {
|
|
37
|
+
return ENCODER.encode(`event: ${type}\ndata: ${JSON.stringify({ type, sequence_number: sequenceNumber, ...data })}\n\n`);
|
|
38
|
+
}
|
|
39
|
+
/** The completed output item for an accumulated streamed tool call. */
|
|
40
|
+
function streamedToolItem(tool) {
|
|
41
|
+
switch (tool.kind) {
|
|
42
|
+
case "custom":
|
|
43
|
+
return {
|
|
44
|
+
type: "custom_tool_call",
|
|
45
|
+
id: tool.itemId,
|
|
46
|
+
call_id: tool.callId,
|
|
47
|
+
name: tool.name,
|
|
48
|
+
input: customToolInput(tool.args),
|
|
49
|
+
status: "completed"
|
|
50
|
+
};
|
|
51
|
+
case "typed":
|
|
52
|
+
return typedToolCallItem({
|
|
53
|
+
name: tool.name,
|
|
54
|
+
itemId: tool.itemId,
|
|
55
|
+
callId: tool.callId,
|
|
56
|
+
args: tool.args
|
|
57
|
+
});
|
|
58
|
+
case "function":
|
|
59
|
+
return {
|
|
60
|
+
type: "function_call",
|
|
61
|
+
id: tool.itemId,
|
|
62
|
+
call_id: tool.callId,
|
|
63
|
+
name: tool.name,
|
|
64
|
+
...(tool.namespace !== undefined ? { namespace: tool.namespace } : {}),
|
|
65
|
+
arguments: tool.args,
|
|
66
|
+
status: "completed"
|
|
67
|
+
};
|
|
68
|
+
case "server":
|
|
69
|
+
// Unreachable in practice: the server-tool loop intercepts these calls
|
|
70
|
+
// before they reach the translator. Render the native item shape so a
|
|
71
|
+
// stray call never surfaces as a function_call nobody dispatches.
|
|
72
|
+
return {
|
|
73
|
+
type: "web_search_call",
|
|
74
|
+
id: tool.itemId,
|
|
75
|
+
status: "completed",
|
|
76
|
+
action: { type: "search", query: webSearchQueryOf(tool.args) }
|
|
77
|
+
};
|
|
78
|
+
default: {
|
|
79
|
+
const exhaustive = tool.kind;
|
|
80
|
+
throw new Error(`unknown tool kind: ${String(exhaustive)}`);
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
/** The `query` from a web_search call's JSON arguments (raw args as fallback). */
|
|
85
|
+
function webSearchQueryOf(args) {
|
|
86
|
+
try {
|
|
87
|
+
const parsed = JSON.parse(args);
|
|
88
|
+
if (typeof parsed.query === "string")
|
|
89
|
+
return parsed.query;
|
|
90
|
+
}
|
|
91
|
+
catch {
|
|
92
|
+
// fall through to the raw argument string
|
|
93
|
+
}
|
|
94
|
+
return args;
|
|
95
|
+
}
|
|
96
|
+
export function openAiSseToResponses(upstream, model, toolRegistry = new Map()) {
|
|
97
|
+
const reader = upstream.getReader();
|
|
98
|
+
const sseDecoder = new SseDecoder();
|
|
99
|
+
const responseId = `resp_${randomId()}`;
|
|
100
|
+
const messageItemId = `msg_${randomId()}`;
|
|
101
|
+
const reasoningItemId = `rs_${randomId()}`;
|
|
102
|
+
// Tool fragments keyed by `index`, falling back to `id`, with id/index-less
|
|
103
|
+
// fragments appended to the last open call (parallel index-less calls no
|
|
104
|
+
// longer collapse into one — the same fix the shared assembler encodes).
|
|
105
|
+
// `toolList` preserves open order for the finalize/assemble passes.
|
|
106
|
+
const toolByIndex = new Map();
|
|
107
|
+
const toolById = new Map();
|
|
108
|
+
const toolList = [];
|
|
109
|
+
let lastTool;
|
|
110
|
+
let created = false;
|
|
111
|
+
let keepaliveTimer;
|
|
112
|
+
let textOpen = false;
|
|
113
|
+
let textValue = "";
|
|
114
|
+
let reasoningOpen = false;
|
|
115
|
+
let reasoningClosed = false;
|
|
116
|
+
const reasoningParts = [];
|
|
117
|
+
let reasoningOutputIndex = -1;
|
|
118
|
+
/** Index into `reasoningParts` of the open token-accumulating part, or -1. */
|
|
119
|
+
let tokenPartIndex = -1;
|
|
120
|
+
let nextOutputIndex = 0;
|
|
121
|
+
let messageOutputIndex = -1;
|
|
122
|
+
let finished = false;
|
|
123
|
+
let inputTokens;
|
|
124
|
+
let outputTokens;
|
|
125
|
+
let providerCost;
|
|
126
|
+
let sequenceNumber = 0;
|
|
127
|
+
const emit = (type, data) => {
|
|
128
|
+
const encoded = sse(type, data, sequenceNumber);
|
|
129
|
+
sequenceNumber += 1;
|
|
130
|
+
return encoded;
|
|
131
|
+
};
|
|
132
|
+
// Gateway-executed web searches (server-tool loop markers): open item per
|
|
133
|
+
// search, completed items collected for the terminal response payload.
|
|
134
|
+
const openSearches = new Map();
|
|
135
|
+
const completedSearchItems = [];
|
|
136
|
+
const baseResponse = (status, output) => ({
|
|
137
|
+
id: responseId,
|
|
138
|
+
object: "response",
|
|
139
|
+
created_at: Math.floor(Date.now() / 1000),
|
|
140
|
+
status,
|
|
141
|
+
model,
|
|
142
|
+
output,
|
|
143
|
+
usage: status === "completed"
|
|
144
|
+
? inputTokens !== undefined || outputTokens !== undefined
|
|
145
|
+
? {
|
|
146
|
+
...(inputTokens !== undefined ? { input_tokens: inputTokens } : {}),
|
|
147
|
+
...(outputTokens !== undefined ? { output_tokens: outputTokens } : {}),
|
|
148
|
+
...(inputTokens !== undefined && outputTokens !== undefined
|
|
149
|
+
? { total_tokens: inputTokens + outputTokens }
|
|
150
|
+
: {})
|
|
151
|
+
}
|
|
152
|
+
: null
|
|
153
|
+
: null,
|
|
154
|
+
...(status === "completed" && providerCost !== undefined ? { provider_cost: providerCost } : {})
|
|
155
|
+
});
|
|
156
|
+
const ensureCreated = (controller) => {
|
|
157
|
+
if (created)
|
|
158
|
+
return;
|
|
159
|
+
created = true;
|
|
160
|
+
controller.enqueue(emit("response.created", { response: baseResponse("in_progress", []) }));
|
|
161
|
+
};
|
|
162
|
+
// Reasoning summary item lifecycle. The item opens on the first reasoning
|
|
163
|
+
// delta and closes as soon as the first real output (text or tool call)
|
|
164
|
+
// begins. Two delta flavors share the item:
|
|
165
|
+
// - `reasoning_content` (reasoning narration): each delta is a complete beat,
|
|
166
|
+
// so each becomes its OWN summary part (added -> delta -> done). Codex
|
|
167
|
+
// flushes reasoning to the transcript on summary-part boundaries and
|
|
168
|
+
// promotes the newest part's bold header to its live status, so per-beat
|
|
169
|
+
// parts are what make the narration visible as it happens.
|
|
170
|
+
// - `reasoning` (the model's raw thinking tokens): deltas are token
|
|
171
|
+
// fragments, so they accumulate into ONE summary part that stays open
|
|
172
|
+
// until a beat arrives or the reasoning item closes.
|
|
173
|
+
const ensureReasoningItem = (controller) => {
|
|
174
|
+
ensureCreated(controller);
|
|
175
|
+
if (reasoningOpen || reasoningClosed)
|
|
176
|
+
return;
|
|
177
|
+
reasoningOpen = true;
|
|
178
|
+
reasoningOutputIndex = nextOutputIndex++;
|
|
179
|
+
controller.enqueue(emit("response.output_item.added", {
|
|
180
|
+
output_index: reasoningOutputIndex,
|
|
181
|
+
item: { type: "reasoning", id: reasoningItemId, summary: [] }
|
|
182
|
+
}));
|
|
183
|
+
};
|
|
184
|
+
const emitReasoningPart = (controller, text) => {
|
|
185
|
+
ensureReasoningItem(controller);
|
|
186
|
+
if (reasoningClosed)
|
|
187
|
+
return;
|
|
188
|
+
closeTokenPart(controller);
|
|
189
|
+
const summaryIndex = reasoningParts.length;
|
|
190
|
+
reasoningParts.push(text);
|
|
191
|
+
const base = { item_id: reasoningItemId, output_index: reasoningOutputIndex, summary_index: summaryIndex };
|
|
192
|
+
controller.enqueue(emit("response.reasoning_summary_part.added", { ...base, part: { type: "summary_text", text: "" } }));
|
|
193
|
+
controller.enqueue(emit("response.reasoning_summary_text.delta", { ...base, delta: text }));
|
|
194
|
+
controller.enqueue(emit("response.reasoning_summary_text.done", { ...base, text }));
|
|
195
|
+
controller.enqueue(emit("response.reasoning_summary_part.done", { ...base, part: { type: "summary_text", text } }));
|
|
196
|
+
};
|
|
197
|
+
// The single accumulating part for raw thinking tokens (`delta.reasoning`).
|
|
198
|
+
const emitReasoningTokenDelta = (controller, text) => {
|
|
199
|
+
ensureReasoningItem(controller);
|
|
200
|
+
if (reasoningClosed)
|
|
201
|
+
return;
|
|
202
|
+
if (tokenPartIndex === -1) {
|
|
203
|
+
tokenPartIndex = reasoningParts.length;
|
|
204
|
+
reasoningParts.push("");
|
|
205
|
+
controller.enqueue(emit("response.reasoning_summary_part.added", {
|
|
206
|
+
item_id: reasoningItemId,
|
|
207
|
+
output_index: reasoningOutputIndex,
|
|
208
|
+
summary_index: tokenPartIndex,
|
|
209
|
+
part: { type: "summary_text", text: "" }
|
|
210
|
+
}));
|
|
211
|
+
}
|
|
212
|
+
reasoningParts[tokenPartIndex] += text;
|
|
213
|
+
controller.enqueue(emit("response.reasoning_summary_text.delta", {
|
|
214
|
+
item_id: reasoningItemId,
|
|
215
|
+
output_index: reasoningOutputIndex,
|
|
216
|
+
summary_index: tokenPartIndex,
|
|
217
|
+
delta: text
|
|
218
|
+
}));
|
|
219
|
+
};
|
|
220
|
+
const closeTokenPart = (controller) => {
|
|
221
|
+
if (tokenPartIndex === -1)
|
|
222
|
+
return;
|
|
223
|
+
const text = reasoningParts[tokenPartIndex] ?? "";
|
|
224
|
+
const base = { item_id: reasoningItemId, output_index: reasoningOutputIndex, summary_index: tokenPartIndex };
|
|
225
|
+
tokenPartIndex = -1;
|
|
226
|
+
controller.enqueue(emit("response.reasoning_summary_text.done", { ...base, text }));
|
|
227
|
+
controller.enqueue(emit("response.reasoning_summary_part.done", { ...base, part: { type: "summary_text", text } }));
|
|
228
|
+
};
|
|
229
|
+
const reasoningSummary = () => reasoningParts.map((text) => ({ type: "summary_text", text }));
|
|
230
|
+
const closeReasoning = (controller) => {
|
|
231
|
+
if (!reasoningOpen || reasoningClosed)
|
|
232
|
+
return;
|
|
233
|
+
closeTokenPart(controller);
|
|
234
|
+
reasoningClosed = true;
|
|
235
|
+
controller.enqueue(emit("response.output_item.done", {
|
|
236
|
+
output_index: reasoningOutputIndex,
|
|
237
|
+
item: { type: "reasoning", id: reasoningItemId, summary: reasoningSummary() }
|
|
238
|
+
}));
|
|
239
|
+
};
|
|
240
|
+
const ensureText = (controller) => {
|
|
241
|
+
ensureCreated(controller);
|
|
242
|
+
closeReasoning(controller);
|
|
243
|
+
if (textOpen)
|
|
244
|
+
return;
|
|
245
|
+
textOpen = true;
|
|
246
|
+
messageOutputIndex = nextOutputIndex++;
|
|
247
|
+
controller.enqueue(emit("response.output_item.added", {
|
|
248
|
+
output_index: messageOutputIndex,
|
|
249
|
+
item: { type: "message", id: messageItemId, status: "in_progress", role: "assistant", content: [] }
|
|
250
|
+
}));
|
|
251
|
+
controller.enqueue(emit("response.content_part.added", {
|
|
252
|
+
item_id: messageItemId,
|
|
253
|
+
output_index: messageOutputIndex,
|
|
254
|
+
content_index: 0,
|
|
255
|
+
part: { type: "output_text", text: "", annotations: [] }
|
|
256
|
+
}));
|
|
257
|
+
};
|
|
258
|
+
// The server-tool loop injects marker chunks around each gateway-executed
|
|
259
|
+
// web search; render them as the native web_search_call item lifecycle.
|
|
260
|
+
const handleServerToolMarker = (controller, marker) => {
|
|
261
|
+
if (marker.phase === "start") {
|
|
262
|
+
ensureCreated(controller);
|
|
263
|
+
closeReasoning(controller);
|
|
264
|
+
const outputIndex = nextOutputIndex++;
|
|
265
|
+
openSearches.set(marker.item_id, { outputIndex });
|
|
266
|
+
controller.enqueue(emit("response.output_item.added", {
|
|
267
|
+
output_index: outputIndex,
|
|
268
|
+
item: {
|
|
269
|
+
type: "web_search_call",
|
|
270
|
+
id: marker.item_id,
|
|
271
|
+
status: "in_progress",
|
|
272
|
+
action: { type: "search", query: marker.query }
|
|
273
|
+
}
|
|
274
|
+
}));
|
|
275
|
+
controller.enqueue(emit("response.web_search_call.in_progress", { output_index: outputIndex, item_id: marker.item_id }));
|
|
276
|
+
controller.enqueue(emit("response.web_search_call.searching", { output_index: outputIndex, item_id: marker.item_id }));
|
|
277
|
+
return;
|
|
278
|
+
}
|
|
279
|
+
const outputIndex = openSearches.get(marker.item_id)?.outputIndex ?? nextOutputIndex++;
|
|
280
|
+
openSearches.delete(marker.item_id);
|
|
281
|
+
const item = {
|
|
282
|
+
type: "web_search_call",
|
|
283
|
+
id: marker.item_id,
|
|
284
|
+
status: marker.status === "failed" ? "failed" : "completed",
|
|
285
|
+
action: { type: "search", query: marker.query }
|
|
286
|
+
};
|
|
287
|
+
completedSearchItems.push(item);
|
|
288
|
+
controller.enqueue(emit("response.web_search_call.completed", { output_index: outputIndex, item_id: marker.item_id }));
|
|
289
|
+
controller.enqueue(emit("response.output_item.done", { output_index: outputIndex, item }));
|
|
290
|
+
};
|
|
291
|
+
const assembleOutput = () => {
|
|
292
|
+
const output = [];
|
|
293
|
+
if (reasoningParts.length > 0) {
|
|
294
|
+
output.push({ type: "reasoning", id: reasoningItemId, summary: reasoningSummary() });
|
|
295
|
+
}
|
|
296
|
+
output.push(...completedSearchItems);
|
|
297
|
+
if (textOpen) {
|
|
298
|
+
output.push({
|
|
299
|
+
type: "message",
|
|
300
|
+
id: messageItemId,
|
|
301
|
+
status: "completed",
|
|
302
|
+
role: "assistant",
|
|
303
|
+
content: [{ type: "output_text", text: textValue, annotations: [] }]
|
|
304
|
+
});
|
|
305
|
+
}
|
|
306
|
+
for (const tool of toolList) {
|
|
307
|
+
output.push(streamedToolItem(tool));
|
|
308
|
+
}
|
|
309
|
+
return output;
|
|
310
|
+
};
|
|
311
|
+
const finalize = (controller, terminal = "completed") => {
|
|
312
|
+
if (finished)
|
|
313
|
+
return;
|
|
314
|
+
finished = true;
|
|
315
|
+
if (keepaliveTimer !== undefined)
|
|
316
|
+
clearInterval(keepaliveTimer);
|
|
317
|
+
closeReasoning(controller);
|
|
318
|
+
if (textOpen) {
|
|
319
|
+
controller.enqueue(emit("response.output_text.done", {
|
|
320
|
+
item_id: messageItemId,
|
|
321
|
+
output_index: messageOutputIndex,
|
|
322
|
+
content_index: 0,
|
|
323
|
+
text: textValue
|
|
324
|
+
}));
|
|
325
|
+
controller.enqueue(emit("response.content_part.done", {
|
|
326
|
+
item_id: messageItemId,
|
|
327
|
+
output_index: messageOutputIndex,
|
|
328
|
+
content_index: 0,
|
|
329
|
+
part: { type: "output_text", text: textValue, annotations: [] }
|
|
330
|
+
}));
|
|
331
|
+
controller.enqueue(emit("response.output_item.done", {
|
|
332
|
+
output_index: messageOutputIndex,
|
|
333
|
+
item: {
|
|
334
|
+
type: "message",
|
|
335
|
+
id: messageItemId,
|
|
336
|
+
status: "completed",
|
|
337
|
+
role: "assistant",
|
|
338
|
+
content: [{ type: "output_text", text: textValue, annotations: [] }]
|
|
339
|
+
}
|
|
340
|
+
}));
|
|
341
|
+
}
|
|
342
|
+
for (const tool of toolList) {
|
|
343
|
+
if (tool.kind === "custom") {
|
|
344
|
+
// The raw input is only extractable from the completed JSON arguments,
|
|
345
|
+
// so a custom call flushes its whole input here in one delta + done.
|
|
346
|
+
const input = customToolInput(tool.args);
|
|
347
|
+
const base = { item_id: tool.itemId, output_index: tool.outputIndex };
|
|
348
|
+
controller.enqueue(emit("response.custom_tool_call_input.delta", { ...base, delta: input }));
|
|
349
|
+
controller.enqueue(emit("response.custom_tool_call_input.done", { ...base, input }));
|
|
350
|
+
controller.enqueue(emit("response.output_item.done", { output_index: tool.outputIndex, item: streamedToolItem(tool) }));
|
|
351
|
+
continue;
|
|
352
|
+
}
|
|
353
|
+
if (tool.kind === "typed" || tool.kind === "server") {
|
|
354
|
+
// A typed/server tool's native item carries its arguments as a
|
|
355
|
+
// completed JSON value, so it flushes whole in the item.done (no
|
|
356
|
+
// argument deltas).
|
|
357
|
+
controller.enqueue(emit("response.output_item.done", { output_index: tool.outputIndex, item: streamedToolItem(tool) }));
|
|
358
|
+
continue;
|
|
359
|
+
}
|
|
360
|
+
controller.enqueue(emit("response.function_call_arguments.done", {
|
|
361
|
+
item_id: tool.itemId,
|
|
362
|
+
output_index: tool.outputIndex,
|
|
363
|
+
arguments: tool.args
|
|
364
|
+
}));
|
|
365
|
+
controller.enqueue(emit("response.output_item.done", { output_index: tool.outputIndex, item: streamedToolItem(tool) }));
|
|
366
|
+
}
|
|
367
|
+
// Truncation is an error, not a clean stop: an upstream that ended without a
|
|
368
|
+
// finish_reason terminates as `response.incomplete`, never a fabricated
|
|
369
|
+
// `response.completed` (WS5.2), so callers that meter/persist see the turn
|
|
370
|
+
// as incomplete.
|
|
371
|
+
controller.enqueue(terminal === "completed"
|
|
372
|
+
? emit("response.completed", { response: baseResponse("completed", assembleOutput()) })
|
|
373
|
+
: emit("response.incomplete", { response: baseResponse("incomplete", assembleOutput()) }));
|
|
374
|
+
};
|
|
375
|
+
// A mid-stream provider failure (`data: {"error": {...}}` — e.g. the
|
|
376
|
+
// router's classified provider_error) becomes a `response.failed` event
|
|
377
|
+
// carrying the upstream message, so the consumer (codex) reports the real
|
|
378
|
+
// provider/API error instead of a bare "stream disconnected".
|
|
379
|
+
const failStream = (controller, error) => {
|
|
380
|
+
if (finished)
|
|
381
|
+
return;
|
|
382
|
+
finished = true;
|
|
383
|
+
if (keepaliveTimer !== undefined)
|
|
384
|
+
clearInterval(keepaliveTimer);
|
|
385
|
+
ensureCreated(controller);
|
|
386
|
+
controller.enqueue(emit("response.failed", {
|
|
387
|
+
response: {
|
|
388
|
+
...baseResponse("failed", assembleOutput()),
|
|
389
|
+
error: {
|
|
390
|
+
code: error.code ?? error.type ?? "upstream_error",
|
|
391
|
+
message: error.message ?? "upstream provider error"
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
}));
|
|
395
|
+
};
|
|
396
|
+
const process = (controller, chunk) => {
|
|
397
|
+
if (chunk.error != null) {
|
|
398
|
+
failStream(controller, chunk.error);
|
|
399
|
+
return;
|
|
400
|
+
}
|
|
401
|
+
// Real OpenAI streams carry `"usage": null` on every chunk except the
|
|
402
|
+
// final usage chunk, so a null must read as "absent".
|
|
403
|
+
inputTokens = chunk.usage?.prompt_tokens ?? inputTokens;
|
|
404
|
+
outputTokens = chunk.usage?.completion_tokens ?? outputTokens;
|
|
405
|
+
if (chunk.provider_cost !== undefined)
|
|
406
|
+
providerCost = chunk.provider_cost;
|
|
407
|
+
const choice = chunk.choices?.[0];
|
|
408
|
+
if (choice === undefined)
|
|
409
|
+
return;
|
|
410
|
+
const delta = choice.delta ?? {};
|
|
411
|
+
if (typeof delta.reasoning_content === "string" && delta.reasoning_content.length > 0 && !reasoningClosed) {
|
|
412
|
+
emitReasoningPart(controller, delta.reasoning_content);
|
|
413
|
+
}
|
|
414
|
+
if (typeof delta.reasoning === "string" && delta.reasoning.length > 0 && !reasoningClosed) {
|
|
415
|
+
emitReasoningTokenDelta(controller, delta.reasoning);
|
|
416
|
+
}
|
|
417
|
+
if (typeof delta.content === "string" && delta.content.length > 0) {
|
|
418
|
+
ensureText(controller);
|
|
419
|
+
textValue += delta.content;
|
|
420
|
+
controller.enqueue(emit("response.output_text.delta", {
|
|
421
|
+
item_id: messageItemId,
|
|
422
|
+
output_index: messageOutputIndex,
|
|
423
|
+
content_index: 0,
|
|
424
|
+
delta: delta.content
|
|
425
|
+
}));
|
|
426
|
+
}
|
|
427
|
+
if (Array.isArray(delta.tool_calls)) {
|
|
428
|
+
for (const call of delta.tool_calls) {
|
|
429
|
+
const indexKey = typeof call.index === "number" ? call.index : undefined;
|
|
430
|
+
const idKey = typeof call.id === "string" && call.id.length > 0 ? call.id : undefined;
|
|
431
|
+
let tool = indexKey !== undefined
|
|
432
|
+
? toolByIndex.get(indexKey)
|
|
433
|
+
: idKey !== undefined
|
|
434
|
+
? toolById.get(idKey)
|
|
435
|
+
: lastTool;
|
|
436
|
+
if (tool === undefined) {
|
|
437
|
+
ensureCreated(controller);
|
|
438
|
+
closeReasoning(controller);
|
|
439
|
+
const name = call.function?.name ?? "";
|
|
440
|
+
const entry = toolRegistry.get(name) ?? { kind: "function" };
|
|
441
|
+
const kind = entry.kind;
|
|
442
|
+
tool = {
|
|
443
|
+
outputIndex: nextOutputIndex++,
|
|
444
|
+
itemId: kind === "custom"
|
|
445
|
+
? `ctc_${randomId()}`
|
|
446
|
+
: kind === "typed"
|
|
447
|
+
? `ttc_${randomId()}`
|
|
448
|
+
: kind === "server"
|
|
449
|
+
? `ws_${randomId()}`
|
|
450
|
+
: `fc_${randomId()}`,
|
|
451
|
+
callId: call.id ?? `call_${randomId()}`,
|
|
452
|
+
name,
|
|
453
|
+
args: "",
|
|
454
|
+
kind,
|
|
455
|
+
...(entry.namespace !== undefined ? { namespace: entry.namespace } : {})
|
|
456
|
+
};
|
|
457
|
+
toolList.push(tool);
|
|
458
|
+
controller.enqueue(emit("response.output_item.added", {
|
|
459
|
+
output_index: tool.outputIndex,
|
|
460
|
+
item: kind === "custom"
|
|
461
|
+
? { type: "custom_tool_call", id: tool.itemId, call_id: tool.callId, name: tool.name, input: "" }
|
|
462
|
+
: kind === "typed"
|
|
463
|
+
? { type: `${tool.name}_call`, id: tool.itemId, call_id: tool.callId, status: "in_progress", execution: "client", arguments: {} }
|
|
464
|
+
: kind === "server"
|
|
465
|
+
? { type: "web_search_call", id: tool.itemId, status: "in_progress", action: { type: "search" } }
|
|
466
|
+
: {
|
|
467
|
+
type: "function_call",
|
|
468
|
+
id: tool.itemId,
|
|
469
|
+
call_id: tool.callId,
|
|
470
|
+
name: tool.name,
|
|
471
|
+
...(tool.namespace !== undefined ? { namespace: tool.namespace } : {}),
|
|
472
|
+
arguments: ""
|
|
473
|
+
}
|
|
474
|
+
}));
|
|
475
|
+
}
|
|
476
|
+
if (indexKey !== undefined && !toolByIndex.has(indexKey))
|
|
477
|
+
toolByIndex.set(indexKey, tool);
|
|
478
|
+
if (idKey !== undefined && !toolById.has(idKey))
|
|
479
|
+
toolById.set(idKey, tool);
|
|
480
|
+
lastTool = tool;
|
|
481
|
+
if (call.function?.name !== undefined && tool.name.length === 0)
|
|
482
|
+
tool.name = call.function.name;
|
|
483
|
+
const args = call.function?.arguments;
|
|
484
|
+
if (typeof args === "string" && args.length > 0) {
|
|
485
|
+
tool.args += args;
|
|
486
|
+
// Custom and typed calls buffer their arguments (extracted at finalize).
|
|
487
|
+
if (tool.kind === "function") {
|
|
488
|
+
controller.enqueue(emit("response.function_call_arguments.delta", {
|
|
489
|
+
item_id: tool.itemId,
|
|
490
|
+
output_index: tool.outputIndex,
|
|
491
|
+
delta: args
|
|
492
|
+
}));
|
|
493
|
+
}
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
}
|
|
497
|
+
if (choice.finish_reason !== null && choice.finish_reason !== undefined) {
|
|
498
|
+
finalize(controller);
|
|
499
|
+
}
|
|
500
|
+
};
|
|
501
|
+
// Backpressure handshake: the pump awaits `resumePull` while the consumer's
|
|
502
|
+
// queue is full; `pull` resolves it. This replaces the "return when
|
|
503
|
+
// desiredSize changed" hack with an explicit pump that drains the upstream
|
|
504
|
+
// reader to completion while honoring backpressure.
|
|
505
|
+
let resumePull;
|
|
506
|
+
const awaitPull = () => new Promise((resolve) => {
|
|
507
|
+
resumePull = resolve;
|
|
508
|
+
});
|
|
509
|
+
const handleEvent = (controller, data) => {
|
|
510
|
+
if (data.length === 0)
|
|
511
|
+
return;
|
|
512
|
+
if (data === "[DONE]") {
|
|
513
|
+
// A `[DONE]` with no prior finish_reason is truncation, not a clean stop.
|
|
514
|
+
if (!finished)
|
|
515
|
+
finalize(controller, "incomplete");
|
|
516
|
+
return;
|
|
517
|
+
}
|
|
518
|
+
let chunk;
|
|
519
|
+
try {
|
|
520
|
+
chunk = JSON.parse(data);
|
|
521
|
+
}
|
|
522
|
+
catch (error) {
|
|
523
|
+
// The live upstream stream is authoritative: a malformed payload is a
|
|
524
|
+
// stream error, never silently skipped (WS5).
|
|
525
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
526
|
+
throw new SseParseError(`malformed OpenAI SSE payload in Responses translation: ${detail}`, data.slice(0, 200));
|
|
527
|
+
}
|
|
528
|
+
const marker = serverToolMarkerOf(chunk);
|
|
529
|
+
if (marker !== undefined) {
|
|
530
|
+
handleServerToolMarker(controller, marker);
|
|
531
|
+
return;
|
|
532
|
+
}
|
|
533
|
+
process(controller, chunk);
|
|
534
|
+
};
|
|
535
|
+
const pump = async (controller) => {
|
|
536
|
+
try {
|
|
537
|
+
for (;;) {
|
|
538
|
+
if ((controller.desiredSize ?? 1) <= 0)
|
|
539
|
+
await awaitPull();
|
|
540
|
+
const { done, value } = await reader.read();
|
|
541
|
+
if (done) {
|
|
542
|
+
for (const event of sseDecoder.flush())
|
|
543
|
+
handleEvent(controller, event.data);
|
|
544
|
+
// Upstream closed with no finish_reason: incomplete, not completed.
|
|
545
|
+
if (!finished)
|
|
546
|
+
finalize(controller, "incomplete");
|
|
547
|
+
controller.close();
|
|
548
|
+
return;
|
|
549
|
+
}
|
|
550
|
+
if (value !== undefined) {
|
|
551
|
+
for (const event of sseDecoder.feed(value))
|
|
552
|
+
handleEvent(controller, event.data);
|
|
553
|
+
}
|
|
554
|
+
}
|
|
555
|
+
}
|
|
556
|
+
catch (error) {
|
|
557
|
+
if (keepaliveTimer !== undefined)
|
|
558
|
+
clearInterval(keepaliveTimer);
|
|
559
|
+
controller.error(error);
|
|
560
|
+
void reader.cancel(error).catch(() => undefined);
|
|
561
|
+
}
|
|
562
|
+
};
|
|
563
|
+
return new ReadableStream({
|
|
564
|
+
start(controller) {
|
|
565
|
+
// Emit `response.created` immediately and keep the connection alive with
|
|
566
|
+
// SSE comments while the upstream is still producing its first event. Real
|
|
567
|
+
// CLIs (codex) reconnect if they see nothing for a while — which happens
|
|
568
|
+
// during a slow upstream phase before the first token.
|
|
569
|
+
ensureCreated(controller);
|
|
570
|
+
keepaliveTimer = setInterval(() => {
|
|
571
|
+
if (finished)
|
|
572
|
+
return;
|
|
573
|
+
// Honor backpressure: skip the keepalive if the consumer's queue is full.
|
|
574
|
+
if ((controller.desiredSize ?? 1) <= 0)
|
|
575
|
+
return;
|
|
576
|
+
try {
|
|
577
|
+
controller.enqueue(ENCODER.encode(": keepalive\n\n"));
|
|
578
|
+
}
|
|
579
|
+
catch {
|
|
580
|
+
// controller closed
|
|
581
|
+
}
|
|
582
|
+
}, 3000);
|
|
583
|
+
void pump(controller);
|
|
584
|
+
},
|
|
585
|
+
pull() {
|
|
586
|
+
resumePull?.();
|
|
587
|
+
resumePull = undefined;
|
|
588
|
+
},
|
|
589
|
+
cancel(reason) {
|
|
590
|
+
if (keepaliveTimer !== undefined)
|
|
591
|
+
clearInterval(keepaliveTimer);
|
|
592
|
+
resumePull?.();
|
|
593
|
+
resumePull = undefined;
|
|
594
|
+
return reader.cancel(reason);
|
|
595
|
+
}
|
|
596
|
+
});
|
|
597
|
+
}
|