@maci0/dsh-google-vertex 0.0.0-stage → 0.12.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +166 -2
- package/cordis.patch.yml +34 -0
- package/icon.svg +6 -0
- package/lib/adapter.js +398 -0
- package/lib/auth.js +246 -0
- package/lib/client.js +206 -0
- package/lib/discovery.js +180 -0
- package/lib/gemini.js +561 -0
- package/lib/gemini_adapter.js +61 -0
- package/lib/host.js +18 -0
- package/lib/index.js +196 -0
- package/lib/types/adapter.d.ts +219 -0
- package/lib/types/auth.d.ts +111 -0
- package/lib/types/discovery.d.ts +57 -0
- package/lib/types/gemini.d.ts +210 -0
- package/lib/types/gemini_adapter.d.ts +54 -0
- package/lib/types/host.d.ts +202 -0
- package/lib/types/index.d.ts +161 -0
- package/lib/types/wire-shared.d.ts +19 -0
- package/lib/types/wire.d.ts +271 -0
- package/lib/wire-shared.js +59 -0
- package/lib/wire.js +722 -0
- package/locale/en.json +6 -0
- package/locale/zh.json +6 -0
- package/package.json +93 -4
package/lib/wire.js
ADDED
|
@@ -0,0 +1,722 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Wire translation for Google-hosted Anthropic models on Vertex AI: the request
|
|
3
|
+
* body Vertex accepts, the two endpoint shapes (global and regional), the SSE
|
|
4
|
+
* event stream it answers with, and the failure envelope it refuses with.
|
|
5
|
+
*
|
|
6
|
+
* Everything here is pure: the adapter feeds it bytes and yields the chunks it
|
|
7
|
+
* returns, so the protocol is testable without a harness or a network.
|
|
8
|
+
*
|
|
9
|
+
* Vertex serves Claude through the publisher endpoint rather than Anthropic's
|
|
10
|
+
* own API, which changes three things this module owns: the path
|
|
11
|
+
* (`…/publishers/anthropic/models/{model}:streamRawPredict`), a body carrying
|
|
12
|
+
* `anthropic_version` and no `model` field, and a bearer token instead of an
|
|
13
|
+
* `x-api-key` header.
|
|
14
|
+
*
|
|
15
|
+
* @module dsh-google-vertex/wire
|
|
16
|
+
*/
|
|
17
|
+
import { systemParts as collectSystemParts, takeCounters, toolInput } from './wire-shared.js';
|
|
18
|
+
// The three provider-neutral codes are the harness's own; imported for local use
|
|
19
|
+
// and re-exported so callers keep importing them from here.
|
|
20
|
+
import { CONTEXT_WINDOW_EXCEEDED_CODE, EMPTY_RESPONSE_CODE, QUOTA_EXCEEDED_CODE, } from '@deepseek-ai/dsh-llm';
|
|
21
|
+
export { CONTEXT_WINDOW_EXCEEDED_CODE, EMPTY_RESPONSE_CODE, QUOTA_EXCEEDED_CODE };
|
|
22
|
+
/** Vertex's Anthropic API version marker; required in every request body. */
|
|
23
|
+
const VERTEX_ANTHROPIC_VERSION = 'vertex-2023-10-16';
|
|
24
|
+
/** Endpoint host used when no region is configured. */
|
|
25
|
+
export const DEFAULT_LOCATION = 'global';
|
|
26
|
+
/**
|
|
27
|
+
* Default bound on the interval between two stream reads, in milliseconds.
|
|
28
|
+
* Matches the shipped remote adapters (`llm-deepseek/src/common/defaults.ts`).
|
|
29
|
+
*/
|
|
30
|
+
export const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 300_000;
|
|
31
|
+
/** Largest delay `setTimeout` schedules without clamping it to one millisecond. */
|
|
32
|
+
export const MAX_TIMER_DELAY_MS = 2_147_483_647;
|
|
33
|
+
/**
|
|
34
|
+
* Host for a location. `global` has no region prefix: the documented global
|
|
35
|
+
* endpoint is `aiplatform.googleapis.com`, and `global-aiplatform…` is not a
|
|
36
|
+
* host Vertex answers on.
|
|
37
|
+
* @param location - configured or defaulted region, or `global`.
|
|
38
|
+
* @returns the origin, without a trailing slash.
|
|
39
|
+
*/
|
|
40
|
+
export function endpointOrigin(location) {
|
|
41
|
+
return location === DEFAULT_LOCATION
|
|
42
|
+
? 'https://aiplatform.googleapis.com'
|
|
43
|
+
: `https://${location}-aiplatform.googleapis.com`;
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* The streaming publisher path for one model.
|
|
47
|
+
* @param project - Google Cloud project id.
|
|
48
|
+
* @param location - region, or `global`.
|
|
49
|
+
* @param model - publisher model id, e.g. `claude-sonnet-4-5`.
|
|
50
|
+
* @returns the absolute request URL.
|
|
51
|
+
*/
|
|
52
|
+
export function endpointFor(project, location, model) {
|
|
53
|
+
const path = `/v1/projects/${encodeURIComponent(project)}/locations/${encodeURIComponent(location)}`
|
|
54
|
+
+ `/publishers/anthropic/models/${encodeURIComponent(model)}:streamRawPredict`;
|
|
55
|
+
return `${endpointOrigin(location)}${path}`;
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* One cache breakpoint is only useful where Vertex can store a prefix, which is
|
|
59
|
+
* every block type this adapter emits: the request never carries the provider's
|
|
60
|
+
* own non-cacheable blocks.
|
|
61
|
+
*/
|
|
62
|
+
const CACHE_CONTROL = { type: 'ephemeral' };
|
|
63
|
+
/**
|
|
64
|
+
* Flatten a tool-role message's content to the text Vertex accepts.
|
|
65
|
+
*
|
|
66
|
+
* The Gemini route sends a tool result the same way, so this is exported rather
|
|
67
|
+
* than written twice.
|
|
68
|
+
*/
|
|
69
|
+
export function resultText(blocks) {
|
|
70
|
+
const parts = [];
|
|
71
|
+
for (const block of blocks) {
|
|
72
|
+
if (block.type === 'text' && typeof block.text === 'string')
|
|
73
|
+
parts.push(block.text);
|
|
74
|
+
else if (block.type === 'tool-call')
|
|
75
|
+
parts.push(`[tool call] ${String(block.name)}(${String(block.arguments)})`);
|
|
76
|
+
}
|
|
77
|
+
const joined = parts.join('');
|
|
78
|
+
// An empty tool result is still a result: the provider rejects empty content.
|
|
79
|
+
return joined.length > 0 ? joined : '(no output)';
|
|
80
|
+
}
|
|
81
|
+
/** Ensure a tool's JSON Schema is one Anthropic accepts. */
|
|
82
|
+
function inputSchema(parameters) {
|
|
83
|
+
return {
|
|
84
|
+
type: 'object',
|
|
85
|
+
...parameters,
|
|
86
|
+
properties: typeof parameters['properties'] === 'object' && parameters['properties'] !== null
|
|
87
|
+
? parameters['properties']
|
|
88
|
+
: {},
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
/** Project one non-system harness message onto its wire content. */
|
|
92
|
+
function messageContent(message) {
|
|
93
|
+
const content = [];
|
|
94
|
+
for (const block of message.content) {
|
|
95
|
+
switch (block.type) {
|
|
96
|
+
case 'text':
|
|
97
|
+
if (typeof block.text === 'string' && block.text.length > 0)
|
|
98
|
+
content.push({ type: 'text', text: block.text });
|
|
99
|
+
break;
|
|
100
|
+
case 'tool-call':
|
|
101
|
+
content.push({
|
|
102
|
+
type: 'tool_use',
|
|
103
|
+
id: String(block.id),
|
|
104
|
+
name: String(block.name),
|
|
105
|
+
input: toolInput(String(block.arguments)),
|
|
106
|
+
});
|
|
107
|
+
break;
|
|
108
|
+
default:
|
|
109
|
+
// Reasoning blocks carry no reusable text: Vertex requires a signed
|
|
110
|
+
// thinking block to replay one, and this adapter never enables
|
|
111
|
+
// extended thinking, so a stray block is history from another route.
|
|
112
|
+
break;
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
return content;
|
|
116
|
+
}
|
|
117
|
+
/**
|
|
118
|
+
* Project one harness `tool`-role message onto the `tool_result` block Vertex
|
|
119
|
+
* requires in the user turn that follows the assistant's `tool_use`.
|
|
120
|
+
*
|
|
121
|
+
* The harness attributes a tool result to the call on the message itself
|
|
122
|
+
* (`toolCallId`, `isError`); the provider reads both from the block.
|
|
123
|
+
*/
|
|
124
|
+
function toolResultContent(message) {
|
|
125
|
+
const callId = message.toolCallId;
|
|
126
|
+
if (callId === undefined || callId.length === 0)
|
|
127
|
+
return messageContent(message);
|
|
128
|
+
return [{
|
|
129
|
+
type: 'tool_result',
|
|
130
|
+
tool_use_id: callId,
|
|
131
|
+
content: resultText(message.content),
|
|
132
|
+
...message.isError === true ? { is_error: true } : {},
|
|
133
|
+
}];
|
|
134
|
+
}
|
|
135
|
+
/**
|
|
136
|
+
* Build the request body for one model call.
|
|
137
|
+
*
|
|
138
|
+
* History is projected block by block, consecutive same-role messages are
|
|
139
|
+
* merged (the provider reads one turn per role), and three cache breakpoints are
|
|
140
|
+
* placed: end of tools, end of system, and the final block of the conversation.
|
|
141
|
+
* Those three make the static prefix and every earlier turn cacheable while the
|
|
142
|
+
* growing tail is re-read.
|
|
143
|
+
* @param options - the harness request.
|
|
144
|
+
* @param config - project, location, and default output cap.
|
|
145
|
+
* @returns the wire body.
|
|
146
|
+
*/
|
|
147
|
+
export function buildRequestBody(options, config) {
|
|
148
|
+
const messages = [];
|
|
149
|
+
for (const message of options.messages) {
|
|
150
|
+
if (message.role === 'system' || message.role === 'developer')
|
|
151
|
+
continue;
|
|
152
|
+
const role = message.role === 'assistant' ? 'assistant' : 'user';
|
|
153
|
+
const content = message.role === 'tool' ? toolResultContent(message) : messageContent(message);
|
|
154
|
+
if (content.length === 0)
|
|
155
|
+
continue;
|
|
156
|
+
const previous = messages.at(-1);
|
|
157
|
+
if (previous !== undefined && previous.role === role)
|
|
158
|
+
previous.content.push(...content);
|
|
159
|
+
else
|
|
160
|
+
messages.push({ role, content });
|
|
161
|
+
}
|
|
162
|
+
const systemParts = collectSystemParts(options);
|
|
163
|
+
const system = systemParts.map(text => ({ type: 'text', text }));
|
|
164
|
+
const tools = (options.tools ?? []).map((tool) => ({
|
|
165
|
+
name: tool.name,
|
|
166
|
+
description: tool.description,
|
|
167
|
+
input_schema: inputSchema(tool.parameters),
|
|
168
|
+
}));
|
|
169
|
+
const lastTool = tools.at(-1);
|
|
170
|
+
if (lastTool !== undefined)
|
|
171
|
+
lastTool.cache_control = CACHE_CONTROL;
|
|
172
|
+
const lastSystem = system.at(-1);
|
|
173
|
+
if (lastSystem !== undefined)
|
|
174
|
+
lastSystem.cache_control = CACHE_CONTROL;
|
|
175
|
+
const lastMessage = messages.at(-1);
|
|
176
|
+
const lastBlock = lastMessage?.content.at(-1);
|
|
177
|
+
if (lastBlock !== undefined)
|
|
178
|
+
lastBlock.cache_control = CACHE_CONTROL;
|
|
179
|
+
return {
|
|
180
|
+
anthropic_version: VERTEX_ANTHROPIC_VERSION,
|
|
181
|
+
stream: true,
|
|
182
|
+
max_tokens: options.maxTokens ?? config.maxTokens,
|
|
183
|
+
messages,
|
|
184
|
+
...system.length === 0 ? {} : { system },
|
|
185
|
+
...tools.length === 0 ? {} : { tools },
|
|
186
|
+
...options.temperature === undefined ? {} : { temperature: options.temperature },
|
|
187
|
+
...options.stop === undefined || options.stop.length === 0 ? {} : { stop_sequences: [...options.stop] },
|
|
188
|
+
};
|
|
189
|
+
}
|
|
190
|
+
/**
|
|
191
|
+
* Map Vertex's usage counters onto harness accounting.
|
|
192
|
+
*
|
|
193
|
+
* Vertex splits prompt tokens the way the harness does (`input_tokens` counts
|
|
194
|
+
* only uncached input, with cache reads and writes reported separately), so the
|
|
195
|
+
* fields transfer without arithmetic, and the total is their sum.
|
|
196
|
+
* @param usage - the latest cumulative counters seen on the stream.
|
|
197
|
+
* @returns disjoint harness counts.
|
|
198
|
+
*/
|
|
199
|
+
export function mapUsage(usage) {
|
|
200
|
+
const input = usage.input_tokens ?? 0;
|
|
201
|
+
const output = usage.output_tokens ?? 0;
|
|
202
|
+
const cacheRead = usage.cache_read_input_tokens ?? 0;
|
|
203
|
+
const cacheWrite = usage.cache_creation_input_tokens ?? 0;
|
|
204
|
+
return {
|
|
205
|
+
inputTokens: input,
|
|
206
|
+
outputTokens: output,
|
|
207
|
+
totalTokens: input + output + cacheRead + cacheWrite,
|
|
208
|
+
...cacheRead > 0 ? { cacheReadTokens: cacheRead } : {},
|
|
209
|
+
...cacheWrite > 0 ? { cacheWriteTokens: cacheWrite } : {},
|
|
210
|
+
};
|
|
211
|
+
}
|
|
212
|
+
/**
|
|
213
|
+
* Map one provider stop reason onto the harness vocabulary.
|
|
214
|
+
*
|
|
215
|
+
* `pause_turn` needs no case of its own: it cannot recur here, because this
|
|
216
|
+
* adapter declares no server-executed tools, so the answer is complete, which
|
|
217
|
+
* is what the default already reports for any reason the provider adds.
|
|
218
|
+
* @param reason - the `stop_reason` Vertex reported, if any.
|
|
219
|
+
* @returns the harness finish reason.
|
|
220
|
+
*/
|
|
221
|
+
export function mapStopReason(reason) {
|
|
222
|
+
switch (reason) {
|
|
223
|
+
case 'max_tokens':
|
|
224
|
+
return { kind: 'max-tokens' };
|
|
225
|
+
case 'tool_use':
|
|
226
|
+
return { kind: 'tool-calls' };
|
|
227
|
+
case 'refusal':
|
|
228
|
+
return {
|
|
229
|
+
kind: 'error',
|
|
230
|
+
failure: { message: 'google-vertex: the model refused to answer', code: 'REFUSAL' },
|
|
231
|
+
};
|
|
232
|
+
case 'model_context_window_exceeded':
|
|
233
|
+
return {
|
|
234
|
+
kind: 'error',
|
|
235
|
+
failure: {
|
|
236
|
+
message: 'google-vertex: the request exceeded the model context window',
|
|
237
|
+
code: CONTEXT_WINDOW_EXCEEDED_CODE,
|
|
238
|
+
},
|
|
239
|
+
};
|
|
240
|
+
default:
|
|
241
|
+
return { kind: 'stop' };
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
/**
|
|
245
|
+
* Extract the provider's message, and Google's canonical `status` when the
|
|
246
|
+
* envelope carries one, from a failure body.
|
|
247
|
+
*/
|
|
248
|
+
function failureDetail(body) {
|
|
249
|
+
try {
|
|
250
|
+
const parsed = JSON.parse(body);
|
|
251
|
+
if (typeof parsed === 'object' && parsed !== null) {
|
|
252
|
+
const error = parsed['error'];
|
|
253
|
+
if (typeof error === 'object' && error !== null) {
|
|
254
|
+
const fields = error;
|
|
255
|
+
const status = typeof fields['status'] === 'string' ? fields['status'] : '';
|
|
256
|
+
if (typeof fields['message'] === 'string')
|
|
257
|
+
return { message: fields['message'], status };
|
|
258
|
+
}
|
|
259
|
+
const message = parsed['message'];
|
|
260
|
+
if (typeof message === 'string')
|
|
261
|
+
return { message, status: '' };
|
|
262
|
+
}
|
|
263
|
+
}
|
|
264
|
+
catch {
|
|
265
|
+
// Not JSON: the raw body is the most specific thing available.
|
|
266
|
+
}
|
|
267
|
+
return { message: body.slice(0, 300).replace(/\s+/g, ' ').trim(), status: '' };
|
|
268
|
+
}
|
|
269
|
+
/**
|
|
270
|
+
* Classify a refused HTTP response.
|
|
271
|
+
*
|
|
272
|
+
* Two refusals carry a code of their own because the harness treats them
|
|
273
|
+
* differently: an oversized request must not be retried, and an exhausted quota
|
|
274
|
+
* is a capacity problem rather than an invalid one. Google names the quota in
|
|
275
|
+
* the envelope's `status` (`RESOURCE_EXHAUSTED`), so that is read as well as
|
|
276
|
+
* the message, exactly as {@link failureForEvent} reads an in-band envelope.
|
|
277
|
+
* @param status - HTTP status.
|
|
278
|
+
* @param body - response body text.
|
|
279
|
+
* @param subject - route description named in the failure.
|
|
280
|
+
* @param headers - HTTP response headers carrying an optional Retry-After.
|
|
281
|
+
* @returns the failure to report.
|
|
282
|
+
*/
|
|
283
|
+
export function failureForStatus(status, body, subject, headers) {
|
|
284
|
+
const detail = failureDetail(body);
|
|
285
|
+
const message = `google-vertex: ${subject}: HTTP ${status}${detail.message.length > 0 ? `: ${detail.message}` : ''}`;
|
|
286
|
+
const retry = headers?.get('retry-after')?.trim();
|
|
287
|
+
const delay = retry === undefined ? NaN : /^\d+(?:\.\d+)?$/u.test(retry)
|
|
288
|
+
? Number(retry) * 1000 : /^[A-Za-z]/u.test(retry) ? Date.parse(retry) - Date.now() : NaN;
|
|
289
|
+
return {
|
|
290
|
+
message, code: codeForDetail(`${detail.status} ${detail.message}`) ?? codeForStatus(status), status,
|
|
291
|
+
...(Number.isFinite(delay) && delay > 0 ? { providerRetryAfterMs: delay } : {}),
|
|
292
|
+
};
|
|
293
|
+
}
|
|
294
|
+
/**
|
|
295
|
+
* Canonical code for one HTTP status, shared by a refused response and by an
|
|
296
|
+
* in-band error envelope naming the same condition numerically.
|
|
297
|
+
* @param status - HTTP status, or the provider's numeric error code.
|
|
298
|
+
* @returns the harness code.
|
|
299
|
+
*/
|
|
300
|
+
function codeForStatus(status) {
|
|
301
|
+
return status === 400 || status === 413 || status === 422
|
|
302
|
+
? 'INVALID_REQUEST'
|
|
303
|
+
: status === 401 || status === 403
|
|
304
|
+
? 'AUTH'
|
|
305
|
+
: status === 404
|
|
306
|
+
? 'NOT_FOUND'
|
|
307
|
+
: status === 408 || status === 504
|
|
308
|
+
? 'TIMEOUT'
|
|
309
|
+
: status === 429
|
|
310
|
+
? 'RATE_LIMIT'
|
|
311
|
+
: status >= 500
|
|
312
|
+
? 'SERVER'
|
|
313
|
+
: 'TRANSPORT';
|
|
314
|
+
}
|
|
315
|
+
/**
|
|
316
|
+
* Canonical code for a refusal whose own words name a cause.
|
|
317
|
+
* @param detail - the provider's message or error status text.
|
|
318
|
+
* @returns the code, or undefined when the text names no special cause.
|
|
319
|
+
*/
|
|
320
|
+
function codeForDetail(detail) {
|
|
321
|
+
if (/prompt is too long|input (?:is )?too long|maximum context length|context length|too many tokens|exceeds the maximum/i.test(detail)) {
|
|
322
|
+
return CONTEXT_WINDOW_EXCEEDED_CODE;
|
|
323
|
+
}
|
|
324
|
+
if (/RESOURCE_EXHAUSTED|quota/i.test(detail))
|
|
325
|
+
return QUOTA_EXCEEDED_CODE;
|
|
326
|
+
return undefined;
|
|
327
|
+
}
|
|
328
|
+
/** Canonical code for Anthropic's named error type. */
|
|
329
|
+
function codeForErrorType(type) {
|
|
330
|
+
return type === 'overloaded_error'
|
|
331
|
+
? 'SERVER'
|
|
332
|
+
: type === 'rate_limit_error'
|
|
333
|
+
? 'RATE_LIMIT'
|
|
334
|
+
: type === 'authentication_error' || type === 'permission_error'
|
|
335
|
+
? 'AUTH'
|
|
336
|
+
: type === 'invalid_request_error'
|
|
337
|
+
? 'INVALID_REQUEST'
|
|
338
|
+
: 'SERVER';
|
|
339
|
+
}
|
|
340
|
+
/**
|
|
341
|
+
* Canonical code for Google's `{code, message, status}` error envelope, which
|
|
342
|
+
* classifies exactly as a refused body does: a named cause first, then the
|
|
343
|
+
* numeric code.
|
|
344
|
+
* @param fields - the envelope's own members.
|
|
345
|
+
* @param message - its message, already read.
|
|
346
|
+
* @returns the harness code.
|
|
347
|
+
*/
|
|
348
|
+
function codeForGoogleError(fields, message) {
|
|
349
|
+
const status = typeof fields['status'] === 'string' ? fields['status'] : '';
|
|
350
|
+
const detail = codeForDetail(`${status} ${message}`);
|
|
351
|
+
return detail ?? codeForStatus(typeof fields['code'] === 'number' ? fields['code'] : 500);
|
|
352
|
+
}
|
|
353
|
+
/**
|
|
354
|
+
* Classify a stream-level `error` payload.
|
|
355
|
+
*
|
|
356
|
+
* Both publishers deliver one mid-stream: Anthropic's envelope names the
|
|
357
|
+
* condition in `type`, and Google's in a canonical `status` plus a numeric
|
|
358
|
+
* `code`. A provider error is what turns a mid-stream refusal into a terminal
|
|
359
|
+
* failure instead of a truncated response.
|
|
360
|
+
* @param error - the payload's `error` member.
|
|
361
|
+
* @returns the failure to report.
|
|
362
|
+
*/
|
|
363
|
+
export function failureForEvent(error) {
|
|
364
|
+
const fields = (typeof error === 'object' && error !== null ? error : {});
|
|
365
|
+
const message = typeof fields['message'] === 'string' ? fields['message'] : 'the provider reported an error';
|
|
366
|
+
const type = fields['type'];
|
|
367
|
+
const code = typeof type === 'string'
|
|
368
|
+
? codeForErrorType(type)
|
|
369
|
+
: codeForGoogleError(fields, message);
|
|
370
|
+
return { message: `google-vertex: ${message}`, code };
|
|
371
|
+
}
|
|
372
|
+
/**
|
|
373
|
+
* Decode one SSE record (everything up to a blank line) into its JSON payload.
|
|
374
|
+
*
|
|
375
|
+
* `event:` lines are ignored because every Vertex payload carries its own
|
|
376
|
+
* `type`; `data:` lines are concatenated, which is what the SSE specification
|
|
377
|
+
* requires for a multi-line payload.
|
|
378
|
+
* @param record - one raw record, without its trailing blank line.
|
|
379
|
+
* @returns the parsed payload, or undefined for a comment, ping, or malformed record.
|
|
380
|
+
*/
|
|
381
|
+
export function parseSseRecord(record) {
|
|
382
|
+
let payload = '';
|
|
383
|
+
let found = false;
|
|
384
|
+
const length = record.length;
|
|
385
|
+
for (let cursor = 0; cursor <= length;) {
|
|
386
|
+
let end = record.indexOf('\n', cursor);
|
|
387
|
+
if (end === -1)
|
|
388
|
+
end = length;
|
|
389
|
+
if (record.startsWith('data:', cursor)) {
|
|
390
|
+
const value = record.slice(record.charCodeAt(cursor + 5) === 0x20 ? cursor + 6 : cursor + 5, end);
|
|
391
|
+
payload = found ? `${payload}\n${value}` : value;
|
|
392
|
+
found = true;
|
|
393
|
+
}
|
|
394
|
+
if (end === length)
|
|
395
|
+
break;
|
|
396
|
+
cursor = end + 1;
|
|
397
|
+
}
|
|
398
|
+
if (!found || payload.length === 0)
|
|
399
|
+
return undefined;
|
|
400
|
+
try {
|
|
401
|
+
const parsed = JSON.parse(payload);
|
|
402
|
+
return typeof parsed === 'object' && parsed !== null ? parsed : undefined;
|
|
403
|
+
}
|
|
404
|
+
catch {
|
|
405
|
+
return undefined;
|
|
406
|
+
}
|
|
407
|
+
}
|
|
408
|
+
/**
|
|
409
|
+
* Reassembles SSE records from arbitrarily split transport chunks.
|
|
410
|
+
*
|
|
411
|
+
* Line endings are normalized as text arrives, so a `\r\n` split across two
|
|
412
|
+
* chunks still frames one record: a chunk-final CR is carried over and resolved
|
|
413
|
+
* against the next chunk's first byte rather than normalized chunk by chunk.
|
|
414
|
+
*/
|
|
415
|
+
export class SseBuffer {
|
|
416
|
+
#buffer = '';
|
|
417
|
+
/** A trailing CR held back: the next chunk may carry the LF that pairs with it. */
|
|
418
|
+
#pendingCr = false;
|
|
419
|
+
/**
|
|
420
|
+
* Absorb one decoded chunk.
|
|
421
|
+
* @param text - newly decoded text.
|
|
422
|
+
* @returns every complete record it completes, in order.
|
|
423
|
+
*/
|
|
424
|
+
push(text) {
|
|
425
|
+
// A CR at the end of a chunk cannot be classified yet: it is either the
|
|
426
|
+
// first half of a CRLF whose LF opens the next chunk, or a lone CR. Holding
|
|
427
|
+
// it back is what keeps a split CRLF one line ending: normalizing the
|
|
428
|
+
// chunk on its own would turn the pair into two, framing a phantom record
|
|
429
|
+
// and cutting a multi-line payload in half.
|
|
430
|
+
let chunk = text;
|
|
431
|
+
if (this.#pendingCr) {
|
|
432
|
+
// The held CR is a line ending either way; its LF, if this chunk opens
|
|
433
|
+
// with one, is consumed by the pair and not a second ending.
|
|
434
|
+
chunk = `\n${chunk.startsWith('\n') ? chunk.slice(1) : chunk}`;
|
|
435
|
+
this.#pendingCr = false;
|
|
436
|
+
}
|
|
437
|
+
this.#pendingCr = chunk.endsWith('\r');
|
|
438
|
+
const content = this.#pendingCr ? chunk.slice(0, -1) : chunk;
|
|
439
|
+
// The regex engine is only worth starting when the chunk carries a CR; a
|
|
440
|
+
// body served with bare LF (the common case) is scanned once and kept.
|
|
441
|
+
this.#buffer += content.includes('\r') ? content.replace(/\r\n?/g, '\n') : content;
|
|
442
|
+
const records = this.#buffer.split('\n\n');
|
|
443
|
+
this.#buffer = records.pop() ?? '';
|
|
444
|
+
return records;
|
|
445
|
+
}
|
|
446
|
+
/**
|
|
447
|
+
* Release whatever a completed body ended with.
|
|
448
|
+
* @returns the trailing record, or undefined when the body ended on a boundary.
|
|
449
|
+
*/
|
|
450
|
+
flush() {
|
|
451
|
+
// A held CR was a lone line ending after all: the body ended before any LF.
|
|
452
|
+
const rest = this.#pendingCr ? `${this.#buffer}\n` : this.#buffer;
|
|
453
|
+
this.#buffer = '';
|
|
454
|
+
this.#pendingCr = false;
|
|
455
|
+
return rest.trim().length > 0 ? rest : undefined;
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
/**
|
|
459
|
+
* Frames one streaming response body into SSE records, one transport read at a
|
|
460
|
+
* time.
|
|
461
|
+
*
|
|
462
|
+
* Both publisher routes answer with the same framing and differ only in what
|
|
463
|
+
* the payload means, so the read loop, the decoder and record buffers, and the
|
|
464
|
+
* end-of-body flush live here once. The adapter supplies the idle watchdog.
|
|
465
|
+
*
|
|
466
|
+
* This is a reader rather than an async generator because a generator costs a
|
|
467
|
+
* suspended frame, a promise, and a microtask per framed record; the adapter
|
|
468
|
+
* drives this directly, so a record is framed, parsed, and translated in the
|
|
469
|
+
* same turn.
|
|
470
|
+
*
|
|
471
|
+
* The idle bound covers one outstanding read: it is armed before every read and
|
|
472
|
+
* cleared as soon as that read resolves: a consumer holding a yielded event is
|
|
473
|
+
* not a stalled provider. A read that throws is left for the caller to
|
|
474
|
+
* classify.
|
|
475
|
+
*/
|
|
476
|
+
export class SseRecordReader {
|
|
477
|
+
#body;
|
|
478
|
+
#hooks;
|
|
479
|
+
#buffer = new SseBuffer();
|
|
480
|
+
#decoder = new TextDecoder();
|
|
481
|
+
#done = false;
|
|
482
|
+
/**
|
|
483
|
+
* @param body - the response body's byte stream.
|
|
484
|
+
* @param hooks - the caller's idle-watchdog controls.
|
|
485
|
+
*/
|
|
486
|
+
constructor(body, hooks) {
|
|
487
|
+
this.#body = body[Symbol.asyncIterator]();
|
|
488
|
+
this.#hooks = hooks;
|
|
489
|
+
}
|
|
490
|
+
/**
|
|
491
|
+
* Await the next transport read and frame the records it completes.
|
|
492
|
+
* @returns the records this read completed (an empty array when it completed
|
|
493
|
+
* none, including the final read that drains the decoder), or undefined once
|
|
494
|
+
* the body has ended and been drained.
|
|
495
|
+
*/
|
|
496
|
+
async read() {
|
|
497
|
+
if (this.#done)
|
|
498
|
+
return undefined;
|
|
499
|
+
this.#hooks.armIdle();
|
|
500
|
+
const next = await this.#body.next();
|
|
501
|
+
this.#hooks.clearIdle();
|
|
502
|
+
if (next.done !== true) {
|
|
503
|
+
return this.#buffer.push(this.#decoder.decode(next.value, { stream: true }));
|
|
504
|
+
}
|
|
505
|
+
this.#done = true;
|
|
506
|
+
// The decoder holds a partial character and the buffer a partial record;
|
|
507
|
+
// both belong to the body that just ended.
|
|
508
|
+
const records = this.#buffer.push(this.#decoder.decode());
|
|
509
|
+
const trailing = this.#buffer.flush();
|
|
510
|
+
if (trailing !== undefined)
|
|
511
|
+
records.push(trailing);
|
|
512
|
+
return records;
|
|
513
|
+
}
|
|
514
|
+
/**
|
|
515
|
+
* Release the body iterator, so a caller that stops early tears down the
|
|
516
|
+
* transport instead of leaving it reading into a buffer nobody drains.
|
|
517
|
+
*/
|
|
518
|
+
async close() {
|
|
519
|
+
if (this.#done)
|
|
520
|
+
return;
|
|
521
|
+
this.#done = true;
|
|
522
|
+
await this.#body.return?.();
|
|
523
|
+
}
|
|
524
|
+
}
|
|
525
|
+
/** The index of a stream event, or undefined when absent. */
|
|
526
|
+
function eventIndex(event) {
|
|
527
|
+
const index = event['index'];
|
|
528
|
+
return typeof index === 'number' && Number.isSafeInteger(index) && index >= 0 ? index : undefined;
|
|
529
|
+
}
|
|
530
|
+
/**
|
|
531
|
+
* Translate Vertex's Anthropic event stream into harness chunks.
|
|
532
|
+
*
|
|
533
|
+
* The translator is stateful because the two protocols disagree about
|
|
534
|
+
* granularity: Vertex announces a content block and then streams deltas, while
|
|
535
|
+
* the harness wants a start, the deltas, and an authoritative close. It is
|
|
536
|
+
* tolerant of the events it does not project (`ping`, thinking, citations,
|
|
537
|
+
* server tool use), because a provider that adds one must not break the turn.
|
|
538
|
+
*/
|
|
539
|
+
export class StreamTranslator {
|
|
540
|
+
#blocks = new Map();
|
|
541
|
+
#usage = {};
|
|
542
|
+
#usageReported = false;
|
|
543
|
+
#stopReason;
|
|
544
|
+
#blocksEmitted = false;
|
|
545
|
+
#done = false;
|
|
546
|
+
/**
|
|
547
|
+
* Feed one decoded event.
|
|
548
|
+
* @param event - the parsed SSE payload.
|
|
549
|
+
* @returns the chunks this event completes, in order.
|
|
550
|
+
*/
|
|
551
|
+
handle(event) {
|
|
552
|
+
if (this.#done)
|
|
553
|
+
return [];
|
|
554
|
+
switch (event['type']) {
|
|
555
|
+
case 'message_start': {
|
|
556
|
+
const message = event['message'];
|
|
557
|
+
if (typeof message === 'object' && message !== null) {
|
|
558
|
+
this.#mergeUsage(message['usage']);
|
|
559
|
+
}
|
|
560
|
+
return [];
|
|
561
|
+
}
|
|
562
|
+
case 'content_block_start':
|
|
563
|
+
return this.#startBlock(event);
|
|
564
|
+
case 'content_block_delta':
|
|
565
|
+
return this.#delta(event);
|
|
566
|
+
case 'content_block_stop': {
|
|
567
|
+
const index = eventIndex(event);
|
|
568
|
+
if (index === undefined)
|
|
569
|
+
return [];
|
|
570
|
+
const partial = this.#blocks.get(index);
|
|
571
|
+
this.#blocks.delete(index);
|
|
572
|
+
if (partial === undefined || partial.ignored)
|
|
573
|
+
return [];
|
|
574
|
+
this.#blocksEmitted = true;
|
|
575
|
+
return [{
|
|
576
|
+
type: 'block-end',
|
|
577
|
+
index,
|
|
578
|
+
block: partial.blockType === 'text'
|
|
579
|
+
? { type: 'text', text: partial.text }
|
|
580
|
+
: {
|
|
581
|
+
type: 'tool-call',
|
|
582
|
+
id: partial.id,
|
|
583
|
+
name: partial.name,
|
|
584
|
+
arguments: partial.arguments.length > 0 ? partial.arguments : '{}',
|
|
585
|
+
},
|
|
586
|
+
}];
|
|
587
|
+
}
|
|
588
|
+
case 'message_delta': {
|
|
589
|
+
const delta = event['delta'];
|
|
590
|
+
if (typeof delta === 'object' && delta !== null) {
|
|
591
|
+
const reason = delta['stop_reason'];
|
|
592
|
+
if (typeof reason === 'string')
|
|
593
|
+
this.#stopReason = reason;
|
|
594
|
+
}
|
|
595
|
+
this.#mergeUsage(event['usage']);
|
|
596
|
+
return [];
|
|
597
|
+
}
|
|
598
|
+
case 'message_stop':
|
|
599
|
+
return this.#finish();
|
|
600
|
+
case 'error':
|
|
601
|
+
this.#done = true;
|
|
602
|
+
return [{ type: 'finish', reason: { kind: 'error', failure: failureForEvent(event['error']) } }];
|
|
603
|
+
default:
|
|
604
|
+
return [];
|
|
605
|
+
}
|
|
606
|
+
}
|
|
607
|
+
/** Start one content block, or mark it as one this adapter does not project. */
|
|
608
|
+
#startBlock(event) {
|
|
609
|
+
const index = eventIndex(event);
|
|
610
|
+
const raw = event['content_block'];
|
|
611
|
+
if (index === undefined || typeof raw !== 'object' || raw === null)
|
|
612
|
+
return [];
|
|
613
|
+
const block = raw;
|
|
614
|
+
switch (block['type']) {
|
|
615
|
+
case 'text':
|
|
616
|
+
this.#blocks.set(index, { blockType: 'text', text: '', arguments: '', id: '', name: '', ignored: false });
|
|
617
|
+
return [{ type: 'block-start', index, blockType: 'text' }];
|
|
618
|
+
case 'tool_use': {
|
|
619
|
+
// Vertex streams a tool call as an empty input object followed by
|
|
620
|
+
// `input_json_delta` fragments; a populated one is a complete call and
|
|
621
|
+
// any later delta would duplicate it.
|
|
622
|
+
const input = block['input'];
|
|
623
|
+
const complete = typeof input === 'object' && input !== null && Object.keys(input).length > 0;
|
|
624
|
+
const partial = {
|
|
625
|
+
blockType: 'tool-call',
|
|
626
|
+
text: '',
|
|
627
|
+
arguments: complete ? JSON.stringify(input) : '',
|
|
628
|
+
id: typeof block['id'] === 'string' && block['id'].length > 0 ? block['id'] : `toolu_${index}`,
|
|
629
|
+
name: typeof block['name'] === 'string' ? block['name'] : '',
|
|
630
|
+
ignored: false,
|
|
631
|
+
};
|
|
632
|
+
this.#blocks.set(index, partial);
|
|
633
|
+
return [
|
|
634
|
+
{ type: 'block-start', index, blockType: 'tool-call' },
|
|
635
|
+
{
|
|
636
|
+
type: 'tool-call-delta',
|
|
637
|
+
index,
|
|
638
|
+
id: partial.id,
|
|
639
|
+
name: partial.name,
|
|
640
|
+
argumentsDelta: partial.arguments,
|
|
641
|
+
},
|
|
642
|
+
];
|
|
643
|
+
}
|
|
644
|
+
default:
|
|
645
|
+
this.#blocks.set(index, { blockType: 'text', text: '', arguments: '', id: '', name: '', ignored: true });
|
|
646
|
+
return [];
|
|
647
|
+
}
|
|
648
|
+
}
|
|
649
|
+
/** Absorb one content delta. */
|
|
650
|
+
#delta(event) {
|
|
651
|
+
const index = eventIndex(event);
|
|
652
|
+
const raw = event['delta'];
|
|
653
|
+
if (index === undefined || typeof raw !== 'object' || raw === null)
|
|
654
|
+
return [];
|
|
655
|
+
const partial = this.#blocks.get(index);
|
|
656
|
+
if (partial === undefined || partial.ignored)
|
|
657
|
+
return [];
|
|
658
|
+
const delta = raw;
|
|
659
|
+
switch (delta['type']) {
|
|
660
|
+
case 'text_delta': {
|
|
661
|
+
const text = typeof delta['text'] === 'string' ? delta['text'] : '';
|
|
662
|
+
if (text.length === 0)
|
|
663
|
+
return [];
|
|
664
|
+
partial.text += text;
|
|
665
|
+
return [{ type: 'text-delta', index, text }];
|
|
666
|
+
}
|
|
667
|
+
case 'input_json_delta': {
|
|
668
|
+
const fragment = typeof delta['partial_json'] === 'string' ? delta['partial_json'] : '';
|
|
669
|
+
if (fragment.length === 0)
|
|
670
|
+
return [];
|
|
671
|
+
partial.arguments += fragment;
|
|
672
|
+
return [{ type: 'tool-call-delta', index, id: partial.id, argumentsDelta: fragment }];
|
|
673
|
+
}
|
|
674
|
+
default:
|
|
675
|
+
return [];
|
|
676
|
+
}
|
|
677
|
+
}
|
|
678
|
+
/** Terminal chunks: usage, then the mapped finish reason. */
|
|
679
|
+
#finish() {
|
|
680
|
+
this.#done = true;
|
|
681
|
+
const reason = mapStopReason(this.#stopReason);
|
|
682
|
+
// A completed turn that produced nothing is a degenerate provider response,
|
|
683
|
+
// not a successful empty answer; a provider-reported failure keeps its own
|
|
684
|
+
// reason, which is the more specific account of the same turn.
|
|
685
|
+
if (!this.#blocksEmitted && reason.kind !== 'error') {
|
|
686
|
+
return [{
|
|
687
|
+
type: 'finish',
|
|
688
|
+
reason: {
|
|
689
|
+
kind: 'error',
|
|
690
|
+
failure: {
|
|
691
|
+
message: 'google-vertex: the model completed the response with no content',
|
|
692
|
+
code: EMPTY_RESPONSE_CODE,
|
|
693
|
+
},
|
|
694
|
+
},
|
|
695
|
+
}];
|
|
696
|
+
}
|
|
697
|
+
// Usage accompanies a real provider report; a synthesized zero would claim
|
|
698
|
+
// a measurement that never happened.
|
|
699
|
+
return [
|
|
700
|
+
...this.#usageReported ? [{ type: 'usage', usage: mapUsage(this.#usage) }] : [],
|
|
701
|
+
{ type: 'finish', reason },
|
|
702
|
+
];
|
|
703
|
+
}
|
|
704
|
+
/** Merge the newest cumulative usage counters. */
|
|
705
|
+
#mergeUsage(raw) {
|
|
706
|
+
const merged = takeCounters(raw, [
|
|
707
|
+
'input_tokens',
|
|
708
|
+
'output_tokens',
|
|
709
|
+
'cache_read_input_tokens',
|
|
710
|
+
'cache_creation_input_tokens',
|
|
711
|
+
]);
|
|
712
|
+
// Only a counter the provider actually sent counts as a report.
|
|
713
|
+
if (Object.keys(merged).length === 0)
|
|
714
|
+
return;
|
|
715
|
+
this.#usageReported = true;
|
|
716
|
+
this.#usage = { ...this.#usage, ...merged };
|
|
717
|
+
}
|
|
718
|
+
/** True once a terminal event arrived, so the adapter can tell truncation. */
|
|
719
|
+
get terminal() {
|
|
720
|
+
return this.#done;
|
|
721
|
+
}
|
|
722
|
+
}
|