@maci0/dsh-google-vertex 0.0.0-stage → 0.12.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +166 -2
- package/cordis.patch.yml +34 -0
- package/icon.svg +6 -0
- package/lib/adapter.js +398 -0
- package/lib/auth.js +246 -0
- package/lib/client.js +206 -0
- package/lib/discovery.js +180 -0
- package/lib/gemini.js +561 -0
- package/lib/gemini_adapter.js +61 -0
- package/lib/host.js +18 -0
- package/lib/index.js +196 -0
- package/lib/types/adapter.d.ts +219 -0
- package/lib/types/auth.d.ts +111 -0
- package/lib/types/discovery.d.ts +57 -0
- package/lib/types/gemini.d.ts +210 -0
- package/lib/types/gemini_adapter.d.ts +54 -0
- package/lib/types/host.d.ts +202 -0
- package/lib/types/index.d.ts +161 -0
- package/lib/types/wire-shared.d.ts +19 -0
- package/lib/types/wire.d.ts +271 -0
- package/lib/wire-shared.js +59 -0
- package/lib/wire.js +722 -0
- package/locale/en.json +6 -0
- package/locale/zh.json +6 -0
- package/package.json +93 -4
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Wire translation for Google-hosted Anthropic models on Vertex AI: the request
|
|
3
|
+
* body Vertex accepts, the two endpoint shapes (global and regional), the SSE
|
|
4
|
+
* event stream it answers with, and the failure envelope it refuses with.
|
|
5
|
+
*
|
|
6
|
+
* Everything here is pure: the adapter feeds it bytes and yields the chunks it
|
|
7
|
+
* returns, so the protocol is testable without a harness or a network.
|
|
8
|
+
*
|
|
9
|
+
* Vertex serves Claude through the publisher endpoint rather than Anthropic's
|
|
10
|
+
* own API, which changes three things this module owns: the path
|
|
11
|
+
* (`…/publishers/anthropic/models/{model}:streamRawPredict`), a body carrying
|
|
12
|
+
* `anthropic_version` and no `model` field, and a bearer token instead of an
|
|
13
|
+
* `x-api-key` header.
|
|
14
|
+
*
|
|
15
|
+
* @module dsh-google-vertex/wire
|
|
16
|
+
*/
|
|
17
|
+
import { CONTEXT_WINDOW_EXCEEDED_CODE, EMPTY_RESPONSE_CODE, QUOTA_EXCEEDED_CODE } from '@deepseek-ai/dsh-llm';
|
|
18
|
+
export { CONTEXT_WINDOW_EXCEEDED_CODE, EMPTY_RESPONSE_CODE, QUOTA_EXCEEDED_CODE };
|
|
19
|
+
import type { ContentBlock, FinishReason, GenerateOptions, LlmFailure, StreamChunk, TokenUsage } from './host.ts';
|
|
20
|
+
/** Endpoint host used when no region is configured. */
|
|
21
|
+
export declare const DEFAULT_LOCATION = "global";
|
|
22
|
+
/**
|
|
23
|
+
* Default bound on the interval between two stream reads, in milliseconds.
|
|
24
|
+
* Matches the shipped remote adapters (`llm-deepseek/src/common/defaults.ts`).
|
|
25
|
+
*/
|
|
26
|
+
export declare const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 300000;
|
|
27
|
+
/** Largest delay `setTimeout` schedules without clamping it to one millisecond. */
|
|
28
|
+
export declare const MAX_TIMER_DELAY_MS = 2147483647;
|
|
29
|
+
/**
|
|
30
|
+
* Host for a location. `global` has no region prefix: the documented global
|
|
31
|
+
* endpoint is `aiplatform.googleapis.com`, and `global-aiplatform…` is not a
|
|
32
|
+
* host Vertex answers on.
|
|
33
|
+
* @param location - configured or defaulted region, or `global`.
|
|
34
|
+
* @returns the origin, without a trailing slash.
|
|
35
|
+
*/
|
|
36
|
+
export declare function endpointOrigin(location: string): string;
|
|
37
|
+
/**
|
|
38
|
+
* The streaming publisher path for one model.
|
|
39
|
+
* @param project - Google Cloud project id.
|
|
40
|
+
* @param location - region, or `global`.
|
|
41
|
+
* @param model - publisher model id, e.g. `claude-sonnet-4-5`.
|
|
42
|
+
* @returns the absolute request URL.
|
|
43
|
+
*/
|
|
44
|
+
export declare function endpointFor(project: string, location: string, model: string): string;
|
|
45
|
+
/** Cache marker Vertex honours on tools, system blocks, and message blocks. */
|
|
46
|
+
interface CacheControl {
|
|
47
|
+
readonly type: 'ephemeral';
|
|
48
|
+
}
|
|
49
|
+
/** One text block on the wire. */
|
|
50
|
+
interface WireTextBlock {
|
|
51
|
+
type: 'text';
|
|
52
|
+
text: string;
|
|
53
|
+
cache_control?: CacheControl;
|
|
54
|
+
}
|
|
55
|
+
/** One tool-result block on the wire. */
|
|
56
|
+
interface WireToolResultBlock {
|
|
57
|
+
type: 'tool_result';
|
|
58
|
+
tool_use_id: string;
|
|
59
|
+
content: string;
|
|
60
|
+
is_error?: boolean;
|
|
61
|
+
cache_control?: CacheControl;
|
|
62
|
+
}
|
|
63
|
+
/** One tool-use block on the wire. */
|
|
64
|
+
interface WireToolUseBlock {
|
|
65
|
+
type: 'tool_use';
|
|
66
|
+
id: string;
|
|
67
|
+
name: string;
|
|
68
|
+
input: unknown;
|
|
69
|
+
cache_control?: CacheControl;
|
|
70
|
+
}
|
|
71
|
+
/** Any wire content block this adapter emits. */
|
|
72
|
+
type WireContentBlock = WireTextBlock | WireToolResultBlock | WireToolUseBlock;
|
|
73
|
+
/** One wire message. */
|
|
74
|
+
export interface WireMessage {
|
|
75
|
+
role: 'user' | 'assistant';
|
|
76
|
+
content: WireContentBlock[];
|
|
77
|
+
}
|
|
78
|
+
/** One wire tool declaration. */
|
|
79
|
+
interface WireTool {
|
|
80
|
+
name: string;
|
|
81
|
+
description: string;
|
|
82
|
+
input_schema: Record<string, unknown>;
|
|
83
|
+
cache_control?: CacheControl;
|
|
84
|
+
}
|
|
85
|
+
/** The request body `:streamRawPredict` accepts. */
|
|
86
|
+
interface WireRequestBody {
|
|
87
|
+
anthropic_version: string;
|
|
88
|
+
stream: true;
|
|
89
|
+
max_tokens: number;
|
|
90
|
+
messages: WireMessage[];
|
|
91
|
+
system?: WireTextBlock[];
|
|
92
|
+
tools?: WireTool[];
|
|
93
|
+
temperature?: number;
|
|
94
|
+
stop_sequences?: string[];
|
|
95
|
+
}
|
|
96
|
+
/** What the adapter needs to build a request and address the endpoint. */
|
|
97
|
+
export interface VertexWireConfig {
|
|
98
|
+
project: string;
|
|
99
|
+
location: string;
|
|
100
|
+
/** Output cap materialized when a caller omits `maxTokens`. */
|
|
101
|
+
maxTokens: number;
|
|
102
|
+
}
|
|
103
|
+
/**
|
|
104
|
+
* Flatten a tool-role message's content to the text Vertex accepts.
|
|
105
|
+
*
|
|
106
|
+
* The Gemini route sends a tool result the same way, so this is exported rather
|
|
107
|
+
* than written twice.
|
|
108
|
+
*/
|
|
109
|
+
export declare function resultText(blocks: readonly ContentBlock[]): string;
|
|
110
|
+
/**
|
|
111
|
+
* Build the request body for one model call.
|
|
112
|
+
*
|
|
113
|
+
* History is projected block by block, consecutive same-role messages are
|
|
114
|
+
* merged (the provider reads one turn per role), and three cache breakpoints are
|
|
115
|
+
* placed: end of tools, end of system, and the final block of the conversation.
|
|
116
|
+
* Those three make the static prefix and every earlier turn cacheable while the
|
|
117
|
+
* growing tail is re-read.
|
|
118
|
+
* @param options - the harness request.
|
|
119
|
+
* @param config - project, location, and default output cap.
|
|
120
|
+
* @returns the wire body.
|
|
121
|
+
*/
|
|
122
|
+
export declare function buildRequestBody(options: GenerateOptions, config: VertexWireConfig): WireRequestBody;
|
|
123
|
+
/** Raw usage counters as Vertex reports them. */
|
|
124
|
+
interface WireUsage {
|
|
125
|
+
input_tokens?: number;
|
|
126
|
+
output_tokens?: number;
|
|
127
|
+
cache_read_input_tokens?: number;
|
|
128
|
+
cache_creation_input_tokens?: number;
|
|
129
|
+
}
|
|
130
|
+
/**
|
|
131
|
+
* Map Vertex's usage counters onto harness accounting.
|
|
132
|
+
*
|
|
133
|
+
* Vertex splits prompt tokens the way the harness does (`input_tokens` counts
|
|
134
|
+
* only uncached input, with cache reads and writes reported separately), so the
|
|
135
|
+
* fields transfer without arithmetic, and the total is their sum.
|
|
136
|
+
* @param usage - the latest cumulative counters seen on the stream.
|
|
137
|
+
* @returns disjoint harness counts.
|
|
138
|
+
*/
|
|
139
|
+
export declare function mapUsage(usage: WireUsage): TokenUsage;
|
|
140
|
+
/**
|
|
141
|
+
* Map one provider stop reason onto the harness vocabulary.
|
|
142
|
+
*
|
|
143
|
+
* `pause_turn` needs no case of its own: it cannot recur here, because this
|
|
144
|
+
* adapter declares no server-executed tools, so the answer is complete, which
|
|
145
|
+
* is what the default already reports for any reason the provider adds.
|
|
146
|
+
* @param reason - the `stop_reason` Vertex reported, if any.
|
|
147
|
+
* @returns the harness finish reason.
|
|
148
|
+
*/
|
|
149
|
+
export declare function mapStopReason(reason: string | undefined): FinishReason;
|
|
150
|
+
/**
|
|
151
|
+
* Classify a refused HTTP response.
|
|
152
|
+
*
|
|
153
|
+
* Two refusals carry a code of their own because the harness treats them
|
|
154
|
+
* differently: an oversized request must not be retried, and an exhausted quota
|
|
155
|
+
* is a capacity problem rather than an invalid one. Google names the quota in
|
|
156
|
+
* the envelope's `status` (`RESOURCE_EXHAUSTED`), so that is read as well as
|
|
157
|
+
* the message, exactly as {@link failureForEvent} reads an in-band envelope.
|
|
158
|
+
* @param status - HTTP status.
|
|
159
|
+
* @param body - response body text.
|
|
160
|
+
* @param subject - route description named in the failure.
|
|
161
|
+
* @param headers - HTTP response headers carrying an optional Retry-After.
|
|
162
|
+
* @returns the failure to report.
|
|
163
|
+
*/
|
|
164
|
+
export declare function failureForStatus(status: number, body: string, subject: string, headers?: Headers): LlmFailure;
|
|
165
|
+
/**
|
|
166
|
+
* Classify a stream-level `error` payload.
|
|
167
|
+
*
|
|
168
|
+
* Both publishers deliver one mid-stream: Anthropic's envelope names the
|
|
169
|
+
* condition in `type`, and Google's in a canonical `status` plus a numeric
|
|
170
|
+
* `code`. A provider error is what turns a mid-stream refusal into a terminal
|
|
171
|
+
* failure instead of a truncated response.
|
|
172
|
+
* @param error - the payload's `error` member.
|
|
173
|
+
* @returns the failure to report.
|
|
174
|
+
*/
|
|
175
|
+
export declare function failureForEvent(error: unknown): LlmFailure;
|
|
176
|
+
/**
|
|
177
|
+
* Decode one SSE record (everything up to a blank line) into its JSON payload.
|
|
178
|
+
*
|
|
179
|
+
* `event:` lines are ignored because every Vertex payload carries its own
|
|
180
|
+
* `type`; `data:` lines are concatenated, which is what the SSE specification
|
|
181
|
+
* requires for a multi-line payload.
|
|
182
|
+
* @param record - one raw record, without its trailing blank line.
|
|
183
|
+
* @returns the parsed payload, or undefined for a comment, ping, or malformed record.
|
|
184
|
+
*/
|
|
185
|
+
export declare function parseSseRecord(record: string): Record<string, unknown> | undefined;
|
|
186
|
+
/**
|
|
187
|
+
* Reassembles SSE records from arbitrarily split transport chunks.
|
|
188
|
+
*
|
|
189
|
+
* Line endings are normalized as text arrives, so a `\r\n` split across two
|
|
190
|
+
* chunks still frames one record: a chunk-final CR is carried over and resolved
|
|
191
|
+
* against the next chunk's first byte rather than normalized chunk by chunk.
|
|
192
|
+
*/
|
|
193
|
+
export declare class SseBuffer {
|
|
194
|
+
#private;
|
|
195
|
+
/**
|
|
196
|
+
* Absorb one decoded chunk.
|
|
197
|
+
* @param text - newly decoded text.
|
|
198
|
+
* @returns every complete record it completes, in order.
|
|
199
|
+
*/
|
|
200
|
+
push(text: string): string[];
|
|
201
|
+
/**
|
|
202
|
+
* Release whatever a completed body ended with.
|
|
203
|
+
* @returns the trailing record, or undefined when the body ended on a boundary.
|
|
204
|
+
*/
|
|
205
|
+
flush(): string | undefined;
|
|
206
|
+
}
|
|
207
|
+
/** Per-read watchdog hooks the shared pump calls around every outstanding read. */
|
|
208
|
+
interface SsePumpHooks {
|
|
209
|
+
/** Arm the idle bound for the read about to start. */
|
|
210
|
+
armIdle(): void;
|
|
211
|
+
/** Clear the armed idle bound; called as soon as a read resolves, and on exit. */
|
|
212
|
+
clearIdle(): void;
|
|
213
|
+
}
|
|
214
|
+
/**
|
|
215
|
+
* Frames one streaming response body into SSE records, one transport read at a
|
|
216
|
+
* time.
|
|
217
|
+
*
|
|
218
|
+
* Both publisher routes answer with the same framing and differ only in what
|
|
219
|
+
* the payload means, so the read loop, the decoder and record buffers, and the
|
|
220
|
+
* end-of-body flush live here once. The adapter supplies the idle watchdog.
|
|
221
|
+
*
|
|
222
|
+
* This is a reader rather than an async generator because a generator costs a
|
|
223
|
+
* suspended frame, a promise, and a microtask per framed record; the adapter
|
|
224
|
+
* drives this directly, so a record is framed, parsed, and translated in the
|
|
225
|
+
* same turn.
|
|
226
|
+
*
|
|
227
|
+
* The idle bound covers one outstanding read: it is armed before every read and
|
|
228
|
+
* cleared as soon as that read resolves: a consumer holding a yielded event is
|
|
229
|
+
* not a stalled provider. A read that throws is left for the caller to
|
|
230
|
+
* classify.
|
|
231
|
+
*/
|
|
232
|
+
export declare class SseRecordReader {
|
|
233
|
+
#private;
|
|
234
|
+
/**
|
|
235
|
+
* @param body - the response body's byte stream.
|
|
236
|
+
* @param hooks - the caller's idle-watchdog controls.
|
|
237
|
+
*/
|
|
238
|
+
constructor(body: AsyncIterable<Uint8Array>, hooks: SsePumpHooks);
|
|
239
|
+
/**
|
|
240
|
+
* Await the next transport read and frame the records it completes.
|
|
241
|
+
* @returns the records this read completed (an empty array when it completed
|
|
242
|
+
* none, including the final read that drains the decoder), or undefined once
|
|
243
|
+
* the body has ended and been drained.
|
|
244
|
+
*/
|
|
245
|
+
read(): Promise<readonly string[] | undefined>;
|
|
246
|
+
/**
|
|
247
|
+
* Release the body iterator, so a caller that stops early tears down the
|
|
248
|
+
* transport instead of leaving it reading into a buffer nobody drains.
|
|
249
|
+
*/
|
|
250
|
+
close(): Promise<void>;
|
|
251
|
+
}
|
|
252
|
+
/**
|
|
253
|
+
* Translate Vertex's Anthropic event stream into harness chunks.
|
|
254
|
+
*
|
|
255
|
+
* The translator is stateful because the two protocols disagree about
|
|
256
|
+
* granularity: Vertex announces a content block and then streams deltas, while
|
|
257
|
+
* the harness wants a start, the deltas, and an authoritative close. It is
|
|
258
|
+
* tolerant of the events it does not project (`ping`, thinking, citations,
|
|
259
|
+
* server tool use), because a provider that adds one must not break the turn.
|
|
260
|
+
*/
|
|
261
|
+
export declare class StreamTranslator {
|
|
262
|
+
#private;
|
|
263
|
+
/**
|
|
264
|
+
* Feed one decoded event.
|
|
265
|
+
* @param event - the parsed SSE payload.
|
|
266
|
+
* @returns the chunks this event completes, in order.
|
|
267
|
+
*/
|
|
268
|
+
handle(event: Record<string, unknown>): StreamChunk[];
|
|
269
|
+
/** True once a terminal event arrived, so the adapter can tell truncation. */
|
|
270
|
+
get terminal(): boolean;
|
|
271
|
+
}
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Helpers the two publisher wire translators (Anthropic and Gemini) share:
|
|
3
|
+
* tool-argument parsing, system-text collection, and usage-counter merging.
|
|
4
|
+
*
|
|
5
|
+
* @module dsh-google-vertex/wire-shared
|
|
6
|
+
*/
|
|
7
|
+
/** Parse a model-produced arguments string into the object the provider requires. */
|
|
8
|
+
export function toolInput(argumentsJson) {
|
|
9
|
+
try {
|
|
10
|
+
const parsed = JSON.parse(argumentsJson);
|
|
11
|
+
return typeof parsed === 'object' && parsed !== null && !Array.isArray(parsed)
|
|
12
|
+
? parsed
|
|
13
|
+
: {};
|
|
14
|
+
}
|
|
15
|
+
catch {
|
|
16
|
+
// The harness assembles these strings from the model's own deltas; an
|
|
17
|
+
// unparseable one is a provider-side truncation, and an empty object keeps
|
|
18
|
+
// the turn revisable instead of failing the whole request.
|
|
19
|
+
return {};
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
/** System-role text the request carries, in assembly order. */
|
|
23
|
+
export function systemParts(options) {
|
|
24
|
+
const parts = [];
|
|
25
|
+
if (options.system !== undefined && options.system.length > 0)
|
|
26
|
+
parts.push(options.system);
|
|
27
|
+
for (const message of options.messages) {
|
|
28
|
+
if (message.role !== 'system' && message.role !== 'developer')
|
|
29
|
+
continue;
|
|
30
|
+
for (const block of message.content) {
|
|
31
|
+
if (block.type === 'text' && typeof block.text === 'string' && block.text.length > 0)
|
|
32
|
+
parts.push(block.text);
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
return parts;
|
|
36
|
+
}
|
|
37
|
+
/** A usage counter, ignoring anything the provider sends that is not a number. */
|
|
38
|
+
function count(value) {
|
|
39
|
+
return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined;
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Read the named counters out of one usage payload, keeping only the ones the
|
|
43
|
+
* provider actually sent.
|
|
44
|
+
* @param raw - the payload's usage member.
|
|
45
|
+
* @param names - the counter names this route reads.
|
|
46
|
+
* @returns only the present, well-formed counters.
|
|
47
|
+
*/
|
|
48
|
+
export function takeCounters(raw, names) {
|
|
49
|
+
const merged = {};
|
|
50
|
+
if (typeof raw !== 'object' || raw === null)
|
|
51
|
+
return merged;
|
|
52
|
+
const usage = raw;
|
|
53
|
+
for (const name of names) {
|
|
54
|
+
const value = count(usage[name]);
|
|
55
|
+
if (value !== undefined)
|
|
56
|
+
merged[name] = value;
|
|
57
|
+
}
|
|
58
|
+
return merged;
|
|
59
|
+
}
|