@maci0/dsh-google-vertex 0.0.0-stage → 0.12.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/gemini.js ADDED
@@ -0,0 +1,561 @@
1
+ /**
2
+ * Wire translation for Google's own Gemini models on Vertex AI: the request
3
+ * body `publishers/google` accepts, the SSE response shape, and (the part that
4
+ * makes tool calling work at all on Gemini 3) the thought signature that has to
5
+ * travel back with every replayed function call.
6
+ *
7
+ * Everything here is pure: the adapter feeds it bytes and yields the chunks it
8
+ * returns, so the protocol is testable without a harness or a network.
9
+ *
10
+ * @module dsh-google-vertex/gemini
11
+ */
12
+ import { systemParts, takeCounters, toolInput } from './wire-shared.js';
13
+ import { EMPTY_RESPONSE_CODE, endpointOrigin, failureForEvent, resultText } from './wire.js';
14
+ /** Harness code reported when the provider refused on safety grounds. */
15
+ export const SAFETY_BLOCKED_CODE = 'SAFETY';
16
+ /** Vertex path version this adapter speaks. */
17
+ const API_VERSION = 'v1';
18
+ /**
19
+ * Gemini context capacity: every current Gemini model serves roughly a million
20
+ * tokens, and the catalog below narrows nothing.
21
+ */
22
+ export const DEFAULT_GEMINI_CONTEXT_WINDOW = 1_048_576;
23
+ /**
24
+ * Output cap applied when a caller omits one.
25
+ *
26
+ * Vertex's ceiling here is EXCLUSIVE: `maxOutputTokens: 65536` is refused with
27
+ * "supported range is from 1 (inclusive) to 65536 (exclusive)", so the highest
28
+ * accepted value is one less.
29
+ */
30
+ export const DEFAULT_GEMINI_MAX_TOKENS = 65_535;
31
+ /**
32
+ * Gemini models the global endpoint serves, in picker order. Ids are the
33
+ * provider's own aliases, so a promoted release needs no edit here.
34
+ */
35
+ export const DEFAULT_GEMINI_MODELS = [
36
+ { id: 'gemini-3.5-flash', name: 'Gemini 3.5 Flash (Vertex)' },
37
+ { id: 'gemini-3.1-pro-preview', name: 'Gemini 3.1 Pro Preview (Vertex)' },
38
+ { id: 'gemini-3-flash-preview', name: 'Gemini 3 Flash Preview (Vertex)' },
39
+ { id: 'gemini-2.5-pro', name: 'Gemini 2.5 Pro (Vertex)' },
40
+ { id: 'gemini-2.5-flash', name: 'Gemini 2.5 Flash (Vertex)' },
41
+ { id: 'gemini-2.5-flash-lite', name: 'Gemini 2.5 Flash-Lite (Vertex)' },
42
+ ];
43
+ /**
44
+ * The streaming publisher path for one Gemini model.
45
+ * @param project - Google Cloud project id.
46
+ * @param location - region, or `global`.
47
+ * @param model - publisher model id, e.g. `gemini-3.5-flash`.
48
+ * @returns the absolute request URL, with the SSE response encoding.
49
+ */
50
+ export function geminiEndpointFor(project, location, model) {
51
+ const origin = endpointOrigin(location);
52
+ const path = `/${API_VERSION}/projects/${encodeURIComponent(project)}/locations/${encodeURIComponent(location)}`
53
+ + `/publishers/google/models/${encodeURIComponent(model)}:streamGenerateContent?alt=sse`;
54
+ return `${origin}${path}`;
55
+ }
56
+ /**
57
+ * Build the replay envelope for one finished response.
58
+ * @param model - the model that produced the response.
59
+ * @param blocks - per-block metadata in emitted order.
60
+ * @returns the envelope for the terminal finish chunk.
61
+ */
62
+ export function geminiReplayState(model, blocks) {
63
+ return { response: { kind: 'google-vertex-gemini', version: 1, model }, blocks };
64
+ }
65
+ /** A JSON object, or undefined for anything else. */
66
+ function asObject(value) {
67
+ return typeof value === 'object' && value !== null && !Array.isArray(value) ? value : undefined;
68
+ }
69
+ /**
70
+ * Convert one harness parameter schema (or nested schema) into one Vertex's
71
+ * `Schema` message can carry.
72
+ *
73
+ * Vertex accepts only `anyOf`, `default`, `description`, `enum`, `example`,
74
+ * `format`, `items`, the min/max bounds, `nullable`, `pattern`, `properties`,
75
+ * `propertyOrdering`, `required`, `title`, and `type`. Two members the harness's
76
+ * value-schema subset emits are not among them, and the endpoint refuses the
77
+ * whole request when one arrives: `Invalid JSON payload received. Unknown name
78
+ * "additionalProperties" at 'tools[0].function_declarations[0].parameters'`:
79
+ *
80
+ * - `additionalProperties` (the harness requires it on every object) is dropped.
81
+ * - `oneOf` has no `Schema` member; `anyOf` does. The backend also has no null
82
+ * type, so a `null` branch becomes `nullable: true` and the remaining branch
83
+ * is inlined (one branch left) or kept as `anyOf` (several). That is the same
84
+ * rewrite the official `@google/genai` converter applies to `anyOf`, and it
85
+ * only loosens the provider-side hint, since the harness validates the model's
86
+ * arguments against the original schema itself.
87
+ *
88
+ * Every other member is copied verbatim.
89
+ * @param schema - one parameter schema, or one nested schema.
90
+ * @returns the schema Vertex accepts; the input is not mutated.
91
+ */
92
+ function geminiSchema(schema) {
93
+ const copy = {};
94
+ for (const [key, value] of Object.entries(schema)) {
95
+ if (key === 'additionalProperties')
96
+ continue;
97
+ if (key === 'properties') {
98
+ const properties = asObject(value);
99
+ if (properties !== undefined) {
100
+ const rebuilt = {};
101
+ for (const [name, property] of Object.entries(properties)) {
102
+ const nested = asObject(property);
103
+ rebuilt[name] = nested === undefined ? property : geminiSchema(nested);
104
+ }
105
+ copy[key] = rebuilt;
106
+ continue;
107
+ }
108
+ }
109
+ if (key === 'items' || key === 'anyOf') {
110
+ copy[key] = geminiSchemaValue(value);
111
+ continue;
112
+ }
113
+ copy[key] = value;
114
+ }
115
+ const union = copy['oneOf'];
116
+ if (!Array.isArray(union))
117
+ return copy;
118
+ delete copy['oneOf'];
119
+ const branches = [];
120
+ for (const branch of union) {
121
+ const node = asObject(branch);
122
+ if (node === undefined) {
123
+ branches.push(branch);
124
+ continue;
125
+ }
126
+ if (node['type'] === 'null') {
127
+ copy['nullable'] = true;
128
+ continue;
129
+ }
130
+ branches.push(geminiSchema(node));
131
+ }
132
+ if (branches.length === 1) {
133
+ const only = asObject(branches[0]);
134
+ // One branch left: the node itself, its annotations kept, plus the branch.
135
+ if (only !== undefined) {
136
+ for (const [key, value] of Object.entries(only))
137
+ copy[key] ??= value;
138
+ return copy;
139
+ }
140
+ }
141
+ if (branches.length > 0)
142
+ copy['anyOf'] = branches;
143
+ return copy;
144
+ }
145
+ /** Convert one schema-valued member: a single schema, or a list of them. */
146
+ function geminiSchemaValue(value) {
147
+ if (Array.isArray(value)) {
148
+ return value.map((entry) => {
149
+ const nested = asObject(entry);
150
+ return nested === undefined ? entry : geminiSchema(nested);
151
+ });
152
+ }
153
+ const nested = asObject(value);
154
+ return nested === undefined ? value : geminiSchema(nested);
155
+ }
156
+ /**
157
+ * Read back the replay metadata for one assistant message.
158
+ *
159
+ * Anything unexpected (a foreign envelope, another model, a block count that no
160
+ * longer lines up with the content) yields undefined, which degrades to sending
161
+ * the call without its signature rather than throwing: the provider then decides,
162
+ * and a cross-provider history stays replayable.
163
+ * @param message - the assistant message from history.
164
+ * @param model - the model about to be called; signatures do not cross models.
165
+ * @returns index-aligned metadata, or undefined when unusable.
166
+ */
167
+ export function readGeminiReplay(message, model) {
168
+ const source = message.source;
169
+ if (source?.kind !== 'model' || source.replayState === undefined)
170
+ return undefined;
171
+ const envelope = asObject(source.replayState);
172
+ const response = asObject(envelope?.['response']);
173
+ if (response?.['kind'] !== 'google-vertex-gemini' || response['version'] !== 1)
174
+ return undefined;
175
+ if (response['model'] !== model)
176
+ return undefined;
177
+ const raw = envelope?.['blocks'];
178
+ if (!Array.isArray(raw) || raw.length !== message.content.length)
179
+ return undefined;
180
+ const blocks = [];
181
+ for (const [index, value] of raw.entries()) {
182
+ const block = asObject(value);
183
+ const type = block?.['type'];
184
+ if (type !== message.content[index]?.type)
185
+ return undefined;
186
+ if (type !== 'text' && type !== 'tool-call')
187
+ return undefined;
188
+ const signature = block?.['thoughtSignature'];
189
+ if (signature !== undefined && typeof signature !== 'string')
190
+ return undefined;
191
+ blocks.push({ type, ...signature === undefined ? {} : { thoughtSignature: signature } });
192
+ }
193
+ return blocks;
194
+ }
195
+ /** Every tool-call id in this request, mapped to the tool's name. */
196
+ function toolNames(messages) {
197
+ const names = new Map();
198
+ for (const message of messages) {
199
+ for (const block of message.content) {
200
+ if (block.type === 'tool-call' && typeof block.id === 'string')
201
+ names.set(block.id, String(block.name));
202
+ }
203
+ }
204
+ return names;
205
+ }
206
+ /** Project one assistant message onto Gemini parts, signatures included. */
207
+ function assistantParts(message, model) {
208
+ const replay = readGeminiReplay(message, model);
209
+ const parts = [];
210
+ message.content.forEach((block, index) => {
211
+ const signature = replay?.[index]?.thoughtSignature;
212
+ if (block.type === 'text') {
213
+ if (typeof block.text === 'string' && block.text.length > 0) {
214
+ parts.push({ text: block.text, ...signature === undefined ? {} : { thoughtSignature: signature } });
215
+ }
216
+ return;
217
+ }
218
+ if (block.type === 'tool-call') {
219
+ const call = {
220
+ name: String(block.name),
221
+ args: toolInput(String(block.arguments)),
222
+ ...typeof block.id === 'string' && block.id.length > 0 ? { id: block.id } : {},
223
+ };
224
+ parts.push({ functionCall: call, ...signature === undefined ? {} : { thoughtSignature: signature } });
225
+ }
226
+ // Reasoning blocks carry text this adapter never asks for; a Gemini thought
227
+ // part needs its own signature to replay, so a foreign one is dropped.
228
+ });
229
+ return parts;
230
+ }
231
+ /** Project one user message onto Gemini parts. */
232
+ function userParts(message) {
233
+ const parts = [];
234
+ for (const block of message.content) {
235
+ if (block.type === 'text' && typeof block.text === 'string' && block.text.length > 0)
236
+ parts.push({ text: block.text });
237
+ }
238
+ return parts;
239
+ }
240
+ /**
241
+ * Project one harness `tool`-role message onto the `functionResponse` part
242
+ * Gemini requires for the `functionCall` it answers.
243
+ *
244
+ * The harness puts the call identity on the message, the provider reads it from
245
+ * the part, and the function name is what Vertex matches a response by.
246
+ */
247
+ function toolResultParts(message, names) {
248
+ const callId = message.toolCallId;
249
+ if (callId === undefined || callId.length === 0)
250
+ return userParts(message);
251
+ return [{
252
+ functionResponse: {
253
+ name: names.get(callId) ?? callId,
254
+ id: callId,
255
+ response: { result: resultText((message.content ?? [])) },
256
+ },
257
+ }];
258
+ }
259
+ /**
260
+ * Build the request body for one model call.
261
+ *
262
+ * History is projected part by part (tool results become `functionResponse`
263
+ * parts, replayed tool calls carry their thought signature), and consecutive
264
+ * same-role turns are merged, which is the one turn shape Gemini accepts.
265
+ *
266
+ * No `thinkingConfig` is ever sent: Gemini's own default (dynamic thinking) is
267
+ * what keeps 2.5 Pro working, since that model refuses a zero thinking budget.
268
+ * @param options - the harness request.
269
+ * @param config - project, location, and default output cap.
270
+ * @returns the wire body.
271
+ */
272
+ export function buildGeminiRequest(options, config) {
273
+ const model = options.model;
274
+ const names = toolNames(options.messages);
275
+ const contents = [];
276
+ for (const message of options.messages) {
277
+ if (message.role === 'system' || message.role === 'developer')
278
+ continue;
279
+ const role = message.role === 'assistant' ? 'model' : 'user';
280
+ const parts = message.role === 'tool'
281
+ ? toolResultParts(message, names)
282
+ : role === 'model'
283
+ ? assistantParts(message, model)
284
+ : userParts(message);
285
+ if (parts.length === 0)
286
+ continue;
287
+ const previous = contents.at(-1);
288
+ if (previous !== undefined && previous.role === role)
289
+ previous.parts.push(...parts);
290
+ else
291
+ contents.push({ role, parts });
292
+ }
293
+ const system = systemParts(options).join('\n\n');
294
+ const tools = (options.tools ?? []).map(tool => ({
295
+ name: tool.name,
296
+ description: tool.description,
297
+ parameters: geminiSchema(tool.parameters),
298
+ }));
299
+ const generationConfig = {
300
+ maxOutputTokens: options.maxTokens ?? config.maxTokens,
301
+ ...options.temperature === undefined ? {} : { temperature: options.temperature },
302
+ ...options.stop === undefined || options.stop.length === 0 ? {} : { stopSequences: [...options.stop] },
303
+ };
304
+ return {
305
+ contents,
306
+ ...system.length === 0 ? {} : { systemInstruction: { parts: [{ text: system }] } },
307
+ ...tools.length === 0 ? {} : { tools: [{ functionDeclarations: tools }] },
308
+ generationConfig,
309
+ };
310
+ }
311
+ /**
312
+ * Map Gemini's counters onto harness accounting.
313
+ *
314
+ * Gemini folds cached input into `promptTokenCount` and reports it separately as
315
+ * well, so the harness's disjoint rule means subtracting it out; thinking tokens
316
+ * are billed as output, exactly as the harness's own pi-ai mapping treats them.
317
+ * @param usage - the latest cumulative counters seen on the stream.
318
+ * @returns disjoint harness counts.
319
+ */
320
+ export function mapGeminiUsage(usage) {
321
+ const cached = usage.cachedContentTokenCount ?? 0;
322
+ const prompt = usage.promptTokenCount;
323
+ const thoughts = usage.thoughtsTokenCount ?? 0;
324
+ const candidates = usage.candidatesTokenCount;
325
+ const output = (candidates ?? 0) + thoughts;
326
+ // An exact total is the provider's own; the arithmetic fallback is only as
327
+ // complete as the two counters it needs, so a missing one omits the total
328
+ // rather than reporting a partial sum as the whole request.
329
+ const total = usage.totalTokenCount
330
+ ?? (prompt === undefined || candidates === undefined ? undefined : prompt + output);
331
+ return {
332
+ inputTokens: Math.max(0, (prompt ?? 0) - cached),
333
+ outputTokens: output,
334
+ ...total === undefined ? {} : { totalTokens: total },
335
+ ...cached > 0 ? { cacheReadTokens: cached } : {},
336
+ ...thoughts > 0 ? { reasoningTokens: thoughts } : {},
337
+ };
338
+ }
339
+ /**
340
+ * Map Gemini's finish reason onto the harness vocabulary.
341
+ *
342
+ * `STOP`, `OTHER`, and an absent reason all mean the same thing here, and so
343
+ * does any reason a provider release adds: the turn ended, and only a tool call
344
+ * in it changes what that is called.
345
+ * @param reason - the `finishReason` Vertex reported.
346
+ * @param sawToolCall - whether the response contained a function call, which
347
+ * Gemini reports as an ordinary `STOP`.
348
+ * @returns the harness finish reason.
349
+ */
350
+ export function mapGeminiFinishReason(reason, sawToolCall) {
351
+ switch (reason) {
352
+ case 'MAX_TOKENS':
353
+ return { kind: 'max-tokens' };
354
+ case 'SAFETY':
355
+ case 'RECITATION':
356
+ case 'BLOCKLIST':
357
+ case 'PROHIBITED_CONTENT':
358
+ case 'SPII':
359
+ case 'IMAGE_SAFETY':
360
+ return {
361
+ kind: 'error',
362
+ failure: {
363
+ message: `google-vertex: the model refused to answer (finish reason ${reason})`,
364
+ code: SAFETY_BLOCKED_CODE,
365
+ },
366
+ };
367
+ case 'MALFORMED_FUNCTION_CALL':
368
+ return {
369
+ kind: 'error',
370
+ failure: {
371
+ message: 'google-vertex: the model produced a malformed function call',
372
+ code: 'INVALID_REQUEST',
373
+ },
374
+ };
375
+ default:
376
+ return sawToolCall ? { kind: 'tool-calls' } : { kind: 'stop' };
377
+ }
378
+ }
379
+ /**
380
+ * Translate Gemini's `streamGenerateContent` chunks into harness chunks.
381
+ *
382
+ * Gemini streams whole parts rather than deltas, so the shape of this translator
383
+ * is: text parts append to one open text block, a function call closes that block
384
+ * and is itself closed immediately (arguments arrive complete), and the terminal
385
+ * chunk carries usage plus the stop reason.
386
+ */
387
+ export class GeminiStreamTranslator {
388
+ #model;
389
+ #open;
390
+ #nextIndex = 0;
391
+ #replay = [];
392
+ #usage = {};
393
+ #usageReported = false;
394
+ #finishReason;
395
+ #sawFinish = false;
396
+ #failed = false;
397
+ #sawToolCall = false;
398
+ /**
399
+ * @param model - the model being called, recorded in the replay envelope.
400
+ */
401
+ constructor(model) {
402
+ this.#model = model;
403
+ }
404
+ /**
405
+ * Feed one decoded SSE payload.
406
+ * @param event - the parsed chunk.
407
+ * @returns the chunks this payload completes, in order.
408
+ */
409
+ handle(event) {
410
+ if (this.#failed)
411
+ return [];
412
+ // Vertex can refuse mid-stream with an error envelope instead of a
413
+ // candidate. Without this the body simply ends, and the adapter would
414
+ // report a truncated response and lose the provider's own message and code.
415
+ const error = event['error'];
416
+ if (error !== undefined) {
417
+ this.#failed = true;
418
+ return [{ type: 'finish', reason: { kind: 'error', failure: failureForEvent(error) } }];
419
+ }
420
+ const out = [];
421
+ const candidates = event['candidates'];
422
+ if (Array.isArray(candidates)) {
423
+ for (const candidate of candidates) {
424
+ const fields = asObject(candidate);
425
+ if (fields === undefined)
426
+ continue;
427
+ const content = asObject(fields['content']);
428
+ const parts = content?.['parts'];
429
+ if (Array.isArray(parts))
430
+ for (const part of parts)
431
+ out.push(...this.#part(asObject(part)));
432
+ const reason = fields['finishReason'];
433
+ if (typeof reason === 'string') {
434
+ this.#finishReason = reason;
435
+ this.#sawFinish = true;
436
+ }
437
+ }
438
+ }
439
+ const usage = asObject(event['usageMetadata']);
440
+ if (usage !== undefined)
441
+ this.#mergeUsage(usage);
442
+ return out;
443
+ }
444
+ /** Absorb one content part. */
445
+ #part(part) {
446
+ if (part === undefined)
447
+ return [];
448
+ // A thought summary is reasoning text this adapter never requested; keeping
449
+ // it would need its own signature to replay, which the harness block cannot
450
+ // carry, so it is dropped rather than made unreplayable.
451
+ if (part['thought'] === true)
452
+ return [];
453
+ const signature = typeof part['thoughtSignature'] === 'string' ? part['thoughtSignature'] : undefined;
454
+ const call = asObject(part['functionCall']);
455
+ if (call !== undefined) {
456
+ const out = this.#closeText();
457
+ const index = this.#nextIndex;
458
+ this.#nextIndex += 1;
459
+ const name = typeof call['name'] === 'string' ? call['name'] : '';
460
+ const id = typeof call['id'] === 'string' && call['id'].length > 0 ? call['id'] : `call_${index}`;
461
+ const args = asObject(call['args']) ?? {};
462
+ this.#replay.push({ type: 'tool-call', ...signature === undefined ? {} : { thoughtSignature: signature } });
463
+ this.#sawToolCall = true;
464
+ out.push({ type: 'block-start', index, blockType: 'tool-call' }, { type: 'tool-call-delta', index, id, name, argumentsDelta: JSON.stringify(args) }, { type: 'block-end', index, block: { type: 'tool-call', id, name, arguments: JSON.stringify(args) } });
465
+ return out;
466
+ }
467
+ const text = part['text'];
468
+ if (typeof text === 'string' && text.length > 0) {
469
+ const open = this.#open;
470
+ if (open !== undefined) {
471
+ open.text += text;
472
+ if (signature !== undefined)
473
+ this.#setSignature(signature);
474
+ return [{ type: 'text-delta', index: open.index, text }];
475
+ }
476
+ const index = this.#nextIndex;
477
+ this.#nextIndex += 1;
478
+ this.#open = { index, kind: 'text', text, signature };
479
+ this.#replay.push({ type: 'text', ...signature === undefined ? {} : { thoughtSignature: signature } });
480
+ return [{ type: 'block-start', index, blockType: 'text' }, { type: 'text-delta', index, text }];
481
+ }
482
+ return [];
483
+ }
484
+ /** Replace the open block's recorded signature, if the provider sent one. */
485
+ #setSignature(signature) {
486
+ const open = this.#open;
487
+ if (open === undefined)
488
+ return;
489
+ const index = this.#replay.length - 1;
490
+ const entry = this.#replay[index];
491
+ if (entry === undefined || entry.type !== open.kind)
492
+ return;
493
+ this.#replay[index] = { type: entry.type, thoughtSignature: signature };
494
+ open.signature = signature;
495
+ }
496
+ /** Close an open text block, emitting its authoritative end. */
497
+ #closeText() {
498
+ const open = this.#open;
499
+ if (open === undefined)
500
+ return [];
501
+ this.#open = undefined;
502
+ return [{ type: 'block-end', index: open.index, block: { type: 'text', text: open.text } }];
503
+ }
504
+ /** Merge the newest cumulative usage counters. */
505
+ #mergeUsage(usage) {
506
+ const merged = takeCounters(usage, [
507
+ 'promptTokenCount',
508
+ 'candidatesTokenCount',
509
+ 'cachedContentTokenCount',
510
+ 'thoughtsTokenCount',
511
+ 'totalTokenCount',
512
+ ]);
513
+ // Only a counter the provider actually sent counts as a report.
514
+ if (Object.keys(merged).length === 0)
515
+ return;
516
+ this.#usageReported = true;
517
+ this.#usage = { ...this.#usage, ...merged };
518
+ }
519
+ /** True once the provider reported a finish reason. */
520
+ get sawFinish() {
521
+ return this.#sawFinish;
522
+ }
523
+ /** True once an in-band error ended the stream, which `handle` already reported. */
524
+ get failed() {
525
+ return this.#failed;
526
+ }
527
+ /** {@inheritDoc StreamTranslatorLike.terminal} */
528
+ get terminal() {
529
+ return this.#failed;
530
+ }
531
+ /**
532
+ * Terminal chunks: the closed tail, usage, and the finish reason.
533
+ *
534
+ * Called once, after the body ends. A response that produced nothing at all is
535
+ * reported as an empty response rather than as a silent success.
536
+ * @returns the trailing chunks in emission order.
537
+ */
538
+ finish() {
539
+ const out = this.#closeText();
540
+ const reason = mapGeminiFinishReason(this.#finishReason, this.#sawToolCall);
541
+ if (this.#replay.length === 0 && reason.kind !== 'error') {
542
+ return [{
543
+ type: 'finish',
544
+ reason: {
545
+ kind: 'error',
546
+ failure: {
547
+ message: 'google-vertex: the model completed the response with no content',
548
+ code: EMPTY_RESPONSE_CODE,
549
+ },
550
+ },
551
+ }];
552
+ }
553
+ return [
554
+ ...out,
555
+ // Usage accompanies a real provider report; a synthesized zero would
556
+ // claim a measurement that never happened.
557
+ ...this.#usageReported ? [{ type: 'usage', usage: mapGeminiUsage(this.#usage) }] : [],
558
+ { type: 'finish', reason, replayState: geminiReplayState(this.#model, this.#replay) },
559
+ ];
560
+ }
561
+ }
@@ -0,0 +1,61 @@
1
+ /**
2
+ * Provider adapter for Google's own Gemini models on Vertex AI.
3
+ *
4
+ * Same credential as the Claude route beside it (one service-account file, one
5
+ * bearer token per request) over the `publishers/google` endpoint instead of
6
+ * `publishers/anthropic`. Both live in one plugin row so project, region, and
7
+ * credentials are configured once.
8
+ *
9
+ * The route is text-only: `inputModalities: ['text']` makes `LlmRuntime` project
10
+ * images and files to placeholder text before dispatch. Tool calling is fully
11
+ * supported, including Vertex's required thought-signature replay.
12
+ *
13
+ * @module dsh-google-vertex/gemini-adapter
14
+ */
15
+ import { streamVertex, VertexPublisherAdapter, } from './adapter.js';
16
+ import { buildGeminiRequest, DEFAULT_GEMINI_CONTEXT_WINDOW, DEFAULT_GEMINI_MAX_TOKENS, GeminiStreamTranslator, geminiEndpointFor, } from './gemini.js';
17
+ /**
18
+ * The capacities every Gemini model serves, wherever the id came from. A
19
+ * catalog entry carries no capacities of its own until a model actually needs
20
+ * different ones.
21
+ */
22
+ const GEMINI_CAPACITY = {
23
+ contextWindow: DEFAULT_GEMINI_CONTEXT_WINDOW,
24
+ defaultMaxTokens: DEFAULT_GEMINI_MAX_TOKENS,
25
+ };
26
+ /**
27
+ * Duck-typed adapter over Vertex's Gemini publisher endpoint.
28
+ *
29
+ * The metadata face and the streaming pipeline are the shared ones; this class
30
+ * supplies the Gemini catalog, wording, endpoint, body, and translator.
31
+ */
32
+ export class GoogleVertexGeminiAdapter extends VertexPublisherAdapter {
33
+ /**
34
+ * @param config - the resolved configuration this adapter serves.
35
+ * @param options - transport, token-source, and discovery overrides for tests.
36
+ */
37
+ constructor(config, options = {}) {
38
+ super({
39
+ providerName: 'Google Vertex AI (Gemini)',
40
+ capacity: GEMINI_CAPACITY,
41
+ describe: row => `Google Gemini on Vertex AI (project ${row.project}, ${row.location}).`,
42
+ }, config, options);
43
+ }
44
+ /**
45
+ * Stream one completion through `:streamGenerateContent?alt=sse`.
46
+ *
47
+ * Gemini has no terminal event: the body simply ends, and the finish reason
48
+ * rides the last content chunk. A body that ends without one is therefore a
49
+ * truncated response, which is what {@link GeminiStreamTranslator.sawFinish}
50
+ * distinguishes, unless an in-band error or the idle watchdog already ended
51
+ * the turn. The shared pump owns the watchdog, the token mint, and the SSE
52
+ * loop; this route's finish is built at the end of the body.
53
+ */
54
+ stream(options) {
55
+ return streamVertex(this.config, options, this.fetch, this.tokens, model => new GeminiStreamTranslator(model), {
56
+ endpoint: (model, config) => geminiEndpointFor(config.project, config.location, model),
57
+ body: (request, config) => buildGeminiRequest(request, config),
58
+ truncatedMessage: model => `google-vertex: model "${model}" stream ended before a finish reason`,
59
+ });
60
+ }
61
+ }
package/lib/host.js ADDED
@@ -0,0 +1,18 @@
1
+ /**
2
+ * The slice of the DeepSeek Harness host surface this plugin uses, declared
3
+ * structurally.
4
+ *
5
+ * Like the other plugins under `~/dsh-plugins`, this package is installed
6
+ * outside the harness checkout and cannot resolve `@deepseek-ai/*` from its own
7
+ * directory, so it carries no runtime dependency on them. `LlmRuntime` reaches
8
+ * adapters through plain method calls (there is no `instanceof LlmAdapter`
9
+ * check anywhere in it), so a duck-typed adapter is a supported shape.
10
+ *
11
+ * Each declaration is narrowed to what this adapter reads or emits. Fields the
12
+ * adapter never touches are deliberately absent: a mirrored field that nothing
13
+ * reads is a field whose absence goes unnoticed. Widen a declaration when the
14
+ * adapter starts using it, not before.
15
+ *
16
+ * @module dsh-google-vertex/host
17
+ */
18
+ export {};