@prismshadow/mmsp 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +121 -0
  2. package/dist/ant_messages/client.d.ts +59 -0
  3. package/dist/ant_messages/client.d.ts.map +1 -0
  4. package/dist/ant_messages/client.js +419 -0
  5. package/dist/ant_messages/index.d.ts +2 -0
  6. package/dist/ant_messages/index.d.ts.map +1 -0
  7. package/dist/ant_messages/index.js +18 -0
  8. package/dist/anthropic_official/client.d.ts +63 -0
  9. package/dist/anthropic_official/client.d.ts.map +1 -0
  10. package/dist/anthropic_official/client.js +540 -0
  11. package/dist/anthropic_official/index.d.ts +2 -0
  12. package/dist/anthropic_official/index.d.ts.map +1 -0
  13. package/dist/anthropic_official/index.js +18 -0
  14. package/dist/autoClient.d.ts +96 -0
  15. package/dist/autoClient.d.ts.map +1 -0
  16. package/dist/autoClient.js +258 -0
  17. package/dist/baseClient.d.ts +111 -0
  18. package/dist/baseClient.d.ts.map +1 -0
  19. package/dist/baseClient.js +284 -0
  20. package/dist/deepseek_official/client.d.ts +60 -0
  21. package/dist/deepseek_official/client.d.ts.map +1 -0
  22. package/dist/deepseek_official/client.js +407 -0
  23. package/dist/deepseek_official/index.d.ts +2 -0
  24. package/dist/deepseek_official/index.d.ts.map +1 -0
  25. package/dist/deepseek_official/index.js +18 -0
  26. package/dist/errors.d.ts +84 -0
  27. package/dist/errors.d.ts.map +1 -0
  28. package/dist/errors.js +140 -0
  29. package/dist/gemini_generate_content/client.d.ts +86 -0
  30. package/dist/gemini_generate_content/client.d.ts.map +1 -0
  31. package/dist/gemini_generate_content/client.js +809 -0
  32. package/dist/gemini_generate_content/index.d.ts +2 -0
  33. package/dist/gemini_generate_content/index.d.ts.map +1 -0
  34. package/dist/gemini_generate_content/index.js +18 -0
  35. package/dist/gemini_official/client.d.ts +87 -0
  36. package/dist/gemini_official/client.d.ts.map +1 -0
  37. package/dist/gemini_official/client.js +780 -0
  38. package/dist/gemini_official/index.d.ts +2 -0
  39. package/dist/gemini_official/index.d.ts.map +1 -0
  40. package/dist/gemini_official/index.js +18 -0
  41. package/dist/index.d.ts +6 -0
  42. package/dist/index.d.ts.map +1 -0
  43. package/dist/index.js +44 -0
  44. package/dist/integration/index.d.ts +1 -0
  45. package/dist/integration/index.d.ts.map +1 -0
  46. package/dist/integration/index.js +14 -0
  47. package/dist/integration/playground.d.ts +17 -0
  48. package/dist/integration/playground.d.ts.map +1 -0
  49. package/dist/integration/playground.js +3246 -0
  50. package/dist/integration/tracer.d.ts +188 -0
  51. package/dist/integration/tracer.d.ts.map +1 -0
  52. package/dist/integration/tracer.js +1928 -0
  53. package/dist/legacy.d.ts +13 -0
  54. package/dist/legacy.d.ts.map +1 -0
  55. package/dist/legacy.js +59 -0
  56. package/dist/minimax_official/client.d.ts +43 -0
  57. package/dist/minimax_official/client.d.ts.map +1 -0
  58. package/dist/minimax_official/client.js +346 -0
  59. package/dist/minimax_official/index.d.ts +2 -0
  60. package/dist/minimax_official/index.d.ts.map +1 -0
  61. package/dist/minimax_official/index.js +18 -0
  62. package/dist/moonshot_official/client.d.ts +73 -0
  63. package/dist/moonshot_official/client.d.ts.map +1 -0
  64. package/dist/moonshot_official/client.js +477 -0
  65. package/dist/moonshot_official/index.d.ts +2 -0
  66. package/dist/moonshot_official/index.d.ts.map +1 -0
  67. package/dist/moonshot_official/index.js +18 -0
  68. package/dist/openai_chat/client.d.ts +67 -0
  69. package/dist/openai_chat/client.d.ts.map +1 -0
  70. package/dist/openai_chat/client.js +419 -0
  71. package/dist/openai_chat/index.d.ts +2 -0
  72. package/dist/openai_chat/index.d.ts.map +1 -0
  73. package/dist/openai_chat/index.js +18 -0
  74. package/dist/openai_chat_vllm_adapter/client.d.ts +15 -0
  75. package/dist/openai_chat_vllm_adapter/client.d.ts.map +1 -0
  76. package/dist/openai_chat_vllm_adapter/client.js +118 -0
  77. package/dist/openai_chat_vllm_adapter/index.d.ts +2 -0
  78. package/dist/openai_chat_vllm_adapter/index.d.ts.map +1 -0
  79. package/dist/openai_chat_vllm_adapter/index.js +5 -0
  80. package/dist/openai_embedding/client.d.ts +46 -0
  81. package/dist/openai_embedding/client.d.ts.map +1 -0
  82. package/dist/openai_embedding/client.js +124 -0
  83. package/dist/openai_embedding/index.d.ts +2 -0
  84. package/dist/openai_embedding/index.d.ts.map +1 -0
  85. package/dist/openai_embedding/index.js +18 -0
  86. package/dist/openai_official/client.d.ts +61 -0
  87. package/dist/openai_official/client.d.ts.map +1 -0
  88. package/dist/openai_official/client.js +464 -0
  89. package/dist/openai_official/index.d.ts +2 -0
  90. package/dist/openai_official/index.d.ts.map +1 -0
  91. package/dist/openai_official/index.js +18 -0
  92. package/dist/openai_responses/client.d.ts +61 -0
  93. package/dist/openai_responses/client.d.ts.map +1 -0
  94. package/dist/openai_responses/client.js +449 -0
  95. package/dist/openai_responses/index.d.ts +2 -0
  96. package/dist/openai_responses/index.d.ts.map +1 -0
  97. package/dist/openai_responses/index.js +18 -0
  98. package/dist/registry.d.ts +49 -0
  99. package/dist/registry.d.ts.map +1 -0
  100. package/dist/registry.js +798 -0
  101. package/dist/streamItems.d.ts +36 -0
  102. package/dist/streamItems.d.ts.map +1 -0
  103. package/dist/streamItems.js +183 -0
  104. package/dist/types.d.ts +190 -0
  105. package/dist/types.d.ts.map +1 -0
  106. package/dist/types.js +37 -0
  107. package/dist/utils.d.ts +99 -0
  108. package/dist/utils.d.ts.map +1 -0
  109. package/dist/utils.js +312 -0
  110. package/dist/zai_official/client.d.ts +78 -0
  111. package/dist/zai_official/client.d.ts.map +1 -0
  112. package/dist/zai_official/client.js +420 -0
  113. package/dist/zai_official/index.d.ts +2 -0
  114. package/dist/zai_official/index.d.ts.map +1 -0
  115. package/dist/zai_official/index.js +18 -0
  116. package/package.json +68 -0
@@ -0,0 +1,809 @@
1
+ "use strict";
2
+ // Copyright 2025 Prism Shadow. and/or its affiliates
3
+ //
4
+ // Licensed under the Apache License, Version 2.0 (the "License");
5
+ // you may not use this file except in compliance with the License.
6
+ // You may obtain a copy of the License at
7
+ //
8
+ // http://www.apache.org/licenses/LICENSE-2.0
9
+ //
10
+ // Unless required by applicable law or agreed to in writing, software
11
+ // distributed under the License is distributed on an "AS IS" BASIS,
12
+ // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ // See the License for the specific language governing permissions and
14
+ // limitations under the License.
15
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
16
+ if (k2 === undefined) k2 = k;
17
+ var desc = Object.getOwnPropertyDescriptor(m, k);
18
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
19
+ desc = { enumerable: true, get: function() { return m[k]; } };
20
+ }
21
+ Object.defineProperty(o, k2, desc);
22
+ }) : (function(o, m, k, k2) {
23
+ if (k2 === undefined) k2 = k;
24
+ o[k2] = m[k];
25
+ }));
26
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
27
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
28
+ }) : function(o, v) {
29
+ o["default"] = v;
30
+ });
31
+ var __importStar = (this && this.__importStar) || (function () {
32
+ var ownKeys = function(o) {
33
+ ownKeys = Object.getOwnPropertyNames || function (o) {
34
+ var ar = [];
35
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
36
+ return ar;
37
+ };
38
+ return ownKeys(o);
39
+ };
40
+ return function (mod) {
41
+ if (mod && mod.__esModule) return mod;
42
+ var result = {};
43
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
44
+ __setModuleDefault(result, mod);
45
+ return result;
46
+ };
47
+ })();
48
+ Object.defineProperty(exports, "__esModule", { value: true });
49
+ exports.GeminiGenerateContentClient = void 0;
50
+ const genai_1 = require("@google/genai");
51
+ const path = __importStar(require("path"));
52
+ const baseClient_1 = require("../baseClient");
53
+ const errors_1 = require("../errors");
54
+ const types_1 = require("../types");
55
+ const utils_1 = require("../utils");
56
+ /**
57
+ * Split a message's parts into consecutive runs of functionResponse and
58
+ * non-functionResponse parts, preserving order. Vertex AI requires function
59
+ * responses to sit in a content of their own (see the call site); a message
60
+ * without function responses — or with nothing else — comes back as one run.
61
+ */
62
+ function splitFunctionResponseRuns(parts) {
63
+ const runs = [];
64
+ let lastIsResponse = null;
65
+ for (const part of parts) {
66
+ const isResponse = part.functionResponse !== undefined;
67
+ if (isResponse !== lastIsResponse) {
68
+ runs.push([]);
69
+ lastIsResponse = isResponse;
70
+ }
71
+ runs[runs.length - 1].push(part);
72
+ }
73
+ return runs.length > 0 ? runs : [parts];
74
+ }
75
+ /**
76
+ * Client for the Gemini family through generateContent, named for the newest
77
+ * generation it serves (3.8). It serves Gemini on Vertex AI, whose Interactions
78
+ * endpoint serves none of these models, and gateways that proxy generateContent
79
+ * only: 3.8 back through the 3.x text, image, TTS, and embedding models. It
80
+ * applies the 3.6-generation parameter contract to the whole family:
81
+ * temperature is rejected everywhere.
82
+ *
83
+ * Starting with the 3.6 generation the API deprecates the temperature/top_p/top_k
84
+ * sampling parameters (silently ignored today, HTTP 400 in future
85
+ * generations), so this client rejects them instead of sending a no-op.
86
+ */
87
+ class GeminiGenerateContentClient extends baseClient_1.LLMClient {
88
+ /**
89
+ * Initialize Gemini 3.8 generateContent client with model and API key.
90
+ */
91
+ constructor(options) {
92
+ super();
93
+ this._model = options.model;
94
+ const key = options.apiKey || process.env.GEMINI_API_KEY || undefined;
95
+ const url = options.baseUrl || process.env.GEMINI_BASE_URL || undefined;
96
+ // the Gemini SDK carries connection headers inside httpOptions rather than its own argument
97
+ const httpOptions = {};
98
+ if (url) {
99
+ httpOptions.baseUrl = url;
100
+ }
101
+ if (options.defaultHeaders) {
102
+ httpOptions.headers = options.defaultHeaders;
103
+ }
104
+ if (key && key.startsWith("{")) {
105
+ const credentials = JSON.parse(key);
106
+ const googleAuthOptions = {
107
+ credentials,
108
+ scopes: ["https://www.googleapis.com/auth/cloud-platform"],
109
+ };
110
+ this._client = new genai_1.GoogleGenAI({
111
+ vertexai: true,
112
+ location: "global",
113
+ project: credentials.project_id,
114
+ googleAuthOptions,
115
+ httpOptions,
116
+ });
117
+ }
118
+ else {
119
+ this._client = new genai_1.GoogleGenAI({
120
+ apiKey: key,
121
+ httpOptions,
122
+ });
123
+ }
124
+ }
125
+ /**
126
+ * Detect MIME type from URL extension for image.
127
+ */
128
+ _detectImageMimeType(url) {
129
+ const ext = path.extname(url).toLowerCase();
130
+ const mimeTypes = {
131
+ ".bmp": "image/bmp",
132
+ ".gif": "image/gif",
133
+ ".jpg": "image/jpeg",
134
+ ".jpeg": "image/jpeg",
135
+ ".png": "image/png",
136
+ ".svg": "image/svg+xml",
137
+ ".tiff": "image/tiff",
138
+ ".webp": "image/webp",
139
+ };
140
+ return mimeTypes[ext] || "image/jpeg";
141
+ }
142
+ /**
143
+ * Get image bytes and MIME type from URL.
144
+ */
145
+ async _getImageBytesAndMimeType(url, signal) {
146
+ if (url.startsWith("data:")) {
147
+ const match = url.match(/^data:([^;]+);base64,(.+)$/);
148
+ if (match) {
149
+ const mimeType = match[1];
150
+ const base64Data = match[2];
151
+ const data = Buffer.from(base64Data, "base64");
152
+ return { data, mimeType };
153
+ }
154
+ else {
155
+ throw new Error(`Invalid base64 image: ${url}`);
156
+ }
157
+ }
158
+ else {
159
+ const response = await fetch(url, { signal });
160
+ if (!response.ok) {
161
+ throw new Error(`Failed to fetch image: ${url}`);
162
+ }
163
+ const arrayBuffer = await response.arrayBuffer();
164
+ const data = Buffer.from(arrayBuffer);
165
+ const mimeType = this._detectImageMimeType(url);
166
+ return { data, mimeType };
167
+ }
168
+ }
169
+ /**
170
+ * Thinking levels the target model accepts (llmsdk_docs/gemini3_8/docs/thinking.md).
171
+ *
172
+ * An empty array means the model rejects the thinking_level parameter
173
+ * entirely, so it must be omitted from the request.
174
+ */
175
+ _supportedThinkingLevels() {
176
+ if (this._model.includes("-image")) {
177
+ return [genai_1.ThinkingLevel.MINIMAL, genai_1.ThinkingLevel.HIGH];
178
+ }
179
+ if (this._model.includes("gemini-3-pro")) {
180
+ // The only pro generation without "medium".
181
+ return [genai_1.ThinkingLevel.LOW, genai_1.ThinkingLevel.HIGH];
182
+ }
183
+ if (this._model.includes("-pro")) {
184
+ // Every pro generation rejects "minimal"; matching broadly keeps
185
+ // future pro models on the safe side (clamping a level the model
186
+ // would have accepted costs a little accuracy, forwarding an
187
+ // unsupported one is a 400).
188
+ return [
189
+ genai_1.ThinkingLevel.LOW,
190
+ genai_1.ThinkingLevel.MEDIUM,
191
+ genai_1.ThinkingLevel.HIGH,
192
+ ];
193
+ }
194
+ if (this._model.includes("gemini-3.7") ||
195
+ this._model.includes("gemini-3.8")) {
196
+ // Both generations reject "minimal" with a 400 (3.7 verified live 2026-08-13,
197
+ // 3.8 on Vertex AI 2026-09-17).
198
+ return [
199
+ genai_1.ThinkingLevel.LOW,
200
+ genai_1.ThinkingLevel.MEDIUM,
201
+ genai_1.ThinkingLevel.HIGH,
202
+ ];
203
+ }
204
+ return GeminiGenerateContentClient.GEMINI_LEVEL_ORDER;
205
+ }
206
+ /**
207
+ * Convert ThinkingLevel enum to the closest Gemini ThinkingLevel the model supports.
208
+ */
209
+ _convertThinkingLevel(thinkingLevel) {
210
+ if (!thinkingLevel)
211
+ return undefined;
212
+ const mapping = {
213
+ [types_1.ThinkingLevel.NONE]: genai_1.ThinkingLevel.MINIMAL,
214
+ [types_1.ThinkingLevel.LOW]: genai_1.ThinkingLevel.LOW,
215
+ [types_1.ThinkingLevel.MEDIUM]: genai_1.ThinkingLevel.MEDIUM,
216
+ [types_1.ThinkingLevel.HIGH]: genai_1.ThinkingLevel.HIGH,
217
+ [types_1.ThinkingLevel.XHIGH]: genai_1.ThinkingLevel.HIGH,
218
+ // Gemini stops at "high", so both top levels land there before per-model clamping
219
+ [types_1.ThinkingLevel.MAX]: genai_1.ThinkingLevel.HIGH,
220
+ };
221
+ const level = mapping[thinkingLevel];
222
+ if (level === undefined) {
223
+ return undefined;
224
+ }
225
+ const supported = this._supportedThinkingLevels();
226
+ if (supported.length === 0) {
227
+ // A model that takes no thinking_level at all has nothing to clamp onto, so the
228
+ // parameter is omitted rather than turned into a failed request. thinking_summary
229
+ // is unaffected -- includeThoughts still rides along.
230
+ return undefined;
231
+ }
232
+ if (supported.includes(level)) {
233
+ return level;
234
+ }
235
+ // Degrade silently to the nearest supported level; ties round up,
236
+ // e.g. MEDIUM becomes HIGH on gemini-3-pro and NONE maps to LOW on
237
+ // gemini-3.7-flash. `supported` is non-empty here, so the
238
+ // initial-value-less reduce cannot throw.
239
+ const order = GeminiGenerateContentClient.GEMINI_LEVEL_ORDER;
240
+ const index = order.indexOf(level);
241
+ return supported.reduce((best, candidate) => {
242
+ const bestDistance = Math.abs(order.indexOf(best) - index);
243
+ const candidateDistance = Math.abs(order.indexOf(candidate) - index);
244
+ if (candidateDistance !== bestDistance) {
245
+ return candidateDistance < bestDistance ? candidate : best;
246
+ }
247
+ return order.indexOf(candidate) > order.indexOf(best) ? candidate : best;
248
+ });
249
+ }
250
+ /**
251
+ * Convert ToolChoice to Gemini's tool config.
252
+ */
253
+ _convertToolChoice(toolChoice) {
254
+ if (Array.isArray(toolChoice)) {
255
+ return {
256
+ mode: "ANY",
257
+ allowedFunctionNames: toolChoice,
258
+ };
259
+ }
260
+ else if (toolChoice === "none") {
261
+ return { mode: "NONE" };
262
+ }
263
+ else if (toolChoice === "auto") {
264
+ return { mode: "AUTO" };
265
+ }
266
+ else if (toolChoice === "required") {
267
+ return { mode: "ANY" };
268
+ }
269
+ return undefined;
270
+ }
271
+ _withAbortSignal(config, signal) {
272
+ if (!signal) {
273
+ return config;
274
+ }
275
+ return { ...(config ?? {}), abortSignal: signal };
276
+ }
277
+ /**
278
+ * Transform universal configuration to Gemini-specific configuration.
279
+ */
280
+ transformUniConfigToModelConfig(config) {
281
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
282
+ const configParams = {};
283
+ if (config.temperature !== undefined) {
284
+ throw new errors_1.UnsupportedParameterError({
285
+ client: this.constructor.name,
286
+ parameter: "temperature",
287
+ message: "Gemini models do not support setting temperature; the API deprecated " +
288
+ "sampling parameters starting with the 3.6 generation.",
289
+ });
290
+ }
291
+ if (config.fast_mode) {
292
+ throw new errors_1.UnsupportedParameterError({
293
+ client: this.constructor.name,
294
+ parameter: "fast_mode",
295
+ message: "Gemini does not support fast mode.",
296
+ });
297
+ }
298
+ if (config.prompt_caching !== undefined &&
299
+ config.prompt_caching !== types_1.PromptCaching.ENABLE) {
300
+ throw new errors_1.UnsupportedParameterError({
301
+ client: this.constructor.name,
302
+ parameter: "prompt_caching",
303
+ message: "prompt_caching must be ENABLE for Gemini.",
304
+ });
305
+ }
306
+ if (config.max_tokens !== undefined) {
307
+ configParams.maxOutputTokens = config.max_tokens;
308
+ }
309
+ // A TTS model takes the speech settings and nothing else: a system instruction, a
310
+ // thinking config, or a tool declaration each comes back as a 400 (verified live
311
+ // 2026-08-20), so the rest of the universal config never reaches the request.
312
+ if (this._model.toLowerCase().includes("tts")) {
313
+ configParams.responseModalities = ["AUDIO"];
314
+ const ttsConfig = config.tts_config ?? [{ voice: "Kore" }];
315
+ if (![1, 2].includes(ttsConfig.length)) {
316
+ throw new Error("tts_config must contain 1 or 2 entries.");
317
+ }
318
+ if (ttsConfig.length === 1) {
319
+ configParams.speechConfig = {
320
+ voiceConfig: {
321
+ prebuiltVoiceConfig: {
322
+ voiceName: ttsConfig[0].voice,
323
+ },
324
+ },
325
+ };
326
+ }
327
+ else {
328
+ const speakerVoiceConfigs = ttsConfig.map((speakerConfig) => {
329
+ if (!speakerConfig.speaker) {
330
+ throw new Error("speaker is required when tts_config has 2 entries.");
331
+ }
332
+ return {
333
+ speaker: speakerConfig.speaker,
334
+ voiceConfig: {
335
+ prebuiltVoiceConfig: {
336
+ voiceName: speakerConfig.voice,
337
+ },
338
+ },
339
+ };
340
+ });
341
+ configParams.speechConfig = {
342
+ multiSpeakerVoiceConfig: {
343
+ speakerVoiceConfigs,
344
+ },
345
+ };
346
+ }
347
+ return configParams;
348
+ }
349
+ if (config.system_prompt !== undefined) {
350
+ configParams.systemInstruction = config.system_prompt;
351
+ }
352
+ // includeThoughts asks for thought summaries, but whether generateContent returns any
353
+ // is model-dependent (llmsdk_docs/gemini3_8/docs/thinking.md)
354
+ const thinkingSummary = config.thinking_summary;
355
+ const thinkingLevel = config.thinking_level;
356
+ if (thinkingSummary !== undefined || thinkingLevel !== undefined) {
357
+ configParams.thinkingConfig = {
358
+ includeThoughts: thinkingSummary,
359
+ thinkingLevel: this._convertThinkingLevel(thinkingLevel),
360
+ };
361
+ }
362
+ if (config.tools !== undefined) {
363
+ configParams.tools = [{ functionDeclarations: config.tools }];
364
+ const toolChoice = config.tool_choice;
365
+ if (toolChoice !== undefined) {
366
+ const toolConfig = this._convertToolChoice(toolChoice);
367
+ if (toolConfig) {
368
+ configParams.toolConfig = {
369
+ functionCallingConfig: toolConfig,
370
+ };
371
+ }
372
+ }
373
+ }
374
+ if (config.image_config !== undefined) {
375
+ configParams.imageConfig = {
376
+ aspectRatio: config.image_config.aspect_ratio,
377
+ imageSize: config.image_config.image_size,
378
+ };
379
+ }
380
+ return Object.keys(configParams).length > 0
381
+ ? configParams
382
+ : undefined;
383
+ }
384
+ /**
385
+ * Transform universal message format to Gemini's Content format.
386
+ */
387
+ async transformUniMessageToModelInput(messages, signal) {
388
+ const mapping = {
389
+ user: "user",
390
+ assistant: "model",
391
+ };
392
+ const contents = [];
393
+ // The generateContent API wants both the call id and the function name on a function
394
+ // response, but a universal tool_result carries only the id, so remember each call's name.
395
+ const callNames = new Map();
396
+ for (const msg of messages) {
397
+ let parts = [];
398
+ for (const item of msg.content_items) {
399
+ const thoughtSignature = "fidelity" in item && item.fidelity?.signature
400
+ ? { thoughtSignature: item.fidelity.signature }
401
+ : {};
402
+ if (item.type === "text.done") {
403
+ parts.push({ text: item.text, ...thoughtSignature });
404
+ }
405
+ else if (item.type === "image_url.done") {
406
+ const urlValue = item.image_url;
407
+ const imageData = await this._getImageBytesAndMimeType(urlValue, signal);
408
+ parts.push({
409
+ inlineData: {
410
+ mimeType: imageData.mimeType,
411
+ data: imageData.data.toString("base64"),
412
+ },
413
+ });
414
+ }
415
+ else if (item.type === "inline_data.done") {
416
+ parts.push({
417
+ inlineData: {
418
+ mimeType: item.mime_type,
419
+ data: item.data.toString("base64"),
420
+ },
421
+ ...thoughtSignature,
422
+ });
423
+ }
424
+ else if (item.type === "thinking.done") {
425
+ parts.push({
426
+ text: item.thinking,
427
+ thought: true,
428
+ ...thoughtSignature,
429
+ });
430
+ }
431
+ else if (item.type === "inline_thinking.done") {
432
+ parts.push({
433
+ inlineData: {
434
+ mimeType: item.mime_type,
435
+ data: item.data.toString("base64"),
436
+ },
437
+ thought: true,
438
+ ...thoughtSignature,
439
+ });
440
+ }
441
+ else if (item.type === "tool_call.done") {
442
+ callNames.set(item.tool_call_id, item.name);
443
+ // Histories from before ids were stored carry the name as the tool_call_id;
444
+ // replay those without an id, exactly as they arrived.
445
+ const functionCall = {
446
+ ...(item.tool_call_id !== item.name
447
+ ? { id: item.tool_call_id }
448
+ : {}),
449
+ name: item.name,
450
+ args: item.arguments,
451
+ };
452
+ parts.push({ functionCall: functionCall, ...thoughtSignature });
453
+ }
454
+ else if (item.type === "tool_result.done") {
455
+ if (!item.tool_call_id) {
456
+ throw new Error("tool_call_id is required for tool result.");
457
+ }
458
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
459
+ const toolResult = { result: item.text };
460
+ const multimodalParts = [];
461
+ if (item.images) {
462
+ for (const imageUrl of item.images) {
463
+ const imageData = await this._getImageBytesAndMimeType(imageUrl, signal);
464
+ multimodalParts.push({
465
+ inlineData: {
466
+ mimeType: imageData.mimeType,
467
+ data: imageData.data.toString("base64"),
468
+ },
469
+ });
470
+ }
471
+ }
472
+ const functionName = callNames.get(item.tool_call_id) ?? item.tool_call_id;
473
+ parts.push({
474
+ functionResponse: {
475
+ ...(item.tool_call_id !== functionName
476
+ ? { id: item.tool_call_id }
477
+ : {}),
478
+ name: functionName,
479
+ response: toolResult,
480
+ parts: multimodalParts.length > 0 ? multimodalParts : undefined,
481
+ },
482
+ });
483
+ }
484
+ else {
485
+ throw new Error(`Unknown item: ${JSON.stringify(item)}`);
486
+ }
487
+ }
488
+ if (msg.role === "assistant") {
489
+ // generateContent validates a single signature, the one on the first function call of
490
+ // a step (llmsdk_docs/gemini3/docs/thought-signatures.md); a response without a call
491
+ // signs a later part, which goes back unvalidated. The Interactions client records a
492
+ // signature on the thinking item in front of what it signs instead, so a text thought's
493
+ // signature moves onto the first call, or onto the next part when the turn makes no
494
+ // call, and a thought left empty is dropped; an image thought keeps its own. A call
495
+ // nobody signed, one another provider made, carries the placeholder Google documents
496
+ // for calls the API did not produce.
497
+ const firstCall = parts.findIndex((part) => part.functionCall);
498
+ if (firstCall >= 0 && !parts[firstCall].thoughtSignature) {
499
+ let donor;
500
+ for (let i = firstCall - 1; i >= 0 && !donor; i--) {
501
+ if (parts[i].thought &&
502
+ parts[i].text !== undefined &&
503
+ parts[i].thoughtSignature) {
504
+ donor = parts[i];
505
+ }
506
+ }
507
+ if (donor) {
508
+ parts[firstCall].thoughtSignature = donor.thoughtSignature;
509
+ delete donor.thoughtSignature;
510
+ }
511
+ else {
512
+ parts[firstCall].thoughtSignature =
513
+ "skip_thought_signature_validator";
514
+ }
515
+ }
516
+ for (const [i, part] of parts.entries()) {
517
+ if (part.thought &&
518
+ part.text !== undefined &&
519
+ part.thoughtSignature) {
520
+ const next = parts
521
+ .slice(i + 1)
522
+ .find((later) => !later.thought &&
523
+ !later.functionResponse &&
524
+ !later.thoughtSignature);
525
+ if (next) {
526
+ next.thoughtSignature = part.thoughtSignature;
527
+ delete part.thoughtSignature;
528
+ }
529
+ }
530
+ }
531
+ parts = parts.filter((part) => !(part.thought &&
532
+ part.text === "" &&
533
+ !part.inlineData &&
534
+ !part.thoughtSignature));
535
+ }
536
+ // Vertex AI rejects a content that mixes functionResponse parts with any other
537
+ // part kind — the request fails with a misleading 400, "Requests ending with a
538
+ // model turn are not supported" (the Gemini API endpoint accepts the mix). Split
539
+ // such a message into consecutive same-role contents: each run of function
540
+ // responses becomes its own content, the surrounding parts keep theirs, and the
541
+ // part order is preserved. Homogeneous messages stay a single content.
542
+ for (const runParts of splitFunctionResponseRuns(parts)) {
543
+ contents.push({
544
+ role: mapping[msg.role],
545
+ parts: runParts,
546
+ });
547
+ }
548
+ }
549
+ return contents;
550
+ }
551
+ /**
552
+ * Transform one generateContent stream chunk into a universal event.
553
+ *
554
+ * generateContent gives a part no identity, so each delta's item_id is the kind of wire part
555
+ * that carried it: an item runs until a part of another kind arrives, except that every
556
+ * function call and every image is an item of its own, while audio chunks and consecutive text
557
+ * parts share one.
558
+ */
559
+ transformModelOutputToUniEvent(modelOutput) {
560
+ let eventType = "delta";
561
+ const contentItems = [];
562
+ let usageMetadata = null;
563
+ let finishReason = null;
564
+ const candidate = modelOutput.candidates?.[0];
565
+ if (candidate) {
566
+ for (const part of candidate.content?.parts ?? []) {
567
+ const fidelity = part.thoughtSignature
568
+ ? { signature: part.thoughtSignature }
569
+ : {};
570
+ if (part.functionCall) {
571
+ // generateContent sends a call whole, so it streams as one complete delta
572
+ contentItems.push({
573
+ type: "tool_call.delta",
574
+ name: part.functionCall.name ?? "",
575
+ arguments: JSON.stringify(part.functionCall.args ?? {}),
576
+ tool_call_id: part.functionCall.id || part.functionCall.name || "",
577
+ fidelity: { item_id: "function_call", ...fidelity },
578
+ });
579
+ }
580
+ else if (part.thought && part.text != null) {
581
+ if (part.text || part.thoughtSignature) {
582
+ contentItems.push({
583
+ type: "thinking.delta",
584
+ thinking: part.text,
585
+ fidelity: { item_id: "thought", ...fidelity },
586
+ });
587
+ }
588
+ }
589
+ else if (part.thought && part.inlineData) {
590
+ contentItems.push({
591
+ type: "inline_thinking.delta",
592
+ data: Buffer.from(part.inlineData.data || "", "base64"),
593
+ mime_type: part.inlineData.mimeType || "application/octet-stream",
594
+ fidelity: { item_id: "inline_thinking", ...fidelity },
595
+ });
596
+ }
597
+ else if (part.inlineData) {
598
+ contentItems.push({
599
+ type: "inline_data.delta",
600
+ data: Buffer.from(part.inlineData.data || "", "base64"),
601
+ mime_type: part.inlineData.mimeType || "application/octet-stream",
602
+ fidelity: { item_id: "inline_data", ...fidelity },
603
+ });
604
+ }
605
+ else if (part.text != null) {
606
+ // a response ends on an empty text part, which carries something only when it brings
607
+ // the signature
608
+ if (part.text || part.thoughtSignature) {
609
+ contentItems.push({
610
+ type: "text.delta",
611
+ text: part.text,
612
+ fidelity: { item_id: "text", ...fidelity },
613
+ });
614
+ }
615
+ }
616
+ else if ((0, utils_1.isDebugEnabled)()) {
617
+ throw new Error(`Unknown output: ${JSON.stringify(part)}`);
618
+ }
619
+ }
620
+ if (candidate.finishReason) {
621
+ eventType = "stop";
622
+ const stopReasonMapping = {
623
+ [genai_1.FinishReason.STOP]: "stop",
624
+ [genai_1.FinishReason.MAX_TOKENS]: "length",
625
+ };
626
+ finishReason = stopReasonMapping[candidate.finishReason] || "unknown";
627
+ }
628
+ }
629
+ // Vertex AI puts a usage object carrying only its traffic type on every chunk; the counts
630
+ // arrive with the last one
631
+ const usage = modelOutput.usageMetadata;
632
+ if (usage?.promptTokenCount != null) {
633
+ eventType = "stop";
634
+ usageMetadata = {
635
+ cached_tokens: usage.cachedContentTokenCount || null,
636
+ prompt_tokens: (usage.promptTokenCount || 0) - (usage.cachedContentTokenCount || 0),
637
+ thoughts_tokens: usage.thoughtsTokenCount || null,
638
+ response_tokens: usage.candidatesTokenCount || null,
639
+ };
640
+ }
641
+ return {
642
+ role: "assistant",
643
+ event_type: eventType,
644
+ content_items: contentItems,
645
+ usage_metadata: usageMetadata,
646
+ finish_reason: finishReason,
647
+ };
648
+ }
649
+ async *_embedMessagesInternal(options) {
650
+ const geminiConfig = this._withAbortSignal(options.config.embedding_config?.dimensions != null
651
+ ? {
652
+ outputDimensionality: options.config.embedding_config.dimensions,
653
+ }
654
+ : undefined, options.signal);
655
+ // Vertex AI embeds one content per call: a second content is a 400 there, and both SDKs refuse
656
+ // to send one. It reports no billable characters either, only a token count per embedding.
657
+ let promptTokens = null;
658
+ for (const msg of options.messages) {
659
+ const parts = [];
660
+ for (const item of msg.content_items) {
661
+ if (item.type === "text.done") {
662
+ parts.push({ text: item.text });
663
+ }
664
+ else if (item.type === "image_url.done") {
665
+ const imageData = await this._getImageBytesAndMimeType(item.image_url, options.signal);
666
+ parts.push({
667
+ inlineData: {
668
+ mimeType: imageData.mimeType,
669
+ data: imageData.data.toString("base64"),
670
+ },
671
+ });
672
+ }
673
+ else if (item.type === "inline_data.done") {
674
+ parts.push({
675
+ inlineData: {
676
+ mimeType: item.mime_type,
677
+ data: item.data.toString("base64"),
678
+ },
679
+ });
680
+ }
681
+ else {
682
+ throw new Error(`Unknown item: ${JSON.stringify(item)}`);
683
+ }
684
+ }
685
+ const result = await this._client.models.embedContent({
686
+ model: this._model,
687
+ contents: [{ role: msg.role === "user" ? "user" : "model", parts }],
688
+ config: geminiConfig,
689
+ });
690
+ const embedding = result.embeddings?.[0];
691
+ // a vector streams once its call returns; the usage, summed over the calls, follows the last one
692
+ yield {
693
+ role: "assistant",
694
+ event_type: "delta",
695
+ content_items: [
696
+ { type: "embedding.delta", embedding: embedding?.values ?? [] },
697
+ ],
698
+ usage_metadata: null,
699
+ finish_reason: null,
700
+ };
701
+ const tokenCount = embedding?.statistics?.tokenCount ??
702
+ result.metadata?.billableCharacterCount;
703
+ if (tokenCount != null) {
704
+ promptTokens = (promptTokens ?? 0) + tokenCount;
705
+ }
706
+ }
707
+ yield {
708
+ role: "assistant",
709
+ event_type: "stop",
710
+ content_items: [],
711
+ usage_metadata: {
712
+ cached_tokens: null,
713
+ prompt_tokens: promptTokens,
714
+ thoughts_tokens: null,
715
+ response_tokens: null,
716
+ },
717
+ finish_reason: "stop",
718
+ };
719
+ }
720
+ /**
721
+ * Stream generate using Gemini SDK with unified conversion methods.
722
+ */
723
+ async *_streamingResponseInternal(options) {
724
+ if (this._model.toLowerCase().includes("embedding")) {
725
+ yield* this._embedMessagesInternal(options);
726
+ return;
727
+ }
728
+ // A TTS model synthesizes a single text turn: a conversation comes back as "Multiturn chat
729
+ // is not enabled for this model" and an audio part as "Audio input modality is not enabled
730
+ // for this model" (verified live 2026-08-20), so only the newest message is sent and the
731
+ // audio a stateful session records stays out of the request.
732
+ let messages = options.messages;
733
+ if (this._model.toLowerCase().includes("tts")) {
734
+ messages = messages.slice(-1);
735
+ const invalidItem = messages
736
+ .flatMap((message) => message.content_items)
737
+ .find((item) => item.type !== "text.done");
738
+ if (invalidItem) {
739
+ throw new Error(`Gemini TTS only supports text input, got content item type=${JSON.stringify(invalidItem.type)}.`);
740
+ }
741
+ }
742
+ const geminiConfig = this._withAbortSignal(this.transformUniConfigToModelConfig(options.config), options.signal);
743
+ let contents = await this.transformUniMessageToModelInput(messages, options.signal);
744
+ // Gemini 3.8 TTS takes each speaker's turn as its own part carrying the speaker as
745
+ // speechMetadata (the SDK drops fields it does not know, so this needs @google/genai
746
+ // 2.24), and rejects "Name: line" labels in a two-speaker request, while 3.1 TTS
747
+ // rejects the metadata (both verified live 2026-09-30), so only 3.8 gets the script split.
748
+ const speakers = (options.config.tts_config ?? []).map((entry) => entry.speaker ?? "");
749
+ if (this._model.toLowerCase().includes("tts") &&
750
+ this._model.includes("gemini-3.8") &&
751
+ speakers.length === 2) {
752
+ const script = messages[messages.length - 1].content_items
753
+ .map((item) => (item.type === "text.done" ? item.text : ""))
754
+ .join("\n");
755
+ contents = [
756
+ {
757
+ role: "user",
758
+ parts: (0, utils_1.speakerTurns)(script, speakers).map(([speaker, turn]) => ({
759
+ text: turn,
760
+ speechMetadata: { speaker },
761
+ })),
762
+ },
763
+ ];
764
+ }
765
+ const responseStream = await this._client.models.generateContentStream({
766
+ model: this._model,
767
+ contents: contents,
768
+ config: geminiConfig,
769
+ });
770
+ let sawFunctionCall = false;
771
+ for await (const chunk of responseStream) {
772
+ const event = this.transformModelOutputToUniEvent(chunk);
773
+ if (event.content_items.some((item) => item.type === "tool_call.delta")) {
774
+ sawFunctionCall = true;
775
+ }
776
+ // generateContent reports STOP for a turn that stopped to call tools
777
+ const finishReason = event.finish_reason === "stop" && sawFunctionCall
778
+ ? "tool_call"
779
+ : event.finish_reason;
780
+ yield { ...event, finish_reason: finishReason };
781
+ }
782
+ }
783
+ /**
784
+ * List the model ids the configured endpoint serves.
785
+ *
786
+ * @returns The model ids, in the order the endpoint returned them.
787
+ */
788
+ async listModels() {
789
+ const models = [];
790
+ for await (const model of await this._client.models.list()) {
791
+ // the API returns path-qualified names: models/gemini-3.7-flash,
792
+ // publishers/google/models/gemini-3.7-flash
793
+ const id = model.name?.split("/").pop();
794
+ if (id) {
795
+ models.push(id);
796
+ }
797
+ }
798
+ return models;
799
+ }
800
+ }
801
+ exports.GeminiGenerateContentClient = GeminiGenerateContentClient;
802
+ // Gemini thinking levels from weakest to strongest, used to pick the
803
+ // closest supported level when a model rejects the requested one.
804
+ GeminiGenerateContentClient.GEMINI_LEVEL_ORDER = [
805
+ genai_1.ThinkingLevel.MINIMAL,
806
+ genai_1.ThinkingLevel.LOW,
807
+ genai_1.ThinkingLevel.MEDIUM,
808
+ genai_1.ThinkingLevel.HIGH,
809
+ ];