@prismshadow/mmsp 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +121 -0
  2. package/dist/ant_messages/client.d.ts +59 -0
  3. package/dist/ant_messages/client.d.ts.map +1 -0
  4. package/dist/ant_messages/client.js +419 -0
  5. package/dist/ant_messages/index.d.ts +2 -0
  6. package/dist/ant_messages/index.d.ts.map +1 -0
  7. package/dist/ant_messages/index.js +18 -0
  8. package/dist/anthropic_official/client.d.ts +63 -0
  9. package/dist/anthropic_official/client.d.ts.map +1 -0
  10. package/dist/anthropic_official/client.js +540 -0
  11. package/dist/anthropic_official/index.d.ts +2 -0
  12. package/dist/anthropic_official/index.d.ts.map +1 -0
  13. package/dist/anthropic_official/index.js +18 -0
  14. package/dist/autoClient.d.ts +96 -0
  15. package/dist/autoClient.d.ts.map +1 -0
  16. package/dist/autoClient.js +258 -0
  17. package/dist/baseClient.d.ts +111 -0
  18. package/dist/baseClient.d.ts.map +1 -0
  19. package/dist/baseClient.js +284 -0
  20. package/dist/deepseek_official/client.d.ts +60 -0
  21. package/dist/deepseek_official/client.d.ts.map +1 -0
  22. package/dist/deepseek_official/client.js +407 -0
  23. package/dist/deepseek_official/index.d.ts +2 -0
  24. package/dist/deepseek_official/index.d.ts.map +1 -0
  25. package/dist/deepseek_official/index.js +18 -0
  26. package/dist/errors.d.ts +84 -0
  27. package/dist/errors.d.ts.map +1 -0
  28. package/dist/errors.js +140 -0
  29. package/dist/gemini_generate_content/client.d.ts +86 -0
  30. package/dist/gemini_generate_content/client.d.ts.map +1 -0
  31. package/dist/gemini_generate_content/client.js +809 -0
  32. package/dist/gemini_generate_content/index.d.ts +2 -0
  33. package/dist/gemini_generate_content/index.d.ts.map +1 -0
  34. package/dist/gemini_generate_content/index.js +18 -0
  35. package/dist/gemini_official/client.d.ts +87 -0
  36. package/dist/gemini_official/client.d.ts.map +1 -0
  37. package/dist/gemini_official/client.js +780 -0
  38. package/dist/gemini_official/index.d.ts +2 -0
  39. package/dist/gemini_official/index.d.ts.map +1 -0
  40. package/dist/gemini_official/index.js +18 -0
  41. package/dist/index.d.ts +6 -0
  42. package/dist/index.d.ts.map +1 -0
  43. package/dist/index.js +44 -0
  44. package/dist/integration/index.d.ts +1 -0
  45. package/dist/integration/index.d.ts.map +1 -0
  46. package/dist/integration/index.js +14 -0
  47. package/dist/integration/playground.d.ts +17 -0
  48. package/dist/integration/playground.d.ts.map +1 -0
  49. package/dist/integration/playground.js +3246 -0
  50. package/dist/integration/tracer.d.ts +188 -0
  51. package/dist/integration/tracer.d.ts.map +1 -0
  52. package/dist/integration/tracer.js +1928 -0
  53. package/dist/legacy.d.ts +13 -0
  54. package/dist/legacy.d.ts.map +1 -0
  55. package/dist/legacy.js +59 -0
  56. package/dist/minimax_official/client.d.ts +43 -0
  57. package/dist/minimax_official/client.d.ts.map +1 -0
  58. package/dist/minimax_official/client.js +346 -0
  59. package/dist/minimax_official/index.d.ts +2 -0
  60. package/dist/minimax_official/index.d.ts.map +1 -0
  61. package/dist/minimax_official/index.js +18 -0
  62. package/dist/moonshot_official/client.d.ts +73 -0
  63. package/dist/moonshot_official/client.d.ts.map +1 -0
  64. package/dist/moonshot_official/client.js +477 -0
  65. package/dist/moonshot_official/index.d.ts +2 -0
  66. package/dist/moonshot_official/index.d.ts.map +1 -0
  67. package/dist/moonshot_official/index.js +18 -0
  68. package/dist/openai_chat/client.d.ts +67 -0
  69. package/dist/openai_chat/client.d.ts.map +1 -0
  70. package/dist/openai_chat/client.js +419 -0
  71. package/dist/openai_chat/index.d.ts +2 -0
  72. package/dist/openai_chat/index.d.ts.map +1 -0
  73. package/dist/openai_chat/index.js +18 -0
  74. package/dist/openai_chat_vllm_adapter/client.d.ts +15 -0
  75. package/dist/openai_chat_vllm_adapter/client.d.ts.map +1 -0
  76. package/dist/openai_chat_vllm_adapter/client.js +118 -0
  77. package/dist/openai_chat_vllm_adapter/index.d.ts +2 -0
  78. package/dist/openai_chat_vllm_adapter/index.d.ts.map +1 -0
  79. package/dist/openai_chat_vllm_adapter/index.js +5 -0
  80. package/dist/openai_embedding/client.d.ts +46 -0
  81. package/dist/openai_embedding/client.d.ts.map +1 -0
  82. package/dist/openai_embedding/client.js +124 -0
  83. package/dist/openai_embedding/index.d.ts +2 -0
  84. package/dist/openai_embedding/index.d.ts.map +1 -0
  85. package/dist/openai_embedding/index.js +18 -0
  86. package/dist/openai_official/client.d.ts +61 -0
  87. package/dist/openai_official/client.d.ts.map +1 -0
  88. package/dist/openai_official/client.js +464 -0
  89. package/dist/openai_official/index.d.ts +2 -0
  90. package/dist/openai_official/index.d.ts.map +1 -0
  91. package/dist/openai_official/index.js +18 -0
  92. package/dist/openai_responses/client.d.ts +61 -0
  93. package/dist/openai_responses/client.d.ts.map +1 -0
  94. package/dist/openai_responses/client.js +449 -0
  95. package/dist/openai_responses/index.d.ts +2 -0
  96. package/dist/openai_responses/index.d.ts.map +1 -0
  97. package/dist/openai_responses/index.js +18 -0
  98. package/dist/registry.d.ts +49 -0
  99. package/dist/registry.d.ts.map +1 -0
  100. package/dist/registry.js +798 -0
  101. package/dist/streamItems.d.ts +36 -0
  102. package/dist/streamItems.d.ts.map +1 -0
  103. package/dist/streamItems.js +183 -0
  104. package/dist/types.d.ts +190 -0
  105. package/dist/types.d.ts.map +1 -0
  106. package/dist/types.js +37 -0
  107. package/dist/utils.d.ts +99 -0
  108. package/dist/utils.d.ts.map +1 -0
  109. package/dist/utils.js +312 -0
  110. package/dist/zai_official/client.d.ts +78 -0
  111. package/dist/zai_official/client.d.ts.map +1 -0
  112. package/dist/zai_official/client.js +420 -0
  113. package/dist/zai_official/index.d.ts +2 -0
  114. package/dist/zai_official/index.d.ts.map +1 -0
  115. package/dist/zai_official/index.js +18 -0
  116. package/package.json +68 -0
@@ -0,0 +1,780 @@
1
+ "use strict";
2
+ // Copyright 2025 Prism Shadow. and/or its affiliates
3
+ //
4
+ // Licensed under the Apache License, Version 2.0 (the "License");
5
+ // you may not use this file except in compliance with the License.
6
+ // You may obtain a copy of the License at
7
+ //
8
+ // http://www.apache.org/licenses/LICENSE-2.0
9
+ //
10
+ // Unless required by applicable law or agreed to in writing, software
11
+ // distributed under the License is distributed on an "AS IS" BASIS,
12
+ // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ // See the License for the specific language governing permissions and
14
+ // limitations under the License.
15
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
16
+ if (k2 === undefined) k2 = k;
17
+ var desc = Object.getOwnPropertyDescriptor(m, k);
18
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
19
+ desc = { enumerable: true, get: function() { return m[k]; } };
20
+ }
21
+ Object.defineProperty(o, k2, desc);
22
+ }) : (function(o, m, k, k2) {
23
+ if (k2 === undefined) k2 = k;
24
+ o[k2] = m[k];
25
+ }));
26
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
27
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
28
+ }) : function(o, v) {
29
+ o["default"] = v;
30
+ });
31
+ var __importStar = (this && this.__importStar) || (function () {
32
+ var ownKeys = function(o) {
33
+ ownKeys = Object.getOwnPropertyNames || function (o) {
34
+ var ar = [];
35
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
36
+ return ar;
37
+ };
38
+ return ownKeys(o);
39
+ };
40
+ return function (mod) {
41
+ if (mod && mod.__esModule) return mod;
42
+ var result = {};
43
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
44
+ __setModuleDefault(result, mod);
45
+ return result;
46
+ };
47
+ })();
48
+ Object.defineProperty(exports, "__esModule", { value: true });
49
+ exports.GeminiOfficialClient = void 0;
50
+ const genai_1 = require("@google/genai");
51
+ const path = __importStar(require("path"));
52
+ const baseClient_1 = require("../baseClient");
53
+ const errors_1 = require("../errors");
54
+ const types_1 = require("../types");
55
+ const utils_1 = require("../utils");
56
+ /**
57
+ * Unified client for the Gemini family, named for the newest generation it
58
+ * serves (3.8). It speaks the Interactions API statelessly (store=false, the
59
+ * whole history in every request) for 3.8 back through the 3.x text, image,
60
+ * and TTS models with an API key; Vertex AI is served by
61
+ * gemini_generate_content. It embeds through embedContent, because the
62
+ * Interactions API does not serve the embedding models, and applies the
63
+ * 3.6-generation parameter contract to the whole family: temperature is
64
+ * rejected everywhere.
65
+ *
66
+ * Starting with the 3.6 generation the API deprecates the temperature/top_p/top_k
67
+ * sampling parameters (silently ignored today, HTTP 400 in future
68
+ * generations), so this client rejects them instead of sending a no-op.
69
+ */
70
+ class GeminiOfficialClient extends baseClient_1.LLMClient {
71
+ /**
72
+ * Initialize Gemini 3.8 client with model and API key.
73
+ */
74
+ constructor(options) {
75
+ super();
76
+ this._model = options.model;
77
+ const key = options.apiKey || process.env.GEMINI_API_KEY || undefined;
78
+ const url = options.baseUrl || process.env.GEMINI_BASE_URL || undefined;
79
+ // the Gemini SDK carries connection headers inside httpOptions rather than its own argument
80
+ const httpOptions = {};
81
+ if (url) {
82
+ httpOptions.baseUrl = url;
83
+ }
84
+ if (options.defaultHeaders) {
85
+ httpOptions.headers = options.defaultHeaders;
86
+ }
87
+ if (key && key.startsWith("{")) {
88
+ const credentials = JSON.parse(key);
89
+ const googleAuthOptions = {
90
+ credentials,
91
+ scopes: ["https://www.googleapis.com/auth/cloud-platform"],
92
+ };
93
+ this._client = new genai_1.GoogleGenAI({
94
+ vertexai: true,
95
+ location: "global",
96
+ project: credentials.project_id,
97
+ googleAuthOptions,
98
+ httpOptions,
99
+ });
100
+ }
101
+ else {
102
+ this._client = new genai_1.GoogleGenAI({
103
+ apiKey: key,
104
+ httpOptions,
105
+ });
106
+ }
107
+ }
108
+ /**
109
+ * Detect MIME type from URL extension for image.
110
+ */
111
+ _detectImageMimeType(url) {
112
+ const ext = path.extname(url).toLowerCase();
113
+ const mimeTypes = {
114
+ ".bmp": "image/bmp",
115
+ ".gif": "image/gif",
116
+ ".jpg": "image/jpeg",
117
+ ".jpeg": "image/jpeg",
118
+ ".png": "image/png",
119
+ ".svg": "image/svg+xml",
120
+ ".tiff": "image/tiff",
121
+ ".webp": "image/webp",
122
+ };
123
+ return mimeTypes[ext] || "image/jpeg";
124
+ }
125
+ /**
126
+ * Get image bytes and MIME type from URL.
127
+ */
128
+ async _getImageBytesAndMimeType(url, signal) {
129
+ if (url.startsWith("data:")) {
130
+ const match = url.match(/^data:([^;]+);base64,(.+)$/);
131
+ if (match) {
132
+ const mimeType = match[1];
133
+ const base64Data = match[2];
134
+ const data = Buffer.from(base64Data, "base64");
135
+ return { data, mimeType };
136
+ }
137
+ else {
138
+ throw new Error(`Invalid base64 image: ${url}`);
139
+ }
140
+ }
141
+ else {
142
+ const response = await fetch(url, { signal });
143
+ if (!response.ok) {
144
+ throw new Error(`Failed to fetch image: ${url}`);
145
+ }
146
+ const arrayBuffer = await response.arrayBuffer();
147
+ const data = Buffer.from(arrayBuffer);
148
+ const mimeType = this._detectImageMimeType(url);
149
+ return { data, mimeType };
150
+ }
151
+ }
152
+ /**
153
+ * Thinking levels the target model accepts (llmsdk_docs/gemini_interactions/docs/thinking.md).
154
+ *
155
+ * An empty array means the model rejects the thinking_level parameter
156
+ * entirely, so it must be omitted from the request.
157
+ */
158
+ _supportedThinkingLevels() {
159
+ if (this._model.includes("-image")) {
160
+ return ["minimal", "high"];
161
+ }
162
+ if (this._model.includes("gemini-3-pro")) {
163
+ // The only pro generation without "medium".
164
+ return ["low", "high"];
165
+ }
166
+ if (this._model.includes("-pro")) {
167
+ // Every pro generation rejects "minimal"; matching broadly keeps
168
+ // future pro models on the safe side (clamping a level the model
169
+ // would have accepted costs a little accuracy, forwarding an
170
+ // unsupported one is a 400).
171
+ return ["low", "medium", "high"];
172
+ }
173
+ if (this._model.includes("gemini-3.7") ||
174
+ this._model.includes("gemini-3.8")) {
175
+ // Both generations reject "minimal" with a 400 (3.7 verified live 2026-08-13,
176
+ // 3.7 and 3.8 again through the Interactions API 2026-09-16).
177
+ return ["low", "medium", "high"];
178
+ }
179
+ return GeminiOfficialClient.GEMINI_LEVEL_ORDER;
180
+ }
181
+ /**
182
+ * Convert ThinkingLevel enum to the closest Gemini thinking level the model supports.
183
+ */
184
+ _convertThinkingLevel(thinkingLevel) {
185
+ if (!thinkingLevel)
186
+ return undefined;
187
+ const mapping = {
188
+ [types_1.ThinkingLevel.NONE]: "minimal",
189
+ [types_1.ThinkingLevel.LOW]: "low",
190
+ [types_1.ThinkingLevel.MEDIUM]: "medium",
191
+ [types_1.ThinkingLevel.HIGH]: "high",
192
+ [types_1.ThinkingLevel.XHIGH]: "high",
193
+ // Gemini stops at "high", so both top levels land there before per-model clamping
194
+ [types_1.ThinkingLevel.MAX]: "high",
195
+ };
196
+ const level = mapping[thinkingLevel];
197
+ if (level === undefined) {
198
+ return undefined;
199
+ }
200
+ const supported = this._supportedThinkingLevels();
201
+ if (supported.length === 0) {
202
+ // A model that takes no thinking_level at all has nothing to clamp onto, so the
203
+ // parameter is omitted rather than turned into a failed request. thinking_summary
204
+ // is unaffected -- thinking_summaries still rides along.
205
+ return undefined;
206
+ }
207
+ if (supported.includes(level)) {
208
+ return level;
209
+ }
210
+ // Degrade silently to the nearest supported level; ties round up,
211
+ // e.g. MEDIUM becomes HIGH on gemini-3-pro and NONE maps to LOW on
212
+ // gemini-3.7-flash. `supported` is non-empty here, so the
213
+ // initial-value-less reduce cannot throw.
214
+ const order = GeminiOfficialClient.GEMINI_LEVEL_ORDER;
215
+ const index = order.indexOf(level);
216
+ return supported.reduce((best, candidate) => {
217
+ const bestDistance = Math.abs(order.indexOf(best) - index);
218
+ const candidateDistance = Math.abs(order.indexOf(candidate) - index);
219
+ if (candidateDistance !== bestDistance) {
220
+ return candidateDistance < bestDistance ? candidate : best;
221
+ }
222
+ return order.indexOf(candidate) > order.indexOf(best) ? candidate : best;
223
+ });
224
+ }
225
+ /**
226
+ * Convert ToolChoice to the Interactions API tool_choice.
227
+ */
228
+ _convertToolChoice(toolChoice) {
229
+ if (Array.isArray(toolChoice)) {
230
+ // allowed_tools takes only the "any" and "validated" modes (verified live 2026-09-16)
231
+ return { allowed_tools: { mode: "any", tools: toolChoice } };
232
+ }
233
+ else if (toolChoice === "none") {
234
+ return "none";
235
+ }
236
+ else if (toolChoice === "auto") {
237
+ return "auto";
238
+ }
239
+ else if (toolChoice === "required") {
240
+ return "any";
241
+ }
242
+ return undefined;
243
+ }
244
+ /**
245
+ * Transform universal configuration to an Interactions API request without its input.
246
+ */
247
+ transformUniConfigToModelConfig(config) {
248
+ if (config.temperature !== undefined) {
249
+ throw new errors_1.UnsupportedParameterError({
250
+ client: this.constructor.name,
251
+ parameter: "temperature",
252
+ message: "Gemini models do not support setting temperature; the API deprecated " +
253
+ "sampling parameters starting with the 3.6 generation.",
254
+ });
255
+ }
256
+ if (config.prompt_caching !== undefined &&
257
+ config.prompt_caching !== types_1.PromptCaching.ENABLE) {
258
+ throw new errors_1.UnsupportedParameterError({
259
+ client: this.constructor.name,
260
+ parameter: "prompt_caching",
261
+ message: "prompt_caching must be ENABLE for Gemini.",
262
+ });
263
+ }
264
+ // the history travels in every request, so nothing needs to be stored server-side
265
+ const geminiConfig = {
266
+ model: this._model,
267
+ stream: true,
268
+ store: false,
269
+ };
270
+ const generationConfig = {};
271
+ if (config.max_tokens !== undefined) {
272
+ generationConfig.max_output_tokens = config.max_tokens;
273
+ }
274
+ if (config.fast_mode) {
275
+ geminiConfig.service_tier = "priority";
276
+ }
277
+ // A TTS model takes the speech settings and nothing else: a system instruction, a
278
+ // thinking config, or a tool declaration each comes back as a 400 (verified live
279
+ // 2026-08-20, again through the Interactions API 2026-09-16), so the rest of the
280
+ // universal config never reaches the request.
281
+ if (this._model.toLowerCase().includes("tts")) {
282
+ const ttsConfig = config.tts_config ?? [{ voice: "Kore" }];
283
+ if (![1, 2].includes(ttsConfig.length)) {
284
+ throw new Error("tts_config must contain 1 or 2 entries.");
285
+ }
286
+ geminiConfig.response_format = { type: "audio" };
287
+ generationConfig.speech_config =
288
+ ttsConfig.length === 1
289
+ ? [{ voice: ttsConfig[0].voice }]
290
+ : ttsConfig.map((speakerConfig) => {
291
+ if (!speakerConfig.speaker) {
292
+ throw new Error("speaker is required when tts_config has 2 entries.");
293
+ }
294
+ return {
295
+ speaker: speakerConfig.speaker,
296
+ voice: speakerConfig.voice,
297
+ };
298
+ });
299
+ geminiConfig.generation_config = generationConfig;
300
+ return geminiConfig;
301
+ }
302
+ if (config.system_prompt !== undefined) {
303
+ geminiConfig.system_instruction = config.system_prompt;
304
+ }
305
+ const thinkingLevel = this._convertThinkingLevel(config.thinking_level);
306
+ if (thinkingLevel !== undefined) {
307
+ generationConfig.thinking_level = thinkingLevel;
308
+ }
309
+ if (config.thinking_summary !== undefined) {
310
+ generationConfig.thinking_summaries = config.thinking_summary
311
+ ? "auto"
312
+ : "none";
313
+ }
314
+ if (config.tools !== undefined) {
315
+ geminiConfig.tools = config.tools.map((tool) => ({
316
+ type: "function",
317
+ ...tool,
318
+ }));
319
+ if (config.tool_choice !== undefined) {
320
+ generationConfig.tool_choice = this._convertToolChoice(config.tool_choice);
321
+ }
322
+ }
323
+ if (config.image_config !== undefined) {
324
+ // an image entry alone suppresses the text the model writes beside its images
325
+ geminiConfig.response_format = [
326
+ { type: "text" },
327
+ { type: "image", ...config.image_config },
328
+ ];
329
+ }
330
+ if (Object.keys(generationConfig).length > 0) {
331
+ geminiConfig.generation_config = generationConfig;
332
+ }
333
+ return geminiConfig;
334
+ }
335
+ /**
336
+ * Transform universal message format to Interactions API input steps.
337
+ */
338
+ async transformUniMessageToModelInput(messages, signal) {
339
+ const steps = [];
340
+ // A function_result must name its function (HTTP 400 without it), but a universal
341
+ // tool_result carries only the call id, so remember each call's name.
342
+ const callNames = new Map();
343
+ for (const msg of messages) {
344
+ const messageStart = steps.length;
345
+ // consecutive text and media of a message share one user_input or model_output step
346
+ let content = null;
347
+ // consecutive thinking items share one thought step, which ends at the item carrying
348
+ // the signature: a stream closes every thought step with its signature
349
+ let thought = null;
350
+ for (const item of msg.content_items) {
351
+ if (item.type === "thinking.done" ||
352
+ item.type === "inline_thinking.done") {
353
+ content = null;
354
+ const signature = item.fidelity?.signature;
355
+ const summary = [];
356
+ if (item.type === "inline_thinking.done") {
357
+ summary.push({
358
+ type: "image",
359
+ data: item.data.toString("base64"),
360
+ mime_type: item.mime_type,
361
+ });
362
+ }
363
+ else if (item.thinking) {
364
+ summary.push({ type: "text", text: item.thinking });
365
+ }
366
+ if (summary.length === 0 && !signature) {
367
+ continue;
368
+ }
369
+ if (thought === null) {
370
+ thought = { type: "thought", summary: [] };
371
+ steps.push(thought);
372
+ }
373
+ thought.summary.push(...summary);
374
+ if (signature) {
375
+ thought.signature = signature;
376
+ thought = null;
377
+ }
378
+ continue;
379
+ }
380
+ thought = null;
381
+ if ("fidelity" in item && item.fidelity?.signature) {
382
+ // Histories recorded through generateContent carry the signature on the text,
383
+ // image or call it came with and hold no thinking item; the Interactions API takes
384
+ // it back as a thought step in front of that item (verified live 2026-09-16).
385
+ steps.push({ type: "thought", signature: item.fidelity.signature });
386
+ content = null;
387
+ // A thought summary such a history holds is unsigned, and a turn opening with an unsigned
388
+ // thought is rejected ("Request contains an invalid argument") while the same signature
389
+ // on two thoughts is accepted (verified live 2026-09-17), so the opening thought takes it too.
390
+ const first = steps[messageStart];
391
+ if (first.type === "thought" && !first.signature) {
392
+ first.signature = item.fidelity.signature;
393
+ }
394
+ }
395
+ if (item.type === "text.done" ||
396
+ item.type === "image_url.done" ||
397
+ item.type === "inline_data.done") {
398
+ let block;
399
+ if (item.type === "text.done") {
400
+ // an empty text block is rejected: "Missing text in content of type text"
401
+ if (!item.text) {
402
+ continue;
403
+ }
404
+ block = { type: "text", text: item.text };
405
+ }
406
+ else {
407
+ const { data, mimeType } = item.type === "image_url.done"
408
+ ? await this._getImageBytesAndMimeType(item.image_url, signal)
409
+ : { data: item.data, mimeType: item.mime_type };
410
+ // the block type follows the MIME type: image/jpeg is an image, application/pdf a document
411
+ const kind = mimeType.split("/")[0];
412
+ block = {
413
+ type: ["image", "audio", "video"].includes(kind)
414
+ ? kind
415
+ : "document",
416
+ data: data.toString("base64"),
417
+ mime_type: mimeType,
418
+ };
419
+ }
420
+ if (content === null) {
421
+ content = [];
422
+ steps.push({
423
+ type: msg.role === "user" ? "user_input" : "model_output",
424
+ content,
425
+ });
426
+ }
427
+ content.push(block);
428
+ }
429
+ else if (item.type === "tool_call.done") {
430
+ content = null;
431
+ callNames.set(item.tool_call_id, item.name);
432
+ // Histories from before ids were stored carry the name as the tool_call_id; replay
433
+ // those without an id, because parallel calls sharing one id are rejected with a
434
+ // 400 (verified live 2026-09-16).
435
+ steps.push({
436
+ type: "function_call",
437
+ ...(item.tool_call_id !== item.name
438
+ ? { id: item.tool_call_id }
439
+ : {}),
440
+ name: item.name,
441
+ arguments: item.arguments,
442
+ });
443
+ }
444
+ else if (item.type === "tool_result.done") {
445
+ content = null;
446
+ if (!item.tool_call_id) {
447
+ throw new Error("tool_call_id is required for tool result.");
448
+ }
449
+ let result = item.text;
450
+ if (item.images) {
451
+ // an empty text block is rejected, while a result of images alone is accepted
452
+ const resultContent = item.text
453
+ ? [{ type: "text", text: item.text }]
454
+ : [];
455
+ for (const imageUrl of item.images) {
456
+ const imageData = await this._getImageBytesAndMimeType(imageUrl, signal);
457
+ resultContent.push({
458
+ type: "image",
459
+ data: imageData.data.toString("base64"),
460
+ mime_type: imageData.mimeType,
461
+ });
462
+ }
463
+ result = resultContent;
464
+ }
465
+ const functionName = callNames.get(item.tool_call_id) ?? item.tool_call_id;
466
+ steps.push({
467
+ type: "function_result",
468
+ ...(item.tool_call_id !== functionName
469
+ ? { call_id: item.tool_call_id }
470
+ : {}),
471
+ name: functionName,
472
+ result,
473
+ });
474
+ }
475
+ else {
476
+ throw new Error(`Unknown item: ${JSON.stringify(item)}`);
477
+ }
478
+ }
479
+ // An image generation model (gemini-*-image) sometimes streams its text before its first
480
+ // thought step, but the API takes a turn holding a thought back only when the turn opens
481
+ // with one: "Model turns with images must start with a thought block" (verified live
482
+ // 2026-09-16). A turn another provider produced holds no signed thought at all, which the
483
+ // API rejects once the turn continues with its tool results (verified live 2026-09-16).
484
+ // Both open with the placeholder signature Google documents for thoughts it did not
485
+ // produce.
486
+ const turn = steps.slice(messageStart);
487
+ if ((turn.some((step) => step.type === "thought") &&
488
+ turn[0].type !== "thought") ||
489
+ (turn.some((step) => step.type === "model_output" || step.type === "function_call") &&
490
+ !turn.some((step) => step.type === "thought" && step.signature))) {
491
+ steps.splice(messageStart, 0, {
492
+ type: "thought",
493
+ signature: "skip_thought_signature_validator",
494
+ });
495
+ }
496
+ }
497
+ return steps;
498
+ }
499
+ /**
500
+ * Transform one Interactions API stream event into a universal event, its items identified by
501
+ * step index. A step streams one item per run of a content kind: an image generation model's
502
+ * thought summary can go text, image, text, which is three items. Every image delta is a whole
503
+ * image and an item of its own, while audio streams in chunks of one item.
504
+ */
505
+ transformModelOutputToUniEvent(modelOutput) {
506
+ let eventType = "delta";
507
+ const contentItems = [];
508
+ let usageMetadata = null;
509
+ let finishReason = null;
510
+ if (modelOutput.event_type === "step.start") {
511
+ const step = modelOutput.step;
512
+ if (step.type === "function_call") {
513
+ // the start names the call; its arguments stream as deltas behind an empty object
514
+ const startArguments = Object.keys(step.arguments ?? {}).length > 0
515
+ ? JSON.stringify(step.arguments)
516
+ : "";
517
+ contentItems.push({
518
+ type: "tool_call.delta",
519
+ name: step.name,
520
+ arguments: startArguments,
521
+ tool_call_id: step.id,
522
+ fidelity: { item_id: String(modelOutput.index) },
523
+ });
524
+ }
525
+ else if (step.type === "thought" || step.type === "model_output") {
526
+ // their content arrives in the step's deltas
527
+ }
528
+ else if ((0, utils_1.isDebugEnabled)()) {
529
+ throw new Error(`Unknown output: ${JSON.stringify(modelOutput)}`);
530
+ }
531
+ }
532
+ else if (modelOutput.event_type === "step.delta") {
533
+ const itemId = String(modelOutput.index);
534
+ const delta = modelOutput.delta;
535
+ if (delta.type === "thought_summary" && delta.content?.type === "text") {
536
+ contentItems.push({
537
+ type: "thinking.delta",
538
+ thinking: delta.content.text,
539
+ fidelity: { item_id: itemId },
540
+ });
541
+ }
542
+ else if (delta.type === "thought_summary" &&
543
+ delta.content?.type === "image") {
544
+ // image generation models summarize their thinking with interim images too
545
+ contentItems.push({
546
+ type: "inline_thinking.delta",
547
+ data: Buffer.from(delta.content.data || "", "base64"),
548
+ mime_type: delta.content.mime_type || "image/jpeg",
549
+ fidelity: { item_id: itemId },
550
+ });
551
+ }
552
+ else if (delta.type === "thought_signature") {
553
+ // the signature is the last delta of its thought step, and belongs to the item the
554
+ // step ends with, an image one included
555
+ contentItems.push({
556
+ type: "thinking.delta",
557
+ thinking: "",
558
+ fidelity: { item_id: itemId, signature: delta.signature },
559
+ });
560
+ }
561
+ else if (delta.type === "arguments_delta") {
562
+ contentItems.push({
563
+ type: "tool_call.delta",
564
+ name: "",
565
+ arguments: delta.arguments || "",
566
+ tool_call_id: "",
567
+ fidelity: { item_id: itemId },
568
+ });
569
+ }
570
+ else if (delta.type === "text") {
571
+ contentItems.push({
572
+ type: "text.delta",
573
+ text: delta.text,
574
+ fidelity: { item_id: itemId },
575
+ });
576
+ }
577
+ else if (delta.type === "image") {
578
+ contentItems.push({
579
+ type: "inline_data.delta",
580
+ data: Buffer.from(delta.data || "", "base64"),
581
+ mime_type: delta.mime_type || "image/jpeg",
582
+ fidelity: { item_id: itemId },
583
+ });
584
+ }
585
+ else if (delta.type === "audio") {
586
+ // TTS streams raw PCM in 40 ms chunks; the MIME type carries the format a player needs
587
+ contentItems.push({
588
+ type: "inline_data.delta",
589
+ data: Buffer.from(delta.data || "", "base64"),
590
+ mime_type: `${delta.mime_type}; rate=${delta.sample_rate}; channels=${delta.channels}`,
591
+ fidelity: { item_id: itemId },
592
+ });
593
+ }
594
+ else if ((0, utils_1.isDebugEnabled)()) {
595
+ throw new Error(`Unknown output: ${JSON.stringify(modelOutput)}`);
596
+ }
597
+ }
598
+ else if (modelOutput.event_type === "interaction.completed") {
599
+ eventType = "stop";
600
+ const statusMapping = {
601
+ completed: "stop",
602
+ requires_action: "tool_call",
603
+ incomplete: "length",
604
+ };
605
+ finishReason = statusMapping[modelOutput.interaction.status] || "unknown";
606
+ const usage = modelOutput.interaction.usage;
607
+ // total_input_tokens includes the cached tokens; total_output_tokens excludes the thoughts
608
+ usageMetadata = {
609
+ cached_tokens: usage?.total_cached_tokens || null,
610
+ prompt_tokens: (usage?.total_input_tokens || 0) - (usage?.total_cached_tokens || 0),
611
+ thoughts_tokens: usage?.total_thought_tokens || null,
612
+ response_tokens: usage?.total_output_tokens || null,
613
+ };
614
+ }
615
+ else if (modelOutput.event_type === "error" && modelOutput.error) {
616
+ // Neither Interactions SDK raises on an error event inside an open stream, so the provider's
617
+ // failure is raised here rather than lost; an error event without an error, which the Python
618
+ // SDK makes of a gateway heartbeat, stays with the unknown-event guard.
619
+ throw new Error(`Gemini stream error ${modelOutput.error.code}: ${modelOutput.error.message}`);
620
+ }
621
+ else if ([
622
+ "interaction.created",
623
+ "interaction.status_update",
624
+ "step.stop",
625
+ ].includes(modelOutput.event_type)) {
626
+ // the interaction's lifecycle carries nothing universal, and a step needs no stop: its
627
+ // last item is done when the next step begins or the stream ends
628
+ }
629
+ else if ((0, utils_1.isDebugEnabled)()) {
630
+ throw new Error(`Unknown output: ${JSON.stringify(modelOutput)}`);
631
+ }
632
+ // the API adds event, step and delta types over time, and killing a long generation
633
+ // over one costs more than dropping it
634
+ return {
635
+ role: "assistant",
636
+ event_type: eventType,
637
+ content_items: contentItems,
638
+ usage_metadata: usageMetadata,
639
+ finish_reason: finishReason,
640
+ };
641
+ }
642
+ async *_embedMessagesInternal(options) {
643
+ // the Interactions API does not serve embedding models (HTTP 404, verified live
644
+ // 2026-09-16), so they stay on embedContent
645
+ const contents = [];
646
+ for (const msg of options.messages) {
647
+ const parts = [];
648
+ for (const item of msg.content_items) {
649
+ if (item.type === "text.done") {
650
+ parts.push({ text: item.text });
651
+ }
652
+ else if (item.type === "image_url.done") {
653
+ const imageData = await this._getImageBytesAndMimeType(item.image_url, options.signal);
654
+ parts.push({
655
+ inlineData: {
656
+ mimeType: imageData.mimeType,
657
+ data: imageData.data.toString("base64"),
658
+ },
659
+ });
660
+ }
661
+ else if (item.type === "inline_data.done") {
662
+ parts.push({
663
+ inlineData: {
664
+ mimeType: item.mime_type,
665
+ data: item.data.toString("base64"),
666
+ },
667
+ });
668
+ }
669
+ else {
670
+ throw new Error(`Unknown item: ${JSON.stringify(item)}`);
671
+ }
672
+ }
673
+ contents.push({
674
+ role: msg.role === "user" ? "user" : "model",
675
+ parts,
676
+ });
677
+ }
678
+ const geminiConfig = { abortSignal: options.signal };
679
+ if (options.config.embedding_config?.dimensions != null) {
680
+ geminiConfig.outputDimensionality =
681
+ options.config.embedding_config.dimensions;
682
+ }
683
+ const result = await this._client.models.embedContent({
684
+ model: this._model,
685
+ contents,
686
+ config: geminiConfig,
687
+ });
688
+ yield {
689
+ role: "assistant",
690
+ event_type: "stop",
691
+ content_items: result.embeddings?.map((embedding) => ({
692
+ type: "embedding.delta",
693
+ embedding: embedding.values ?? [],
694
+ })) ?? [],
695
+ usage_metadata: {
696
+ cached_tokens: null,
697
+ prompt_tokens: result.metadata?.billableCharacterCount ?? null,
698
+ thoughts_tokens: null,
699
+ response_tokens: null,
700
+ },
701
+ finish_reason: "stop",
702
+ };
703
+ }
704
+ /**
705
+ * Stream generate through the Interactions API with unified conversion methods.
706
+ */
707
+ async *_streamingResponseInternal(options) {
708
+ if (this._model.toLowerCase().includes("embedding")) {
709
+ yield* this._embedMessagesInternal(options);
710
+ return;
711
+ }
712
+ // A TTS model synthesizes a single text turn: a conversation comes back as "Multiturn chat
713
+ // is not enabled for this model" and an audio part as "Audio input modality is not enabled
714
+ // for this model" (verified live 2026-08-20), so only the newest message is sent and the
715
+ // audio a stateful session records stays out of the request.
716
+ let messages = options.messages;
717
+ if (this._model.toLowerCase().includes("tts")) {
718
+ messages = messages.slice(-1);
719
+ const invalidItem = messages
720
+ .flatMap((message) => message.content_items)
721
+ .find((item) => item.type !== "text.done");
722
+ if (invalidItem) {
723
+ throw new Error(`Gemini TTS only supports text input, got content item type=${JSON.stringify(invalidItem.type)}.`);
724
+ }
725
+ }
726
+ const geminiConfig = this.transformUniConfigToModelConfig(options.config);
727
+ let input = await this.transformUniMessageToModelInput(messages, options.signal);
728
+ // Gemini 3.8 TTS takes each speaker's turn as its own text block carrying the speaker as
729
+ // metadata, and rejects "Name: line" labels in a two-speaker request, while 3.1 TTS rejects
730
+ // the metadata (both verified live 2026-09-30), so only 3.8 gets the script split into turns.
731
+ const speakers = (options.config.tts_config ?? []).map((entry) => entry.speaker ?? "");
732
+ if (this._model.toLowerCase().includes("tts") &&
733
+ this._model.includes("gemini-3.8") &&
734
+ speakers.length === 2) {
735
+ const script = messages[messages.length - 1].content_items
736
+ .map((item) => (item.type === "text.done" ? item.text : ""))
737
+ .join("\n");
738
+ input = [
739
+ {
740
+ type: "user_input",
741
+ content: (0, utils_1.speakerTurns)(script, speakers).map(([speaker, turn]) => ({
742
+ type: "text",
743
+ text: turn,
744
+ annotations: [{ type: "speech_metadata", speaker }],
745
+ })),
746
+ },
747
+ ];
748
+ }
749
+ const stream = await this._client.interactions.create({ ...geminiConfig, input, stream: true }, { signal: options.signal });
750
+ for await (const event of stream) {
751
+ yield this.transformModelOutputToUniEvent(event);
752
+ }
753
+ }
754
+ /**
755
+ * List the model ids the configured endpoint serves.
756
+ *
757
+ * @returns The model ids, in the order the endpoint returned them.
758
+ */
759
+ async listModels() {
760
+ const models = [];
761
+ for await (const model of await this._client.models.list()) {
762
+ // the API returns path-qualified names: models/gemini-3.7-flash,
763
+ // publishers/google/models/gemini-3.7-flash
764
+ const id = model.name?.split("/").pop();
765
+ if (id) {
766
+ models.push(id);
767
+ }
768
+ }
769
+ return models;
770
+ }
771
+ }
772
+ exports.GeminiOfficialClient = GeminiOfficialClient;
773
+ // Gemini thinking levels from weakest to strongest, used to pick the
774
+ // closest supported level when a model rejects the requested one.
775
+ GeminiOfficialClient.GEMINI_LEVEL_ORDER = [
776
+ "minimal",
777
+ "low",
778
+ "medium",
779
+ "high",
780
+ ];