@zhivex-ai/gemini 0.10.5 → 0.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,4 +1,5 @@
1
1
  import { toJSONSchema } from "zod";
2
+ import { capabilities, groundedCapabilities, imageGenerationCapabilities, isGeminiLiveTranslateModel, isGeminiLiveTranscribeModel, musicGenerationCapabilities, realtimeCapabilities, speechCapabilities, transcriptionCapabilities, videoGenerationCapabilities, } from "./capabilities.js";
2
3
  import { CallbackRealtimeSession, ConfigurationError, ProviderHTTPError, UnsupportedFeatureError, assertTrustedEndpoint, createMcpToolSet, createProviderAdapter, decodeBase64WithLimit, encodeAudioFrame, encodeMediaFrame, isCallableToolDefinition, isHostedToolDefinition, normalizeFinishReason, openWebSocketConnection, readBodyWithLimit, readErrorBodyWithLimit, readJsonWithLimit, streamSSE, toToolSet, toolResultPayload, withRetry, withTimeoutSignal } from "@zhivex-ai/core";
3
4
  const encodeGeminiPathSegment = (value, label) => {
4
5
  if (!value || value === "." || value === ".." || /[\\/?#\s]/.test(value)) {
@@ -16,196 +17,6 @@ const encodeGeminiResourceName = (value, label) => {
16
17
  }
17
18
  return segments.map(encodeURIComponent).join("/");
18
19
  };
19
- const capabilities = {
20
- streaming: true,
21
- tools: true,
22
- structuredOutput: true,
23
- jsonMode: true,
24
- toolChoice: true,
25
- parallelToolCalls: false,
26
- vision: true,
27
- files: true,
28
- audioInput: true,
29
- audioOutput: false,
30
- embeddings: true,
31
- fileSearch: true,
32
- urlContext: true,
33
- contextCaching: true,
34
- batch: true,
35
- interactions: true,
36
- rawPrediction: true,
37
- computerUse: true,
38
- reasoning: true,
39
- webSearch: true,
40
- agentCapabilities: {
41
- supportTier: "tier-b",
42
- toolChoiceNone: true,
43
- approvalRequests: false,
44
- hostedWebSearch: true,
45
- hostedFileSearch: true,
46
- remoteMcp: false,
47
- computerUse: true,
48
- codeExecution: true,
49
- toolsets: false
50
- }
51
- };
52
- const transcriptionCapabilities = {
53
- ...capabilities,
54
- streaming: false,
55
- tools: false,
56
- structuredOutput: false,
57
- jsonMode: false,
58
- toolChoice: false,
59
- parallelToolCalls: false,
60
- audioInput: true,
61
- audioOutput: false,
62
- embeddings: false,
63
- reasoning: false,
64
- webSearch: false,
65
- agentCapabilities: {
66
- supportTier: "tier-c",
67
- toolChoiceNone: false,
68
- approvalRequests: false,
69
- hostedWebSearch: false,
70
- hostedFileSearch: false,
71
- remoteMcp: false,
72
- computerUse: false,
73
- codeExecution: false,
74
- toolsets: false
75
- }
76
- };
77
- const speechCapabilities = {
78
- ...transcriptionCapabilities,
79
- streaming: true,
80
- audioInput: false,
81
- audioOutput: true
82
- };
83
- const groundedCapabilities = {
84
- ...capabilities,
85
- webSearch: true
86
- };
87
- const imageGenerationCapabilities = {
88
- ...capabilities,
89
- streaming: false,
90
- tools: false,
91
- structuredOutput: false,
92
- jsonMode: false,
93
- toolChoice: false,
94
- parallelToolCalls: false,
95
- embeddings: false,
96
- imageGeneration: true,
97
- videoGeneration: false,
98
- musicGeneration: false,
99
- reasoning: false,
100
- webSearch: false,
101
- agentCapabilities: {
102
- supportTier: "tier-c",
103
- toolChoiceNone: false,
104
- approvalRequests: false,
105
- hostedWebSearch: false,
106
- hostedFileSearch: false,
107
- remoteMcp: false,
108
- computerUse: false,
109
- codeExecution: false,
110
- toolsets: false
111
- }
112
- };
113
- const videoGenerationCapabilities = {
114
- ...capabilities,
115
- streaming: false,
116
- tools: false,
117
- structuredOutput: false,
118
- jsonMode: false,
119
- toolChoice: false,
120
- parallelToolCalls: false,
121
- vision: false,
122
- embeddings: false,
123
- imageGeneration: false,
124
- videoGeneration: true,
125
- musicGeneration: false,
126
- reasoning: false,
127
- webSearch: false,
128
- agentCapabilities: {
129
- supportTier: "tier-c",
130
- toolChoiceNone: false,
131
- approvalRequests: false,
132
- hostedWebSearch: false,
133
- hostedFileSearch: false,
134
- remoteMcp: false,
135
- computerUse: false,
136
- codeExecution: false,
137
- toolsets: false
138
- }
139
- };
140
- const musicGenerationCapabilities = {
141
- ...capabilities,
142
- streaming: false,
143
- tools: false,
144
- structuredOutput: false,
145
- jsonMode: false,
146
- toolChoice: false,
147
- parallelToolCalls: false,
148
- embeddings: false,
149
- imageGeneration: false,
150
- videoGeneration: false,
151
- musicGeneration: true,
152
- reasoning: false,
153
- webSearch: false,
154
- agentCapabilities: {
155
- supportTier: "tier-c",
156
- toolChoiceNone: false,
157
- approvalRequests: false,
158
- hostedWebSearch: false,
159
- hostedFileSearch: false,
160
- remoteMcp: false,
161
- computerUse: false,
162
- codeExecution: false,
163
- toolsets: false
164
- }
165
- };
166
- const realtimeCapabilities = (modelId) => {
167
- const translation = isGeminiLiveTranslateModel(modelId);
168
- return {
169
- ...capabilities,
170
- streaming: false,
171
- tools: !translation,
172
- structuredOutput: false,
173
- jsonMode: false,
174
- toolChoice: false,
175
- parallelToolCalls: false,
176
- vision: !translation,
177
- files: false,
178
- audioInput: true,
179
- audioOutput: true,
180
- embeddings: false,
181
- fileSearch: false,
182
- urlContext: false,
183
- contextCaching: false,
184
- batch: false,
185
- interactions: false,
186
- rawPrediction: false,
187
- computerUse: false,
188
- reasoning: !translation,
189
- webSearch: !translation,
190
- agentCapabilities: {
191
- ...capabilities.agentCapabilities,
192
- toolChoiceNone: !translation,
193
- hostedWebSearch: !translation,
194
- hostedFileSearch: false,
195
- remoteMcp: false,
196
- computerUse: false,
197
- codeExecution: false
198
- },
199
- realtime: {
200
- sessions: true,
201
- audioInput: true,
202
- audioOutput: true,
203
- imageInput: !translation,
204
- tools: !translation,
205
- browserTokens: true
206
- }
207
- };
208
- };
209
20
  const MIB = 1024 * 1024;
210
21
  const MAX_JSON_RESPONSE_BYTES = 128 * MIB;
211
22
  const MAX_MEDIA_RESPONSE_BYTES = 64 * MIB;
@@ -749,7 +560,6 @@ const mapRealtimeProviderOptions = (providerOptions) => providerOptions
749
560
  "translationConfig"
750
561
  ].includes(key)))
751
562
  : {};
752
- const isGeminiLiveTranslateModel = (modelId) => /^gemini-3\.5-live-translate(?:-preview)?$/i.test(modelId.trim());
753
563
  const isGemini31FlashLiveModel = (modelId) => /^gemini-3\.1-flash-live(?:-preview)?$/i.test(modelId.trim());
754
564
  const geminiRealtimeURL = (baseURL, apiKey, providerOptions) => {
755
565
  const override = providerOptions?.realtime_url;
@@ -832,8 +642,30 @@ const assertGeminiRealtimeTranslateConfig = (config, modelId) => {
832
642
  throw new UnsupportedFeatureError('Model "gemini/gemini-3.5-live-translate-preview" does not support realtime system instructions.');
833
643
  }
834
644
  };
645
+ const assertGeminiRealtimeTranscribeConfig = (config, modelId) => {
646
+ if (!isGeminiLiveTranscribeModel(modelId)) {
647
+ return;
648
+ }
649
+ if (config.mode && config.mode !== "transcription") {
650
+ throw new UnsupportedFeatureError('Model "gemini/gemini-3.5-transcribe-live" only supports realtime transcription mode.');
651
+ }
652
+ if (config.voice || config.outputAudioTranscription) {
653
+ throw new UnsupportedFeatureError('Model "gemini/gemini-3.5-transcribe-live" produces text transcripts and does not support audio output.');
654
+ }
655
+ if (config.translation) {
656
+ throw new UnsupportedFeatureError('Model "gemini/gemini-3.5-transcribe-live" does not support realtime translation.');
657
+ }
658
+ const tools = toToolSet(config.tools);
659
+ if (tools && Object.keys(tools).length > 0) {
660
+ throw new UnsupportedFeatureError('Model "gemini/gemini-3.5-transcribe-live" does not support realtime tools.');
661
+ }
662
+ if (config.reasoning || config.instructions) {
663
+ throw new UnsupportedFeatureError('Model "gemini/gemini-3.5-transcribe-live" does not support reasoning or system instructions.');
664
+ }
665
+ };
835
666
  const assertGeminiRealtimeConfig = (config, modelId) => {
836
667
  assertGeminiRealtimeTranslateConfig(config, modelId);
668
+ assertGeminiRealtimeTranscribeConfig(config, modelId);
837
669
  if (config.toolChoice !== undefined && !["auto", "none"].includes(String(config.toolChoice))) {
838
670
  throw new UnsupportedFeatureError("Gemini Live supports automatic tool selection or tool disabling, but not required or named tool choice.");
839
671
  }
@@ -851,7 +683,7 @@ const geminiRealtimeSetup = (config, modelId) => ({
851
683
  setup: {
852
684
  model: `models/${modelId}`,
853
685
  generationConfig: {
854
- responseModalities: ["AUDIO"],
686
+ responseModalities: [isGeminiLiveTranscribeModel(modelId) ? "TEXT" : "AUDIO"],
855
687
  ...(config.voice
856
688
  ? {
857
689
  speechConfig: {
@@ -866,9 +698,11 @@ const geminiRealtimeSetup = (config, modelId) => ({
866
698
  ...(mapRealtimeThinkingConfig(config) ? { thinkingConfig: mapRealtimeThinkingConfig(config) } : {})
867
699
  },
868
700
  ...(mapRealtimeTranslationConfig(config) ? { translationConfig: mapRealtimeTranslationConfig(config) } : {}),
869
- ...(mapRealtimeTranscriptionConfig(config.inputAudioTranscription ?? (config.inputTranscription ? true : undefined))
701
+ ...(mapRealtimeTranscriptionConfig(config.inputAudioTranscription ??
702
+ (config.inputTranscription || isGeminiLiveTranscribeModel(modelId) ? true : undefined))
870
703
  ? {
871
- inputAudioTranscription: mapRealtimeTranscriptionConfig(config.inputAudioTranscription ?? (config.inputTranscription ? true : undefined))
704
+ inputAudioTranscription: mapRealtimeTranscriptionConfig(config.inputAudioTranscription ??
705
+ (config.inputTranscription || isGeminiLiveTranscribeModel(modelId) ? true : undefined))
872
706
  }
873
707
  : {}),
874
708
  ...(mapRealtimeTranscriptionConfig(config.outputAudioTranscription)
@@ -1097,17 +931,23 @@ const createGeminiRealtimeEventParser = () => {
1097
931
  };
1098
932
  const isGemini3Model = (modelId) => /^gemini-3([.-]|$)/.test(modelId);
1099
933
  const isGemini3ProModel = (modelId) => /^gemini-3([.-].*)?pro([.-]|$)/.test(modelId);
1100
- const usesCurrentGeminiRequestRules = (modelId) => modelId === "gemini-3.6-flash" || modelId === "gemini-3.5-flash-lite";
934
+ const usesCurrentGeminiRequestRules = (modelId) => modelId === "gemini-3.8-flash" ||
935
+ modelId === "gemini-3.7-flash" ||
936
+ modelId === "gemini-3.6-flash" ||
937
+ modelId === "gemini-3.5-flash-lite";
1101
938
  const currentGeminiReasoningEfforts = [
1102
939
  "minimal",
1103
940
  "low",
1104
941
  "medium",
1105
942
  "high"
1106
943
  ];
944
+ const reasoningEffortsForModel = (modelId) => (modelId === "gemini-3.8-flash" || modelId === "gemini-3.7-flash")
945
+ ? ["low", "medium", "high"]
946
+ : currentGeminiReasoningEfforts;
1107
947
  const modelCapabilities = (modelId, baseCapabilities = capabilities) => usesCurrentGeminiRequestRules(modelId)
1108
948
  ? {
1109
949
  ...baseCapabilities,
1110
- reasoningEfforts: [...currentGeminiReasoningEfforts]
950
+ reasoningEfforts: [...reasoningEffortsForModel(modelId)]
1111
951
  }
1112
952
  : baseCapabilities;
1113
953
  const currentGeminiGenerationControlKeys = [
@@ -1144,7 +984,23 @@ const assertCurrentGeminiGenerateInput = (provider, modelId, input) => {
1144
984
  if (!usesCurrentGeminiRequestRules(modelId)) {
1145
985
  return;
1146
986
  }
987
+ if (input.reasoning?.effort !== undefined && !reasoningEffortsForModel(modelId).includes(input.reasoning.effort)) {
988
+ throw new UnsupportedFeatureError(`Provider "${provider}" does not support reasoning effort "${input.reasoning.effort}" for model "${modelId}".`);
989
+ }
1147
990
  const providerOptions = input.providerOptions;
991
+ const rawConfig = providerOptions?.generationConfig ?? providerOptions?.generation_config;
992
+ const config = rawConfig && typeof rawConfig === "object" ? rawConfig : {};
993
+ const rawThinking = config.thinkingConfig ?? config.thinking_config ?? providerOptions?.thinkingConfig ?? providerOptions?.thinking_config;
994
+ if (rawThinking && typeof rawThinking === "object") {
995
+ const thinking = rawThinking;
996
+ const effort = thinking.thinkingLevel ?? thinking.thinking_level;
997
+ if (effort !== undefined && !reasoningEffortsForModel(modelId).some((supported) => supported === String(effort).toLowerCase())) {
998
+ throw new UnsupportedFeatureError(`Provider "${provider}" does not support thinking level "${effort}" for model "${modelId}".`);
999
+ }
1000
+ if (thinking.thinkingBudget !== undefined || thinking.thinking_budget !== undefined) {
1001
+ throw new UnsupportedFeatureError(`Provider "${provider}" requires thinking levels instead of budgets for model "${modelId}".`);
1002
+ }
1003
+ }
1148
1004
  const unsupportedControl = firstUnsupportedGenerationControl(input.temperature === undefined ? undefined : { temperature: input.temperature }, providerOptions, providerOptions?.generationConfig, providerOptions?.generation_config);
1149
1005
  if (unsupportedControl) {
1150
1006
  throw new UnsupportedFeatureError(`Provider "${provider}" does not support generation control "${unsupportedControl}" for model "${modelId}". ` +
@@ -2194,15 +2050,84 @@ class GeminiTranscriptionModel {
2194
2050
  apiKey;
2195
2051
  baseURL;
2196
2052
  fetcher;
2053
+ allowUnsafeEndpoints;
2197
2054
  provider = "gemini";
2198
2055
  capabilities = transcriptionCapabilities;
2199
- constructor(modelId, apiKey, baseURL, fetcher) {
2056
+ constructor(modelId, apiKey, baseURL, fetcher, allowUnsafeEndpoints = false) {
2200
2057
  this.modelId = modelId;
2201
2058
  this.apiKey = apiKey;
2202
2059
  this.baseURL = baseURL;
2203
2060
  this.fetcher = fetcher;
2061
+ this.allowUnsafeEndpoints = allowUnsafeEndpoints;
2062
+ }
2063
+ isDedicatedTranscriptionModel() {
2064
+ return /^gemini-3\.5-transcribe$/i.test(this.modelId.trim());
2065
+ }
2066
+ async transcribeDedicated(input) {
2067
+ if (input.prompt) {
2068
+ throw new UnsupportedFeatureError('Model "gemini/gemini-3.5-transcribe" does not support a free-form transcription prompt. Use providerOptions.custom_vocabulary for terminology hints.');
2069
+ }
2070
+ const audioBytes = typeof input.audio.data === "string"
2071
+ ? Buffer.from(input.audio.data, "base64")
2072
+ : input.audio.data instanceof Uint8Array
2073
+ ? input.audio.data
2074
+ : new Uint8Array(input.audio.data);
2075
+ const requestOptions = {
2076
+ abortSignal: input.abortSignal,
2077
+ timeoutMs: input.timeoutMs,
2078
+ maxRetries: input.maxRetries,
2079
+ retryBackoffMs: input.retryBackoffMs
2080
+ };
2081
+ const files = new GeminiFilesClient(this.apiKey, this.baseURL, this.fetcher, this.allowUnsafeEndpoints);
2082
+ const uploaded = await files.upload({
2083
+ ...requestOptions,
2084
+ data: audioBytes,
2085
+ mediaType: input.audio.mediaType,
2086
+ displayName: input.audio.filename
2087
+ });
2088
+ if (!uploaded.name) {
2089
+ throw new ProviderHTTPError('Gemini Files API did not return a name for the transcription upload.', 500);
2090
+ }
2091
+ try {
2092
+ if (!uploaded.uri) {
2093
+ throw new ProviderHTTPError('Gemini Files API did not return a URI for the transcription upload.', 500);
2094
+ }
2095
+ const interaction = await new GeminiInteractionsClient(this.apiKey, this.baseURL, this.fetcher).create({
2096
+ ...requestOptions,
2097
+ modelId: this.modelId,
2098
+ input: [
2099
+ {
2100
+ type: "audio",
2101
+ uri: uploaded.uri,
2102
+ mime_type: uploaded.mimeType ?? input.audio.mediaType
2103
+ }
2104
+ ],
2105
+ generationConfig: {
2106
+ transcription_config: {
2107
+ ...(input.providerOptions ?? {}),
2108
+ ...(input.language ? { language_codes: [input.language] } : {})
2109
+ }
2110
+ },
2111
+ store: false
2112
+ });
2113
+ return {
2114
+ text: interaction.outputText ?? "",
2115
+ rawResponse: interaction.rawResponse
2116
+ };
2117
+ }
2118
+ finally {
2119
+ await files.delete({
2120
+ name: uploaded.name,
2121
+ timeoutMs: input.timeoutMs,
2122
+ maxRetries: input.maxRetries,
2123
+ retryBackoffMs: input.retryBackoffMs
2124
+ });
2125
+ }
2204
2126
  }
2205
2127
  async transcribe(input) {
2128
+ if (this.isDedicatedTranscriptionModel()) {
2129
+ return this.transcribeDedicated(input);
2130
+ }
2206
2131
  const { signal, cleanup } = withTimeoutSignal(input);
2207
2132
  try {
2208
2133
  const response = await withRetry(() => this.fetcher(`${this.baseURL}/models/${encodeGeminiPathSegment(this.modelId, "Gemini model ID")}:generateContent?key=${this.apiKey}`, {
@@ -2622,8 +2547,8 @@ class GeminiRealtimeModel {
2622
2547
  }
2623
2548
  ],
2624
2549
  buildMediaPayloads: (frame) => {
2625
- if (isGeminiLiveTranslateModel(this.modelId)) {
2626
- throw new UnsupportedFeatureError('Model "gemini/gemini-3.5-live-translate-preview" only supports audio input.');
2550
+ if (isGeminiLiveTranslateModel(this.modelId) || isGeminiLiveTranscribeModel(this.modelId)) {
2551
+ throw new UnsupportedFeatureError(`Model "gemini/${this.modelId}" only supports audio input.`);
2627
2552
  }
2628
2553
  return [
2629
2554
  {
@@ -2637,8 +2562,8 @@ class GeminiRealtimeModel {
2637
2562
  ];
2638
2563
  },
2639
2564
  buildTextPayloads: (text) => {
2640
- if (isGeminiLiveTranslateModel(this.modelId)) {
2641
- throw new UnsupportedFeatureError('Model "gemini/gemini-3.5-live-translate-preview" only supports audio input.');
2565
+ if (isGeminiLiveTranslateModel(this.modelId) || isGeminiLiveTranscribeModel(this.modelId)) {
2566
+ throw new UnsupportedFeatureError(`Model "gemini/${this.modelId}" only supports audio input.`);
2642
2567
  }
2643
2568
  return [
2644
2569
  {
@@ -2761,7 +2686,7 @@ export const createGemini = (options = {}) => {
2761
2686
  name: "gemini",
2762
2687
  languageModel: (modelId) => new GeminiLanguageModel(modelId, apiKey, baseURL, fetcher),
2763
2688
  embeddingModel: (modelId) => new GeminiEmbeddingModel(modelId, apiKey, baseURL, fetcher),
2764
- transcriptionModel: (modelId) => new GeminiTranscriptionModel(modelId, apiKey, baseURL, fetcher),
2689
+ transcriptionModel: (modelId) => new GeminiTranscriptionModel(modelId, apiKey, baseURL, fetcher, options.allowUnsafeEndpoints),
2765
2690
  speechModel: (modelId) => new GeminiSpeechModel(modelId, apiKey, baseURL, fetcher),
2766
2691
  imageGenerationModel: (modelId) => new GeminiImageGenerationModel(modelId, apiKey, baseURL, fetcher),
2767
2692
  videoGenerationModel: (modelId) => new GeminiVideoGenerationModel(modelId, apiKey, baseURL, fetcher),