@tanstack/ai-gemini 0.3.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,40 +22,39 @@ This will be enforced on the GenerateContentRequest.contents and GenerateContent
22
22
  safetySettings?: Array<SafetySetting>
23
23
  }
24
24
 
25
- export interface GeminiGenerationConfigOptions {
25
+ export interface GeminiCommonConfigOptions {
26
26
  /**
27
27
  * Configuration options for model generation and outputs.
28
28
  */
29
- generationConfig?: {
30
- /**
31
- * The set of character sequences (up to 5) that will stop output generation. If specified, the API will stop at the first appearance of a stop_sequence. The stop sequence will not be included as part of the response.
32
- */
33
- stopSequences?: Array<string>
34
- /**
29
+ /**
30
+ * The set of character sequences (up to 5) that will stop output generation. If specified, the API will stop at the first appearance of a stop_sequence. The stop sequence will not be included as part of the response.
31
+ */
32
+ stopSequences?: Array<string>
33
+ /**
35
34
  * The requested modalities of the response. Represents the set of modalities that the model can return, and should be expected in the response. This is an exact match to the modalities of the response.
36
35
 
37
36
  A model may have multiple combinations of supported modalities. If the requested modalities do not match any of the supported combinations, an error will be returned.
38
37
  */
39
- responseModalities?: Array<
40
- 'MODALITY_UNSPECIFIED' | 'TEXT' | 'IMAGE' | 'AUDIO'
41
- >
42
- /**
43
- * Number of generated responses to return. If unset, this will default to 1. Please note that this doesn't work for previous generation models (Gemini 1.0 family)
44
- */
45
- candidateCount?: number
46
- /**
38
+ responseModalities?: Array<
39
+ 'MODALITY_UNSPECIFIED' | 'TEXT' | 'IMAGE' | 'AUDIO'
40
+ >
41
+ /**
42
+ * Number of generated responses to return. If unset, this will default to 1. Please note that this doesn't work for previous generation models (Gemini 1.0 family)
43
+ */
44
+ candidateCount?: number
45
+ /**
47
46
  * The maximum number of tokens to consider when sampling.
48
47
 
49
48
  Gemini models use Top-p (nucleus) sampling or a combination of Top-k and nucleus sampling. Top-k sampling considers the set of topK most probable tokens. Models running with nucleus sampling don't allow topK setting.
50
49
 
51
50
  Note: The default value varies by Model and is specified by theModel.top_p attribute returned from the getModel function. An empty topK attribute indicates that the model doesn't apply top-k sampling and doesn't allow setting topK on requests.
52
51
  */
53
- topK?: number
54
- /**
55
- * Seed used in decoding. If not set, the request uses a randomly generated seed.
56
- */
57
- seed?: number
58
- /**
52
+ topK?: number
53
+ /**
54
+ * Seed used in decoding. If not set, the request uses a randomly generated seed.
55
+ */
56
+ seed?: number
57
+ /**
59
58
  * Presence penalty applied to the next token's logprobs if the token has already been seen in the response.
60
59
 
61
60
  This penalty is binary on/off and not dependant on the number of times the token is used (after the first). Use frequencyPenalty for a penalty that increases with each use.
@@ -64,107 +63,105 @@ A positive penalty will discourage the use of tokens that have already been used
64
63
 
65
64
  A negative penalty will encourage the use of tokens that have already been used in the response, decreasing the vocabulary.
66
65
  */
67
- presencePenalty?: number
68
- /**
66
+ presencePenalty?: number
67
+ /**
69
68
  * Frequency penalty applied to the next token's logprobs, multiplied by the number of times each token has been seen in the respponse so far.
70
69
 
71
70
  A positive penalty will discourage the use of tokens that have already been used, proportional to the number of times the token has been used: The more a token is used, the more difficult it is for the model to use that token again increasing the vocabulary of responses.
72
71
 
73
72
  Caution: A negative penalty will encourage the model to reuse tokens proportional to the number of times the token has been used. Small negative values will reduce the vocabulary of a response. Larger negative values will cause the model to start repeating a common token until it hits the maxOutputTokens limit.
74
73
  */
75
- frequencyPenalty?: number
76
- /**
77
- * If true, export the logprobs results in response.
78
- */
79
- responseLogprobs?: boolean
74
+ frequencyPenalty?: number
75
+ /**
76
+ * If true, export the logprobs results in response.
77
+ */
78
+ responseLogprobs?: boolean
80
79
 
81
- /**
82
- * Only valid if responseLogprobs=True. This sets the number of top logprobs to return at each decoding step in the Candidate.logprobs_result. The number must be in the range of [0, 20].
83
- */
84
- logprobs?: number
80
+ /**
81
+ * Only valid if responseLogprobs=True. This sets the number of top logprobs to return at each decoding step in the Candidate.logprobs_result. The number must be in the range of [0, 20].
82
+ */
83
+ logprobs?: number
85
84
 
86
- /**
87
- * Enables enhanced civic answers. It may not be available for all models.
88
- */
89
- enableEnhancedCivicAnswers?: boolean
85
+ /**
86
+ * Enables enhanced civic answers. It may not be available for all models.
87
+ */
88
+ enableEnhancedCivicAnswers?: boolean
90
89
 
91
- /**
92
- * The speech generation config.
93
- */
94
- speechConfig?: {
95
- voiceConfig: {
96
- prebuiltVoiceConfig: {
97
- voiceName: string
98
- }
90
+ /**
91
+ * The speech generation config.
92
+ */
93
+ speechConfig?: {
94
+ voiceConfig: {
95
+ prebuiltVoiceConfig: {
96
+ voiceName: string
99
97
  }
98
+ }
100
99
 
101
- multiSpeakerVoiceConfig?: {
102
- speakerVoiceConfigs?: Array<{
103
- speaker: string
104
- voiceConfig: {
105
- prebuiltVoiceConfig: {
106
- voiceName: string
107
- }
100
+ multiSpeakerVoiceConfig?: {
101
+ speakerVoiceConfigs?: Array<{
102
+ speaker: string
103
+ voiceConfig: {
104
+ prebuiltVoiceConfig: {
105
+ voiceName: string
108
106
  }
109
- }>
110
- }
111
- /**
107
+ }
108
+ }>
109
+ }
110
+ /**
112
111
  * Language code (in BCP 47 format, e.g. "en-US") for speech synthesis.
113
112
 
114
113
  Valid values are: de-DE, en-AU, en-GB, en-IN, en-US, es-US, fr-FR, hi-IN, pt-BR, ar-XA, es-ES, fr-CA, id-ID, it-IT, ja-JP, tr-TR, vi-VN, bn-IN, gu-IN, kn-IN, ml-IN, mr-IN, ta-IN, te-IN, nl-NL, ko-KR, cmn-CN, pl-PL, ru-RU, and th-TH.
115
114
  */
116
- languageCode?:
117
- | 'de-DE'
118
- | 'en-AU'
119
- | 'en-GB'
120
- | 'en-IN'
121
- | 'en-US'
122
- | 'es-US'
123
- | 'fr-FR'
124
- | 'hi-IN'
125
- | 'pt-BR'
126
- | 'ar-XA'
127
- | 'es-ES'
128
- | 'fr-CA'
129
- | 'id-ID'
130
- | 'it-IT'
131
- | 'ja-JP'
132
- | 'tr-TR'
133
- | 'vi-VN'
134
- | 'bn-IN'
135
- | 'gu-IN'
136
- | 'kn-IN'
137
- | 'ml-IN'
138
- | 'mr-IN'
139
- | 'ta-IN'
140
- | 'te-IN'
141
- | 'nl-NL'
142
- | 'ko-KR'
143
- | 'cmn-CN'
144
- | 'pl-PL'
145
- | 'ru-RU'
146
- | 'th-TH'
147
- }
148
- /**
149
- * Config for image generation. An error will be returned if this field is set for models that don't support these config options.
150
- */
151
- imageConfig?: {
152
- aspectRatio?:
153
- | '1:1'
154
- | '2:3'
155
- | '3:2'
156
- | '3:4'
157
- | '4:3'
158
- | '9:16'
159
- | '16:9'
160
- | '21:9'
161
- }
162
- /**
163
- * If specified, the media resolution specified will be used.
164
- */
165
- mediaResolution?: MediaResolution
166
- } & GeminiThinkingOptions &
167
- GeminiStructuredOutputOptions
115
+ languageCode?:
116
+ | 'de-DE'
117
+ | 'en-AU'
118
+ | 'en-GB'
119
+ | 'en-IN'
120
+ | 'en-US'
121
+ | 'es-US'
122
+ | 'fr-FR'
123
+ | 'hi-IN'
124
+ | 'pt-BR'
125
+ | 'ar-XA'
126
+ | 'es-ES'
127
+ | 'fr-CA'
128
+ | 'id-ID'
129
+ | 'it-IT'
130
+ | 'ja-JP'
131
+ | 'tr-TR'
132
+ | 'vi-VN'
133
+ | 'bn-IN'
134
+ | 'gu-IN'
135
+ | 'kn-IN'
136
+ | 'ml-IN'
137
+ | 'mr-IN'
138
+ | 'ta-IN'
139
+ | 'te-IN'
140
+ | 'nl-NL'
141
+ | 'ko-KR'
142
+ | 'cmn-CN'
143
+ | 'pl-PL'
144
+ | 'ru-RU'
145
+ | 'th-TH'
146
+ }
147
+ /**
148
+ * Config for image generation. An error will be returned if this field is set for models that don't support these config options.
149
+ */
150
+ imageConfig?: {
151
+ aspectRatio?:
152
+ | '1:1'
153
+ | '2:3'
154
+ | '3:2'
155
+ | '3:4'
156
+ | '4:3'
157
+ | '9:16'
158
+ | '16:9'
159
+ | '21:9'
160
+ }
161
+ /**
162
+ * If specified, the media resolution specified will be used.
163
+ */
164
+ mediaResolution?: MediaResolution
168
165
  }
169
166
 
170
167
  export interface GeminiCachedContentOptions {
@@ -232,15 +229,26 @@ export interface GeminiThinkingOptions {
232
229
  /**
233
230
  * The number of thoughts tokens that the model should generate.
234
231
  */
235
- thinkingBudget: number
232
+ thinkingBudget?: number
233
+ }
234
+ }
235
+
236
+ export interface GeminiThinkingAdvancedOptions {
237
+ /**
238
+ * Config for thinking features. An error will be returned if this field is set for models that don't support thinking.
239
+ */
240
+ thinkingConfig?: {
236
241
  /**
237
242
  * The level of thoughts tokens that the model should generate.
238
243
  */
239
- thinkingLevel?: ThinkingLevel
244
+ thinkingLevel?: keyof typeof ThinkingLevel
240
245
  }
241
246
  }
242
247
 
243
248
  export type ExternalTextProviderOptions = GeminiToolConfigOptions &
244
249
  GeminiSafetyOptions &
245
- GeminiGenerationConfigOptions &
246
- GeminiCachedContentOptions
250
+ GeminiCommonConfigOptions &
251
+ GeminiCachedContentOptions &
252
+ GeminiThinkingOptions &
253
+ GeminiThinkingAdvancedOptions &
254
+ GeminiStructuredOutputOptions