@gullabs/google 0.13.0 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/NOTICE +6 -0
- package/README.md +446 -31
- package/dist/index.cjs +2248 -632
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +680 -278
- package/dist/index.d.ts +680 -278
- package/dist/index.js +2246 -634
- package/dist/index.js.map +1 -1
- package/package.json +18 -10
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,267 @@
|
|
|
1
|
-
import { AuthMaterial, ProviderAdapter, ProviderPlugin, ModelRegistry, ModelDescriptor, PricingSource, ModelRates,
|
|
2
|
-
import { Content } from '@google/genai';
|
|
1
|
+
import { AuthMaterial, Logger, ProviderAdapter, LlmError, ProviderPlugin, ModelRegistry, ModelDescriptor, PricingSource, ModelRates, Scheduler, Message, JsonValue } from '@gullabs/core';
|
|
2
|
+
import { Content, Tool, ToolConfig } from '@google/genai';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* GoogleCacheStore — thin wrapper over the Gemini Context Cache API.
|
|
6
|
+
*
|
|
7
|
+
* Manages create / get-or-create / refresh / delete lifecycle for cached
|
|
8
|
+
* contents. In-memory cache entries are PROCESS-SCOPED; they are not shared
|
|
9
|
+
* across processes, workers, or restarts.
|
|
10
|
+
*
|
|
11
|
+
* Injectable client and `now` function keep tests free of network and clock.
|
|
12
|
+
*
|
|
13
|
+
* @module
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* A handle to a cached content resource in the Gemini Context Cache API.
|
|
18
|
+
*
|
|
19
|
+
* Pass `cacheName` as `providerOptions.google.cachedContent` in LlmRequest.
|
|
20
|
+
*/
|
|
21
|
+
interface GoogleCacheHandle {
|
|
22
|
+
/** Resource name, e.g. "cachedContents/xyz". */
|
|
23
|
+
cacheName: string;
|
|
24
|
+
/**
|
|
25
|
+
* Local view of when the cache expires (server-authoritative; computed from
|
|
26
|
+
* ttl or parsed from the server response `expireTime`).
|
|
27
|
+
*/
|
|
28
|
+
expiresAt: Date;
|
|
29
|
+
/** Caches are model-bound; never use a handle with a different model. */
|
|
30
|
+
model: string;
|
|
31
|
+
/**
|
|
32
|
+
* Tokens the cache holds, from `usageMetadata.totalTokenCount` of the create
|
|
33
|
+
* response. Cache storage is billed per token-hour (see Google's pricing page)
|
|
34
|
+
* and no usage record carries it, so a host that prices storage reads this.
|
|
35
|
+
* Absent when the response had no `usageMetadata`. Kept across a TTL refresh.
|
|
36
|
+
*/
|
|
37
|
+
totalTokenCount?: number;
|
|
38
|
+
/**
|
|
39
|
+
* The kinds of tool the cache holds, from the `tools` given to `create` (the
|
|
40
|
+
* keys of each `Tool`: `googleSearch`, `functionDeclarations`, ...); empty when
|
|
41
|
+
* it holds none. Pass it with the name as
|
|
42
|
+
* `providerOptions.google.cachedContent: { cacheName, toolKinds }` so a cached
|
|
43
|
+
* `googleSearch` is priced as Search: the request carries no tool, so nothing
|
|
44
|
+
* else says Search runs. Absent on a handle built by hand, where the content of
|
|
45
|
+
* the cache is unknown. Kept across a TTL refresh.
|
|
46
|
+
*/
|
|
47
|
+
toolKinds?: readonly string[];
|
|
48
|
+
}
|
|
49
|
+
/** Key used to look up or create an entry in the in-process cache map. */
|
|
50
|
+
interface CacheKey {
|
|
51
|
+
model: string;
|
|
52
|
+
/** Caller-chosen stable identifier, e.g. a hash of the cached content. */
|
|
53
|
+
stableKey: string;
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* Minimal structural interface for the Gemini Caches client surface we use.
|
|
57
|
+
* Satisfied by the real `ai.caches` object or a test fake.
|
|
58
|
+
*/
|
|
59
|
+
interface GeminiCachesClientLike {
|
|
60
|
+
create(params: {
|
|
61
|
+
model: string;
|
|
62
|
+
config: {
|
|
63
|
+
contents?: Content[];
|
|
64
|
+
systemInstruction?: Content | string;
|
|
65
|
+
tools?: Tool[];
|
|
66
|
+
toolConfig?: ToolConfig;
|
|
67
|
+
ttl?: string;
|
|
68
|
+
displayName?: string;
|
|
69
|
+
};
|
|
70
|
+
}): Promise<{
|
|
71
|
+
name?: string;
|
|
72
|
+
model?: string;
|
|
73
|
+
expireTime?: string;
|
|
74
|
+
usageMetadata?: {
|
|
75
|
+
totalTokenCount?: number;
|
|
76
|
+
};
|
|
77
|
+
}>;
|
|
78
|
+
update(params: {
|
|
79
|
+
name: string;
|
|
80
|
+
config: {
|
|
81
|
+
ttl?: string;
|
|
82
|
+
expireTime?: string;
|
|
83
|
+
};
|
|
84
|
+
}): Promise<{
|
|
85
|
+
name?: string;
|
|
86
|
+
expireTime?: string;
|
|
87
|
+
}>;
|
|
88
|
+
delete(params: {
|
|
89
|
+
name: string;
|
|
90
|
+
}): Promise<unknown>;
|
|
91
|
+
}
|
|
92
|
+
interface GoogleCacheStoreOptions {
|
|
93
|
+
auth: AuthMaterial;
|
|
94
|
+
/** Injectable client for tests; skips SDK import when provided. */
|
|
95
|
+
client?: GeminiCachesClientLike;
|
|
96
|
+
/**
|
|
97
|
+
* When true, concurrent `getOrCreate` calls for the same key share one
|
|
98
|
+
* in-flight create. Default false.
|
|
99
|
+
*/
|
|
100
|
+
coalesce?: boolean;
|
|
101
|
+
/**
|
|
102
|
+
* Subtracted from the local expiry so we stop using a cache slightly before
|
|
103
|
+
* the server evicts it. Default 30 s.
|
|
104
|
+
*/
|
|
105
|
+
expirySkewSeconds?: number;
|
|
106
|
+
/** Called on delete failures instead of rethrowing. */
|
|
107
|
+
onDeleteError?: (cacheName: string, err: unknown) => void;
|
|
108
|
+
/** Optional structured logger. When provided, routes delete failures to logger.error. */
|
|
109
|
+
logger?: Logger;
|
|
110
|
+
/** Injectable clock for deterministic tests. Default: `Date.now`. */
|
|
111
|
+
now?: () => number;
|
|
112
|
+
/**
|
|
113
|
+
* Opt-in pre-flight token-count gate applied before every cache create
|
|
114
|
+
* (both `create()` directly and the `getOrCreate()` path it delegates to,
|
|
115
|
+
* including the coalesced path — enforced once, inside `create()`, so
|
|
116
|
+
* there is no separate "in-flight" gap to close).
|
|
117
|
+
*/
|
|
118
|
+
preflight?: {
|
|
119
|
+
/** Minimum token count required before a cache create is allowed to proceed. */
|
|
120
|
+
minTokens: number;
|
|
121
|
+
/**
|
|
122
|
+
* Counts tokens for the exact token-bearing payload of the impending
|
|
123
|
+
* create — `model` + `contents` + `systemInstruction` + `tools` only.
|
|
124
|
+
* `toolConfig`, `ttl` and `displayName` are excluded: they carry no tokens
|
|
125
|
+
* and are irrelevant to the pre-flight check.
|
|
126
|
+
*
|
|
127
|
+
* This callback receives genai-native `Content[]`/`Content|string` — it
|
|
128
|
+
* does NOT receive the library's `Message[]` shape and there is no
|
|
129
|
+
* automatic conversion between the two (explicit seam, by design). Hosts
|
|
130
|
+
* using genai-native content directly can wire this to a raw
|
|
131
|
+
* `client.models.countTokens` call; hosts building from the library's
|
|
132
|
+
* `Message[]` should call `@gullabs/core`'s `client.countTokens` rather
|
|
133
|
+
* than expecting this callback to convert for them.
|
|
134
|
+
*/
|
|
135
|
+
countTokens: (payload: {
|
|
136
|
+
model: string;
|
|
137
|
+
contents?: Content[];
|
|
138
|
+
systemInstruction?: Content | string;
|
|
139
|
+
tools?: Tool[];
|
|
140
|
+
}) => Promise<number>;
|
|
141
|
+
};
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* Process-scoped helper for the Gemini Context Cache API.
|
|
145
|
+
*
|
|
146
|
+
* NOTE: `getOrCreate` reuse is PROCESS-SCOPED only. Entries survive only for
|
|
147
|
+
* the lifetime of this `GoogleCacheStore` instance. Across restarts, new
|
|
148
|
+
* caches will be created (and old ones will be server-evicted after their TTL).
|
|
149
|
+
*
|
|
150
|
+
* **Auth snapshot note:** this store captures the `AuthMaterial` at construction
|
|
151
|
+
* time and memoizes a single SDK client from it (`clientPromise`). This is
|
|
152
|
+
* correct and sufficient for static API keys. If refreshable credentials
|
|
153
|
+
* (short-lived OAuth/STS tokens) are added in the future, this memoization is
|
|
154
|
+
* the seam that will need rework: the cached client would hold stale credentials
|
|
155
|
+
* for the lifetime of a long-lived store instance. At that point, the store
|
|
156
|
+
* will need to either rebuild the client on each operation or accept a
|
|
157
|
+
* credential-resolver callback rather than a plain `AuthMaterial` value.
|
|
158
|
+
* See ADR-020 in DECISIONS.md.
|
|
159
|
+
*/
|
|
160
|
+
declare class GoogleCacheStore {
|
|
161
|
+
private readonly auth;
|
|
162
|
+
private readonly clientOverride;
|
|
163
|
+
private readonly coalesce;
|
|
164
|
+
private readonly skewMs;
|
|
165
|
+
private readonly onDeleteError;
|
|
166
|
+
private readonly logger;
|
|
167
|
+
private readonly now;
|
|
168
|
+
private readonly preflight;
|
|
169
|
+
/** Memoised client promise — built at most once per store instance. */
|
|
170
|
+
private clientPromise;
|
|
171
|
+
/** In-process cache of live handles, keyed by `${model}:${stableKey}`. */
|
|
172
|
+
private readonly entries;
|
|
173
|
+
/** In-flight create promises when coalescing is enabled. */
|
|
174
|
+
private readonly inflight;
|
|
175
|
+
constructor(opts: GoogleCacheStoreOptions);
|
|
176
|
+
private getClient;
|
|
177
|
+
private isLive;
|
|
178
|
+
/**
|
|
179
|
+
* Create a new cached content resource.
|
|
180
|
+
*
|
|
181
|
+
* The `expiresAt` on the returned handle is computed from the server's
|
|
182
|
+
* `expireTime` when available, with a local-clock fallback of
|
|
183
|
+
* `now + ttlSeconds * 1000`.
|
|
184
|
+
*/
|
|
185
|
+
create(input: {
|
|
186
|
+
model: string;
|
|
187
|
+
ttlSeconds: number;
|
|
188
|
+
contents?: Content[];
|
|
189
|
+
systemInstruction?: Content | string;
|
|
190
|
+
/**
|
|
191
|
+
* Tool declarations to store in the cache. Gemini rejects a request that
|
|
192
|
+
* sends `tools` or `toolConfig` together with `cachedContent`, so a cache
|
|
193
|
+
* used by a tool-calling call must hold them (the adapter rejects the
|
|
194
|
+
* combination before dispatch).
|
|
195
|
+
*/
|
|
196
|
+
tools?: Tool[];
|
|
197
|
+
toolConfig?: ToolConfig;
|
|
198
|
+
displayName?: string;
|
|
199
|
+
}): Promise<GoogleCacheHandle>;
|
|
200
|
+
/**
|
|
201
|
+
* Return a live cached-content handle, creating one if none exists or the
|
|
202
|
+
* cached entry has expired (accounting for skew).
|
|
203
|
+
*
|
|
204
|
+
* Reuse is PROCESS-SCOPED only — this store instance's in-memory map.
|
|
205
|
+
*
|
|
206
|
+
* When `coalesce` is enabled, concurrent calls for the same key share one
|
|
207
|
+
* in-flight create.
|
|
208
|
+
*/
|
|
209
|
+
getOrCreate(key: CacheKey, factory: () => Promise<{
|
|
210
|
+
ttlSeconds: number;
|
|
211
|
+
contents?: Content[];
|
|
212
|
+
systemInstruction?: Content | string;
|
|
213
|
+
tools?: Tool[];
|
|
214
|
+
toolConfig?: ToolConfig;
|
|
215
|
+
}>): Promise<GoogleCacheHandle>;
|
|
216
|
+
/**
|
|
217
|
+
* Extend the cache TTL if it will expire within `thresholdSeconds`.
|
|
218
|
+
*
|
|
219
|
+
* Fail-open: if the update call throws, the original handle is returned
|
|
220
|
+
* unchanged. This method NEVER throws.
|
|
221
|
+
*
|
|
222
|
+
* @param handle - Handle to potentially refresh.
|
|
223
|
+
* @param opts.thresholdSeconds - Extend if expiry is within this many seconds.
|
|
224
|
+
* Default 300.
|
|
225
|
+
* @param opts.extensionSeconds - New TTL to set. Default: original TTL from
|
|
226
|
+
* entries map, or 3600 s when the handle was not created via `getOrCreate`.
|
|
227
|
+
*/
|
|
228
|
+
refreshIfExpiringSoon(handle: GoogleCacheHandle, opts?: {
|
|
229
|
+
thresholdSeconds?: number;
|
|
230
|
+
extensionSeconds?: number;
|
|
231
|
+
}): Promise<GoogleCacheHandle>;
|
|
232
|
+
/**
|
|
233
|
+
* Delete a cached content resource. Idempotent: a cache that is already gone
|
|
234
|
+
* (HTTP 404, or the 403 `CachedContent not found` Google sends for an expired
|
|
235
|
+
* one) is success, not an error.
|
|
236
|
+
*
|
|
237
|
+
* Any other error is forwarded to `onDeleteError` and NOT rethrown.
|
|
238
|
+
* The handle is removed from the in-process entries map regardless.
|
|
239
|
+
*/
|
|
240
|
+
delete(handle: GoogleCacheHandle): Promise<void>;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
/**
|
|
244
|
+
* The `safetySettings` categories and thresholds Gemini documents.
|
|
245
|
+
*
|
|
246
|
+
* Sources, read 2026-10-03:
|
|
247
|
+
*
|
|
248
|
+
* - Categories: ai.google.dev/api/generate-content (`HarmCategory`: hate speech,
|
|
249
|
+
* sexually explicit, dangerous content, harassment, civic integrity,
|
|
250
|
+
* jailbreak) and ai.google.dev/gemini-api/docs/safety-settings (page dated
|
|
251
|
+
* 2026-09-17, the adjustable filters). The SDK's `HARM_CATEGORY_IMAGE_*`
|
|
252
|
+
* values are marked "not supported in Gemini API" and are not admitted.
|
|
253
|
+
* - Thresholds: ai.google.dev/gemini-api/docs/safety-settings (page dated
|
|
254
|
+
* 2026-09-17), cross-checked against the SDK's `HarmBlockThreshold` enum.
|
|
255
|
+
*
|
|
256
|
+
* The adapter and every model's config schema read these lists, so a typo is
|
|
257
|
+
* rejected before dispatch instead of costing a round trip.
|
|
258
|
+
*
|
|
259
|
+
* @module
|
|
260
|
+
*/
|
|
261
|
+
declare const GOOGLE_SAFETY_CATEGORIES: readonly ["HARM_CATEGORY_HARASSMENT", "HARM_CATEGORY_HATE_SPEECH", "HARM_CATEGORY_SEXUALLY_EXPLICIT", "HARM_CATEGORY_DANGEROUS_CONTENT", "HARM_CATEGORY_CIVIC_INTEGRITY", "HARM_CATEGORY_JAILBREAK"];
|
|
262
|
+
declare const GOOGLE_SAFETY_THRESHOLDS: readonly ["HARM_BLOCK_THRESHOLD_UNSPECIFIED", "BLOCK_LOW_AND_ABOVE", "BLOCK_MEDIUM_AND_ABOVE", "BLOCK_ONLY_HIGH", "BLOCK_NONE", "OFF"];
|
|
263
|
+
type GoogleSafetyCategory = (typeof GOOGLE_SAFETY_CATEGORIES)[number];
|
|
264
|
+
type GoogleSafetyThreshold = (typeof GOOGLE_SAFETY_THRESHOLDS)[number];
|
|
3
265
|
|
|
4
266
|
/**
|
|
5
267
|
* Google-specific provider options for `@gullabs/google`.
|
|
@@ -12,16 +274,34 @@ import { Content } from '@google/genai';
|
|
|
12
274
|
*
|
|
13
275
|
* @module
|
|
14
276
|
*/
|
|
277
|
+
|
|
15
278
|
type GoogleSafetySetting = {
|
|
16
|
-
|
|
17
|
-
|
|
279
|
+
/** A documented `HarmCategory`; see `safety-settings.ts` for the source. */
|
|
280
|
+
category: GoogleSafetyCategory;
|
|
281
|
+
/** A documented `HarmBlockThreshold`. */
|
|
282
|
+
threshold: GoogleSafetyThreshold;
|
|
18
283
|
};
|
|
19
284
|
type GoogleSearchTool = {
|
|
20
285
|
googleSearch: Record<string, never>;
|
|
21
286
|
};
|
|
287
|
+
/**
|
|
288
|
+
* A cached-content reference: the resource name, or `{ cacheName, toolKinds }`
|
|
289
|
+
* taken from a `GoogleCacheHandle` (`handle.toolKinds` records which tools the
|
|
290
|
+
* cache holds). Pass those two fields, not the whole handle: the other fields
|
|
291
|
+
* are typed `never` here and the schema is strict. The request sent to Google
|
|
292
|
+
* carries only the name. A bare name says nothing about what the cache holds,
|
|
293
|
+
* so a Search fee in the response is priced from the observed queries and the
|
|
294
|
+
* cost is `estimated`; a handle whose `toolKinds` lists `googleSearch` marks the
|
|
295
|
+
* call as a Search call up front.
|
|
296
|
+
*/
|
|
297
|
+
type GoogleCachedContentRef = string | (Pick<GoogleCacheHandle, 'cacheName' | 'toolKinds'> & {
|
|
298
|
+
expiresAt?: never;
|
|
299
|
+
model?: never;
|
|
300
|
+
totalTokenCount?: never;
|
|
301
|
+
});
|
|
22
302
|
type GoogleProviderOptions = {
|
|
23
|
-
/** Google cached content resource name. */
|
|
24
|
-
cachedContent?:
|
|
303
|
+
/** Google cached content: the resource name, or a handle that records the cache's tool kinds. Explicit-caching models only. */
|
|
304
|
+
cachedContent?: GoogleCachedContentRef;
|
|
25
305
|
/** Allowlisted Google safety settings. */
|
|
26
306
|
safetySettings?: GoogleSafetySetting[];
|
|
27
307
|
/** Exact Google tool declarations admitted by the selected model schema. */
|
|
@@ -33,6 +313,32 @@ type GoogleProviderOptions = {
|
|
|
33
313
|
};
|
|
34
314
|
/** Allow provider fallback from flex when flex was explicitly selected. */
|
|
35
315
|
flexFallback?: boolean;
|
|
316
|
+
/**
|
|
317
|
+
* Admit `googleSearch` together with `output.jsonSchema` on a model that does
|
|
318
|
+
* not admit the pair by default. Opting in turns on {@link requireGrounding}
|
|
319
|
+
* unless it is set to `false`, because Search can be skipped silently when a
|
|
320
|
+
* response schema is attached. Requires both Search and a schema; Search is
|
|
321
|
+
* `googleSearch` in `tools` or a `cachedContent` handle whose `toolKinds` lists
|
|
322
|
+
* it. A schema call on such a model with a handle that lists `googleSearch`
|
|
323
|
+
* needs this flag, like an inline `googleSearch`; a bare cache name is not
|
|
324
|
+
* blocked, and a response of a schema call that reports search queries without
|
|
325
|
+
* a declared Search carries a warning.
|
|
326
|
+
*/
|
|
327
|
+
allowSchemaWithSearch?: boolean;
|
|
328
|
+
/**
|
|
329
|
+
* Fail the call unless the response proves Search ran: `groundingMetadata`
|
|
330
|
+
* present with at least one `webSearchQueries` entry. The check judges only a
|
|
331
|
+
* candidate that finished normally (`STOP`, or no finish reason). Without the
|
|
332
|
+
* proof it throws a `server` error with reason `grounding_missing` and the
|
|
333
|
+
* attempt's usage is recorded; the error is retryable when no response schema
|
|
334
|
+
* is attached and not retryable when one is (the same request keeps missing).
|
|
335
|
+
* A `MAX_TOKENS` or other abnormal finish with no proof is not judged: it
|
|
336
|
+
* returns, with its own finish reason (`length`), and a filtered candidate
|
|
337
|
+
* throws its `content_filter` error. Requires Search (`googleSearch` in `tools`
|
|
338
|
+
* or a `cachedContent` handle whose `toolKinds` lists it). Defaults to
|
|
339
|
+
* `true` when {@link allowSchemaWithSearch} is `true`, else `false`.
|
|
340
|
+
*/
|
|
341
|
+
requireGrounding?: boolean;
|
|
36
342
|
};
|
|
37
343
|
declare module '@gullabs/core' {
|
|
38
344
|
interface ProviderOptionsMap {
|
|
@@ -102,6 +408,11 @@ interface GeminiPartShape {
|
|
|
102
408
|
name?: string;
|
|
103
409
|
args?: unknown;
|
|
104
410
|
};
|
|
411
|
+
/**
|
|
412
|
+
* Opaque signature Gemini 3.x attaches to the first function call of a turn
|
|
413
|
+
* (and sometimes to text parts). Real field: `Part.thoughtSignature`.
|
|
414
|
+
*/
|
|
415
|
+
thoughtSignature?: string;
|
|
105
416
|
}
|
|
106
417
|
/** A candidate returned by Gemini generateContent. */
|
|
107
418
|
interface GeminiCandidateShape {
|
|
@@ -114,6 +425,14 @@ interface GeminiCandidateShape {
|
|
|
114
425
|
* "RECITATION", "BLOCKLIST", "PROHIBITED_CONTENT", etc.
|
|
115
426
|
*/
|
|
116
427
|
finishReason?: string;
|
|
428
|
+
/** Human-readable detail Google sends with some finish reasons. */
|
|
429
|
+
finishMessage?: string;
|
|
430
|
+
/** Per-category safety ratings of the candidate. */
|
|
431
|
+
safetyRatings?: unknown[];
|
|
432
|
+
/** Source-attribution metadata for recited content. */
|
|
433
|
+
citationMetadata?: unknown;
|
|
434
|
+
/** Retrieval status of each URL the model was asked to read. */
|
|
435
|
+
urlContextMetadata?: unknown;
|
|
117
436
|
/**
|
|
118
437
|
* Grounding metadata returned when Google Search grounding is active.
|
|
119
438
|
* Real SDK type: GroundingMetadata. Kept as `unknown` to avoid a hard
|
|
@@ -133,9 +452,23 @@ interface GeminiUsageMetadataShape {
|
|
|
133
452
|
candidatesTokenCount?: number;
|
|
134
453
|
cachedContentTokenCount?: number;
|
|
135
454
|
thoughtsTokenCount?: number;
|
|
455
|
+
/** Tokens of tool results fed back to the model (Search results on Gemini 2.5). */
|
|
456
|
+
toolUsePromptTokenCount?: number;
|
|
136
457
|
totalTokenCount?: number;
|
|
137
458
|
/** Provider-echoed actual tier; can differ from the requested tier. */
|
|
138
459
|
serviceTier?: string;
|
|
460
|
+
/**
|
|
461
|
+
* Prompt tokens per modality (`TEXT`, `IMAGE`, `VIDEO`, `AUDIO`, `DOCUMENT`);
|
|
462
|
+
* the counts sum to `promptTokenCount` and include the cached part.
|
|
463
|
+
*/
|
|
464
|
+
promptTokensDetails?: GeminiModalityTokenCount[];
|
|
465
|
+
/** The cached part of the prompt per modality, when a cache was used. */
|
|
466
|
+
cacheTokensDetails?: GeminiModalityTokenCount[];
|
|
467
|
+
}
|
|
468
|
+
/** One entry of `usageMetadata.promptTokensDetails` / `cacheTokensDetails`. */
|
|
469
|
+
interface GeminiModalityTokenCount {
|
|
470
|
+
modality?: string;
|
|
471
|
+
tokenCount?: number;
|
|
139
472
|
}
|
|
140
473
|
/**
|
|
141
474
|
* Structural equivalent of @google/genai's GenerateContentResponse.
|
|
@@ -172,6 +505,8 @@ interface GeminiPartMediaResolution {
|
|
|
172
505
|
/** A text part in a content object we construct. */
|
|
173
506
|
interface GeminiTextContentPart {
|
|
174
507
|
text: string;
|
|
508
|
+
/** Replayed signature for this part (real field: `Part.thoughtSignature`). */
|
|
509
|
+
thoughtSignature?: string;
|
|
175
510
|
}
|
|
176
511
|
/**
|
|
177
512
|
* An inline binary media part in a content object we construct.
|
|
@@ -212,6 +547,8 @@ interface GeminiFunctionCallPart {
|
|
|
212
547
|
name: string;
|
|
213
548
|
args?: unknown;
|
|
214
549
|
};
|
|
550
|
+
/** Replayed signature for this call (real field: `Part.thoughtSignature`). */
|
|
551
|
+
thoughtSignature?: string;
|
|
215
552
|
}
|
|
216
553
|
interface GeminiFunctionResponsePart {
|
|
217
554
|
functionResponse: {
|
|
@@ -226,20 +563,6 @@ interface GeminiContent {
|
|
|
226
563
|
role: string;
|
|
227
564
|
parts: GeminiContentPart[];
|
|
228
565
|
}
|
|
229
|
-
/**
|
|
230
|
-
* Schema shape we pass as responseSchema.
|
|
231
|
-
* Structurally compatible with @google/genai Schema.
|
|
232
|
-
*/
|
|
233
|
-
interface GeminiSchema {
|
|
234
|
-
type?: string;
|
|
235
|
-
description?: string;
|
|
236
|
-
properties?: Record<string, GeminiSchema>;
|
|
237
|
-
required?: string[];
|
|
238
|
-
items?: GeminiSchema;
|
|
239
|
-
enum?: string[];
|
|
240
|
-
nullable?: boolean;
|
|
241
|
-
format?: string;
|
|
242
|
-
}
|
|
243
566
|
/**
|
|
244
567
|
* Thinking configuration.
|
|
245
568
|
* Real type: ThinkingConfig in @google/genai.
|
|
@@ -270,7 +593,12 @@ interface GeminiGenerateConfig {
|
|
|
270
593
|
maxOutputTokens?: number;
|
|
271
594
|
stopSequences?: string[];
|
|
272
595
|
responseMimeType?: string;
|
|
273
|
-
|
|
596
|
+
/**
|
|
597
|
+
* Standard JSON Schema for the response, sent verbatim (real field:
|
|
598
|
+
* `GenerateContentConfig.responseJsonSchema`). Never `responseSchema`, the
|
|
599
|
+
* OpenAPI-dialect field (ADR-034).
|
|
600
|
+
*/
|
|
601
|
+
responseJsonSchema?: unknown;
|
|
274
602
|
thinkingConfig?: GeminiThinkingConfig;
|
|
275
603
|
/** Real type: ServiceTier enum. Values: "flex" | "standard". */
|
|
276
604
|
serviceTier?: string;
|
|
@@ -281,22 +609,18 @@ interface GeminiGenerateConfig {
|
|
|
281
609
|
* We use this to set a transport-level timeout that is >= the AbortSignal
|
|
282
610
|
* deadline so the SDK fetch does not preempt the abort.
|
|
283
611
|
*
|
|
284
|
-
* Real field: GenerateContentConfig.httpOptions.timeout (milliseconds).
|
|
285
|
-
*
|
|
286
|
-
* (No current use of custom headers here: Vertex AI auth — and the
|
|
287
|
-
* Vertex flex-routing header injection this field once supported — was
|
|
288
|
-
* removed from this library; see ADR-019 in DECISIONS.md. Only
|
|
289
|
-
* API-key auth is supported below.)
|
|
612
|
+
* Real field: GenerateContentConfig.httpOptions.timeout (milliseconds). No
|
|
613
|
+
* other `httpOptions` field is admitted.
|
|
290
614
|
*/
|
|
291
615
|
httpOptions?: {
|
|
292
616
|
timeout?: number;
|
|
293
|
-
headers?: Record<string, string>;
|
|
294
617
|
};
|
|
295
618
|
tools?: Array<{
|
|
296
619
|
functionDeclarations?: Array<{
|
|
297
620
|
name: string;
|
|
298
621
|
description: string;
|
|
299
|
-
|
|
622
|
+
/** Standard JSON Schema, verbatim (real field: `parametersJsonSchema`). */
|
|
623
|
+
parametersJsonSchema?: unknown;
|
|
300
624
|
}>;
|
|
301
625
|
googleSearch?: Record<string, never>;
|
|
302
626
|
}>;
|
|
@@ -317,23 +641,28 @@ interface GeminiGenerateParams {
|
|
|
317
641
|
config?: GeminiGenerateConfig;
|
|
318
642
|
}
|
|
319
643
|
/**
|
|
320
|
-
* Parameters for
|
|
321
|
-
*
|
|
644
|
+
* Parameters for counting tokens.
|
|
645
|
+
*
|
|
646
|
+
* With only `model` and `contents` the call is the SDK's `models.countTokens`.
|
|
647
|
+
* With `systemInstruction` or `tools` the Developer API's SDK method cannot
|
|
648
|
+
* carry them (it throws), so `buildGoogleClient` sends the REST `countTokens`
|
|
649
|
+
* with a full `generateContentRequest` instead; the two forms are mutually
|
|
650
|
+
* exclusive on the wire, so `contents` then travels inside the request.
|
|
322
651
|
*/
|
|
323
652
|
interface GeminiCountTokensParams {
|
|
324
653
|
model: string;
|
|
325
654
|
contents: GeminiContent[];
|
|
655
|
+
systemInstruction?: {
|
|
656
|
+
parts: GeminiContentPart[];
|
|
657
|
+
};
|
|
658
|
+
tools?: NonNullable<GeminiGenerateConfig['tools']>;
|
|
326
659
|
config?: {
|
|
327
|
-
systemInstruction?: {
|
|
328
|
-
parts: GeminiContentPart[];
|
|
329
|
-
};
|
|
330
660
|
/**
|
|
331
661
|
* Real field: CountTokensConfig.abortSignal. countTokens has no
|
|
332
|
-
* tier-timeout dance (no flex/standard default ceilings)
|
|
662
|
+
* tier-timeout dance (no flex/standard default ceilings): `ctx.signal`
|
|
333
663
|
* is forwarded here directly, unlike `run()`'s combined timer signal.
|
|
334
664
|
*/
|
|
335
665
|
abortSignal?: AbortSignal;
|
|
336
|
-
tools?: GeminiGenerateConfig['tools'];
|
|
337
666
|
};
|
|
338
667
|
}
|
|
339
668
|
/**
|
|
@@ -385,14 +714,6 @@ interface GeminiAdapterOptions {
|
|
|
385
714
|
* as a typed `LlmError`.
|
|
386
715
|
*/
|
|
387
716
|
client?: GeminiClientLike;
|
|
388
|
-
/**
|
|
389
|
-
* @internal Testing-only.
|
|
390
|
-
*
|
|
391
|
-
* Override the default `buildGoogleClient` factory. Allows unit tests to
|
|
392
|
-
* simulate construction failures (e.g. bad credentials) without importing
|
|
393
|
-
* the real `@google/genai` SDK. Never set this in production code.
|
|
394
|
-
*/
|
|
395
|
-
_clientFactory?: (auth: AuthMaterial) => GeminiClientLike | Promise<GeminiClientLike>;
|
|
396
717
|
}
|
|
397
718
|
/**
|
|
398
719
|
* Create a Gemini provider adapter.
|
|
@@ -401,6 +722,59 @@ interface GeminiAdapterOptions {
|
|
|
401
722
|
*/
|
|
402
723
|
declare function geminiAdapter(opts?: GeminiAdapterOptions): ProviderAdapter;
|
|
403
724
|
|
|
725
|
+
/**
|
|
726
|
+
* classifyGoogleError — reclassify a raw thrown value into a typed
|
|
727
|
+
* {@link LlmError} for @gullabs/google.
|
|
728
|
+
*
|
|
729
|
+
* Thin wrapper around `@gullabs/core`'s `classifyError`, which already routes
|
|
730
|
+
* by HTTP status first, treats a transport failure (`fetch failed`, an errno on
|
|
731
|
+
* the cause chain) as a retryable `server` error and maps 404/413 to
|
|
732
|
+
* `bad_request`. This module adds the overlays Gemini's structured error body
|
|
733
|
+
* supports (ADR-028 style: only the parsed body, never free text):
|
|
734
|
+
*
|
|
735
|
+
* - `RetryInfo.retryDelay` becomes `retryAfterMs` (the SDK's `ApiError` keeps no
|
|
736
|
+
* headers, so the body is the only place Google puts the delay).
|
|
737
|
+
* - A `QuotaFailure` whose quota id contains `PerDay` is a daily quota: it
|
|
738
|
+
* cannot recover before the next day, so it is `rate_limited` with
|
|
739
|
+
* `retryable: false` and `reason: 'daily_quota'`.
|
|
740
|
+
* - `ErrorInfo.reason` `API_KEY_INVALID` / `API_KEY_EXPIRED` is `invalid_auth`.
|
|
741
|
+
* Google sends the invalid-key case as HTTP 400, which would otherwise read
|
|
742
|
+
* as a caller bug (live capture, probe P6, 2026-10-03). `API_KEY_EXPIRED` is
|
|
743
|
+
* mapped from Google's documented reason set; an expired key could not be
|
|
744
|
+
* produced to capture it.
|
|
745
|
+
* - A 403 whose body message is `CachedContent not found (or permission
|
|
746
|
+
* denied)` is a stale `cachedContent` reference: `bad_request` with
|
|
747
|
+
* `reason: 'cache_not_found'` (probe P6). Google sends no structured reason
|
|
748
|
+
* for it, so the body's `message` field is the only signal, and a genuine
|
|
749
|
+
* permission failure cannot be told apart.
|
|
750
|
+
*
|
|
751
|
+
* @module
|
|
752
|
+
*/
|
|
753
|
+
|
|
754
|
+
/** Optional extra fields threaded onto the returned {@link LlmError}. */
|
|
755
|
+
interface ClassifyGoogleErrorExtra {
|
|
756
|
+
/** Service tier actually attempted by the provider when known. */
|
|
757
|
+
servedServiceTier?: string;
|
|
758
|
+
/**
|
|
759
|
+
* Set by the adapter when its own client-side ceiling, or the SDK's transport
|
|
760
|
+
* timer, ended the call (the caller and the engine's deadline had not
|
|
761
|
+
* aborted): what happened, in words. The error is then a `timeout` that is not
|
|
762
|
+
* retryable, `reason: 'transport_timeout'`.
|
|
763
|
+
*/
|
|
764
|
+
transportTimeout?: string;
|
|
765
|
+
}
|
|
766
|
+
/**
|
|
767
|
+
* Classify a raw error thrown from a `@google/genai` client call into a
|
|
768
|
+
* typed {@link LlmError} always tagged `provider: 'google'`.
|
|
769
|
+
*
|
|
770
|
+
* Delegates to `@gullabs/core`'s `classifyError` (an already-classified
|
|
771
|
+
* `LlmError` passes through unchanged), applies the structured-body overlays
|
|
772
|
+
* described in the module header, and rebuilds the result with
|
|
773
|
+
* `provider: 'google'` forced on, so every error surfaced by this adapter is
|
|
774
|
+
* tagged even one injected pre-classified.
|
|
775
|
+
*/
|
|
776
|
+
declare function classifyGoogleError(rawErr: unknown, extra?: ClassifyGoogleErrorExtra): LlmError;
|
|
777
|
+
|
|
404
778
|
/**
|
|
405
779
|
* `googleProvider` — {@link ProviderPlugin} factory for @gullabs/google.
|
|
406
780
|
*
|
|
@@ -458,8 +832,8 @@ declare const defaultGeminiRegistry: ModelRegistry;
|
|
|
458
832
|
* Provides `geminiPricingSource` — a factory returning a `PricingSource` port
|
|
459
833
|
* implementation backed by the frozen Gemini pricing snapshot ({@link
|
|
460
834
|
* GEMINI_PRICING}). Uses exact priced model identifiers,
|
|
461
|
-
* resolves the concrete per-tier rates, and delegates the arithmetic to
|
|
462
|
-
* `@gullabs/core`'s `computeCost
|
|
835
|
+
* resolves the concrete per-tier rates, and delegates the token arithmetic to
|
|
836
|
+
* `@gullabs/core`'s `computeCost` (audio input and grounding are priced here). Core itself carries zero Gemini pricing
|
|
463
837
|
* knowledge and applies no tier multiplier.
|
|
464
838
|
*
|
|
465
839
|
* @module
|
|
@@ -492,12 +866,13 @@ declare function geminiPricingSource(): PricingSource;
|
|
|
492
866
|
* All rates are in **micro-USD per million tokens** (µUSD/M).
|
|
493
867
|
* To get the cost for N tokens: `cost_µUSD = N * ratePerM / 1_000_000`.
|
|
494
868
|
*
|
|
495
|
-
* **Service tiers.** Each model stores concrete `standard
|
|
496
|
-
*
|
|
497
|
-
*
|
|
498
|
-
*
|
|
499
|
-
*
|
|
500
|
-
*
|
|
869
|
+
* **Service tiers.** Each model stores concrete `standard` and `flex` rates
|
|
870
|
+
* transcribed from Google's pricing page. Flex is not a flat 50% of standard: on
|
|
871
|
+
* several models the cached lane stays at the standard cached rate (or a
|
|
872
|
+
* published rate that is not half). A tier that is not one of those two is
|
|
873
|
+
* unpriced (reject-don't-map). `priority` is intentionally absent: it needs
|
|
874
|
+
* downgrade accounting and is a backlog item. The Batch API has no path in this
|
|
875
|
+
* library (no schema admits a batch tier), so its rates are not carried.
|
|
501
876
|
*
|
|
502
877
|
* **Long-context tier.** Gemini Pro models charge a premium when the GROSS
|
|
503
878
|
* input token count exceeds 200,000. Selected by `inputTokens` (incl. cached),
|
|
@@ -506,14 +881,30 @@ declare function geminiPricingSource(): PricingSource;
|
|
|
506
881
|
* **Thinking tokens.** Already inside `outputTokens` (GROSS convention) and
|
|
507
882
|
* billed at the output rate — no separate thinking lane.
|
|
508
883
|
*
|
|
509
|
-
* **
|
|
510
|
-
*
|
|
511
|
-
*
|
|
512
|
-
*
|
|
884
|
+
* **Audio input.** Gemini 2.5 Flash, 2.5 Flash-Lite and 3.1 Flash-Lite charge
|
|
885
|
+
* more for audio input than for text, image and video tokens ({@link GeminiRates.audio}).
|
|
886
|
+
* The adapter records the prompt's per-modality token counts
|
|
887
|
+
* (`usageMetadata.promptTokensDetails` and `cacheTokensDetails`) as
|
|
888
|
+
* `usage.details.input_<modality>` / `cached_<modality>`, and the pricing source
|
|
889
|
+
* bills the audio tokens at the audio rates and the rest at the text rate. Every
|
|
890
|
+
* other model is billed one input rate for all modalities on the page. Models with
|
|
891
|
+
* an audio rate have no `gt200k` band, so the long-context band never needs the
|
|
892
|
+
* audio split.
|
|
513
893
|
*
|
|
514
894
|
* Re-verified against https://ai.google.dev/gemini-api/docs/pricing on
|
|
515
895
|
* 2026-09-25. Standard token rates for already-registered models were
|
|
516
|
-
* unchanged from the 2026-08-12 snapshot; flex
|
|
896
|
+
* unchanged from the 2026-08-12 snapshot; flex cached rates were not. The audio
|
|
897
|
+
* input and cached-audio rates (standard and flex) were read from the same page
|
|
898
|
+
* on 2026-10-03; the page showed "Last Updated 2026-10-01 UTC".
|
|
899
|
+
*
|
|
900
|
+
* **Grounding with Google Search** is a tool lane, not a token rate; see
|
|
901
|
+
* {@link GEMINI_GROUNDING_PRICING}. It was added from the same pricing page
|
|
902
|
+
* on 2026-10-03. {@link pricingVersion} is `gemini-2026-10-03` because the
|
|
903
|
+
* snapshot gained the grounding lane and the audio lane that day (a grounded or
|
|
904
|
+
* audio call prices differently under it). The 2026-10-03 read of the page also
|
|
905
|
+
* matched the standard text and cached rates of the models it listed (and the
|
|
906
|
+
* flex text and cached rates of the three audio models); only the models the page
|
|
907
|
+
* summary did not list keep their 2026-09-25 verification.
|
|
517
908
|
*
|
|
518
909
|
* @module
|
|
519
910
|
*/
|
|
@@ -525,30 +916,143 @@ declare function geminiPricingSource(): PricingSource;
|
|
|
525
916
|
* core concept — it lives here (not `@gullabs/core`) alongside the rates it
|
|
526
917
|
* dates.
|
|
527
918
|
*/
|
|
528
|
-
declare const pricingVersion: "gemini-2026-
|
|
919
|
+
declare const pricingVersion: "gemini-2026-10-03";
|
|
529
920
|
/** Tiers this snapshot prices. Anything else is unpriced. */
|
|
530
|
-
declare const GEMINI_PRICED_TIERS: readonly ["standard", "flex"
|
|
921
|
+
declare const GEMINI_PRICED_TIERS: readonly ["standard", "flex"];
|
|
531
922
|
type GeminiPricedTier = (typeof GEMINI_PRICED_TIERS)[number];
|
|
923
|
+
/** Audio input rates for a model that prices audio apart from text (µUSD per million tokens). */
|
|
924
|
+
interface GeminiAudioRates {
|
|
925
|
+
/** Non-cached audio input tokens. */
|
|
926
|
+
inputPerM: number;
|
|
927
|
+
/** Cached audio input tokens. */
|
|
928
|
+
cachedPerM: number;
|
|
929
|
+
}
|
|
930
|
+
/** {@link ModelRates} plus the audio input rates, for models that publish them. */
|
|
931
|
+
interface GeminiRates extends ModelRates {
|
|
932
|
+
/**
|
|
933
|
+
* Present only on a model whose pricing page lists a separate audio input
|
|
934
|
+
* price. Priced on the audio tokens `promptTokensDetails` reports; every other
|
|
935
|
+
* input token uses the text/image/video rates above.
|
|
936
|
+
*/
|
|
937
|
+
audio?: GeminiAudioRates;
|
|
938
|
+
}
|
|
532
939
|
/** Concrete per-tier rates for one model. */
|
|
533
940
|
interface GeminiTierRates {
|
|
534
|
-
standard:
|
|
535
|
-
flex:
|
|
536
|
-
batch: ModelRates;
|
|
941
|
+
standard: GeminiRates;
|
|
942
|
+
flex: GeminiRates;
|
|
537
943
|
}
|
|
538
944
|
/**
|
|
539
|
-
*
|
|
540
|
-
* by priced tier
|
|
945
|
+
* Deep-frozen Gemini pricing snapshot (per-1M in µUSD), keyed by model id, then
|
|
946
|
+
* by priced tier; no rate object, `gt200k` band or `audio` rate can be changed.
|
|
947
|
+
* Every number is transcribed from the pricing page.
|
|
541
948
|
*
|
|
542
949
|
* Keys are exact priced model identifiers. Unlisted variants are unpriced.
|
|
543
950
|
*
|
|
544
|
-
* Source: https://ai.google.dev/gemini-api/docs/pricing (re-verified 2026-09-25
|
|
951
|
+
* Source: https://ai.google.dev/gemini-api/docs/pricing (re-verified 2026-09-25;
|
|
952
|
+
* the audio input and cached-audio rates read 2026-10-03, page last updated
|
|
953
|
+
* 2026-10-01).
|
|
545
954
|
*/
|
|
546
955
|
declare const GEMINI_PRICING: Readonly<Record<string, GeminiTierRates>>;
|
|
547
|
-
/**
|
|
548
|
-
|
|
956
|
+
/**
|
|
957
|
+
* How a model bills grounding with Google Search, in µUSD per unit.
|
|
958
|
+
*
|
|
959
|
+
* - `'query'`: Gemini 3 bills each search query the model performed. The unit
|
|
960
|
+
* count is `usage.details.web_search_calls`, counted as occurrences in
|
|
961
|
+
* `webSearchQueries` (a repeated query counts each time; whether Google bills
|
|
962
|
+
* a repeat is not established, so the count is the conservative one).
|
|
963
|
+
* - `'prompt'`: Gemini 2.5 bills each prompt that was grounded, once however
|
|
964
|
+
* many queries it ran. A grounded prompt is one whose response reports at
|
|
965
|
+
* least one query.
|
|
966
|
+
*
|
|
967
|
+
* Transcribed from https://ai.google.dev/gemini-api/docs/pricing, grounding
|
|
968
|
+
* with Google Search, read 2026-10-03 (Gemini 3: $14 per 1,000 queries;
|
|
969
|
+
* Gemini 2.5: $35 per 1,000 grounded prompts). The page also publishes a free
|
|
970
|
+
* allowance (as read 2026-10: 5,000 requests per month shared
|
|
971
|
+
* across Gemini 3.x, 1,500 requests per day on Gemini 2.5). It is shared across
|
|
972
|
+
* a project's calls, so no single call can know whether it was free: every
|
|
973
|
+
* grounding fee is charged in full here, which is why a call that ran Search is
|
|
974
|
+
* never reported as exact.
|
|
975
|
+
*
|
|
976
|
+
* Keys are exact priced model identifiers. Gemma has no token price in this
|
|
977
|
+
* snapshot, so it has no grounding price either.
|
|
978
|
+
*/
|
|
979
|
+
interface GeminiGroundingRate {
|
|
980
|
+
readonly unit: 'query' | 'prompt';
|
|
981
|
+
readonly microUsdPerUnit: number;
|
|
982
|
+
}
|
|
983
|
+
declare const GEMINI_GROUNDING_PRICING: Readonly<Record<string, GeminiGroundingRate>>;
|
|
984
|
+
/** Resolve the concrete {@link GeminiRates} for `(model, tier)`. */
|
|
985
|
+
declare function resolveGeminiRates(model: string, tier: string | undefined): GeminiRates | undefined;
|
|
549
986
|
|
|
987
|
+
/**
|
|
988
|
+
* True when a failed Flex call may be retried once on the Standard tier
|
|
989
|
+
* because Flex capacity, not the caller's quota, ran out.
|
|
990
|
+
*
|
|
991
|
+
* Only HTTP 503 counts. Google's Flex page (ai.google.dev/gemini-api/docs/
|
|
992
|
+
* flex-inference, read 2026-10-03, page dated 2026-09-23) lists two failures
|
|
993
|
+
* when capacity is unavailable, 503 "The system is currently at capacity" and
|
|
994
|
+
* 429 "Rate limits or resource exhaustion", but documents no field that tells
|
|
995
|
+
* a capacity 429 from a quota 429, and no capture of either 429 exists. A 429
|
|
996
|
+
* is therefore the ordinary rate-limit path (the provider's `RetryInfo` delay
|
|
997
|
+
* is honoured, no Standard dispatch, no tier pin): dispatching Standard at once
|
|
998
|
+
* would undercut the delay, add a call to a rate-limited project and bill the
|
|
999
|
+
* logical call at the Standard rate. Nothing in the message text is read.
|
|
1000
|
+
*
|
|
1001
|
+
* `err` is the classified error.
|
|
1002
|
+
*/
|
|
550
1003
|
declare function isGeminiCapacityError(err: LlmError): boolean;
|
|
551
1004
|
|
|
1005
|
+
/**
|
|
1006
|
+
* Token limits and admitted input media types for the registered Google
|
|
1007
|
+
* models, read from Google's own documentation (re-read 2026-10-03).
|
|
1008
|
+
*
|
|
1009
|
+
* One table feeds both the descriptors (`limits`, `inputMimeTypes`) and the
|
|
1010
|
+
* per-model config schemas (`maxOutputTokens` is capped only when a number is
|
|
1011
|
+
* documented), so a descriptor and its schema cannot drift.
|
|
1012
|
+
*
|
|
1013
|
+
* Sources, all read 2026-10-03:
|
|
1014
|
+
*
|
|
1015
|
+
* - Gemini: `https://ai.google.dev/gemini-api/docs/models/<model-id>`, one page
|
|
1016
|
+
* per model. Every registered Gemini model lists "Input token limit
|
|
1017
|
+
* 1,048,576" and "Output token limit 65,536". The output limit includes
|
|
1018
|
+
* thinking tokens.
|
|
1019
|
+
* - Gemma 4: the model card, `https://ai.google.dev/gemma/docs/core/model_card_4`,
|
|
1020
|
+
* states a "256K tokens" context window (read as 262,144). Neither it nor
|
|
1021
|
+
* `https://ai.google.dev/gemma/docs/core/gemma_on_gemini_api` documents an
|
|
1022
|
+
* output limit, so `maxOutputTokens` is `null` (see
|
|
1023
|
+
* {@link ModelLimits.maxOutputTokens}): no figure is invented and the schema
|
|
1024
|
+
* applies no cap.
|
|
1025
|
+
* - Gemini input media: Google publishes lists for images
|
|
1026
|
+
* (`https://ai.google.dev/gemini-api/docs/image-understanding`: PNG, JPEG,
|
|
1027
|
+
* WebP, HEIC, HEIF), audio (`.../audio`), and video
|
|
1028
|
+
* (`.../video-understanding`), but no closed list for documents:
|
|
1029
|
+
* `.../document-processing` says only that PDF is understood natively and
|
|
1030
|
+
* "you can pass other MIME types for document understanding, like TXT,
|
|
1031
|
+
* Markdown, HTML, XML, etc.", extracted as plain text. `.../files` lists no
|
|
1032
|
+
* types either. The descriptor therefore admits the documented families by
|
|
1033
|
+
* prefix (`text/*`, `image/*`, `audio/*`, `video/*`) plus `application/pdf`,
|
|
1034
|
+
* and leaves a type inside a family that Google does not accept to Google's
|
|
1035
|
+
* own error. `application/json`, `application/xml` and other `application/*`
|
|
1036
|
+
* types are not in any documented family and stay rejected.
|
|
1037
|
+
* - Gemma 4 input media: the model card lists "Supported Modalities: Text,
|
|
1038
|
+
* Image" for the 31B and 26B A4B models and says "All models support image
|
|
1039
|
+
* inputs and can process videos as frames", with video "a maximum of 60
|
|
1040
|
+
* seconds" at one frame per second; audio input is E2B/E4B/12B only. No Gemma
|
|
1041
|
+
* page names image or video media types, so `image/*` and `video/*` are
|
|
1042
|
+
* admitted and nothing else. That the Gemini API's Gemma endpoint takes a
|
|
1043
|
+
* video part (rather than frames sent as images) has not been probed.
|
|
1044
|
+
*
|
|
1045
|
+
* @module
|
|
1046
|
+
*/
|
|
1047
|
+
|
|
1048
|
+
/**
|
|
1049
|
+
* Media types every Gemini model in the registry accepts as input parts: the
|
|
1050
|
+
* documented families (text, image, audio, video) and PDF. See the module
|
|
1051
|
+
* comment for why the families are not closed lists. One rule serves
|
|
1052
|
+
* `generate` and `GoogleFileStore.upload`.
|
|
1053
|
+
*/
|
|
1054
|
+
declare const GEMINI_INPUT_MIME_TYPES: readonly string[];
|
|
1055
|
+
|
|
552
1056
|
/**
|
|
553
1057
|
* GoogleFileStore — thin wrapper over the Gemini File API.
|
|
554
1058
|
*
|
|
@@ -568,6 +1072,20 @@ interface GoogleFileHandle {
|
|
|
568
1072
|
/** Provider auto-deletes ~48 h after upload. Absent when not returned. */
|
|
569
1073
|
expiresAt?: Date;
|
|
570
1074
|
}
|
|
1075
|
+
/** The `File` resource fields the store reads. */
|
|
1076
|
+
type FileResp = {
|
|
1077
|
+
name?: string;
|
|
1078
|
+
uri?: string;
|
|
1079
|
+
mimeType?: string;
|
|
1080
|
+
state?: string;
|
|
1081
|
+
expirationTime?: string;
|
|
1082
|
+
/** Real field `File.error` (`FileStatus`): why processing failed. */
|
|
1083
|
+
error?: {
|
|
1084
|
+
code?: number;
|
|
1085
|
+
message?: string;
|
|
1086
|
+
details?: Record<string, unknown>[];
|
|
1087
|
+
};
|
|
1088
|
+
};
|
|
571
1089
|
/**
|
|
572
1090
|
* Minimal structural interface for the Gemini Files client surface we use.
|
|
573
1091
|
* Satisfied by the real ai.files object or a test fake.
|
|
@@ -578,23 +1096,12 @@ interface GeminiFilesClientLike {
|
|
|
578
1096
|
config?: {
|
|
579
1097
|
mimeType?: string;
|
|
580
1098
|
displayName?: string;
|
|
1099
|
+
abortSignal?: AbortSignal;
|
|
581
1100
|
};
|
|
582
|
-
}): Promise<
|
|
583
|
-
name?: string;
|
|
584
|
-
uri?: string;
|
|
585
|
-
mimeType?: string;
|
|
586
|
-
state?: string;
|
|
587
|
-
expirationTime?: string;
|
|
588
|
-
}>;
|
|
1101
|
+
}): Promise<FileResp>;
|
|
589
1102
|
get(params: {
|
|
590
1103
|
name: string;
|
|
591
|
-
}): Promise<
|
|
592
|
-
name?: string;
|
|
593
|
-
uri?: string;
|
|
594
|
-
mimeType?: string;
|
|
595
|
-
state?: string;
|
|
596
|
-
expirationTime?: string;
|
|
597
|
-
}>;
|
|
1104
|
+
}): Promise<FileResp>;
|
|
598
1105
|
delete(params: {
|
|
599
1106
|
name: string;
|
|
600
1107
|
}): Promise<void>;
|
|
@@ -629,7 +1136,18 @@ interface GoogleFileStoreOptions {
|
|
|
629
1136
|
/** Max time to wait for ACTIVE. Default: 300 000 ms (5 min). */
|
|
630
1137
|
timeoutMs?: number;
|
|
631
1138
|
};
|
|
632
|
-
/**
|
|
1139
|
+
/**
|
|
1140
|
+
* Timer source for the poll wait; pass the client's `FakeClock` in tests so
|
|
1141
|
+
* one `advance` fires the wait. Default: the platform's timers. (`now` is the
|
|
1142
|
+
* poll timeout's clock; pass `() => clock.now()` with it.)
|
|
1143
|
+
*/
|
|
1144
|
+
scheduler?: Scheduler;
|
|
1145
|
+
/**
|
|
1146
|
+
* Replaces the poll wait wholesale (instant polling in tests). When given,
|
|
1147
|
+
* `scheduler` is not used for the wait, and the wait cannot be cancelled: if
|
|
1148
|
+
* the deadline or an abort ends the upload first, a timer your `sleep` set
|
|
1149
|
+
* stays pending until it fires. The default wait is cleared at once.
|
|
1150
|
+
*/
|
|
633
1151
|
sleep?: (ms: number) => Promise<void>;
|
|
634
1152
|
/** Injectable clock for deterministic tests. Default: `Date.now`. */
|
|
635
1153
|
now?: () => number;
|
|
@@ -652,7 +1170,8 @@ declare class GoogleFileStore {
|
|
|
652
1170
|
private readonly logger;
|
|
653
1171
|
private readonly intervalMs;
|
|
654
1172
|
private readonly timeoutMs;
|
|
655
|
-
private readonly
|
|
1173
|
+
private readonly startWait;
|
|
1174
|
+
private readonly scheduler;
|
|
656
1175
|
private readonly now;
|
|
657
1176
|
/** Memoised client promise — built at most once per store instance. */
|
|
658
1177
|
private clientPromise;
|
|
@@ -662,13 +1181,26 @@ declare class GoogleFileStore {
|
|
|
662
1181
|
* Upload bytes to the Gemini File API and wait until the file is ACTIVE.
|
|
663
1182
|
*
|
|
664
1183
|
* @param source - Raw bytes or Blob.
|
|
665
|
-
* @param mimeType - IANA media type, e.g. `"image/png"`.
|
|
1184
|
+
* @param mimeType - IANA media type, e.g. `"image/png"`. It must pass the same
|
|
1185
|
+
* admission rule `generate` applies to a Gemini model's parts (one shared
|
|
1186
|
+
* function, so a file that uploads can be used): an empty or unadmitted type
|
|
1187
|
+
* is `bad_request` before any bytes are sent. The string is sent to Google
|
|
1188
|
+
* unchanged.
|
|
666
1189
|
* @param opts - Optional display name.
|
|
667
1190
|
*/
|
|
668
1191
|
upload(source: Uint8Array | Blob, mimeType: string, opts?: {
|
|
669
1192
|
displayName?: string;
|
|
670
1193
|
signal?: AbortSignal;
|
|
671
1194
|
}): Promise<GoogleFileHandle>;
|
|
1195
|
+
/**
|
|
1196
|
+
* `work` raced against the abort promise and the time left until `deadline`
|
|
1197
|
+
* (the deadline is a non-retryable `server` error). `work` is observed, so a
|
|
1198
|
+
* late rejection after the race is lost is not unhandled; the deadline timer
|
|
1199
|
+
* is always cleared.
|
|
1200
|
+
*/
|
|
1201
|
+
private raceDeadline;
|
|
1202
|
+
/** Polls `name` until ACTIVE; see {@link GoogleFileStore.upload}. */
|
|
1203
|
+
private pollUntilActive;
|
|
672
1204
|
/**
|
|
673
1205
|
* Delete a single uploaded file. Idempotent: not-found → success.
|
|
674
1206
|
*
|
|
@@ -690,208 +1222,62 @@ declare class GoogleFileStore {
|
|
|
690
1222
|
}
|
|
691
1223
|
|
|
692
1224
|
/**
|
|
693
|
-
*
|
|
1225
|
+
* Gemini 3 thought signatures as an overlay on the host's actual history.
|
|
694
1226
|
*
|
|
695
|
-
*
|
|
696
|
-
*
|
|
697
|
-
*
|
|
1227
|
+
* Gemini 3.x returns an opaque `thoughtSignature` on the first function call of
|
|
1228
|
+
* each model turn (and sometimes on text parts) and rejects a replayed turn
|
|
1229
|
+
* whose function call has lost it. The library never keeps a copy of the
|
|
1230
|
+
* history. `result.transientProviderState` is only an overlay saying which part
|
|
1231
|
+
* of the host's own messages gets which signature:
|
|
698
1232
|
*
|
|
699
|
-
*
|
|
1233
|
+
* ```ts
|
|
1234
|
+
* { google: { signatures: [{ messageIndex, partIndex, kind, model, partSha256, signature }] } }
|
|
1235
|
+
* ```
|
|
1236
|
+
*
|
|
1237
|
+
* `messageIndex` indexes the messages the adapter receives (`request.messages`
|
|
1238
|
+
* as the engine hands them on, after any middleware), `partIndex` indexes that
|
|
1239
|
+
* message's `parts` (thought parts are never in a message), `kind` is the part
|
|
1240
|
+
* kind that was signed (`'text'` or `'tool-call'`), `model` is the model string
|
|
1241
|
+
* the request named, and `partSha256` is the SHA-256 of the part's RFC 8785
|
|
1242
|
+
* canonical JSON, so an edited argument or text is detected and key order
|
|
1243
|
+
* (Postgres `jsonb`) does not matter.
|
|
1244
|
+
*
|
|
1245
|
+
* A function-call signature is required on replay, so a stale one is
|
|
1246
|
+
* `bad_request`. A text signature is optional (Google accepts the next turn
|
|
1247
|
+
* without it), so a stale one (edited, moved, removed, or issued for another
|
|
1248
|
+
* model) is dropped with a warning and nothing else is lost.
|
|
700
1249
|
*
|
|
701
1250
|
* @module
|
|
702
1251
|
*/
|
|
703
1252
|
|
|
704
|
-
/**
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
/**
|
|
711
|
-
|
|
712
|
-
/**
|
|
713
|
-
* Local view of when the cache expires (server-authoritative; computed from
|
|
714
|
-
* ttl or parsed from the server response `expireTime`).
|
|
715
|
-
*/
|
|
716
|
-
expiresAt: Date;
|
|
717
|
-
/** Caches are model-bound; never use a handle with a different model. */
|
|
718
|
-
model: string;
|
|
719
|
-
}
|
|
720
|
-
/** Key used to look up or create an entry in the in-process cache map. */
|
|
721
|
-
interface CacheKey {
|
|
1253
|
+
/** The part kinds Google signs. */
|
|
1254
|
+
type GoogleSignedKind = 'text' | 'tool-call';
|
|
1255
|
+
/** One signature, pinned to a part of the host's history. */
|
|
1256
|
+
type GoogleSignatureEntry = {
|
|
1257
|
+
messageIndex: number;
|
|
1258
|
+
partIndex: number;
|
|
1259
|
+
/** The kind of part that was signed; decides whether a stale entry is fatal. */
|
|
1260
|
+
kind: GoogleSignedKind;
|
|
722
1261
|
model: string;
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
}
|
|
726
|
-
/**
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
interface GeminiCachesClientLike {
|
|
731
|
-
create(params: {
|
|
732
|
-
model: string;
|
|
733
|
-
config: {
|
|
734
|
-
contents?: Content[];
|
|
735
|
-
systemInstruction?: Content | string;
|
|
736
|
-
ttl?: string;
|
|
737
|
-
displayName?: string;
|
|
738
|
-
};
|
|
739
|
-
}): Promise<{
|
|
740
|
-
name?: string;
|
|
741
|
-
model?: string;
|
|
742
|
-
expireTime?: string;
|
|
743
|
-
}>;
|
|
744
|
-
update(params: {
|
|
745
|
-
name: string;
|
|
746
|
-
config: {
|
|
747
|
-
ttl?: string;
|
|
748
|
-
expireTime?: string;
|
|
749
|
-
};
|
|
750
|
-
}): Promise<{
|
|
751
|
-
name?: string;
|
|
752
|
-
expireTime?: string;
|
|
753
|
-
}>;
|
|
754
|
-
delete(params: {
|
|
755
|
-
name: string;
|
|
756
|
-
}): Promise<unknown>;
|
|
757
|
-
}
|
|
758
|
-
interface GoogleCacheStoreOptions {
|
|
759
|
-
auth: AuthMaterial;
|
|
760
|
-
/** Injectable client for tests; skips SDK import when provided. */
|
|
761
|
-
client?: GeminiCachesClientLike;
|
|
762
|
-
/**
|
|
763
|
-
* When true, concurrent `getOrCreate` calls for the same key share one
|
|
764
|
-
* in-flight create. Default false.
|
|
765
|
-
*/
|
|
766
|
-
coalesce?: boolean;
|
|
767
|
-
/**
|
|
768
|
-
* Subtracted from the local expiry so we stop using a cache slightly before
|
|
769
|
-
* the server evicts it. Default 30 s.
|
|
770
|
-
*/
|
|
771
|
-
expirySkewSeconds?: number;
|
|
772
|
-
/** Called on delete failures instead of rethrowing. */
|
|
773
|
-
onDeleteError?: (cacheName: string, err: unknown) => void;
|
|
774
|
-
/** Optional structured logger. When provided, routes delete failures to logger.error. */
|
|
775
|
-
logger?: Logger;
|
|
776
|
-
/** Injectable clock for deterministic tests. Default: `Date.now`. */
|
|
777
|
-
now?: () => number;
|
|
778
|
-
/**
|
|
779
|
-
* Opt-in pre-flight token-count gate applied before every cache create
|
|
780
|
-
* (both `create()` directly and the `getOrCreate()` path it delegates to,
|
|
781
|
-
* including the coalesced path — enforced once, inside `create()`, so
|
|
782
|
-
* there is no separate "in-flight" gap to close).
|
|
783
|
-
*/
|
|
784
|
-
preflight?: {
|
|
785
|
-
/** Minimum token count required before a cache create is allowed to proceed. */
|
|
786
|
-
minTokens: number;
|
|
787
|
-
/**
|
|
788
|
-
* Counts tokens for the exact token-bearing payload of the impending
|
|
789
|
-
* create — `model` + `contents` + `systemInstruction` only. `ttl` and
|
|
790
|
-
* `displayName` are excluded: they carry no tokens and are irrelevant to
|
|
791
|
-
* the pre-flight check.
|
|
792
|
-
*
|
|
793
|
-
* This callback receives genai-native `Content[]`/`Content|string` — it
|
|
794
|
-
* does NOT receive the library's `Message[]` shape and there is no
|
|
795
|
-
* automatic conversion between the two (explicit seam, by design). Hosts
|
|
796
|
-
* using genai-native content directly can wire this to a raw
|
|
797
|
-
* `client.models.countTokens` call; hosts building from the library's
|
|
798
|
-
* `Message[]` should call `@gullabs/core`'s `client.countTokens` rather
|
|
799
|
-
* than expecting this callback to convert for them.
|
|
800
|
-
*/
|
|
801
|
-
countTokens: (payload: {
|
|
802
|
-
model: string;
|
|
803
|
-
contents?: Content[];
|
|
804
|
-
systemInstruction?: Content | string;
|
|
805
|
-
}) => Promise<number>;
|
|
1262
|
+
partSha256: string;
|
|
1263
|
+
signature: string;
|
|
1264
|
+
};
|
|
1265
|
+
/** The shape of `transientProviderState` for Gemini 3.x models. */
|
|
1266
|
+
type GoogleSignatureState = {
|
|
1267
|
+
google: {
|
|
1268
|
+
signatures: GoogleSignatureEntry[];
|
|
806
1269
|
};
|
|
807
|
-
}
|
|
1270
|
+
};
|
|
808
1271
|
/**
|
|
809
|
-
*
|
|
810
|
-
*
|
|
811
|
-
*
|
|
812
|
-
* the
|
|
813
|
-
*
|
|
814
|
-
*
|
|
815
|
-
*
|
|
816
|
-
* time and memoizes a single SDK client from it (`clientPromise`). This is
|
|
817
|
-
* correct and sufficient for static API keys. If refreshable credentials
|
|
818
|
-
* (short-lived OAuth/STS tokens) are added in the future, this memoization is
|
|
819
|
-
* the seam that will need rework: the cached client would hold stale credentials
|
|
820
|
-
* for the lifetime of a long-lived store instance. At that point, the store
|
|
821
|
-
* will need to either rebuild the client on each operation or accept a
|
|
822
|
-
* credential-resolver callback rather than a plain `AuthMaterial` value.
|
|
823
|
-
* See ADR-020 in DECISIONS.md.
|
|
1272
|
+
* Remove the entries for messages the host removed from its history, and shift
|
|
1273
|
+
* the `messageIndex` of every later entry down so it still points at the same
|
|
1274
|
+
* message. `indices` are positions in the history the state was issued for
|
|
1275
|
+
* (before the removal). Whole turns only: remove a tool-call message together
|
|
1276
|
+
* with its tool-result message, and never keep a tool-call message while
|
|
1277
|
+
* removing its entry. Returns `undefined` when no entry remains, so the result
|
|
1278
|
+
* can be sent as `transientProviderState` or omitted.
|
|
824
1279
|
*/
|
|
825
|
-
declare
|
|
826
|
-
private readonly auth;
|
|
827
|
-
private readonly clientOverride;
|
|
828
|
-
private readonly coalesce;
|
|
829
|
-
private readonly skewMs;
|
|
830
|
-
private readonly onDeleteError;
|
|
831
|
-
private readonly logger;
|
|
832
|
-
private readonly now;
|
|
833
|
-
private readonly preflight;
|
|
834
|
-
/** Memoised client promise — built at most once per store instance. */
|
|
835
|
-
private clientPromise;
|
|
836
|
-
/** In-process cache of live handles, keyed by `${model}:${stableKey}`. */
|
|
837
|
-
private readonly entries;
|
|
838
|
-
/** In-flight create promises when coalescing is enabled. */
|
|
839
|
-
private readonly inflight;
|
|
840
|
-
constructor(opts: GoogleCacheStoreOptions);
|
|
841
|
-
private getClient;
|
|
842
|
-
private isLive;
|
|
843
|
-
/**
|
|
844
|
-
* Create a new cached content resource.
|
|
845
|
-
*
|
|
846
|
-
* The `expiresAt` on the returned handle is computed from the server's
|
|
847
|
-
* `expireTime` when available, with a local-clock fallback of
|
|
848
|
-
* `now + ttlSeconds * 1000`.
|
|
849
|
-
*/
|
|
850
|
-
create(input: {
|
|
851
|
-
model: string;
|
|
852
|
-
ttlSeconds: number;
|
|
853
|
-
contents?: Content[];
|
|
854
|
-
systemInstruction?: Content | string;
|
|
855
|
-
displayName?: string;
|
|
856
|
-
}): Promise<GoogleCacheHandle>;
|
|
857
|
-
/**
|
|
858
|
-
* Return a live cached-content handle, creating one if none exists or the
|
|
859
|
-
* cached entry has expired (accounting for skew).
|
|
860
|
-
*
|
|
861
|
-
* Reuse is PROCESS-SCOPED only — this store instance's in-memory map.
|
|
862
|
-
*
|
|
863
|
-
* When `coalesce` is enabled, concurrent calls for the same key share one
|
|
864
|
-
* in-flight create.
|
|
865
|
-
*/
|
|
866
|
-
getOrCreate(key: CacheKey, factory: () => Promise<{
|
|
867
|
-
ttlSeconds: number;
|
|
868
|
-
contents?: Content[];
|
|
869
|
-
systemInstruction?: Content | string;
|
|
870
|
-
}>): Promise<GoogleCacheHandle>;
|
|
871
|
-
/**
|
|
872
|
-
* Extend the cache TTL if it will expire within `thresholdSeconds`.
|
|
873
|
-
*
|
|
874
|
-
* Fail-open: if the update call throws, the original handle is returned
|
|
875
|
-
* unchanged. This method NEVER throws.
|
|
876
|
-
*
|
|
877
|
-
* @param handle - Handle to potentially refresh.
|
|
878
|
-
* @param opts.thresholdSeconds - Extend if expiry is within this many seconds.
|
|
879
|
-
* Default 300.
|
|
880
|
-
* @param opts.extensionSeconds - New TTL to set. Default: original TTL from
|
|
881
|
-
* entries map, or 3600 s when the handle was not created via `getOrCreate`.
|
|
882
|
-
*/
|
|
883
|
-
refreshIfExpiringSoon(handle: GoogleCacheHandle, opts?: {
|
|
884
|
-
thresholdSeconds?: number;
|
|
885
|
-
extensionSeconds?: number;
|
|
886
|
-
}): Promise<GoogleCacheHandle>;
|
|
887
|
-
/**
|
|
888
|
-
* Delete a cached content resource.
|
|
889
|
-
*
|
|
890
|
-
* Errors are forwarded to `onDeleteError` and NOT rethrown.
|
|
891
|
-
* The handle is removed from the in-process entries map regardless.
|
|
892
|
-
*/
|
|
893
|
-
delete(handle: GoogleCacheHandle): Promise<void>;
|
|
894
|
-
}
|
|
1280
|
+
declare function dropMessagesFromSignatureState(state: unknown, indices: readonly number[]): GoogleSignatureState | undefined;
|
|
895
1281
|
|
|
896
1282
|
/**
|
|
897
1283
|
* geminiContentToMessages — migration utility: `@google/genai` `Content[]` →
|
|
@@ -910,7 +1296,10 @@ declare class GoogleCacheStore {
|
|
|
910
1296
|
* Validation is an exhaustive own-key scan: the set of defined keys on each
|
|
911
1297
|
* `Part` must be EXACTLY one of the recognized combinations (`['text']`,
|
|
912
1298
|
* `['inlineData']`, `['inlineData', 'mediaResolution']`, `['fileData']`,
|
|
913
|
-
* `['fileData', 'mediaResolution']`)
|
|
1299
|
+
* `['fileData', 'mediaResolution']`), plus `functionCall` / `functionResponse`.
|
|
1300
|
+
* A `thoughtSignature` on a model text or `functionCall` part is imported into
|
|
1301
|
+
* `transientProviderState` (see {@link GeminiContentToMessagesInput.model});
|
|
1302
|
+
* on any other part it throws. Any other key — including unknown
|
|
914
1303
|
* future SDK fields — or any combination outside that set throws. Keys whose
|
|
915
1304
|
* value is `undefined` are treated as absent (genai types are all-optional;
|
|
916
1305
|
* only defined values count).
|
|
@@ -930,6 +1319,12 @@ interface GeminiContentToMessagesInput {
|
|
|
930
1319
|
* non-text part throws).
|
|
931
1320
|
*/
|
|
932
1321
|
systemInstruction?: Content | string;
|
|
1322
|
+
/**
|
|
1323
|
+
* The model string the converted history will be sent to. Required when any
|
|
1324
|
+
* part carries a `thoughtSignature`: signatures are bound to the model that
|
|
1325
|
+
* issued them and are never replayed on another one.
|
|
1326
|
+
*/
|
|
1327
|
+
model?: string;
|
|
933
1328
|
}
|
|
934
1329
|
/**
|
|
935
1330
|
* Output produced by {@link geminiContentToMessages}.
|
|
@@ -939,6 +1334,13 @@ interface GeminiContentToMessagesResult {
|
|
|
939
1334
|
system?: string;
|
|
940
1335
|
/** Normalized any-llm messages, one per input `Content`. */
|
|
941
1336
|
messages: Message[];
|
|
1337
|
+
/**
|
|
1338
|
+
* The thought signatures found on model parts, as the overlay a Gemini 3.x
|
|
1339
|
+
* request takes as `transientProviderState`. Present only when at least one
|
|
1340
|
+
* part carried a `thoughtSignature`. Send it with `messages` unedited and the
|
|
1341
|
+
* same `model`.
|
|
1342
|
+
*/
|
|
1343
|
+
transientProviderState?: JsonValue;
|
|
942
1344
|
}
|
|
943
1345
|
/**
|
|
944
1346
|
* Convert `@google/genai` `Content[]` / `Part[]` into any-llm's normalized
|
|
@@ -974,4 +1376,4 @@ interface GeminiContentToMessagesResult {
|
|
|
974
1376
|
*/
|
|
975
1377
|
declare function geminiContentToMessages(input: GeminiContentToMessagesInput): GeminiContentToMessagesResult;
|
|
976
1378
|
|
|
977
|
-
export { type CacheKey, FLEX_DEFAULT_TIMEOUT_MS, type FileDeleteOptions, GEMINI_PRICED_TIERS, GEMINI_PRICING, type GeminiAdapterOptions, type GeminiCachesClientLike, type GeminiClientLike, type GeminiContentToMessagesInput, type GeminiContentToMessagesResult, type GeminiCountTokensParams, type GeminiCountTokensResponseShape, type GeminiFilesClientLike, type GeminiPricedTier, type GeminiTierRates, type GoogleCacheHandle, GoogleCacheStore, type GoogleCacheStoreOptions, type GoogleFileHandle, GoogleFileStore, type GoogleFileStoreOptions, type GoogleProviderOptions, type GoogleSafetySetting, type GoogleSearchTool, STANDARD_DEFAULT_TIMEOUT_MS, TRANSPORT_TIMEOUT_BUFFER_MS, buildGoogleClient, defaultGeminiRegistry, geminiAdapter, geminiContentToMessages, geminiModelDescriptors, geminiPricingSource, gemmaModelDescriptors, googleProvider, isGeminiCapacityError, pricingVersion, requireApiKey, resolveGeminiRates };
|
|
1379
|
+
export { type CacheKey, type ClassifyGoogleErrorExtra, FLEX_DEFAULT_TIMEOUT_MS, type FileDeleteOptions, GEMINI_GROUNDING_PRICING, GEMINI_INPUT_MIME_TYPES, GEMINI_PRICED_TIERS, GEMINI_PRICING, type GeminiAdapterOptions, type GeminiCachesClientLike, type GeminiClientLike, type GeminiContentToMessagesInput, type GeminiContentToMessagesResult, type GeminiCountTokensParams, type GeminiCountTokensResponseShape, type GeminiFilesClientLike, type GeminiGroundingRate, type GeminiPricedTier, type GeminiTierRates, type GoogleCacheHandle, GoogleCacheStore, type GoogleCacheStoreOptions, type GoogleCachedContentRef, type GoogleFileHandle, GoogleFileStore, type GoogleFileStoreOptions, type GoogleProviderOptions, type GoogleSafetySetting, type GoogleSearchTool, type GoogleSignatureEntry, type GoogleSignatureState, type GoogleSignedKind, STANDARD_DEFAULT_TIMEOUT_MS, TRANSPORT_TIMEOUT_BUFFER_MS, buildGoogleClient, classifyGoogleError, defaultGeminiRegistry, dropMessagesFromSignatureState, geminiAdapter, geminiContentToMessages, geminiModelDescriptors, geminiPricingSource, gemmaModelDescriptors, googleProvider, isGeminiCapacityError, pricingVersion, requireApiKey, resolveGeminiRates };
|