@gullabs/google 0.13.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -1,5 +1,267 @@
1
- import { AuthMaterial, ProviderAdapter, ProviderPlugin, ModelRegistry, ModelDescriptor, PricingSource, ModelRates, LlmError, Logger, Message } from '@gullabs/core';
2
- import { Content } from '@google/genai';
1
+ import { AuthMaterial, Logger, ProviderAdapter, LlmError, ProviderPlugin, ModelRegistry, ModelDescriptor, PricingSource, ModelRates, Scheduler, Message, JsonValue } from '@gullabs/core';
2
+ import { Content, Tool, ToolConfig } from '@google/genai';
3
+
4
+ /**
5
+ * GoogleCacheStore — thin wrapper over the Gemini Context Cache API.
6
+ *
7
+ * Manages create / get-or-create / refresh / delete lifecycle for cached
8
+ * contents. In-memory cache entries are PROCESS-SCOPED; they are not shared
9
+ * across processes, workers, or restarts.
10
+ *
11
+ * Injectable client and `now` function keep tests free of network and clock.
12
+ *
13
+ * @module
14
+ */
15
+
16
+ /**
17
+ * A handle to a cached content resource in the Gemini Context Cache API.
18
+ *
19
+ * Pass `cacheName` as `providerOptions.google.cachedContent` in LlmRequest.
20
+ */
21
+ interface GoogleCacheHandle {
22
+ /** Resource name, e.g. "cachedContents/xyz". */
23
+ cacheName: string;
24
+ /**
25
+ * Local view of when the cache expires (server-authoritative; computed from
26
+ * ttl or parsed from the server response `expireTime`).
27
+ */
28
+ expiresAt: Date;
29
+ /** Caches are model-bound; never use a handle with a different model. */
30
+ model: string;
31
+ /**
32
+ * Tokens the cache holds, from `usageMetadata.totalTokenCount` of the create
33
+ * response. Cache storage is billed per token-hour (see Google's pricing page)
34
+ * and no usage record carries it, so a host that prices storage reads this.
35
+ * Absent when the response had no `usageMetadata`. Kept across a TTL refresh.
36
+ */
37
+ totalTokenCount?: number;
38
+ /**
39
+ * The kinds of tool the cache holds, from the `tools` given to `create` (the
40
+ * keys of each `Tool`: `googleSearch`, `functionDeclarations`, ...); empty when
41
+ * it holds none. Pass it with the name as
42
+ * `providerOptions.google.cachedContent: { cacheName, toolKinds }` so a cached
43
+ * `googleSearch` is priced as Search: the request carries no tool, so nothing
44
+ * else says Search runs. Absent on a handle built by hand, where the content of
45
+ * the cache is unknown. Kept across a TTL refresh.
46
+ */
47
+ toolKinds?: readonly string[];
48
+ }
49
+ /** Key used to look up or create an entry in the in-process cache map. */
50
+ interface CacheKey {
51
+ model: string;
52
+ /** Caller-chosen stable identifier, e.g. a hash of the cached content. */
53
+ stableKey: string;
54
+ }
55
+ /**
56
+ * Minimal structural interface for the Gemini Caches client surface we use.
57
+ * Satisfied by the real `ai.caches` object or a test fake.
58
+ */
59
+ interface GeminiCachesClientLike {
60
+ create(params: {
61
+ model: string;
62
+ config: {
63
+ contents?: Content[];
64
+ systemInstruction?: Content | string;
65
+ tools?: Tool[];
66
+ toolConfig?: ToolConfig;
67
+ ttl?: string;
68
+ displayName?: string;
69
+ };
70
+ }): Promise<{
71
+ name?: string;
72
+ model?: string;
73
+ expireTime?: string;
74
+ usageMetadata?: {
75
+ totalTokenCount?: number;
76
+ };
77
+ }>;
78
+ update(params: {
79
+ name: string;
80
+ config: {
81
+ ttl?: string;
82
+ expireTime?: string;
83
+ };
84
+ }): Promise<{
85
+ name?: string;
86
+ expireTime?: string;
87
+ }>;
88
+ delete(params: {
89
+ name: string;
90
+ }): Promise<unknown>;
91
+ }
92
+ interface GoogleCacheStoreOptions {
93
+ auth: AuthMaterial;
94
+ /** Injectable client for tests; skips SDK import when provided. */
95
+ client?: GeminiCachesClientLike;
96
+ /**
97
+ * When true, concurrent `getOrCreate` calls for the same key share one
98
+ * in-flight create. Default false.
99
+ */
100
+ coalesce?: boolean;
101
+ /**
102
+ * Subtracted from the local expiry so we stop using a cache slightly before
103
+ * the server evicts it. Default 30 s.
104
+ */
105
+ expirySkewSeconds?: number;
106
+ /** Called on delete failures instead of rethrowing. */
107
+ onDeleteError?: (cacheName: string, err: unknown) => void;
108
+ /** Optional structured logger. When provided, routes delete failures to logger.error. */
109
+ logger?: Logger;
110
+ /** Injectable clock for deterministic tests. Default: `Date.now`. */
111
+ now?: () => number;
112
+ /**
113
+ * Opt-in pre-flight token-count gate applied before every cache create
114
+ * (both `create()` directly and the `getOrCreate()` path it delegates to,
115
+ * including the coalesced path — enforced once, inside `create()`, so
116
+ * there is no separate "in-flight" gap to close).
117
+ */
118
+ preflight?: {
119
+ /** Minimum token count required before a cache create is allowed to proceed. */
120
+ minTokens: number;
121
+ /**
122
+ * Counts tokens for the exact token-bearing payload of the impending
123
+ * create — `model` + `contents` + `systemInstruction` + `tools` only.
124
+ * `toolConfig`, `ttl` and `displayName` are excluded: they carry no tokens
125
+ * and are irrelevant to the pre-flight check.
126
+ *
127
+ * This callback receives genai-native `Content[]`/`Content|string` — it
128
+ * does NOT receive the library's `Message[]` shape and there is no
129
+ * automatic conversion between the two (explicit seam, by design). Hosts
130
+ * using genai-native content directly can wire this to a raw
131
+ * `client.models.countTokens` call; hosts building from the library's
132
+ * `Message[]` should call `@gullabs/core`'s `client.countTokens` rather
133
+ * than expecting this callback to convert for them.
134
+ */
135
+ countTokens: (payload: {
136
+ model: string;
137
+ contents?: Content[];
138
+ systemInstruction?: Content | string;
139
+ tools?: Tool[];
140
+ }) => Promise<number>;
141
+ };
142
+ }
143
+ /**
144
+ * Process-scoped helper for the Gemini Context Cache API.
145
+ *
146
+ * NOTE: `getOrCreate` reuse is PROCESS-SCOPED only. Entries survive only for
147
+ * the lifetime of this `GoogleCacheStore` instance. Across restarts, new
148
+ * caches will be created (and old ones will be server-evicted after their TTL).
149
+ *
150
+ * **Auth snapshot note:** this store captures the `AuthMaterial` at construction
151
+ * time and memoizes a single SDK client from it (`clientPromise`). This is
152
+ * correct and sufficient for static API keys. If refreshable credentials
153
+ * (short-lived OAuth/STS tokens) are added in the future, this memoization is
154
+ * the seam that will need rework: the cached client would hold stale credentials
155
+ * for the lifetime of a long-lived store instance. At that point, the store
156
+ * will need to either rebuild the client on each operation or accept a
157
+ * credential-resolver callback rather than a plain `AuthMaterial` value.
158
+ * See ADR-020 in DECISIONS.md.
159
+ */
160
+ declare class GoogleCacheStore {
161
+ private readonly auth;
162
+ private readonly clientOverride;
163
+ private readonly coalesce;
164
+ private readonly skewMs;
165
+ private readonly onDeleteError;
166
+ private readonly logger;
167
+ private readonly now;
168
+ private readonly preflight;
169
+ /** Memoised client promise — built at most once per store instance. */
170
+ private clientPromise;
171
+ /** In-process cache of live handles, keyed by `${model}:${stableKey}`. */
172
+ private readonly entries;
173
+ /** In-flight create promises when coalescing is enabled. */
174
+ private readonly inflight;
175
+ constructor(opts: GoogleCacheStoreOptions);
176
+ private getClient;
177
+ private isLive;
178
+ /**
179
+ * Create a new cached content resource.
180
+ *
181
+ * The `expiresAt` on the returned handle is computed from the server's
182
+ * `expireTime` when available, with a local-clock fallback of
183
+ * `now + ttlSeconds * 1000`.
184
+ */
185
+ create(input: {
186
+ model: string;
187
+ ttlSeconds: number;
188
+ contents?: Content[];
189
+ systemInstruction?: Content | string;
190
+ /**
191
+ * Tool declarations to store in the cache. Gemini rejects a request that
192
+ * sends `tools` or `toolConfig` together with `cachedContent`, so a cache
193
+ * used by a tool-calling call must hold them (the adapter rejects the
194
+ * combination before dispatch).
195
+ */
196
+ tools?: Tool[];
197
+ toolConfig?: ToolConfig;
198
+ displayName?: string;
199
+ }): Promise<GoogleCacheHandle>;
200
+ /**
201
+ * Return a live cached-content handle, creating one if none exists or the
202
+ * cached entry has expired (accounting for skew).
203
+ *
204
+ * Reuse is PROCESS-SCOPED only — this store instance's in-memory map.
205
+ *
206
+ * When `coalesce` is enabled, concurrent calls for the same key share one
207
+ * in-flight create.
208
+ */
209
+ getOrCreate(key: CacheKey, factory: () => Promise<{
210
+ ttlSeconds: number;
211
+ contents?: Content[];
212
+ systemInstruction?: Content | string;
213
+ tools?: Tool[];
214
+ toolConfig?: ToolConfig;
215
+ }>): Promise<GoogleCacheHandle>;
216
+ /**
217
+ * Extend the cache TTL if it will expire within `thresholdSeconds`.
218
+ *
219
+ * Fail-open: if the update call throws, the original handle is returned
220
+ * unchanged. This method NEVER throws.
221
+ *
222
+ * @param handle - Handle to potentially refresh.
223
+ * @param opts.thresholdSeconds - Extend if expiry is within this many seconds.
224
+ * Default 300.
225
+ * @param opts.extensionSeconds - New TTL to set. Default: original TTL from
226
+ * entries map, or 3600 s when the handle was not created via `getOrCreate`.
227
+ */
228
+ refreshIfExpiringSoon(handle: GoogleCacheHandle, opts?: {
229
+ thresholdSeconds?: number;
230
+ extensionSeconds?: number;
231
+ }): Promise<GoogleCacheHandle>;
232
+ /**
233
+ * Delete a cached content resource. Idempotent: a cache that is already gone
234
+ * (HTTP 404, or the 403 `CachedContent not found` Google sends for an expired
235
+ * one) is success, not an error.
236
+ *
237
+ * Any other error is forwarded to `onDeleteError` and NOT rethrown.
238
+ * The handle is removed from the in-process entries map regardless.
239
+ */
240
+ delete(handle: GoogleCacheHandle): Promise<void>;
241
+ }
242
+
243
+ /**
244
+ * The `safetySettings` categories and thresholds Gemini documents.
245
+ *
246
+ * Sources, read 2026-10-03:
247
+ *
248
+ * - Categories: ai.google.dev/api/generate-content (`HarmCategory`: hate speech,
249
+ * sexually explicit, dangerous content, harassment, civic integrity,
250
+ * jailbreak) and ai.google.dev/gemini-api/docs/safety-settings (page dated
251
+ * 2026-09-17, the adjustable filters). The SDK's `HARM_CATEGORY_IMAGE_*`
252
+ * values are marked "not supported in Gemini API" and are not admitted.
253
+ * - Thresholds: ai.google.dev/gemini-api/docs/safety-settings (page dated
254
+ * 2026-09-17), cross-checked against the SDK's `HarmBlockThreshold` enum.
255
+ *
256
+ * The adapter and every model's config schema read these lists, so a typo is
257
+ * rejected before dispatch instead of costing a round trip.
258
+ *
259
+ * @module
260
+ */
261
+ declare const GOOGLE_SAFETY_CATEGORIES: readonly ["HARM_CATEGORY_HARASSMENT", "HARM_CATEGORY_HATE_SPEECH", "HARM_CATEGORY_SEXUALLY_EXPLICIT", "HARM_CATEGORY_DANGEROUS_CONTENT", "HARM_CATEGORY_CIVIC_INTEGRITY", "HARM_CATEGORY_JAILBREAK"];
262
+ declare const GOOGLE_SAFETY_THRESHOLDS: readonly ["HARM_BLOCK_THRESHOLD_UNSPECIFIED", "BLOCK_LOW_AND_ABOVE", "BLOCK_MEDIUM_AND_ABOVE", "BLOCK_ONLY_HIGH", "BLOCK_NONE", "OFF"];
263
+ type GoogleSafetyCategory = (typeof GOOGLE_SAFETY_CATEGORIES)[number];
264
+ type GoogleSafetyThreshold = (typeof GOOGLE_SAFETY_THRESHOLDS)[number];
3
265
 
4
266
  /**
5
267
  * Google-specific provider options for `@gullabs/google`.
@@ -12,16 +274,34 @@ import { Content } from '@google/genai';
12
274
  *
13
275
  * @module
14
276
  */
277
+
15
278
  type GoogleSafetySetting = {
16
- category: string;
17
- threshold: string;
279
+ /** A documented `HarmCategory`; see `safety-settings.ts` for the source. */
280
+ category: GoogleSafetyCategory;
281
+ /** A documented `HarmBlockThreshold`. */
282
+ threshold: GoogleSafetyThreshold;
18
283
  };
19
284
  type GoogleSearchTool = {
20
285
  googleSearch: Record<string, never>;
21
286
  };
287
+ /**
288
+ * A cached-content reference: the resource name, or `{ cacheName, toolKinds }`
289
+ * taken from a `GoogleCacheHandle` (`handle.toolKinds` records which tools the
290
+ * cache holds). Pass those two fields, not the whole handle: the other fields
291
+ * are typed `never` here and the schema is strict. The request sent to Google
292
+ * carries only the name. A bare name says nothing about what the cache holds,
293
+ * so a Search fee in the response is priced from the observed queries and the
294
+ * cost is `estimated`; a handle whose `toolKinds` lists `googleSearch` marks the
295
+ * call as a Search call up front.
296
+ */
297
+ type GoogleCachedContentRef = string | (Pick<GoogleCacheHandle, 'cacheName' | 'toolKinds'> & {
298
+ expiresAt?: never;
299
+ model?: never;
300
+ totalTokenCount?: never;
301
+ });
22
302
  type GoogleProviderOptions = {
23
- /** Google cached content resource name. */
24
- cachedContent?: string;
303
+ /** Google cached content: the resource name, or a handle that records the cache's tool kinds. Explicit-caching models only. */
304
+ cachedContent?: GoogleCachedContentRef;
25
305
  /** Allowlisted Google safety settings. */
26
306
  safetySettings?: GoogleSafetySetting[];
27
307
  /** Exact Google tool declarations admitted by the selected model schema. */
@@ -33,6 +313,32 @@ type GoogleProviderOptions = {
33
313
  };
34
314
  /** Allow provider fallback from flex when flex was explicitly selected. */
35
315
  flexFallback?: boolean;
316
+ /**
317
+ * Admit `googleSearch` together with `output.jsonSchema` on a model that does
318
+ * not admit the pair by default. Opting in turns on {@link requireGrounding}
319
+ * unless it is set to `false`, because Search can be skipped silently when a
320
+ * response schema is attached. Requires both Search and a schema; Search is
321
+ * `googleSearch` in `tools` or a `cachedContent` handle whose `toolKinds` lists
322
+ * it. A schema call on such a model with a handle that lists `googleSearch`
323
+ * needs this flag, like an inline `googleSearch`; a bare cache name is not
324
+ * blocked, and a response of a schema call that reports search queries without
325
+ * a declared Search carries a warning.
326
+ */
327
+ allowSchemaWithSearch?: boolean;
328
+ /**
329
+ * Fail the call unless the response proves Search ran: `groundingMetadata`
330
+ * present with at least one `webSearchQueries` entry. The check judges only a
331
+ * candidate that finished normally (`STOP`, or no finish reason). Without the
332
+ * proof it throws a `server` error with reason `grounding_missing` and the
333
+ * attempt's usage is recorded; the error is retryable when no response schema
334
+ * is attached and not retryable when one is (the same request keeps missing).
335
+ * A `MAX_TOKENS` or other abnormal finish with no proof is not judged: it
336
+ * returns, with its own finish reason (`length`), and a filtered candidate
337
+ * throws its `content_filter` error. Requires Search (`googleSearch` in `tools`
338
+ * or a `cachedContent` handle whose `toolKinds` lists it). Defaults to
339
+ * `true` when {@link allowSchemaWithSearch} is `true`, else `false`.
340
+ */
341
+ requireGrounding?: boolean;
36
342
  };
37
343
  declare module '@gullabs/core' {
38
344
  interface ProviderOptionsMap {
@@ -102,6 +408,11 @@ interface GeminiPartShape {
102
408
  name?: string;
103
409
  args?: unknown;
104
410
  };
411
+ /**
412
+ * Opaque signature Gemini 3.x attaches to the first function call of a turn
413
+ * (and sometimes to text parts). Real field: `Part.thoughtSignature`.
414
+ */
415
+ thoughtSignature?: string;
105
416
  }
106
417
  /** A candidate returned by Gemini generateContent. */
107
418
  interface GeminiCandidateShape {
@@ -114,6 +425,14 @@ interface GeminiCandidateShape {
114
425
  * "RECITATION", "BLOCKLIST", "PROHIBITED_CONTENT", etc.
115
426
  */
116
427
  finishReason?: string;
428
+ /** Human-readable detail Google sends with some finish reasons. */
429
+ finishMessage?: string;
430
+ /** Per-category safety ratings of the candidate. */
431
+ safetyRatings?: unknown[];
432
+ /** Source-attribution metadata for recited content. */
433
+ citationMetadata?: unknown;
434
+ /** Retrieval status of each URL the model was asked to read. */
435
+ urlContextMetadata?: unknown;
117
436
  /**
118
437
  * Grounding metadata returned when Google Search grounding is active.
119
438
  * Real SDK type: GroundingMetadata. Kept as `unknown` to avoid a hard
@@ -133,9 +452,23 @@ interface GeminiUsageMetadataShape {
133
452
  candidatesTokenCount?: number;
134
453
  cachedContentTokenCount?: number;
135
454
  thoughtsTokenCount?: number;
455
+ /** Tokens of tool results fed back to the model (Search results on Gemini 2.5). */
456
+ toolUsePromptTokenCount?: number;
136
457
  totalTokenCount?: number;
137
458
  /** Provider-echoed actual tier; can differ from the requested tier. */
138
459
  serviceTier?: string;
460
+ /**
461
+ * Prompt tokens per modality (`TEXT`, `IMAGE`, `VIDEO`, `AUDIO`, `DOCUMENT`);
462
+ * the counts sum to `promptTokenCount` and include the cached part.
463
+ */
464
+ promptTokensDetails?: GeminiModalityTokenCount[];
465
+ /** The cached part of the prompt per modality, when a cache was used. */
466
+ cacheTokensDetails?: GeminiModalityTokenCount[];
467
+ }
468
+ /** One entry of `usageMetadata.promptTokensDetails` / `cacheTokensDetails`. */
469
+ interface GeminiModalityTokenCount {
470
+ modality?: string;
471
+ tokenCount?: number;
139
472
  }
140
473
  /**
141
474
  * Structural equivalent of @google/genai's GenerateContentResponse.
@@ -172,6 +505,8 @@ interface GeminiPartMediaResolution {
172
505
  /** A text part in a content object we construct. */
173
506
  interface GeminiTextContentPart {
174
507
  text: string;
508
+ /** Replayed signature for this part (real field: `Part.thoughtSignature`). */
509
+ thoughtSignature?: string;
175
510
  }
176
511
  /**
177
512
  * An inline binary media part in a content object we construct.
@@ -212,6 +547,8 @@ interface GeminiFunctionCallPart {
212
547
  name: string;
213
548
  args?: unknown;
214
549
  };
550
+ /** Replayed signature for this call (real field: `Part.thoughtSignature`). */
551
+ thoughtSignature?: string;
215
552
  }
216
553
  interface GeminiFunctionResponsePart {
217
554
  functionResponse: {
@@ -226,20 +563,6 @@ interface GeminiContent {
226
563
  role: string;
227
564
  parts: GeminiContentPart[];
228
565
  }
229
- /**
230
- * Schema shape we pass as responseSchema.
231
- * Structurally compatible with @google/genai Schema.
232
- */
233
- interface GeminiSchema {
234
- type?: string;
235
- description?: string;
236
- properties?: Record<string, GeminiSchema>;
237
- required?: string[];
238
- items?: GeminiSchema;
239
- enum?: string[];
240
- nullable?: boolean;
241
- format?: string;
242
- }
243
566
  /**
244
567
  * Thinking configuration.
245
568
  * Real type: ThinkingConfig in @google/genai.
@@ -270,7 +593,12 @@ interface GeminiGenerateConfig {
270
593
  maxOutputTokens?: number;
271
594
  stopSequences?: string[];
272
595
  responseMimeType?: string;
273
- responseSchema?: GeminiSchema;
596
+ /**
597
+ * Standard JSON Schema for the response, sent verbatim (real field:
598
+ * `GenerateContentConfig.responseJsonSchema`). Never `responseSchema`, the
599
+ * OpenAPI-dialect field (ADR-034).
600
+ */
601
+ responseJsonSchema?: unknown;
274
602
  thinkingConfig?: GeminiThinkingConfig;
275
603
  /** Real type: ServiceTier enum. Values: "flex" | "standard". */
276
604
  serviceTier?: string;
@@ -281,22 +609,18 @@ interface GeminiGenerateConfig {
281
609
  * We use this to set a transport-level timeout that is >= the AbortSignal
282
610
  * deadline so the SDK fetch does not preempt the abort.
283
611
  *
284
- * Real field: GenerateContentConfig.httpOptions.timeout (milliseconds).
285
- * Real field: GenerateContentConfig.httpOptions.headers (Record<string,string>).
286
- * (No current use of custom headers here: Vertex AI auth — and the
287
- * Vertex flex-routing header injection this field once supported — was
288
- * removed from this library; see ADR-019 in DECISIONS.md. Only
289
- * API-key auth is supported below.)
612
+ * Real field: GenerateContentConfig.httpOptions.timeout (milliseconds). No
613
+ * other `httpOptions` field is admitted.
290
614
  */
291
615
  httpOptions?: {
292
616
  timeout?: number;
293
- headers?: Record<string, string>;
294
617
  };
295
618
  tools?: Array<{
296
619
  functionDeclarations?: Array<{
297
620
  name: string;
298
621
  description: string;
299
- parameters?: unknown;
622
+ /** Standard JSON Schema, verbatim (real field: `parametersJsonSchema`). */
623
+ parametersJsonSchema?: unknown;
300
624
  }>;
301
625
  googleSearch?: Record<string, never>;
302
626
  }>;
@@ -317,23 +641,28 @@ interface GeminiGenerateParams {
317
641
  config?: GeminiGenerateConfig;
318
642
  }
319
643
  /**
320
- * Parameters for models.countTokens.
321
- * Real type: CountTokensParameters.
644
+ * Parameters for counting tokens.
645
+ *
646
+ * With only `model` and `contents` the call is the SDK's `models.countTokens`.
647
+ * With `systemInstruction` or `tools` the Developer API's SDK method cannot
648
+ * carry them (it throws), so `buildGoogleClient` sends the REST `countTokens`
649
+ * with a full `generateContentRequest` instead; the two forms are mutually
650
+ * exclusive on the wire, so `contents` then travels inside the request.
322
651
  */
323
652
  interface GeminiCountTokensParams {
324
653
  model: string;
325
654
  contents: GeminiContent[];
655
+ systemInstruction?: {
656
+ parts: GeminiContentPart[];
657
+ };
658
+ tools?: NonNullable<GeminiGenerateConfig['tools']>;
326
659
  config?: {
327
- systemInstruction?: {
328
- parts: GeminiContentPart[];
329
- };
330
660
  /**
331
661
  * Real field: CountTokensConfig.abortSignal. countTokens has no
332
- * tier-timeout dance (no flex/standard default ceilings) — `ctx.signal`
662
+ * tier-timeout dance (no flex/standard default ceilings): `ctx.signal`
333
663
  * is forwarded here directly, unlike `run()`'s combined timer signal.
334
664
  */
335
665
  abortSignal?: AbortSignal;
336
- tools?: GeminiGenerateConfig['tools'];
337
666
  };
338
667
  }
339
668
  /**
@@ -385,14 +714,6 @@ interface GeminiAdapterOptions {
385
714
  * as a typed `LlmError`.
386
715
  */
387
716
  client?: GeminiClientLike;
388
- /**
389
- * @internal Testing-only.
390
- *
391
- * Override the default `buildGoogleClient` factory. Allows unit tests to
392
- * simulate construction failures (e.g. bad credentials) without importing
393
- * the real `@google/genai` SDK. Never set this in production code.
394
- */
395
- _clientFactory?: (auth: AuthMaterial) => GeminiClientLike | Promise<GeminiClientLike>;
396
717
  }
397
718
  /**
398
719
  * Create a Gemini provider adapter.
@@ -401,6 +722,59 @@ interface GeminiAdapterOptions {
401
722
  */
402
723
  declare function geminiAdapter(opts?: GeminiAdapterOptions): ProviderAdapter;
403
724
 
725
+ /**
726
+ * classifyGoogleError — reclassify a raw thrown value into a typed
727
+ * {@link LlmError} for @gullabs/google.
728
+ *
729
+ * Thin wrapper around `@gullabs/core`'s `classifyError`, which already routes
730
+ * by HTTP status first, treats a transport failure (`fetch failed`, an errno on
731
+ * the cause chain) as a retryable `server` error and maps 404/413 to
732
+ * `bad_request`. This module adds the overlays Gemini's structured error body
733
+ * supports (ADR-028 style: only the parsed body, never free text):
734
+ *
735
+ * - `RetryInfo.retryDelay` becomes `retryAfterMs` (the SDK's `ApiError` keeps no
736
+ * headers, so the body is the only place Google puts the delay).
737
+ * - A `QuotaFailure` whose quota id contains `PerDay` is a daily quota: it
738
+ * cannot recover before the next day, so it is `rate_limited` with
739
+ * `retryable: false` and `reason: 'daily_quota'`.
740
+ * - `ErrorInfo.reason` `API_KEY_INVALID` / `API_KEY_EXPIRED` is `invalid_auth`.
741
+ * Google sends the invalid-key case as HTTP 400, which would otherwise read
742
+ * as a caller bug (live capture, probe P6, 2026-10-03). `API_KEY_EXPIRED` is
743
+ * mapped from Google's documented reason set; an expired key could not be
744
+ * produced to capture it.
745
+ * - A 403 whose body message is `CachedContent not found (or permission
746
+ * denied)` is a stale `cachedContent` reference: `bad_request` with
747
+ * `reason: 'cache_not_found'` (probe P6). Google sends no structured reason
748
+ * for it, so the body's `message` field is the only signal, and a genuine
749
+ * permission failure cannot be told apart.
750
+ *
751
+ * @module
752
+ */
753
+
754
+ /** Optional extra fields threaded onto the returned {@link LlmError}. */
755
+ interface ClassifyGoogleErrorExtra {
756
+ /** Service tier actually attempted by the provider when known. */
757
+ servedServiceTier?: string;
758
+ /**
759
+ * Set by the adapter when its own client-side ceiling, or the SDK's transport
760
+ * timer, ended the call (the caller and the engine's deadline had not
761
+ * aborted): what happened, in words. The error is then a `timeout` that is not
762
+ * retryable, `reason: 'transport_timeout'`.
763
+ */
764
+ transportTimeout?: string;
765
+ }
766
+ /**
767
+ * Classify a raw error thrown from a `@google/genai` client call into a
768
+ * typed {@link LlmError} always tagged `provider: 'google'`.
769
+ *
770
+ * Delegates to `@gullabs/core`'s `classifyError` (an already-classified
771
+ * `LlmError` passes through unchanged), applies the structured-body overlays
772
+ * described in the module header, and rebuilds the result with
773
+ * `provider: 'google'` forced on, so every error surfaced by this adapter is
774
+ * tagged even one injected pre-classified.
775
+ */
776
+ declare function classifyGoogleError(rawErr: unknown, extra?: ClassifyGoogleErrorExtra): LlmError;
777
+
404
778
  /**
405
779
  * `googleProvider` — {@link ProviderPlugin} factory for @gullabs/google.
406
780
  *
@@ -458,8 +832,8 @@ declare const defaultGeminiRegistry: ModelRegistry;
458
832
  * Provides `geminiPricingSource` — a factory returning a `PricingSource` port
459
833
  * implementation backed by the frozen Gemini pricing snapshot ({@link
460
834
  * GEMINI_PRICING}). Uses exact priced model identifiers,
461
- * resolves the concrete per-tier rates, and delegates the arithmetic to
462
- * `@gullabs/core`'s `computeCost`. Core itself carries zero Gemini pricing
835
+ * resolves the concrete per-tier rates, and delegates the token arithmetic to
836
+ * `@gullabs/core`'s `computeCost` (audio input and grounding are priced here). Core itself carries zero Gemini pricing
463
837
  * knowledge and applies no tier multiplier.
464
838
  *
465
839
  * @module
@@ -492,12 +866,13 @@ declare function geminiPricingSource(): PricingSource;
492
866
  * All rates are in **micro-USD per million tokens** (µUSD/M).
493
867
  * To get the cost for N tokens: `cost_µUSD = N * ratePerM / 1_000_000`.
494
868
  *
495
- * **Service tiers.** Each model stores concrete `standard`, `flex`, and
496
- * `batch` rates transcribed from Google's pricing page. Flex and batch are
497
- * not a flat 50% of standard: on several models the cached lane stays at the
498
- * standard cached rate (or a published rate that is not half). A tier that
499
- * is not one of those three is unpriced (reject-don't-map). `priority` is
500
- * intentionally absent — it needs downgrade accounting and is a backlog item.
869
+ * **Service tiers.** Each model stores concrete `standard` and `flex` rates
870
+ * transcribed from Google's pricing page. Flex is not a flat 50% of standard: on
871
+ * several models the cached lane stays at the standard cached rate (or a
872
+ * published rate that is not half). A tier that is not one of those two is
873
+ * unpriced (reject-don't-map). `priority` is intentionally absent: it needs
874
+ * downgrade accounting and is a backlog item. The Batch API has no path in this
875
+ * library (no schema admits a batch tier), so its rates are not carried.
501
876
  *
502
877
  * **Long-context tier.** Gemini Pro models charge a premium when the GROSS
503
878
  * input token count exceeds 200,000. Selected by `inputTokens` (incl. cached),
@@ -506,14 +881,30 @@ declare function geminiPricingSource(): PricingSource;
506
881
  * **Thinking tokens.** Already inside `outputTokens` (GROSS convention) and
507
882
  * billed at the output rate — no separate thinking lane.
508
883
  *
509
- * **Modality caveat (v1 = text).** Gemini 2.5 Flash / Flash-Lite / 3.1
510
- * Flash-Lite charge a higher INPUT rate for audio tokens
511
- * than for text/image/video. v1 is text-only and uses the text/img/vid input
512
- * rate. Per-modality input pricing is a deferred seam (see DESIGN.md).
884
+ * **Audio input.** Gemini 2.5 Flash, 2.5 Flash-Lite and 3.1 Flash-Lite charge
885
+ * more for audio input than for text, image and video tokens ({@link GeminiRates.audio}).
886
+ * The adapter records the prompt's per-modality token counts
887
+ * (`usageMetadata.promptTokensDetails` and `cacheTokensDetails`) as
888
+ * `usage.details.input_<modality>` / `cached_<modality>`, and the pricing source
889
+ * bills the audio tokens at the audio rates and the rest at the text rate. Every
890
+ * other model is billed one input rate for all modalities on the page. Models with
891
+ * an audio rate have no `gt200k` band, so the long-context band never needs the
892
+ * audio split.
513
893
  *
514
894
  * Re-verified against https://ai.google.dev/gemini-api/docs/pricing on
515
895
  * 2026-09-25. Standard token rates for already-registered models were
516
- * unchanged from the 2026-08-12 snapshot; flex/batch cached rates were not.
896
+ * unchanged from the 2026-08-12 snapshot; flex cached rates were not. The audio
897
+ * input and cached-audio rates (standard and flex) were read from the same page
898
+ * on 2026-10-03; the page showed "Last Updated 2026-10-01 UTC".
899
+ *
900
+ * **Grounding with Google Search** is a tool lane, not a token rate; see
901
+ * {@link GEMINI_GROUNDING_PRICING}. It was added from the same pricing page
902
+ * on 2026-10-03. {@link pricingVersion} is `gemini-2026-10-03` because the
903
+ * snapshot gained the grounding lane and the audio lane that day (a grounded or
904
+ * audio call prices differently under it). The 2026-10-03 read of the page also
905
+ * matched the standard text and cached rates of the models it listed (and the
906
+ * flex text and cached rates of the three audio models); only the models the page
907
+ * summary did not list keep their 2026-09-25 verification.
517
908
  *
518
909
  * @module
519
910
  */
@@ -525,30 +916,143 @@ declare function geminiPricingSource(): PricingSource;
525
916
  * core concept — it lives here (not `@gullabs/core`) alongside the rates it
526
917
  * dates.
527
918
  */
528
- declare const pricingVersion: "gemini-2026-09-25";
919
+ declare const pricingVersion: "gemini-2026-10-03";
529
920
  /** Tiers this snapshot prices. Anything else is unpriced. */
530
- declare const GEMINI_PRICED_TIERS: readonly ["standard", "flex", "batch"];
921
+ declare const GEMINI_PRICED_TIERS: readonly ["standard", "flex"];
531
922
  type GeminiPricedTier = (typeof GEMINI_PRICED_TIERS)[number];
923
+ /** Audio input rates for a model that prices audio apart from text (µUSD per million tokens). */
924
+ interface GeminiAudioRates {
925
+ /** Non-cached audio input tokens. */
926
+ inputPerM: number;
927
+ /** Cached audio input tokens. */
928
+ cachedPerM: number;
929
+ }
930
+ /** {@link ModelRates} plus the audio input rates, for models that publish them. */
931
+ interface GeminiRates extends ModelRates {
932
+ /**
933
+ * Present only on a model whose pricing page lists a separate audio input
934
+ * price. Priced on the audio tokens `promptTokensDetails` reports; every other
935
+ * input token uses the text/image/video rates above.
936
+ */
937
+ audio?: GeminiAudioRates;
938
+ }
532
939
  /** Concrete per-tier rates for one model. */
533
940
  interface GeminiTierRates {
534
- standard: ModelRates;
535
- flex: ModelRates;
536
- batch: ModelRates;
941
+ standard: GeminiRates;
942
+ flex: GeminiRates;
537
943
  }
538
944
  /**
539
- * Frozen Gemini pricing snapshot (per-1M in µUSD), keyed by model id, then
540
- * by priced tier. Every number is transcribed from the pricing page.
945
+ * Deep-frozen Gemini pricing snapshot (per-1M in µUSD), keyed by model id, then
946
+ * by priced tier; no rate object, `gt200k` band or `audio` rate can be changed.
947
+ * Every number is transcribed from the pricing page.
541
948
  *
542
949
  * Keys are exact priced model identifiers. Unlisted variants are unpriced.
543
950
  *
544
- * Source: https://ai.google.dev/gemini-api/docs/pricing (re-verified 2026-09-25).
951
+ * Source: https://ai.google.dev/gemini-api/docs/pricing (re-verified 2026-09-25;
952
+ * the audio input and cached-audio rates read 2026-10-03, page last updated
953
+ * 2026-10-01).
545
954
  */
546
955
  declare const GEMINI_PRICING: Readonly<Record<string, GeminiTierRates>>;
547
- /** Resolve the concrete {@link ModelRates} `computeCost` should apply. */
548
- declare function resolveGeminiRates(model: string, tier: string | undefined): ModelRates | undefined;
956
+ /**
957
+ * How a model bills grounding with Google Search, in µUSD per unit.
958
+ *
959
+ * - `'query'`: Gemini 3 bills each search query the model performed. The unit
960
+ * count is `usage.details.web_search_calls`, counted as occurrences in
961
+ * `webSearchQueries` (a repeated query counts each time; whether Google bills
962
+ * a repeat is not established, so the count is the conservative one).
963
+ * - `'prompt'`: Gemini 2.5 bills each prompt that was grounded, once however
964
+ * many queries it ran. A grounded prompt is one whose response reports at
965
+ * least one query.
966
+ *
967
+ * Transcribed from https://ai.google.dev/gemini-api/docs/pricing, grounding
968
+ * with Google Search, read 2026-10-03 (Gemini 3: $14 per 1,000 queries;
969
+ * Gemini 2.5: $35 per 1,000 grounded prompts). The page also publishes a free
970
+ * allowance (as read 2026-10: 5,000 requests per month shared
971
+ * across Gemini 3.x, 1,500 requests per day on Gemini 2.5). It is shared across
972
+ * a project's calls, so no single call can know whether it was free: every
973
+ * grounding fee is charged in full here, which is why a call that ran Search is
974
+ * never reported as exact.
975
+ *
976
+ * Keys are exact priced model identifiers. Gemma has no token price in this
977
+ * snapshot, so it has no grounding price either.
978
+ */
979
+ interface GeminiGroundingRate {
980
+ readonly unit: 'query' | 'prompt';
981
+ readonly microUsdPerUnit: number;
982
+ }
983
+ declare const GEMINI_GROUNDING_PRICING: Readonly<Record<string, GeminiGroundingRate>>;
984
+ /** Resolve the concrete {@link GeminiRates} for `(model, tier)`. */
985
+ declare function resolveGeminiRates(model: string, tier: string | undefined): GeminiRates | undefined;
549
986
 
987
+ /**
988
+ * True when a failed Flex call may be retried once on the Standard tier
989
+ * because Flex capacity, not the caller's quota, ran out.
990
+ *
991
+ * Only HTTP 503 counts. Google's Flex page (ai.google.dev/gemini-api/docs/
992
+ * flex-inference, read 2026-10-03, page dated 2026-09-23) lists two failures
993
+ * when capacity is unavailable, 503 "The system is currently at capacity" and
994
+ * 429 "Rate limits or resource exhaustion", but documents no field that tells
995
+ * a capacity 429 from a quota 429, and no capture of either 429 exists. A 429
996
+ * is therefore the ordinary rate-limit path (the provider's `RetryInfo` delay
997
+ * is honoured, no Standard dispatch, no tier pin): dispatching Standard at once
998
+ * would undercut the delay, add a call to a rate-limited project and bill the
999
+ * logical call at the Standard rate. Nothing in the message text is read.
1000
+ *
1001
+ * `err` is the classified error.
1002
+ */
550
1003
  declare function isGeminiCapacityError(err: LlmError): boolean;
551
1004
 
1005
+ /**
1006
+ * Token limits and admitted input media types for the registered Google
1007
+ * models, read from Google's own documentation (re-read 2026-10-03).
1008
+ *
1009
+ * One table feeds both the descriptors (`limits`, `inputMimeTypes`) and the
1010
+ * per-model config schemas (`maxOutputTokens` is capped only when a number is
1011
+ * documented), so a descriptor and its schema cannot drift.
1012
+ *
1013
+ * Sources, all read 2026-10-03:
1014
+ *
1015
+ * - Gemini: `https://ai.google.dev/gemini-api/docs/models/<model-id>`, one page
1016
+ * per model. Every registered Gemini model lists "Input token limit
1017
+ * 1,048,576" and "Output token limit 65,536". The output limit includes
1018
+ * thinking tokens.
1019
+ * - Gemma 4: the model card, `https://ai.google.dev/gemma/docs/core/model_card_4`,
1020
+ * states a "256K tokens" context window (read as 262,144). Neither it nor
1021
+ * `https://ai.google.dev/gemma/docs/core/gemma_on_gemini_api` documents an
1022
+ * output limit, so `maxOutputTokens` is `null` (see
1023
+ * {@link ModelLimits.maxOutputTokens}): no figure is invented and the schema
1024
+ * applies no cap.
1025
+ * - Gemini input media: Google publishes lists for images
1026
+ * (`https://ai.google.dev/gemini-api/docs/image-understanding`: PNG, JPEG,
1027
+ * WebP, HEIC, HEIF), audio (`.../audio`), and video
1028
+ * (`.../video-understanding`), but no closed list for documents:
1029
+ * `.../document-processing` says only that PDF is understood natively and
1030
+ * "you can pass other MIME types for document understanding, like TXT,
1031
+ * Markdown, HTML, XML, etc.", extracted as plain text. `.../files` lists no
1032
+ * types either. The descriptor therefore admits the documented families by
1033
+ * prefix (`text/*`, `image/*`, `audio/*`, `video/*`) plus `application/pdf`,
1034
+ * and leaves a type inside a family that Google does not accept to Google's
1035
+ * own error. `application/json`, `application/xml` and other `application/*`
1036
+ * types are not in any documented family and stay rejected.
1037
+ * - Gemma 4 input media: the model card lists "Supported Modalities: Text,
1038
+ * Image" for the 31B and 26B A4B models and says "All models support image
1039
+ * inputs and can process videos as frames", with video "a maximum of 60
1040
+ * seconds" at one frame per second; audio input is E2B/E4B/12B only. No Gemma
1041
+ * page names image or video media types, so `image/*` and `video/*` are
1042
+ * admitted and nothing else. That the Gemini API's Gemma endpoint takes a
1043
+ * video part (rather than frames sent as images) has not been probed.
1044
+ *
1045
+ * @module
1046
+ */
1047
+
1048
+ /**
1049
+ * Media types every Gemini model in the registry accepts as input parts: the
1050
+ * documented families (text, image, audio, video) and PDF. See the module
1051
+ * comment for why the families are not closed lists. One rule serves
1052
+ * `generate` and `GoogleFileStore.upload`.
1053
+ */
1054
+ declare const GEMINI_INPUT_MIME_TYPES: readonly string[];
1055
+
552
1056
  /**
553
1057
  * GoogleFileStore — thin wrapper over the Gemini File API.
554
1058
  *
@@ -568,6 +1072,20 @@ interface GoogleFileHandle {
568
1072
  /** Provider auto-deletes ~48 h after upload. Absent when not returned. */
569
1073
  expiresAt?: Date;
570
1074
  }
1075
+ /** The `File` resource fields the store reads. */
1076
+ type FileResp = {
1077
+ name?: string;
1078
+ uri?: string;
1079
+ mimeType?: string;
1080
+ state?: string;
1081
+ expirationTime?: string;
1082
+ /** Real field `File.error` (`FileStatus`): why processing failed. */
1083
+ error?: {
1084
+ code?: number;
1085
+ message?: string;
1086
+ details?: Record<string, unknown>[];
1087
+ };
1088
+ };
571
1089
  /**
572
1090
  * Minimal structural interface for the Gemini Files client surface we use.
573
1091
  * Satisfied by the real ai.files object or a test fake.
@@ -578,23 +1096,12 @@ interface GeminiFilesClientLike {
578
1096
  config?: {
579
1097
  mimeType?: string;
580
1098
  displayName?: string;
1099
+ abortSignal?: AbortSignal;
581
1100
  };
582
- }): Promise<{
583
- name?: string;
584
- uri?: string;
585
- mimeType?: string;
586
- state?: string;
587
- expirationTime?: string;
588
- }>;
1101
+ }): Promise<FileResp>;
589
1102
  get(params: {
590
1103
  name: string;
591
- }): Promise<{
592
- name?: string;
593
- uri?: string;
594
- mimeType?: string;
595
- state?: string;
596
- expirationTime?: string;
597
- }>;
1104
+ }): Promise<FileResp>;
598
1105
  delete(params: {
599
1106
  name: string;
600
1107
  }): Promise<void>;
@@ -629,7 +1136,18 @@ interface GoogleFileStoreOptions {
629
1136
  /** Max time to wait for ACTIVE. Default: 300 000 ms (5 min). */
630
1137
  timeoutMs?: number;
631
1138
  };
632
- /** Injectable sleep for tests. Default: real setTimeout. */
1139
+ /**
1140
+ * Timer source for the poll wait; pass the client's `FakeClock` in tests so
1141
+ * one `advance` fires the wait. Default: the platform's timers. (`now` is the
1142
+ * poll timeout's clock; pass `() => clock.now()` with it.)
1143
+ */
1144
+ scheduler?: Scheduler;
1145
+ /**
1146
+ * Replaces the poll wait wholesale (instant polling in tests). When given,
1147
+ * `scheduler` is not used for the wait, and the wait cannot be cancelled: if
1148
+ * the deadline or an abort ends the upload first, a timer your `sleep` set
1149
+ * stays pending until it fires. The default wait is cleared at once.
1150
+ */
633
1151
  sleep?: (ms: number) => Promise<void>;
634
1152
  /** Injectable clock for deterministic tests. Default: `Date.now`. */
635
1153
  now?: () => number;
@@ -652,7 +1170,8 @@ declare class GoogleFileStore {
652
1170
  private readonly logger;
653
1171
  private readonly intervalMs;
654
1172
  private readonly timeoutMs;
655
- private readonly sleep;
1173
+ private readonly startWait;
1174
+ private readonly scheduler;
656
1175
  private readonly now;
657
1176
  /** Memoised client promise — built at most once per store instance. */
658
1177
  private clientPromise;
@@ -662,13 +1181,26 @@ declare class GoogleFileStore {
662
1181
  * Upload bytes to the Gemini File API and wait until the file is ACTIVE.
663
1182
  *
664
1183
  * @param source - Raw bytes or Blob.
665
- * @param mimeType - IANA media type, e.g. `"image/png"`.
1184
+ * @param mimeType - IANA media type, e.g. `"image/png"`. It must pass the same
1185
+ * admission rule `generate` applies to a Gemini model's parts (one shared
1186
+ * function, so a file that uploads can be used): an empty or unadmitted type
1187
+ * is `bad_request` before any bytes are sent. The string is sent to Google
1188
+ * unchanged.
666
1189
  * @param opts - Optional display name.
667
1190
  */
668
1191
  upload(source: Uint8Array | Blob, mimeType: string, opts?: {
669
1192
  displayName?: string;
670
1193
  signal?: AbortSignal;
671
1194
  }): Promise<GoogleFileHandle>;
1195
+ /**
1196
+ * `work` raced against the abort promise and the time left until `deadline`
1197
+ * (the deadline is a non-retryable `server` error). `work` is observed, so a
1198
+ * late rejection after the race is lost is not unhandled; the deadline timer
1199
+ * is always cleared.
1200
+ */
1201
+ private raceDeadline;
1202
+ /** Polls `name` until ACTIVE; see {@link GoogleFileStore.upload}. */
1203
+ private pollUntilActive;
672
1204
  /**
673
1205
  * Delete a single uploaded file. Idempotent: not-found → success.
674
1206
  *
@@ -690,208 +1222,62 @@ declare class GoogleFileStore {
690
1222
  }
691
1223
 
692
1224
  /**
693
- * GoogleCacheStore — thin wrapper over the Gemini Context Cache API.
1225
+ * Gemini 3 thought signatures as an overlay on the host's actual history.
694
1226
  *
695
- * Manages create / get-or-create / refresh / delete lifecycle for cached
696
- * contents. In-memory cache entries are PROCESS-SCOPED; they are not shared
697
- * across processes, workers, or restarts.
1227
+ * Gemini 3.x returns an opaque `thoughtSignature` on the first function call of
1228
+ * each model turn (and sometimes on text parts) and rejects a replayed turn
1229
+ * whose function call has lost it. The library never keeps a copy of the
1230
+ * history. `result.transientProviderState` is only an overlay saying which part
1231
+ * of the host's own messages gets which signature:
698
1232
  *
699
- * Injectable client and `now` function keep tests free of network and clock.
1233
+ * ```ts
1234
+ * { google: { signatures: [{ messageIndex, partIndex, kind, model, partSha256, signature }] } }
1235
+ * ```
1236
+ *
1237
+ * `messageIndex` indexes the messages the adapter receives (`request.messages`
1238
+ * as the engine hands them on, after any middleware), `partIndex` indexes that
1239
+ * message's `parts` (thought parts are never in a message), `kind` is the part
1240
+ * kind that was signed (`'text'` or `'tool-call'`), `model` is the model string
1241
+ * the request named, and `partSha256` is the SHA-256 of the part's RFC 8785
1242
+ * canonical JSON, so an edited argument or text is detected and key order
1243
+ * (Postgres `jsonb`) does not matter.
1244
+ *
1245
+ * A function-call signature is required on replay, so a stale one is
1246
+ * `bad_request`. A text signature is optional (Google accepts the next turn
1247
+ * without it), so a stale one (edited, moved, removed, or issued for another
1248
+ * model) is dropped with a warning and nothing else is lost.
700
1249
  *
701
1250
  * @module
702
1251
  */
703
1252
 
704
- /**
705
- * A handle to a cached content resource in the Gemini Context Cache API.
706
- *
707
- * Pass `cacheName` as `providerOptions.google.cachedContent` in LlmRequest.
708
- */
709
- interface GoogleCacheHandle {
710
- /** Resource name, e.g. "cachedContents/xyz". */
711
- cacheName: string;
712
- /**
713
- * Local view of when the cache expires (server-authoritative; computed from
714
- * ttl or parsed from the server response `expireTime`).
715
- */
716
- expiresAt: Date;
717
- /** Caches are model-bound; never use a handle with a different model. */
718
- model: string;
719
- }
720
- /** Key used to look up or create an entry in the in-process cache map. */
721
- interface CacheKey {
1253
+ /** The part kinds Google signs. */
1254
+ type GoogleSignedKind = 'text' | 'tool-call';
1255
+ /** One signature, pinned to a part of the host's history. */
1256
+ type GoogleSignatureEntry = {
1257
+ messageIndex: number;
1258
+ partIndex: number;
1259
+ /** The kind of part that was signed; decides whether a stale entry is fatal. */
1260
+ kind: GoogleSignedKind;
722
1261
  model: string;
723
- /** Caller-chosen stable identifier, e.g. a hash of the cached content. */
724
- stableKey: string;
725
- }
726
- /**
727
- * Minimal structural interface for the Gemini Caches client surface we use.
728
- * Satisfied by the real `ai.caches` object or a test fake.
729
- */
730
- interface GeminiCachesClientLike {
731
- create(params: {
732
- model: string;
733
- config: {
734
- contents?: Content[];
735
- systemInstruction?: Content | string;
736
- ttl?: string;
737
- displayName?: string;
738
- };
739
- }): Promise<{
740
- name?: string;
741
- model?: string;
742
- expireTime?: string;
743
- }>;
744
- update(params: {
745
- name: string;
746
- config: {
747
- ttl?: string;
748
- expireTime?: string;
749
- };
750
- }): Promise<{
751
- name?: string;
752
- expireTime?: string;
753
- }>;
754
- delete(params: {
755
- name: string;
756
- }): Promise<unknown>;
757
- }
758
- interface GoogleCacheStoreOptions {
759
- auth: AuthMaterial;
760
- /** Injectable client for tests; skips SDK import when provided. */
761
- client?: GeminiCachesClientLike;
762
- /**
763
- * When true, concurrent `getOrCreate` calls for the same key share one
764
- * in-flight create. Default false.
765
- */
766
- coalesce?: boolean;
767
- /**
768
- * Subtracted from the local expiry so we stop using a cache slightly before
769
- * the server evicts it. Default 30 s.
770
- */
771
- expirySkewSeconds?: number;
772
- /** Called on delete failures instead of rethrowing. */
773
- onDeleteError?: (cacheName: string, err: unknown) => void;
774
- /** Optional structured logger. When provided, routes delete failures to logger.error. */
775
- logger?: Logger;
776
- /** Injectable clock for deterministic tests. Default: `Date.now`. */
777
- now?: () => number;
778
- /**
779
- * Opt-in pre-flight token-count gate applied before every cache create
780
- * (both `create()` directly and the `getOrCreate()` path it delegates to,
781
- * including the coalesced path — enforced once, inside `create()`, so
782
- * there is no separate "in-flight" gap to close).
783
- */
784
- preflight?: {
785
- /** Minimum token count required before a cache create is allowed to proceed. */
786
- minTokens: number;
787
- /**
788
- * Counts tokens for the exact token-bearing payload of the impending
789
- * create — `model` + `contents` + `systemInstruction` only. `ttl` and
790
- * `displayName` are excluded: they carry no tokens and are irrelevant to
791
- * the pre-flight check.
792
- *
793
- * This callback receives genai-native `Content[]`/`Content|string` — it
794
- * does NOT receive the library's `Message[]` shape and there is no
795
- * automatic conversion between the two (explicit seam, by design). Hosts
796
- * using genai-native content directly can wire this to a raw
797
- * `client.models.countTokens` call; hosts building from the library's
798
- * `Message[]` should call `@gullabs/core`'s `client.countTokens` rather
799
- * than expecting this callback to convert for them.
800
- */
801
- countTokens: (payload: {
802
- model: string;
803
- contents?: Content[];
804
- systemInstruction?: Content | string;
805
- }) => Promise<number>;
1262
+ partSha256: string;
1263
+ signature: string;
1264
+ };
1265
+ /** The shape of `transientProviderState` for Gemini 3.x models. */
1266
+ type GoogleSignatureState = {
1267
+ google: {
1268
+ signatures: GoogleSignatureEntry[];
806
1269
  };
807
- }
1270
+ };
808
1271
  /**
809
- * Process-scoped helper for the Gemini Context Cache API.
810
- *
811
- * NOTE: `getOrCreate` reuse is PROCESS-SCOPED only. Entries survive only for
812
- * the lifetime of this `GoogleCacheStore` instance. Across restarts, new
813
- * caches will be created (and old ones will be server-evicted after their TTL).
814
- *
815
- * **Auth snapshot note:** this store captures the `AuthMaterial` at construction
816
- * time and memoizes a single SDK client from it (`clientPromise`). This is
817
- * correct and sufficient for static API keys. If refreshable credentials
818
- * (short-lived OAuth/STS tokens) are added in the future, this memoization is
819
- * the seam that will need rework: the cached client would hold stale credentials
820
- * for the lifetime of a long-lived store instance. At that point, the store
821
- * will need to either rebuild the client on each operation or accept a
822
- * credential-resolver callback rather than a plain `AuthMaterial` value.
823
- * See ADR-020 in DECISIONS.md.
1272
+ * Remove the entries for messages the host removed from its history, and shift
1273
+ * the `messageIndex` of every later entry down so it still points at the same
1274
+ * message. `indices` are positions in the history the state was issued for
1275
+ * (before the removal). Whole turns only: remove a tool-call message together
1276
+ * with its tool-result message, and never keep a tool-call message while
1277
+ * removing its entry. Returns `undefined` when no entry remains, so the result
1278
+ * can be sent as `transientProviderState` or omitted.
824
1279
  */
825
- declare class GoogleCacheStore {
826
- private readonly auth;
827
- private readonly clientOverride;
828
- private readonly coalesce;
829
- private readonly skewMs;
830
- private readonly onDeleteError;
831
- private readonly logger;
832
- private readonly now;
833
- private readonly preflight;
834
- /** Memoised client promise — built at most once per store instance. */
835
- private clientPromise;
836
- /** In-process cache of live handles, keyed by `${model}:${stableKey}`. */
837
- private readonly entries;
838
- /** In-flight create promises when coalescing is enabled. */
839
- private readonly inflight;
840
- constructor(opts: GoogleCacheStoreOptions);
841
- private getClient;
842
- private isLive;
843
- /**
844
- * Create a new cached content resource.
845
- *
846
- * The `expiresAt` on the returned handle is computed from the server's
847
- * `expireTime` when available, with a local-clock fallback of
848
- * `now + ttlSeconds * 1000`.
849
- */
850
- create(input: {
851
- model: string;
852
- ttlSeconds: number;
853
- contents?: Content[];
854
- systemInstruction?: Content | string;
855
- displayName?: string;
856
- }): Promise<GoogleCacheHandle>;
857
- /**
858
- * Return a live cached-content handle, creating one if none exists or the
859
- * cached entry has expired (accounting for skew).
860
- *
861
- * Reuse is PROCESS-SCOPED only — this store instance's in-memory map.
862
- *
863
- * When `coalesce` is enabled, concurrent calls for the same key share one
864
- * in-flight create.
865
- */
866
- getOrCreate(key: CacheKey, factory: () => Promise<{
867
- ttlSeconds: number;
868
- contents?: Content[];
869
- systemInstruction?: Content | string;
870
- }>): Promise<GoogleCacheHandle>;
871
- /**
872
- * Extend the cache TTL if it will expire within `thresholdSeconds`.
873
- *
874
- * Fail-open: if the update call throws, the original handle is returned
875
- * unchanged. This method NEVER throws.
876
- *
877
- * @param handle - Handle to potentially refresh.
878
- * @param opts.thresholdSeconds - Extend if expiry is within this many seconds.
879
- * Default 300.
880
- * @param opts.extensionSeconds - New TTL to set. Default: original TTL from
881
- * entries map, or 3600 s when the handle was not created via `getOrCreate`.
882
- */
883
- refreshIfExpiringSoon(handle: GoogleCacheHandle, opts?: {
884
- thresholdSeconds?: number;
885
- extensionSeconds?: number;
886
- }): Promise<GoogleCacheHandle>;
887
- /**
888
- * Delete a cached content resource.
889
- *
890
- * Errors are forwarded to `onDeleteError` and NOT rethrown.
891
- * The handle is removed from the in-process entries map regardless.
892
- */
893
- delete(handle: GoogleCacheHandle): Promise<void>;
894
- }
1280
+ declare function dropMessagesFromSignatureState(state: unknown, indices: readonly number[]): GoogleSignatureState | undefined;
895
1281
 
896
1282
  /**
897
1283
  * geminiContentToMessages — migration utility: `@google/genai` `Content[]` →
@@ -910,7 +1296,10 @@ declare class GoogleCacheStore {
910
1296
  * Validation is an exhaustive own-key scan: the set of defined keys on each
911
1297
  * `Part` must be EXACTLY one of the recognized combinations (`['text']`,
912
1298
  * `['inlineData']`, `['inlineData', 'mediaResolution']`, `['fileData']`,
913
- * `['fileData', 'mediaResolution']`). Any other key — including unknown
1299
+ * `['fileData', 'mediaResolution']`), plus `functionCall` / `functionResponse`.
1300
+ * A `thoughtSignature` on a model text or `functionCall` part is imported into
1301
+ * `transientProviderState` (see {@link GeminiContentToMessagesInput.model});
1302
+ * on any other part it throws. Any other key — including unknown
914
1303
  * future SDK fields — or any combination outside that set throws. Keys whose
915
1304
  * value is `undefined` are treated as absent (genai types are all-optional;
916
1305
  * only defined values count).
@@ -930,6 +1319,12 @@ interface GeminiContentToMessagesInput {
930
1319
  * non-text part throws).
931
1320
  */
932
1321
  systemInstruction?: Content | string;
1322
+ /**
1323
+ * The model string the converted history will be sent to. Required when any
1324
+ * part carries a `thoughtSignature`: signatures are bound to the model that
1325
+ * issued them and are never replayed on another one.
1326
+ */
1327
+ model?: string;
933
1328
  }
934
1329
  /**
935
1330
  * Output produced by {@link geminiContentToMessages}.
@@ -939,6 +1334,13 @@ interface GeminiContentToMessagesResult {
939
1334
  system?: string;
940
1335
  /** Normalized any-llm messages, one per input `Content`. */
941
1336
  messages: Message[];
1337
+ /**
1338
+ * The thought signatures found on model parts, as the overlay a Gemini 3.x
1339
+ * request takes as `transientProviderState`. Present only when at least one
1340
+ * part carried a `thoughtSignature`. Send it with `messages` unedited and the
1341
+ * same `model`.
1342
+ */
1343
+ transientProviderState?: JsonValue;
942
1344
  }
943
1345
  /**
944
1346
  * Convert `@google/genai` `Content[]` / `Part[]` into any-llm's normalized
@@ -974,4 +1376,4 @@ interface GeminiContentToMessagesResult {
974
1376
  */
975
1377
  declare function geminiContentToMessages(input: GeminiContentToMessagesInput): GeminiContentToMessagesResult;
976
1378
 
977
- export { type CacheKey, FLEX_DEFAULT_TIMEOUT_MS, type FileDeleteOptions, GEMINI_PRICED_TIERS, GEMINI_PRICING, type GeminiAdapterOptions, type GeminiCachesClientLike, type GeminiClientLike, type GeminiContentToMessagesInput, type GeminiContentToMessagesResult, type GeminiCountTokensParams, type GeminiCountTokensResponseShape, type GeminiFilesClientLike, type GeminiPricedTier, type GeminiTierRates, type GoogleCacheHandle, GoogleCacheStore, type GoogleCacheStoreOptions, type GoogleFileHandle, GoogleFileStore, type GoogleFileStoreOptions, type GoogleProviderOptions, type GoogleSafetySetting, type GoogleSearchTool, STANDARD_DEFAULT_TIMEOUT_MS, TRANSPORT_TIMEOUT_BUFFER_MS, buildGoogleClient, defaultGeminiRegistry, geminiAdapter, geminiContentToMessages, geminiModelDescriptors, geminiPricingSource, gemmaModelDescriptors, googleProvider, isGeminiCapacityError, pricingVersion, requireApiKey, resolveGeminiRates };
1379
+ export { type CacheKey, type ClassifyGoogleErrorExtra, FLEX_DEFAULT_TIMEOUT_MS, type FileDeleteOptions, GEMINI_GROUNDING_PRICING, GEMINI_INPUT_MIME_TYPES, GEMINI_PRICED_TIERS, GEMINI_PRICING, type GeminiAdapterOptions, type GeminiCachesClientLike, type GeminiClientLike, type GeminiContentToMessagesInput, type GeminiContentToMessagesResult, type GeminiCountTokensParams, type GeminiCountTokensResponseShape, type GeminiFilesClientLike, type GeminiGroundingRate, type GeminiPricedTier, type GeminiTierRates, type GoogleCacheHandle, GoogleCacheStore, type GoogleCacheStoreOptions, type GoogleCachedContentRef, type GoogleFileHandle, GoogleFileStore, type GoogleFileStoreOptions, type GoogleProviderOptions, type GoogleSafetySetting, type GoogleSearchTool, type GoogleSignatureEntry, type GoogleSignatureState, type GoogleSignedKind, STANDARD_DEFAULT_TIMEOUT_MS, TRANSPORT_TIMEOUT_BUFFER_MS, buildGoogleClient, classifyGoogleError, defaultGeminiRegistry, dropMessagesFromSignatureState, geminiAdapter, geminiContentToMessages, geminiModelDescriptors, geminiPricingSource, gemmaModelDescriptors, googleProvider, isGeminiCapacityError, pricingVersion, requireApiKey, resolveGeminiRates };