@gullabs/google 0.12.0 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -134,6 +134,8 @@ interface GeminiUsageMetadataShape {
134
134
  cachedContentTokenCount?: number;
135
135
  thoughtsTokenCount?: number;
136
136
  totalTokenCount?: number;
137
+ /** Provider-echoed actual tier; can differ from the requested tier. */
138
+ serviceTier?: string;
137
139
  }
138
140
  /**
139
141
  * Structural equivalent of @google/genai's GenerateContentResponse.
@@ -455,10 +457,10 @@ declare const defaultGeminiRegistry: ModelRegistry;
455
457
  *
456
458
  * Provides `geminiPricingSource` — a factory returning a `PricingSource` port
457
459
  * implementation backed by the frozen Gemini pricing snapshot ({@link
458
- * GEMINI_PRICING}). Walks the rates table (exact-then-longest-prefix match)
459
- * and delegates the actual arithmetic to `@gullabs/core`'s `computeCost`,
460
- * supplying this package's rates table + tier-factor map as explicit
461
- * parameters — core itself carries zero Gemini pricing knowledge.
460
+ * GEMINI_PRICING}). Uses exact priced model identifiers,
461
+ * resolves the concrete per-tier rates, and delegates the arithmetic to
462
+ * `@gullabs/core`'s `computeCost`. Core itself carries zero Gemini pricing
463
+ * knowledge and applies no tier multiplier.
462
464
  *
463
465
  * @module
464
466
  */
@@ -490,27 +492,28 @@ declare function geminiPricingSource(): PricingSource;
490
492
  * All rates are in **micro-USD per million tokens** (µUSD/M).
491
493
  * To get the cost for N tokens: `cost_µUSD = N * ratePerM / 1_000_000`.
492
494
  *
493
- * **Service tiers.** Rates below are STANDARD-tier. Google's **Batch** tier is a
494
- * flat 50% discount on standard, and **Flex** matches Batch pricing. The cost
495
- * engine applies the {@link TIER_FACTOR} multiplier — this snapshot stores
496
- * standard rates only.
495
+ * **Service tiers.** Each model stores concrete `standard`, `flex`, and
496
+ * `batch` rates transcribed from Google's pricing page. Flex and batch are
497
+ * not a flat 50% of standard: on several models the cached lane stays at the
498
+ * standard cached rate (or a published rate that is not half). A tier that
499
+ * is not one of those three is unpriced (reject-don't-map). `priority` is
500
+ * intentionally absent — it needs downgrade accounting and is a backlog item.
497
501
  *
498
- * **Long-context tier.** Gemini Pro models charge a premium when the GROSS input
499
- * token count exceeds 200,000. Selected by `inputTokens` (incl. cached), not by
500
- * billable input.
502
+ * **Long-context tier.** Gemini Pro models charge a premium when the GROSS
503
+ * input token count exceeds 200,000. Selected by `inputTokens` (incl. cached),
504
+ * not by billable input. Core's selector uses `> 200_000`.
501
505
  *
502
506
  * **Thinking tokens.** Already inside `outputTokens` (GROSS convention) and
503
- * billed at the standard output rate — no separate thinking lane.
507
+ * billed at the output rate — no separate thinking lane.
504
508
  *
505
- * **Modality caveat (v1 = text).** Gemini 2.5 Flash / Flash-Lite charge a higher
506
- * INPUT rate for audio tokens than for text/image/video. v1 is text-only and uses
507
- * the text/img/vid input rate. Per-modality input pricing is a deferred seam
508
- * (see DESIGN.md) — revisit when audio input is supported.
509
+ * **Modality caveat (v1 = text).** Gemini 2.5 Flash / Flash-Lite / 3.1
510
+ * Flash-Lite charge a higher INPUT rate for audio tokens
511
+ * than for text/image/video. v1 is text-only and uses the text/img/vid input
512
+ * rate. Per-modality input pricing is a deferred seam (see DESIGN.md).
509
513
  *
510
- * Re-verified against https://ai.google.dev/gemini-api/docs/pricing on 2026-08-12.
511
- * Existing registered-model standard rates were unchanged from the 2026-06-28
512
- * snapshot. `gemini-3.6-flash` appears on that page but is not a registered
513
- * model in this package.
514
+ * Re-verified against https://ai.google.dev/gemini-api/docs/pricing on
515
+ * 2026-09-25. Standard token rates for already-registered models were
516
+ * unchanged from the 2026-08-12 snapshot; flex/batch cached rates were not.
514
517
  *
515
518
  * @module
516
519
  */
@@ -522,30 +525,27 @@ declare function geminiPricingSource(): PricingSource;
522
525
  * core concept — it lives here (not `@gullabs/core`) alongside the rates it
523
526
  * dates.
524
527
  */
525
- declare const pricingVersion: "gemini-2026-08-12";
526
- /**
527
- * Service-tier price multipliers. Batch and Flex are a flat 50% of standard
528
- * (per Google's pricing page: "Batch API — 50% cost reduction"; Flex matches Batch).
529
- *
530
- * `serviceTier` is an opaque, provider-defined string end-to-end — this map is
531
- * the *only* place a tier name is resolved to a multiplier. A tier key not
532
- * present here is never coerced to `standard`: `computeCost` (`@gullabs/core`)
533
- * treats that as an unpriced call (reject-don't-map), not a mapping to this
534
- * table's default. `undefined` (no tier requested) is the one case that
535
- * legitimately defaults to `standard` — that is documented default behavior,
536
- * not a guess.
537
- */
538
- declare const TIER_FACTOR: Readonly<Record<string, number>>;
528
+ declare const pricingVersion: "gemini-2026-09-25";
529
+ /** Tiers this snapshot prices. Anything else is unpriced. */
530
+ declare const GEMINI_PRICED_TIERS: readonly ["standard", "flex", "batch"];
531
+ type GeminiPricedTier = (typeof GEMINI_PRICED_TIERS)[number];
532
+ /** Concrete per-tier rates for one model. */
533
+ interface GeminiTierRates {
534
+ standard: ModelRates;
535
+ flex: ModelRates;
536
+ batch: ModelRates;
537
+ }
539
538
  /**
540
- * Frozen Gemini pricing snapshot (STANDARD tier; per-1M in µUSD).
539
+ * Frozen Gemini pricing snapshot (per-1M in µUSD), keyed by model id, then
540
+ * by priced tier. Every number is transcribed from the pricing page.
541
541
  *
542
- * Keys are model-string prefixes / exact identifiers used in routing. The cost
543
- * engine matches exact first, then longest-prefix (see `lookupRates` in
544
- * `cost.ts`).
542
+ * Keys are exact priced model identifiers. Unlisted variants are unpriced.
545
543
  *
546
- * Source: https://ai.google.dev/gemini-api/docs/pricing (re-verified 2026-08-12).
544
+ * Source: https://ai.google.dev/gemini-api/docs/pricing (re-verified 2026-09-25).
547
545
  */
548
- declare const GEMINI_PRICING: Readonly<Record<string, ModelRates>>;
546
+ declare const GEMINI_PRICING: Readonly<Record<string, GeminiTierRates>>;
547
+ /** Resolve the concrete {@link ModelRates} `computeCost` should apply. */
548
+ declare function resolveGeminiRates(model: string, tier: string | undefined): ModelRates | undefined;
549
549
 
550
550
  declare function isGeminiCapacityError(err: LlmError): boolean;
551
551
 
@@ -974,4 +974,4 @@ interface GeminiContentToMessagesResult {
974
974
  */
975
975
  declare function geminiContentToMessages(input: GeminiContentToMessagesInput): GeminiContentToMessagesResult;
976
976
 
977
- export { type CacheKey, FLEX_DEFAULT_TIMEOUT_MS, type FileDeleteOptions, GEMINI_PRICING, type GeminiAdapterOptions, type GeminiCachesClientLike, type GeminiClientLike, type GeminiContentToMessagesInput, type GeminiContentToMessagesResult, type GeminiCountTokensParams, type GeminiCountTokensResponseShape, type GeminiFilesClientLike, type GoogleCacheHandle, GoogleCacheStore, type GoogleCacheStoreOptions, type GoogleFileHandle, GoogleFileStore, type GoogleFileStoreOptions, type GoogleProviderOptions, type GoogleSafetySetting, type GoogleSearchTool, STANDARD_DEFAULT_TIMEOUT_MS, TIER_FACTOR, TRANSPORT_TIMEOUT_BUFFER_MS, buildGoogleClient, defaultGeminiRegistry, geminiAdapter, geminiContentToMessages, geminiModelDescriptors, geminiPricingSource, gemmaModelDescriptors, googleProvider, isGeminiCapacityError, pricingVersion, requireApiKey };
977
+ export { type CacheKey, FLEX_DEFAULT_TIMEOUT_MS, type FileDeleteOptions, GEMINI_PRICED_TIERS, GEMINI_PRICING, type GeminiAdapterOptions, type GeminiCachesClientLike, type GeminiClientLike, type GeminiContentToMessagesInput, type GeminiContentToMessagesResult, type GeminiCountTokensParams, type GeminiCountTokensResponseShape, type GeminiFilesClientLike, type GeminiPricedTier, type GeminiTierRates, type GoogleCacheHandle, GoogleCacheStore, type GoogleCacheStoreOptions, type GoogleFileHandle, GoogleFileStore, type GoogleFileStoreOptions, type GoogleProviderOptions, type GoogleSafetySetting, type GoogleSearchTool, STANDARD_DEFAULT_TIMEOUT_MS, TRANSPORT_TIMEOUT_BUFFER_MS, buildGoogleClient, defaultGeminiRegistry, geminiAdapter, geminiContentToMessages, geminiModelDescriptors, geminiPricingSource, gemmaModelDescriptors, googleProvider, isGeminiCapacityError, pricingVersion, requireApiKey, resolveGeminiRates };
package/dist/index.d.ts CHANGED
@@ -134,6 +134,8 @@ interface GeminiUsageMetadataShape {
134
134
  cachedContentTokenCount?: number;
135
135
  thoughtsTokenCount?: number;
136
136
  totalTokenCount?: number;
137
+ /** Provider-echoed actual tier; can differ from the requested tier. */
138
+ serviceTier?: string;
137
139
  }
138
140
  /**
139
141
  * Structural equivalent of @google/genai's GenerateContentResponse.
@@ -455,10 +457,10 @@ declare const defaultGeminiRegistry: ModelRegistry;
455
457
  *
456
458
  * Provides `geminiPricingSource` — a factory returning a `PricingSource` port
457
459
  * implementation backed by the frozen Gemini pricing snapshot ({@link
458
- * GEMINI_PRICING}). Walks the rates table (exact-then-longest-prefix match)
459
- * and delegates the actual arithmetic to `@gullabs/core`'s `computeCost`,
460
- * supplying this package's rates table + tier-factor map as explicit
461
- * parameters — core itself carries zero Gemini pricing knowledge.
460
+ * GEMINI_PRICING}). Uses exact priced model identifiers,
461
+ * resolves the concrete per-tier rates, and delegates the arithmetic to
462
+ * `@gullabs/core`'s `computeCost`. Core itself carries zero Gemini pricing
463
+ * knowledge and applies no tier multiplier.
462
464
  *
463
465
  * @module
464
466
  */
@@ -490,27 +492,28 @@ declare function geminiPricingSource(): PricingSource;
490
492
  * All rates are in **micro-USD per million tokens** (µUSD/M).
491
493
  * To get the cost for N tokens: `cost_µUSD = N * ratePerM / 1_000_000`.
492
494
  *
493
- * **Service tiers.** Rates below are STANDARD-tier. Google's **Batch** tier is a
494
- * flat 50% discount on standard, and **Flex** matches Batch pricing. The cost
495
- * engine applies the {@link TIER_FACTOR} multiplier — this snapshot stores
496
- * standard rates only.
495
+ * **Service tiers.** Each model stores concrete `standard`, `flex`, and
496
+ * `batch` rates transcribed from Google's pricing page. Flex and batch are
497
+ * not a flat 50% of standard: on several models the cached lane stays at the
498
+ * standard cached rate (or a published rate that is not half). A tier that
499
+ * is not one of those three is unpriced (reject-don't-map). `priority` is
500
+ * intentionally absent — it needs downgrade accounting and is a backlog item.
497
501
  *
498
- * **Long-context tier.** Gemini Pro models charge a premium when the GROSS input
499
- * token count exceeds 200,000. Selected by `inputTokens` (incl. cached), not by
500
- * billable input.
502
+ * **Long-context tier.** Gemini Pro models charge a premium when the GROSS
503
+ * input token count exceeds 200,000. Selected by `inputTokens` (incl. cached),
504
+ * not by billable input. Core's selector uses `> 200_000`.
501
505
  *
502
506
  * **Thinking tokens.** Already inside `outputTokens` (GROSS convention) and
503
- * billed at the standard output rate — no separate thinking lane.
507
+ * billed at the output rate — no separate thinking lane.
504
508
  *
505
- * **Modality caveat (v1 = text).** Gemini 2.5 Flash / Flash-Lite charge a higher
506
- * INPUT rate for audio tokens than for text/image/video. v1 is text-only and uses
507
- * the text/img/vid input rate. Per-modality input pricing is a deferred seam
508
- * (see DESIGN.md) — revisit when audio input is supported.
509
+ * **Modality caveat (v1 = text).** Gemini 2.5 Flash / Flash-Lite / 3.1
510
+ * Flash-Lite charge a higher INPUT rate for audio tokens
511
+ * than for text/image/video. v1 is text-only and uses the text/img/vid input
512
+ * rate. Per-modality input pricing is a deferred seam (see DESIGN.md).
509
513
  *
510
- * Re-verified against https://ai.google.dev/gemini-api/docs/pricing on 2026-08-12.
511
- * Existing registered-model standard rates were unchanged from the 2026-06-28
512
- * snapshot. `gemini-3.6-flash` appears on that page but is not a registered
513
- * model in this package.
514
+ * Re-verified against https://ai.google.dev/gemini-api/docs/pricing on
515
+ * 2026-09-25. Standard token rates for already-registered models were
516
+ * unchanged from the 2026-08-12 snapshot; flex/batch cached rates were not.
514
517
  *
515
518
  * @module
516
519
  */
@@ -522,30 +525,27 @@ declare function geminiPricingSource(): PricingSource;
522
525
  * core concept — it lives here (not `@gullabs/core`) alongside the rates it
523
526
  * dates.
524
527
  */
525
- declare const pricingVersion: "gemini-2026-08-12";
526
- /**
527
- * Service-tier price multipliers. Batch and Flex are a flat 50% of standard
528
- * (per Google's pricing page: "Batch API — 50% cost reduction"; Flex matches Batch).
529
- *
530
- * `serviceTier` is an opaque, provider-defined string end-to-end — this map is
531
- * the *only* place a tier name is resolved to a multiplier. A tier key not
532
- * present here is never coerced to `standard`: `computeCost` (`@gullabs/core`)
533
- * treats that as an unpriced call (reject-don't-map), not a mapping to this
534
- * table's default. `undefined` (no tier requested) is the one case that
535
- * legitimately defaults to `standard` — that is documented default behavior,
536
- * not a guess.
537
- */
538
- declare const TIER_FACTOR: Readonly<Record<string, number>>;
528
+ declare const pricingVersion: "gemini-2026-09-25";
529
+ /** Tiers this snapshot prices. Anything else is unpriced. */
530
+ declare const GEMINI_PRICED_TIERS: readonly ["standard", "flex", "batch"];
531
+ type GeminiPricedTier = (typeof GEMINI_PRICED_TIERS)[number];
532
+ /** Concrete per-tier rates for one model. */
533
+ interface GeminiTierRates {
534
+ standard: ModelRates;
535
+ flex: ModelRates;
536
+ batch: ModelRates;
537
+ }
539
538
  /**
540
- * Frozen Gemini pricing snapshot (STANDARD tier; per-1M in µUSD).
539
+ * Frozen Gemini pricing snapshot (per-1M in µUSD), keyed by model id, then
540
+ * by priced tier. Every number is transcribed from the pricing page.
541
541
  *
542
- * Keys are model-string prefixes / exact identifiers used in routing. The cost
543
- * engine matches exact first, then longest-prefix (see `lookupRates` in
544
- * `cost.ts`).
542
+ * Keys are exact priced model identifiers. Unlisted variants are unpriced.
545
543
  *
546
- * Source: https://ai.google.dev/gemini-api/docs/pricing (re-verified 2026-08-12).
544
+ * Source: https://ai.google.dev/gemini-api/docs/pricing (re-verified 2026-09-25).
547
545
  */
548
- declare const GEMINI_PRICING: Readonly<Record<string, ModelRates>>;
546
+ declare const GEMINI_PRICING: Readonly<Record<string, GeminiTierRates>>;
547
+ /** Resolve the concrete {@link ModelRates} `computeCost` should apply. */
548
+ declare function resolveGeminiRates(model: string, tier: string | undefined): ModelRates | undefined;
549
549
 
550
550
  declare function isGeminiCapacityError(err: LlmError): boolean;
551
551
 
@@ -974,4 +974,4 @@ interface GeminiContentToMessagesResult {
974
974
  */
975
975
  declare function geminiContentToMessages(input: GeminiContentToMessagesInput): GeminiContentToMessagesResult;
976
976
 
977
- export { type CacheKey, FLEX_DEFAULT_TIMEOUT_MS, type FileDeleteOptions, GEMINI_PRICING, type GeminiAdapterOptions, type GeminiCachesClientLike, type GeminiClientLike, type GeminiContentToMessagesInput, type GeminiContentToMessagesResult, type GeminiCountTokensParams, type GeminiCountTokensResponseShape, type GeminiFilesClientLike, type GoogleCacheHandle, GoogleCacheStore, type GoogleCacheStoreOptions, type GoogleFileHandle, GoogleFileStore, type GoogleFileStoreOptions, type GoogleProviderOptions, type GoogleSafetySetting, type GoogleSearchTool, STANDARD_DEFAULT_TIMEOUT_MS, TIER_FACTOR, TRANSPORT_TIMEOUT_BUFFER_MS, buildGoogleClient, defaultGeminiRegistry, geminiAdapter, geminiContentToMessages, geminiModelDescriptors, geminiPricingSource, gemmaModelDescriptors, googleProvider, isGeminiCapacityError, pricingVersion, requireApiKey };
977
+ export { type CacheKey, FLEX_DEFAULT_TIMEOUT_MS, type FileDeleteOptions, GEMINI_PRICED_TIERS, GEMINI_PRICING, type GeminiAdapterOptions, type GeminiCachesClientLike, type GeminiClientLike, type GeminiContentToMessagesInput, type GeminiContentToMessagesResult, type GeminiCountTokensParams, type GeminiCountTokensResponseShape, type GeminiFilesClientLike, type GeminiPricedTier, type GeminiTierRates, type GoogleCacheHandle, GoogleCacheStore, type GoogleCacheStoreOptions, type GoogleFileHandle, GoogleFileStore, type GoogleFileStoreOptions, type GoogleProviderOptions, type GoogleSafetySetting, type GoogleSearchTool, STANDARD_DEFAULT_TIMEOUT_MS, TRANSPORT_TIMEOUT_BUFFER_MS, buildGoogleClient, defaultGeminiRegistry, geminiAdapter, geminiContentToMessages, geminiModelDescriptors, geminiPricingSource, gemmaModelDescriptors, googleProvider, isGeminiCapacityError, pricingVersion, requireApiKey, resolveGeminiRates };