@genesislcap/foundation-ai 15.11.0 → 15.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/dts/index.d.ts +4 -2
- package/dist/dts/index.d.ts.map +1 -1
- package/dist/dts/transports/anthropic-transport.d.ts +90 -0
- package/dist/dts/transports/anthropic-transport.d.ts.map +1 -1
- package/dist/dts/transports/gemini-transport.d.ts +16 -0
- package/dist/dts/transports/gemini-transport.d.ts.map +1 -1
- package/dist/dts/types/chat.types.d.ts +93 -2
- package/dist/dts/types/chat.types.d.ts.map +1 -1
- package/dist/dts/types/config.types.d.ts +26 -0
- package/dist/dts/types/config.types.d.ts.map +1 -1
- package/dist/dts/utils/token-cost.d.ts +193 -0
- package/dist/dts/utils/token-cost.d.ts.map +1 -0
- package/dist/esm/index.js +11 -1
- package/dist/esm/transports/anthropic-transport.js +357 -105
- package/dist/esm/transports/gemini-transport.js +84 -81
- package/dist/esm/types/config.types.js +32 -0
- package/dist/esm/utils/token-cost.js +238 -0
- package/dist/foundation-ai.api.json +1127 -73
- package/dist/foundation-ai.d.ts +432 -2
- package/package.json +11 -11
package/dist/foundation-ai.d.ts
CHANGED
|
@@ -36,13 +36,26 @@ export declare type AgentPickerMode = 'disabled' | 'select' | 'segmented-control
|
|
|
36
36
|
* @beta
|
|
37
37
|
*/
|
|
38
38
|
export declare interface AggregateUsage {
|
|
39
|
-
/**
|
|
39
|
+
/**
|
|
40
|
+
* USD cost, provider-reported per request and summed — cache discounts applied.
|
|
41
|
+
*
|
|
42
|
+
* The authoritative total. It is a **sum of per-request costs**, each computed where
|
|
43
|
+
* the full provider usage was in hand; it is never re-derived from the token buckets
|
|
44
|
+
* below, and should not be.
|
|
45
|
+
*/
|
|
40
46
|
costUsd: number;
|
|
41
47
|
/** Prompt tokens billed at the full uncached input rate. */
|
|
42
48
|
uncachedInputTokens: number;
|
|
43
49
|
/** Prompt tokens served from the provider's cache. */
|
|
44
50
|
cacheReadTokens: number;
|
|
45
|
-
/**
|
|
51
|
+
/**
|
|
52
|
+
* Prompt tokens written to the provider's cache.
|
|
53
|
+
*
|
|
54
|
+
* **Volume only — this aggregate can never be priced.** Two reasons, either of which
|
|
55
|
+
* is sufficient: Anthropic's per-TTL write split is already collapsed on each
|
|
56
|
+
* contributing message, and a run may span models (and providers) at different rates,
|
|
57
|
+
* so no single rate applies to the total. Report it, chart it, do not multiply it.
|
|
58
|
+
*/
|
|
46
59
|
cacheWriteTokens: number;
|
|
47
60
|
/** Generated (output) tokens, including any reasoning tokens the provider bills as output. */
|
|
48
61
|
outputTokens: number;
|
|
@@ -227,6 +240,23 @@ declare interface AITransport {
|
|
|
227
240
|
isAvailable?(): Promise<boolean>;
|
|
228
241
|
}
|
|
229
242
|
|
|
243
|
+
/**
|
|
244
|
+
* Prompt-cache pricing multipliers, applied to the model's base input rate
|
|
245
|
+
* (`promptPerMillion`) — https://docs.claude.com/en/docs/build-with-claude/prompt-caching
|
|
246
|
+
* Reads bill at ~0.1× base input; writes bill by TTL — 5-minute at ~1.25× and 1-hour at 2×.
|
|
247
|
+
* Both write TTLs are reachable (`CachePolicy.ttl` is `'5m' | '1h'`), so each TTL bucket is
|
|
248
|
+
* costed from the response's per-TTL `cache_creation` breakdown rather than assuming one rate.
|
|
249
|
+
*
|
|
250
|
+
* @beta
|
|
251
|
+
*/
|
|
252
|
+
export declare const ANTHROPIC_CACHE_READ_MULTIPLIER = 0.1;
|
|
253
|
+
|
|
254
|
+
/** @beta */
|
|
255
|
+
export declare const ANTHROPIC_CACHE_WRITE_1H_MULTIPLIER = 2;
|
|
256
|
+
|
|
257
|
+
/** @beta */
|
|
258
|
+
export declare const ANTHROPIC_CACHE_WRITE_5M_MULTIPLIER = 1.25;
|
|
259
|
+
|
|
230
260
|
/**
|
|
231
261
|
* Anthropic server-proxy AI configuration (client calls your server; server calls Anthropic).
|
|
232
262
|
*
|
|
@@ -279,6 +309,29 @@ declare interface AnthropicProviderConfig {
|
|
|
279
309
|
criteriaInstructions?: string;
|
|
280
310
|
}
|
|
281
311
|
|
|
312
|
+
/**
|
|
313
|
+
* Standard tier pricing per million tokens —
|
|
314
|
+
* https://docs.claude.com/en/docs/about-claude/pricing
|
|
315
|
+
*
|
|
316
|
+
* @remarks
|
|
317
|
+
* Backed by a total {@link https://www.typescriptlang.org/docs/handbook/utility-types.html#recordkeys-type | Record}
|
|
318
|
+
* over `AnthropicModelId` rather than a chain of checks with a default. That is the whole
|
|
319
|
+
* point: adding a model to `SUPPORTED_ANTHROPIC_MODEL_IDS` without pricing it becomes a
|
|
320
|
+
* COMPILE error instead of a silent charge at whatever the fall-through happened to be — the
|
|
321
|
+
* exact failure this module's header warns about, and one a consumer has already been bitten
|
|
322
|
+
* by. The Gemini half of this file has always had the guarantee; this half now matches.
|
|
323
|
+
*
|
|
324
|
+
* @beta
|
|
325
|
+
*/
|
|
326
|
+
export declare function anthropicRatesFor(model: AnthropicModelId): TokenRates;
|
|
327
|
+
|
|
328
|
+
/**
|
|
329
|
+
* Cost one Anthropic request.
|
|
330
|
+
*
|
|
331
|
+
* @beta
|
|
332
|
+
*/
|
|
333
|
+
export declare function anthropicTokenCost(model: AnthropicModelId, usage: AnthropicUsageRecord): TokenCost;
|
|
334
|
+
|
|
282
335
|
/**
|
|
283
336
|
* Transport for Anthropic Claude. Calls the Messages API directly when `apiKey`
|
|
284
337
|
* is provided, otherwise falls back to a server-proxy endpoint (if `serverEndpoint`
|
|
@@ -309,6 +362,26 @@ export declare class AnthropicTransport implements AITransport, ChatTransport, C
|
|
|
309
362
|
* accrue. Surfaced alongside `getLifetimeCost`.
|
|
310
363
|
*/
|
|
311
364
|
private lifetimeSavingsUsd;
|
|
365
|
+
/**
|
|
366
|
+
* Whether we have already told the caller their `thinkingPolicy: 'off'` cannot be honoured
|
|
367
|
+
* on this model. Latched, because it would otherwise fire on every turn of a tool loop.
|
|
368
|
+
*/
|
|
369
|
+
private warnedThinkingClamped;
|
|
370
|
+
/**
|
|
371
|
+
* Serving-model ids already reported as unrecognised. Per-id rather than a single flag, so a
|
|
372
|
+
* second unknown model is still announced. See {@link AnthropicTransport.servingModel}.
|
|
373
|
+
*/
|
|
374
|
+
private readonly warnedUnknownServingModels;
|
|
375
|
+
/**
|
|
376
|
+
* Warn once when a requested policy is silently clamped, in EITHER direction — which is what
|
|
377
|
+
* `ChatThinkingPolicy` promises callers. Both directions cost the caller something they asked
|
|
378
|
+
* for and would otherwise get no signal about: `'off'` on a model that always thinks keeps
|
|
379
|
+
* billing reasoning as output, and `'auto'` on a model without adaptive support means an agent
|
|
380
|
+
* that asked to reason quietly does not. (Gemini's twin deliberately stays silent on `'auto'`,
|
|
381
|
+
* but only because dynamic thinking is already the default there; Haiku defaults to none, so
|
|
382
|
+
* the same silence would hide a real difference.)
|
|
383
|
+
*/
|
|
384
|
+
private warnIfThinkingUnclampable;
|
|
312
385
|
constructor(config?: AnthropicTransportConfig);
|
|
313
386
|
getConfig(): {
|
|
314
387
|
provider: 'anthropic';
|
|
@@ -364,6 +437,76 @@ export declare class AnthropicTransport implements AITransport, ChatTransport, C
|
|
|
364
437
|
* the payload tidy.
|
|
365
438
|
*/
|
|
366
439
|
private toAnthropicMessages;
|
|
440
|
+
/**
|
|
441
|
+
* The blocks to replay ahead of a tool call — fallback boundaries then reasoning, or none.
|
|
442
|
+
*
|
|
443
|
+
* A `signature` is only valid for the model that produced it, so reasoning captured under a
|
|
444
|
+
* *different* model is normally dropped: an agent that varies `provider` by state can switch
|
|
445
|
+
* models mid-loop, and replaying the old model's signatures would send blocks the new one
|
|
446
|
+
* cannot verify.
|
|
447
|
+
*
|
|
448
|
+
* Two ways the producer can be the model that will validate:
|
|
449
|
+
*
|
|
450
|
+
* 1. **It is the model we are asking for.** `state.model === this.model` — the ordinary case.
|
|
451
|
+
* 2. **Sticky routing will send this conversation back to it.** After a conversation falls
|
|
452
|
+
* back, later requests carrying `fallbacks` go straight to the model that served, without
|
|
453
|
+
* re-running the one that declined. That is what makes a fallback producer's reasoning
|
|
454
|
+
* replayable at all (Fable 5 configured, Opus 4.8 serving — the pairing this transport's
|
|
455
|
+
* own constructor warning recommends). But it holds only while the conversation continues
|
|
456
|
+
* under the SAME request configuration, which is why `requestedModel` is compared rather
|
|
457
|
+
* than just checking the chain for the producer.
|
|
458
|
+
*
|
|
459
|
+
* That second condition is deliberately narrow. Chain membership alone is too weak: a
|
|
460
|
+
* fallback target is only *contingently* the server, so an agent that switches `provider` to
|
|
461
|
+
* a model whose own chain happens to contain the old producer would replay foreign
|
|
462
|
+
* signatures to whichever model actually answers. Requiring the requested model to be
|
|
463
|
+
* unchanged separates "this conversation is still going" from "we are somewhere else now".
|
|
464
|
+
*
|
|
465
|
+
* State captured before `requestedModel` existed falls back to condition 1 alone, which is
|
|
466
|
+
* the conservative branch — it can drop reasoning, never misdirect it.
|
|
467
|
+
*
|
|
468
|
+
* Boundaries themselves are always echoed: keeping them in place is the documented rule, and
|
|
469
|
+
* with no thinking blocks around them they are inert rather than harmful.
|
|
470
|
+
*/
|
|
471
|
+
private reasoningToReplay;
|
|
472
|
+
/**
|
|
473
|
+
* The model that actually served this response.
|
|
474
|
+
*
|
|
475
|
+
* A server-side `fallbacks` chain re-runs a declined request on another model and
|
|
476
|
+
* names it in the response's top-level `model`. Pricing must follow that, not the
|
|
477
|
+
* model we asked for: a Fable 5 request served by Opus 4.8 costed at Fable's
|
|
478
|
+
* $10/$50 instead of $5/$25 doubles that part of the bill.
|
|
479
|
+
*
|
|
480
|
+
* An unrecognised model id is NOT priced at a guessed rate — that is precisely how
|
|
481
|
+
* a consumer ended up billing Haiku at Sonnet rates. It warns and falls back to the
|
|
482
|
+
* configured model, which is at least a figure someone chose.
|
|
483
|
+
*/
|
|
484
|
+
private servingModel;
|
|
485
|
+
/**
|
|
486
|
+
* Whether a fallback-chain attempt is a refusal that was **not billed**.
|
|
487
|
+
*
|
|
488
|
+
* Anthropic does not charge for a refusal that arrives before any output: the token counts
|
|
489
|
+
* still appear in `usage`, but they are not on the bill. Pricing them anyway turns the
|
|
490
|
+
* fallback undercount this reducer was written to fix into an overcount — the documented
|
|
491
|
+
* example declines with `input_tokens: 535, output_tokens: 0`, which is real money at
|
|
492
|
+
* Fable 5's prompt rate.
|
|
493
|
+
*
|
|
494
|
+
* A **mid-output** refusal *is* billed for the input and whatever it streamed, so output
|
|
495
|
+
* tokens are the discriminator rather than the refusal itself.
|
|
496
|
+
*
|
|
497
|
+
* Declined attempts appear as `type: 'message'`; the attempt that served the turn is
|
|
498
|
+
* `type: 'fallback_message'`. The serving entry is normally billable — except when every
|
|
499
|
+
* model in the chain declined, where the last entry is both the serving one and a refusal,
|
|
500
|
+
* which `lastAndRefused` covers.
|
|
501
|
+
*
|
|
502
|
+
* Also called for a response with no chain at all, as `isUnbilledRefusal({}, usage, refused)`:
|
|
503
|
+
* a direct request that was declined has no `iterations` array, and the same rule applies to
|
|
504
|
+
* it. That is in fact the common case — `iterations` only appears when `fallbacks` was
|
|
505
|
+
* configured.
|
|
506
|
+
*/
|
|
507
|
+
private isUnbilledRefusal;
|
|
508
|
+
/** Cost one attempt's usage block at `model`'s rates, logging and banking it. */
|
|
509
|
+
private costAttempt;
|
|
367
510
|
private fromAnthropicResponse;
|
|
368
511
|
private buildEndpoint;
|
|
369
512
|
private static readonly RATE_LIMIT_STATUS;
|
|
@@ -407,6 +550,38 @@ declare interface AnthropicTransportConfig {
|
|
|
407
550
|
maxTokens?: number;
|
|
408
551
|
}
|
|
409
552
|
|
|
553
|
+
/**
|
|
554
|
+
* Token counts from an Anthropic `usage` block.
|
|
555
|
+
*
|
|
556
|
+
* @remarks
|
|
557
|
+
* Note the input semantics, which are the opposite of Gemini's: `input_tokens` on
|
|
558
|
+
* this provider is the **uncached remainder**, so the three input buckets ADD to
|
|
559
|
+
* the prompt total rather than breaking it down.
|
|
560
|
+
*
|
|
561
|
+
* @beta
|
|
562
|
+
*/
|
|
563
|
+
export declare interface AnthropicUsageRecord {
|
|
564
|
+
/** `usage.input_tokens` — the uncached remainder of the prompt, billed at full rate. */
|
|
565
|
+
uncachedInputTokens: number;
|
|
566
|
+
/** `usage.output_tokens`. Includes thinking tokens, which this provider bills as output. */
|
|
567
|
+
outputTokens: number;
|
|
568
|
+
/** `usage.cache_read_input_tokens`. */
|
|
569
|
+
cacheReadTokens: number;
|
|
570
|
+
/** `usage.cache_creation_input_tokens` — the total across BOTH TTLs. */
|
|
571
|
+
cacheWriteTokens: number;
|
|
572
|
+
/**
|
|
573
|
+
* `usage.cache_creation.ephemeral_1h_input_tokens` — the 1-hour portion of
|
|
574
|
+
* {@link AnthropicUsageRecord.cacheWriteTokens}. The remainder is treated as
|
|
575
|
+
* 5-minute, which is also the correct fallback when a response omits the
|
|
576
|
+
* per-TTL breakdown entirely.
|
|
577
|
+
*
|
|
578
|
+
* Pass `0` if unknown. Doing so under-bills a request that really did write to
|
|
579
|
+
* the 1-hour cache — by the difference between the two multipliers — so prefer
|
|
580
|
+
* reading the breakdown when the provider sends one.
|
|
581
|
+
*/
|
|
582
|
+
cacheWrite1hTokens: number;
|
|
583
|
+
}
|
|
584
|
+
|
|
410
585
|
/**
|
|
411
586
|
* The vendors whose spend the ai-service proxy meters — i.e. the only vendors
|
|
412
587
|
* that can raise a {@link BudgetExhaustedError}, and therefore the only ones a
|
|
@@ -914,6 +1089,18 @@ export declare interface ChatMessage {
|
|
|
914
1089
|
* {@link ChatMessage.inputTokens} rather than an addition to it. `undefined`
|
|
915
1090
|
* where the provider reports no distinct write bucket (e.g. Gemini's implicit
|
|
916
1091
|
* caching) — read it with `?? 0`.
|
|
1092
|
+
*
|
|
1093
|
+
* **A volume figure, not a pricing input — do not re-price it.** Anthropic bills
|
|
1094
|
+
* cache writes by TTL (5-minute and 1-hour at different multiples of the input
|
|
1095
|
+
* rate) and this field is the combined total, so the split needed to cost it is
|
|
1096
|
+
* not recoverable from a stored message. Pricing this bucket therefore under-bills
|
|
1097
|
+
* a 1-hour write, silently and in the cheap direction. {@link ChatMessage.cost} is
|
|
1098
|
+
* the authoritative figure: the transport computed it from the per-TTL breakdown
|
|
1099
|
+
* before that breakdown was collapsed to this total.
|
|
1100
|
+
*
|
|
1101
|
+
* A host that genuinely must compute a cost itself — a proxy holding a raw provider
|
|
1102
|
+
* usage block, where nothing stamped one — should price from that raw block instead,
|
|
1103
|
+
* which carries the split.
|
|
917
1104
|
*/
|
|
918
1105
|
cacheWriteTokens?: number;
|
|
919
1106
|
/**
|
|
@@ -921,6 +1108,25 @@ export declare interface ChatMessage {
|
|
|
921
1108
|
* derived from the active model's input/output rates. Set by transports.
|
|
922
1109
|
* Hosts can sum this across the message list to display a running session
|
|
923
1110
|
* cost without re-deriving rates.
|
|
1111
|
+
*
|
|
1112
|
+
* @remarks
|
|
1113
|
+
* **The authoritative cost for this request.** Computed at the transport, at the
|
|
1114
|
+
* one point where the full provider usage is in hand — including Anthropic's
|
|
1115
|
+
* per-TTL cache-write breakdown, which the token fields on this message do not
|
|
1116
|
+
* preserve. Prefer it over any figure re-derived from those fields, which cannot
|
|
1117
|
+
* be exact and is wrong in the direction that under-reports spend.
|
|
1118
|
+
*
|
|
1119
|
+
* Already carries every cache discount and premium, so a cache-heavy request shows
|
|
1120
|
+
* a small `cost` against a large token count — that is the discount working, not a
|
|
1121
|
+
* missing charge.
|
|
1122
|
+
*
|
|
1123
|
+
* **When a server-side fallback chain ran, this and the token fields have different
|
|
1124
|
+
* scopes.** `cost` sums every *billed* attempt, possibly across models at different
|
|
1125
|
+
* rates, while `inputTokens` / `outputTokens` / the cache buckets describe only the
|
|
1126
|
+
* attempt that produced the content. So such a turn can carry cost with no volume
|
|
1127
|
+
* behind it to explain the difference — a $/token column or a "cost without tokens"
|
|
1128
|
+
* audit will see an outlier that is correct. Read `cost` as the bill and the token
|
|
1129
|
+
* fields as the size of the answer, not as two views of one thing.
|
|
924
1130
|
*/
|
|
925
1131
|
cost?: number;
|
|
926
1132
|
/**
|
|
@@ -1043,6 +1249,14 @@ export declare interface ChatRequestOptions {
|
|
|
1043
1249
|
* @beta
|
|
1044
1250
|
*/
|
|
1045
1251
|
responseSchema?: object;
|
|
1252
|
+
/**
|
|
1253
|
+
* Extended-thinking posture for this turn. **Omit** to keep each model's own default —
|
|
1254
|
+
* that is the third state, and it is not the same as either value. See
|
|
1255
|
+
* {@link ChatThinkingPolicy}.
|
|
1256
|
+
*
|
|
1257
|
+
* @beta
|
|
1258
|
+
*/
|
|
1259
|
+
thinkingPolicy?: ChatThinkingPolicy;
|
|
1046
1260
|
}
|
|
1047
1261
|
|
|
1048
1262
|
/**
|
|
@@ -1164,6 +1378,46 @@ export declare const ChatTemperature: {
|
|
|
1164
1378
|
readonly Maximum: 1;
|
|
1165
1379
|
};
|
|
1166
1380
|
|
|
1381
|
+
/**
|
|
1382
|
+
* Whether the model should reason before answering, resolved per turn.
|
|
1383
|
+
* **Provider-neutral intent, provider-specific effect**, and — like `tool_choice` — a
|
|
1384
|
+
* *request*, not a guarantee: a model that cannot honour it keeps its default.
|
|
1385
|
+
*
|
|
1386
|
+
* - `'auto'` — the model decides how much to think (Anthropic `thinking: {type:'adaptive'}`,
|
|
1387
|
+
* Gemini `thinkingConfig.thinkingBudget: -1`). On a model whose default is thinking-*off*
|
|
1388
|
+
* this turns it **on**, which is the point: it is the opt-in for models where adaptive
|
|
1389
|
+
* thinking is available but not default.
|
|
1390
|
+
* - `'off'` — no reasoning tokens (Anthropic `thinking: {type:'disabled'}`, Gemini
|
|
1391
|
+
* `thinkingBudget: 0`). The lever for turns that only step a known sequence, where reasoning
|
|
1392
|
+
* is billed as output at the full candidate rate to decide something already determined.
|
|
1393
|
+
*
|
|
1394
|
+
* **Choose it per turn, not per model call.** A tool-use loop is one assistant turn, and
|
|
1395
|
+
* Anthropic requires a single thinking mode for its duration; toggling part-way through does
|
|
1396
|
+
* not error but silently disables thinking for that request, strips blocks that would leave the
|
|
1397
|
+
* turn structure invalid, and invalidates the prompt cache. The driver pins whatever it resolves
|
|
1398
|
+
* on a turn's first model call for the rest of that turn.
|
|
1399
|
+
*
|
|
1400
|
+
* Deliberately has **no token-budget member**. Gemini accepts a numeric `thinkingBudget`, but
|
|
1401
|
+
* Anthropic *removed* `budget_tokens` and returns a 400 for it on Sonnet 5 and every Opus 5-era
|
|
1402
|
+
* model — so a budget field would be unimplementable on half the supported fleet and would exist
|
|
1403
|
+
* only to be ignored. `'low' | 'high'` can be added later if wanted: Anthropic `effort` and
|
|
1404
|
+
* Gemini `thinkingLevel` do line up.
|
|
1405
|
+
*
|
|
1406
|
+
* **Omitting this is a distinct third state** — each model keeps its own default posture
|
|
1407
|
+
* (see `anthropicThinking`), which is *not* uniformly `'auto'` or `'off'`. So a resolver
|
|
1408
|
+
* returning `undefined` on some turns leaves those turns exactly as they were before this
|
|
1409
|
+
* option existed, rather than silently re-pricing them.
|
|
1410
|
+
*
|
|
1411
|
+
* Not every model can honour every value, and the two providers hit the same wall at the top
|
|
1412
|
+
* of their ranges: Anthropic Fable 5 and Gemini 2.5 Pro both think unconditionally and reject
|
|
1413
|
+
* (or ignore) a request to stop. Transports **clamp** rather than forward a request that would
|
|
1414
|
+
* 400 — the turn runs at the model's default and a one-time warning is logged. Never assume
|
|
1415
|
+
* `'off'` means zero reasoning tokens were billed; read the usage back.
|
|
1416
|
+
*
|
|
1417
|
+
* @beta
|
|
1418
|
+
*/
|
|
1419
|
+
export declare type ChatThinkingPolicy = 'auto' | 'off';
|
|
1420
|
+
|
|
1167
1421
|
/**
|
|
1168
1422
|
* A tool call requested by the assistant.
|
|
1169
1423
|
*
|
|
@@ -1774,6 +2028,27 @@ export declare type FieldLike = string | {
|
|
|
1774
2028
|
NAME?: string;
|
|
1775
2029
|
};
|
|
1776
2030
|
|
|
2031
|
+
/**
|
|
2032
|
+
* Cached / context-cache input tokens bill at a flat ~10% of the model's normal input
|
|
2033
|
+
* rate — consistent across models (flash $0.30→$0.03, flash-lite $0.10→$0.01,
|
|
2034
|
+
* pro $1.25→$0.125). Source: https://ai.google.dev/gemini-api/docs/pricing.
|
|
2035
|
+
* (Explicit context caching also has an hourly storage fee; implicit caching — what the
|
|
2036
|
+
* transport relies on — has none, so only this read discount applies. There is no write
|
|
2037
|
+
* premium on this provider, which is why its savings figure is never negative.)
|
|
2038
|
+
*
|
|
2039
|
+
* @beta
|
|
2040
|
+
*/
|
|
2041
|
+
export declare const GEMINI_CACHED_INPUT_MULTIPLIER = 0.1;
|
|
2042
|
+
|
|
2043
|
+
/**
|
|
2044
|
+
* Prompt size (tokens) at or below which the standard pricing tier applies.
|
|
2045
|
+
* Above it, Gemini's long-context tier kicks in. Google applies a single tier
|
|
2046
|
+
* to the whole request based on prompt size — it is not a marginal/blended rate.
|
|
2047
|
+
*
|
|
2048
|
+
* @beta
|
|
2049
|
+
*/
|
|
2050
|
+
export declare const GEMINI_LONG_CONTEXT_THRESHOLD = 200000;
|
|
2051
|
+
|
|
1777
2052
|
/**
|
|
1778
2053
|
* Gemini server-proxy AI configuration (client calls your server; server calls Gemini).
|
|
1779
2054
|
*
|
|
@@ -1818,6 +2093,30 @@ declare interface GeminiProviderConfig {
|
|
|
1818
2093
|
criteriaInstructions?: string;
|
|
1819
2094
|
}
|
|
1820
2095
|
|
|
2096
|
+
/**
|
|
2097
|
+
* Resolve the per-million-token rates for a model given the request's prompt size,
|
|
2098
|
+
* selecting the long-context tier for tiered models when the prompt exceeds
|
|
2099
|
+
* {@link GEMINI_LONG_CONTEXT_THRESHOLD}.
|
|
2100
|
+
*
|
|
2101
|
+
* @remarks
|
|
2102
|
+
* `promptTokens` must be the **full** prompt size, cached slice included — the tier
|
|
2103
|
+
* is chosen on what was sent, not on what was billed at full rate. A mostly-cached
|
|
2104
|
+
* long prompt is still a long prompt.
|
|
2105
|
+
*
|
|
2106
|
+
* Unlike Anthropic's, this accessor is not a pure function of the model, which is
|
|
2107
|
+
* why there is no single `ratesFor(model)` across both providers.
|
|
2108
|
+
*
|
|
2109
|
+
* @beta
|
|
2110
|
+
*/
|
|
2111
|
+
export declare function geminiRatesFor(model: GeminiModelId, promptTokens: number): TokenRates;
|
|
2112
|
+
|
|
2113
|
+
/**
|
|
2114
|
+
* Cost one Gemini request.
|
|
2115
|
+
*
|
|
2116
|
+
* @beta
|
|
2117
|
+
*/
|
|
2118
|
+
export declare function geminiTokenCost(model: GeminiModelId, usage: GeminiUsageRecord): TokenCost;
|
|
2119
|
+
|
|
1821
2120
|
/**
|
|
1822
2121
|
* Transport for Gemini. Calls the Gemini REST API directly when `apiKey` is
|
|
1823
2122
|
* provided, otherwise falls back to a server-proxy endpoint (if `serverEndpoint`
|
|
@@ -1846,6 +2145,22 @@ export declare class GeminiTransport implements AITransport, ChatTransport, Cost
|
|
|
1846
2145
|
* Surfaced alongside `getLifetimeCost`.
|
|
1847
2146
|
*/
|
|
1848
2147
|
private lifetimeSavingsUsd;
|
|
2148
|
+
/**
|
|
2149
|
+
* Whether we have already told the caller their `thinkingPolicy` cannot be honoured on this
|
|
2150
|
+
* model. Latched, because it would otherwise fire on every turn of a tool loop.
|
|
2151
|
+
*/
|
|
2152
|
+
private warnedThinkingClamped;
|
|
2153
|
+
/**
|
|
2154
|
+
* Warn once when a requested policy is silently clamped. Turning thinking off is a *cost*
|
|
2155
|
+
* decision, so a caller who asked for it and kept paying for reasoning tokens needs to hear
|
|
2156
|
+
* about it — but only once, not per turn.
|
|
2157
|
+
*
|
|
2158
|
+
* Only `'off'` can actually be denied. `'auto'` is already what an omitted budget produces on
|
|
2159
|
+
* every model here — dynamic thinking is the documented default — so warning that it was
|
|
2160
|
+
* "ignored" would be false, and latching on it would spend the one warning the genuinely
|
|
2161
|
+
* unhonourable `'off'` needs.
|
|
2162
|
+
*/
|
|
2163
|
+
private warnIfThinkingUnclampable;
|
|
1849
2164
|
constructor(config?: GeminiTransportConfig);
|
|
1850
2165
|
getConfig(): {
|
|
1851
2166
|
provider: 'gemini';
|
|
@@ -1953,6 +2268,31 @@ declare interface GeminiTransportConfig {
|
|
|
1953
2268
|
serverEndpoint?: string;
|
|
1954
2269
|
}
|
|
1955
2270
|
|
|
2271
|
+
/**
|
|
2272
|
+
* Token counts from a Gemini `usageMetadata` block.
|
|
2273
|
+
*
|
|
2274
|
+
* @remarks
|
|
2275
|
+
* Note the input semantics, which are the opposite of Anthropic's:
|
|
2276
|
+
* `promptTokenCount` is the **whole** prompt and already includes
|
|
2277
|
+
* `cachedContentTokenCount`, so uncached input is a subtraction. Adding the buckets
|
|
2278
|
+
* instead double-charges the cached portion.
|
|
2279
|
+
*
|
|
2280
|
+
* @beta
|
|
2281
|
+
*/
|
|
2282
|
+
export declare interface GeminiUsageRecord {
|
|
2283
|
+
/** `usageMetadata.promptTokenCount` — the whole prompt, cached slice INCLUDED. */
|
|
2284
|
+
promptTokens: number;
|
|
2285
|
+
/** `usageMetadata.candidatesTokenCount`. */
|
|
2286
|
+
candidateTokens: number;
|
|
2287
|
+
/**
|
|
2288
|
+
* `usageMetadata.thoughtsTokenCount`. Reported separately from candidates but
|
|
2289
|
+
* billed at the candidate rate, so omitting it undercounts the bill.
|
|
2290
|
+
*/
|
|
2291
|
+
thoughtTokens: number;
|
|
2292
|
+
/** `usageMetadata.cachedContentTokenCount` — 0 when caching was not active. */
|
|
2293
|
+
cachedTokens: number;
|
|
2294
|
+
}
|
|
2295
|
+
|
|
1956
2296
|
/**
|
|
1957
2297
|
* How the host frames an interaction widget in the chat transcript. The
|
|
1958
2298
|
* assistant avatar is always shown; this controls only the bubble and the
|
|
@@ -2500,6 +2840,69 @@ export declare const SUPPORTED_ANTHROPIC_MODEL_IDS: readonly AnthropicModelId[];
|
|
|
2500
2840
|
/** @beta */
|
|
2501
2841
|
export declare const SUPPORTED_GEMINI_MODEL_IDS: readonly GeminiModelId[];
|
|
2502
2842
|
|
|
2843
|
+
/**
|
|
2844
|
+
* What one request cost, and what prompt caching saved on it.
|
|
2845
|
+
*
|
|
2846
|
+
* @beta
|
|
2847
|
+
*/
|
|
2848
|
+
export declare interface TokenCost {
|
|
2849
|
+
/** Total USD for the request, with every cache discount and premium applied. */
|
|
2850
|
+
costUsd: number;
|
|
2851
|
+
/**
|
|
2852
|
+
* USD saved versus paying the full input rate for every cached token, net of any
|
|
2853
|
+
* cache-write premium.
|
|
2854
|
+
*
|
|
2855
|
+
* **Can be negative**, and that is not an error: a write premium is an upfront
|
|
2856
|
+
* cost that later reads pay back, so a cold write-heavy request legitimately
|
|
2857
|
+
* reports a loss. Clamping it at zero destroys the signal that a cache is being
|
|
2858
|
+
* written more often than it is read.
|
|
2859
|
+
*/
|
|
2860
|
+
savedUsd: number;
|
|
2861
|
+
/**
|
|
2862
|
+
* The four cost buckets, summing to {@link TokenCost.costUsd}.
|
|
2863
|
+
*
|
|
2864
|
+
* Reported because a single total cannot show *why* a request was cheap: a
|
|
2865
|
+
* cache-heavy call moves a large number of tokens for very little money, and only
|
|
2866
|
+
* the split says so. Also what a per-call usage row needs, rather than the one
|
|
2867
|
+
* number a dashboard would have to explain.
|
|
2868
|
+
*/
|
|
2869
|
+
breakdown: TokenCostBreakdown;
|
|
2870
|
+
}
|
|
2871
|
+
|
|
2872
|
+
/**
|
|
2873
|
+
* Per-bucket cost split for one request. Each field is already in USD.
|
|
2874
|
+
*
|
|
2875
|
+
* @beta
|
|
2876
|
+
*/
|
|
2877
|
+
export declare interface TokenCostBreakdown {
|
|
2878
|
+
/** Prompt tokens billed at the full uncached input rate. */
|
|
2879
|
+
promptUsd: number;
|
|
2880
|
+
/** Prompt tokens served from the provider's cache. */
|
|
2881
|
+
cacheReadUsd: number;
|
|
2882
|
+
/**
|
|
2883
|
+
* Prompt tokens written to the provider's cache. Always `0` on Gemini, whose
|
|
2884
|
+
* implicit caching has no write charge.
|
|
2885
|
+
*
|
|
2886
|
+
* A per-request charge for *populating* a cache. Not a home for a cache's ongoing
|
|
2887
|
+
* storage cost, which is not per-request — see the module header.
|
|
2888
|
+
*/
|
|
2889
|
+
cacheWriteUsd: number;
|
|
2890
|
+
/** Generated output, including any reasoning tokens billed as output. */
|
|
2891
|
+
candidateUsd: number;
|
|
2892
|
+
}
|
|
2893
|
+
|
|
2894
|
+
/**
|
|
2895
|
+
* USD per million tokens, split by direction. `prompt` covers input/prompt tokens,
|
|
2896
|
+
* `candidate` covers generated output (including any reasoning tokens the provider
|
|
2897
|
+
* bills as output).
|
|
2898
|
+
*
|
|
2899
|
+
* @beta
|
|
2900
|
+
*/
|
|
2901
|
+
export declare interface TokenRates {
|
|
2902
|
+
promptPerMillion: number;
|
|
2903
|
+
candidatePerMillion: number;
|
|
2904
|
+
}
|
|
2905
|
+
|
|
2503
2906
|
/**
|
|
2504
2907
|
* Why a driver turn ended in failure — the typed taxonomy the tool loop already
|
|
2505
2908
|
* records onto its debug-log timeline (`turn.error` / `turn.retry` details),
|
|
@@ -2562,6 +2965,33 @@ export declare type TurnFailureReason = 'exception' | 'malformed-function-call'
|
|
|
2562
2965
|
*/
|
|
2563
2966
|
export declare const VENDOR_LABELS: Readonly<Record<Exclude<AIProviderType, 'none'>, string>>;
|
|
2564
2967
|
|
|
2968
|
+
/**
|
|
2969
|
+
* Which vendor a model id belongs to, or `undefined` when no supported provider claims it.
|
|
2970
|
+
*
|
|
2971
|
+
* @remarks
|
|
2972
|
+
* Derived from {@link SUPPORTED_ANTHROPIC_MODEL_IDS} and {@link SUPPORTED_GEMINI_MODEL_IDS},
|
|
2973
|
+
* so it cannot drift from them the way a hand-maintained map does — adding a model to an
|
|
2974
|
+
* allowlist is enough to teach this too.
|
|
2975
|
+
*
|
|
2976
|
+
* Exists because callers hold a **model id**, not a vendor. A usage ledger reads a message's
|
|
2977
|
+
* `model` and then has to pick the matching pricing function — `anthropicTokenCost` vs
|
|
2978
|
+
* `geminiTokenCost` — whose usage records are deliberately not interchangeable, because the
|
|
2979
|
+
* two providers disagree about whether the prompt figure includes the cached part. Without
|
|
2980
|
+
* this, every such caller writes its own model→vendor map, which is the same duplication that
|
|
2981
|
+
* exporting the pricing removes.
|
|
2982
|
+
*
|
|
2983
|
+
* Returns `undefined` rather than falling back to a default vendor so an unrecognised model
|
|
2984
|
+
* is a decision the caller has to make out loud. Silently defaulting is precisely how an
|
|
2985
|
+
* unlisted model gets priced at some other tier's rates — a consumer has already had Haiku
|
|
2986
|
+
* billed at Sonnet rates that way.
|
|
2987
|
+
*
|
|
2988
|
+
* Takes a plain `string`, not a union: the interesting callers are reading a model id back
|
|
2989
|
+
* off a stored message or a config file, where it is untrusted text.
|
|
2990
|
+
*
|
|
2991
|
+
* @beta
|
|
2992
|
+
*/
|
|
2993
|
+
export declare function vendorOfModel(modelId: string): AIProviderType | undefined;
|
|
2994
|
+
|
|
2565
2995
|
/**
|
|
2566
2996
|
* The {@link AIProviderType} behind a vendor label, or `undefined` for a label
|
|
2567
2997
|
* no vendor claims.
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@genesislcap/foundation-ai",
|
|
3
3
|
"description": "Genesis Foundation AI - Provider-agnostic AI configuration and shared utilities",
|
|
4
|
-
"version": "15.
|
|
4
|
+
"version": "15.12.0",
|
|
5
5
|
"sideEffects": false,
|
|
6
6
|
"license": "SEE LICENSE IN license.txt",
|
|
7
7
|
"main": "dist/esm/index.js",
|
|
@@ -52,17 +52,17 @@
|
|
|
52
52
|
}
|
|
53
53
|
},
|
|
54
54
|
"devDependencies": {
|
|
55
|
-
"@genesislcap/foundation-testing": "15.
|
|
56
|
-
"@genesislcap/genx": "15.
|
|
57
|
-
"@genesislcap/rollup-builder": "15.
|
|
58
|
-
"@genesislcap/ts-builder": "15.
|
|
59
|
-
"@genesislcap/uvu-playwright-builder": "15.
|
|
60
|
-
"@genesislcap/vite-builder": "15.
|
|
61
|
-
"@genesislcap/webpack-builder": "15.
|
|
55
|
+
"@genesislcap/foundation-testing": "15.12.0",
|
|
56
|
+
"@genesislcap/genx": "15.12.0",
|
|
57
|
+
"@genesislcap/rollup-builder": "15.12.0",
|
|
58
|
+
"@genesislcap/ts-builder": "15.12.0",
|
|
59
|
+
"@genesislcap/uvu-playwright-builder": "15.12.0",
|
|
60
|
+
"@genesislcap/vite-builder": "15.12.0",
|
|
61
|
+
"@genesislcap/webpack-builder": "15.12.0"
|
|
62
62
|
},
|
|
63
63
|
"dependencies": {
|
|
64
|
-
"@genesislcap/foundation-logger": "15.
|
|
65
|
-
"@genesislcap/foundation-utils": "15.
|
|
64
|
+
"@genesislcap/foundation-logger": "15.12.0",
|
|
65
|
+
"@genesislcap/foundation-utils": "15.12.0",
|
|
66
66
|
"@microsoft/fast-foundation": "2.50.0"
|
|
67
67
|
},
|
|
68
68
|
"repository": {
|
|
@@ -73,5 +73,5 @@
|
|
|
73
73
|
"publishConfig": {
|
|
74
74
|
"access": "public"
|
|
75
75
|
},
|
|
76
|
-
"gitHead": "
|
|
76
|
+
"gitHead": "06f5ab51086c6616c6b96a7a1aebccddb3af7881"
|
|
77
77
|
}
|