@genesislcap/foundation-ai 15.11.0 → 15.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -36,13 +36,26 @@ export declare type AgentPickerMode = 'disabled' | 'select' | 'segmented-control
36
36
  * @beta
37
37
  */
38
38
  export declare interface AggregateUsage {
39
- /** USD cost, provider-reported per request and summed — cache discounts applied. */
39
+ /**
40
+ * USD cost, provider-reported per request and summed — cache discounts applied.
41
+ *
42
+ * The authoritative total. It is a **sum of per-request costs**, each computed where
43
+ * the full provider usage was in hand; it is never re-derived from the token buckets
44
+ * below, and should not be.
45
+ */
40
46
  costUsd: number;
41
47
  /** Prompt tokens billed at the full uncached input rate. */
42
48
  uncachedInputTokens: number;
43
49
  /** Prompt tokens served from the provider's cache. */
44
50
  cacheReadTokens: number;
45
- /** Prompt tokens written to the provider's cache. */
51
+ /**
52
+ * Prompt tokens written to the provider's cache.
53
+ *
54
+ * **Volume only — this aggregate can never be priced.** Two reasons, either of which
55
+ * is sufficient: Anthropic's per-TTL write split is already collapsed on each
56
+ * contributing message, and a run may span models (and providers) at different rates,
57
+ * so no single rate applies to the total. Report it, chart it, do not multiply it.
58
+ */
46
59
  cacheWriteTokens: number;
47
60
  /** Generated (output) tokens, including any reasoning tokens the provider bills as output. */
48
61
  outputTokens: number;
@@ -227,6 +240,23 @@ declare interface AITransport {
227
240
  isAvailable?(): Promise<boolean>;
228
241
  }
229
242
 
243
+ /**
244
+ * Prompt-cache pricing multipliers, applied to the model's base input rate
245
+ * (`promptPerMillion`) — https://docs.claude.com/en/docs/build-with-claude/prompt-caching
246
+ * Reads bill at ~0.1× base input; writes bill by TTL — 5-minute at ~1.25× and 1-hour at 2×.
247
+ * Both write TTLs are reachable (`CachePolicy.ttl` is `'5m' | '1h'`), so each TTL bucket is
248
+ * costed from the response's per-TTL `cache_creation` breakdown rather than assuming one rate.
249
+ *
250
+ * @beta
251
+ */
252
+ export declare const ANTHROPIC_CACHE_READ_MULTIPLIER = 0.1;
253
+
254
+ /** @beta */
255
+ export declare const ANTHROPIC_CACHE_WRITE_1H_MULTIPLIER = 2;
256
+
257
+ /** @beta */
258
+ export declare const ANTHROPIC_CACHE_WRITE_5M_MULTIPLIER = 1.25;
259
+
230
260
  /**
231
261
  * Anthropic server-proxy AI configuration (client calls your server; server calls Anthropic).
232
262
  *
@@ -279,6 +309,29 @@ declare interface AnthropicProviderConfig {
279
309
  criteriaInstructions?: string;
280
310
  }
281
311
 
312
+ /**
313
+ * Standard tier pricing per million tokens —
314
+ * https://docs.claude.com/en/docs/about-claude/pricing
315
+ *
316
+ * @remarks
317
+ * Backed by a total {@link https://www.typescriptlang.org/docs/handbook/utility-types.html#recordkeys-type | Record}
318
+ * over `AnthropicModelId` rather than a chain of checks with a default. That is the whole
319
+ * point: adding a model to `SUPPORTED_ANTHROPIC_MODEL_IDS` without pricing it becomes a
320
+ * COMPILE error instead of a silent charge at whatever the fall-through happened to be — the
321
+ * exact failure this module's header warns about, and one a consumer has already been bitten
322
+ * by. The Gemini half of this file has always had the guarantee; this half now matches.
323
+ *
324
+ * @beta
325
+ */
326
+ export declare function anthropicRatesFor(model: AnthropicModelId): TokenRates;
327
+
328
+ /**
329
+ * Cost one Anthropic request.
330
+ *
331
+ * @beta
332
+ */
333
+ export declare function anthropicTokenCost(model: AnthropicModelId, usage: AnthropicUsageRecord): TokenCost;
334
+
282
335
  /**
283
336
  * Transport for Anthropic Claude. Calls the Messages API directly when `apiKey`
284
337
  * is provided, otherwise falls back to a server-proxy endpoint (if `serverEndpoint`
@@ -309,6 +362,26 @@ export declare class AnthropicTransport implements AITransport, ChatTransport, C
309
362
  * accrue. Surfaced alongside `getLifetimeCost`.
310
363
  */
311
364
  private lifetimeSavingsUsd;
365
+ /**
366
+ * Whether we have already told the caller their `thinkingPolicy: 'off'` cannot be honoured
367
+ * on this model. Latched, because it would otherwise fire on every turn of a tool loop.
368
+ */
369
+ private warnedThinkingClamped;
370
+ /**
371
+ * Serving-model ids already reported as unrecognised. Per-id rather than a single flag, so a
372
+ * second unknown model is still announced. See {@link AnthropicTransport.servingModel}.
373
+ */
374
+ private readonly warnedUnknownServingModels;
375
+ /**
376
+ * Warn once when a requested policy is silently clamped, in EITHER direction — which is what
377
+ * `ChatThinkingPolicy` promises callers. Both directions cost the caller something they asked
378
+ * for and would otherwise get no signal about: `'off'` on a model that always thinks keeps
379
+ * billing reasoning as output, and `'auto'` on a model without adaptive support means an agent
380
+ * that asked to reason quietly does not. (Gemini's twin deliberately stays silent on `'auto'`,
381
+ * but only because dynamic thinking is already the default there; Haiku defaults to none, so
382
+ * the same silence would hide a real difference.)
383
+ */
384
+ private warnIfThinkingUnclampable;
312
385
  constructor(config?: AnthropicTransportConfig);
313
386
  getConfig(): {
314
387
  provider: 'anthropic';
@@ -364,6 +437,76 @@ export declare class AnthropicTransport implements AITransport, ChatTransport, C
364
437
  * the payload tidy.
365
438
  */
366
439
  private toAnthropicMessages;
440
+ /**
441
+ * The blocks to replay ahead of a tool call — fallback boundaries then reasoning, or none.
442
+ *
443
+ * A `signature` is only valid for the model that produced it, so reasoning captured under a
444
+ * *different* model is normally dropped: an agent that varies `provider` by state can switch
445
+ * models mid-loop, and replaying the old model's signatures would send blocks the new one
446
+ * cannot verify.
447
+ *
448
+ * Two ways the producer can be the model that will validate:
449
+ *
450
+ * 1. **It is the model we are asking for.** `state.model === this.model` — the ordinary case.
451
+ * 2. **Sticky routing will send this conversation back to it.** After a conversation falls
452
+ * back, later requests carrying `fallbacks` go straight to the model that served, without
453
+ * re-running the one that declined. That is what makes a fallback producer's reasoning
454
+ * replayable at all (Fable 5 configured, Opus 4.8 serving — the pairing this transport's
455
+ * own constructor warning recommends). But it holds only while the conversation continues
456
+ * under the SAME request configuration, which is why `requestedModel` is compared rather
457
+ * than just checking the chain for the producer.
458
+ *
459
+ * That second condition is deliberately narrow. Chain membership alone is too weak: a
460
+ * fallback target is only *contingently* the server, so an agent that switches `provider` to
461
+ * a model whose own chain happens to contain the old producer would replay foreign
462
+ * signatures to whichever model actually answers. Requiring the requested model to be
463
+ * unchanged separates "this conversation is still going" from "we are somewhere else now".
464
+ *
465
+ * State captured before `requestedModel` existed falls back to condition 1 alone, which is
466
+ * the conservative branch — it can drop reasoning, never misdirect it.
467
+ *
468
+ * Boundaries themselves are always echoed: keeping them in place is the documented rule, and
469
+ * with no thinking blocks around them they are inert rather than harmful.
470
+ */
471
+ private reasoningToReplay;
472
+ /**
473
+ * The model that actually served this response.
474
+ *
475
+ * A server-side `fallbacks` chain re-runs a declined request on another model and
476
+ * names it in the response's top-level `model`. Pricing must follow that, not the
477
+ * model we asked for: a Fable 5 request served by Opus 4.8 costed at Fable's
478
+ * $10/$50 instead of $5/$25 doubles that part of the bill.
479
+ *
480
+ * An unrecognised model id is NOT priced at a guessed rate — that is precisely how
481
+ * a consumer ended up billing Haiku at Sonnet rates. It warns and falls back to the
482
+ * configured model, which is at least a figure someone chose.
483
+ */
484
+ private servingModel;
485
+ /**
486
+ * Whether a fallback-chain attempt is a refusal that was **not billed**.
487
+ *
488
+ * Anthropic does not charge for a refusal that arrives before any output: the token counts
489
+ * still appear in `usage`, but they are not on the bill. Pricing them anyway turns the
490
+ * fallback undercount this reducer was written to fix into an overcount — the documented
491
+ * example declines with `input_tokens: 535, output_tokens: 0`, which is real money at
492
+ * Fable 5's prompt rate.
493
+ *
494
+ * A **mid-output** refusal *is* billed for the input and whatever it streamed, so output
495
+ * tokens are the discriminator rather than the refusal itself.
496
+ *
497
+ * Declined attempts appear as `type: 'message'`; the attempt that served the turn is
498
+ * `type: 'fallback_message'`. The serving entry is normally billable — except when every
499
+ * model in the chain declined, where the last entry is both the serving one and a refusal,
500
+ * which `lastAndRefused` covers.
501
+ *
502
+ * Also called for a response with no chain at all, as `isUnbilledRefusal({}, usage, refused)`:
503
+ * a direct request that was declined has no `iterations` array, and the same rule applies to
504
+ * it. That is in fact the common case — `iterations` only appears when `fallbacks` was
505
+ * configured.
506
+ */
507
+ private isUnbilledRefusal;
508
+ /** Cost one attempt's usage block at `model`'s rates, logging and banking it. */
509
+ private costAttempt;
367
510
  private fromAnthropicResponse;
368
511
  private buildEndpoint;
369
512
  private static readonly RATE_LIMIT_STATUS;
@@ -407,6 +550,38 @@ declare interface AnthropicTransportConfig {
407
550
  maxTokens?: number;
408
551
  }
409
552
 
553
+ /**
554
+ * Token counts from an Anthropic `usage` block.
555
+ *
556
+ * @remarks
557
+ * Note the input semantics, which are the opposite of Gemini's: `input_tokens` on
558
+ * this provider is the **uncached remainder**, so the three input buckets ADD to
559
+ * the prompt total rather than breaking it down.
560
+ *
561
+ * @beta
562
+ */
563
+ export declare interface AnthropicUsageRecord {
564
+ /** `usage.input_tokens` — the uncached remainder of the prompt, billed at full rate. */
565
+ uncachedInputTokens: number;
566
+ /** `usage.output_tokens`. Includes thinking tokens, which this provider bills as output. */
567
+ outputTokens: number;
568
+ /** `usage.cache_read_input_tokens`. */
569
+ cacheReadTokens: number;
570
+ /** `usage.cache_creation_input_tokens` — the total across BOTH TTLs. */
571
+ cacheWriteTokens: number;
572
+ /**
573
+ * `usage.cache_creation.ephemeral_1h_input_tokens` — the 1-hour portion of
574
+ * {@link AnthropicUsageRecord.cacheWriteTokens}. The remainder is treated as
575
+ * 5-minute, which is also the correct fallback when a response omits the
576
+ * per-TTL breakdown entirely.
577
+ *
578
+ * Pass `0` if unknown. Doing so under-bills a request that really did write to
579
+ * the 1-hour cache — by the difference between the two multipliers — so prefer
580
+ * reading the breakdown when the provider sends one.
581
+ */
582
+ cacheWrite1hTokens: number;
583
+ }
584
+
410
585
  /**
411
586
  * The vendors whose spend the ai-service proxy meters — i.e. the only vendors
412
587
  * that can raise a {@link BudgetExhaustedError}, and therefore the only ones a
@@ -914,6 +1089,18 @@ export declare interface ChatMessage {
914
1089
  * {@link ChatMessage.inputTokens} rather than an addition to it. `undefined`
915
1090
  * where the provider reports no distinct write bucket (e.g. Gemini's implicit
916
1091
  * caching) — read it with `?? 0`.
1092
+ *
1093
+ * **A volume figure, not a pricing input — do not re-price it.** Anthropic bills
1094
+ * cache writes by TTL (5-minute and 1-hour at different multiples of the input
1095
+ * rate) and this field is the combined total, so the split needed to cost it is
1096
+ * not recoverable from a stored message. Pricing this bucket therefore under-bills
1097
+ * a 1-hour write, silently and in the cheap direction. {@link ChatMessage.cost} is
1098
+ * the authoritative figure: the transport computed it from the per-TTL breakdown
1099
+ * before that breakdown was collapsed to this total.
1100
+ *
1101
+ * A host that genuinely must compute a cost itself — a proxy holding a raw provider
1102
+ * usage block, where nothing stamped one — should price from that raw block instead,
1103
+ * which carries the split.
917
1104
  */
918
1105
  cacheWriteTokens?: number;
919
1106
  /**
@@ -921,6 +1108,25 @@ export declare interface ChatMessage {
921
1108
  * derived from the active model's input/output rates. Set by transports.
922
1109
  * Hosts can sum this across the message list to display a running session
923
1110
  * cost without re-deriving rates.
1111
+ *
1112
+ * @remarks
1113
+ * **The authoritative cost for this request.** Computed at the transport, at the
1114
+ * one point where the full provider usage is in hand — including Anthropic's
1115
+ * per-TTL cache-write breakdown, which the token fields on this message do not
1116
+ * preserve. Prefer it over any figure re-derived from those fields, which cannot
1117
+ * be exact and is wrong in the direction that under-reports spend.
1118
+ *
1119
+ * Already carries every cache discount and premium, so a cache-heavy request shows
1120
+ * a small `cost` against a large token count — that is the discount working, not a
1121
+ * missing charge.
1122
+ *
1123
+ * **When a server-side fallback chain ran, this and the token fields have different
1124
+ * scopes.** `cost` sums every *billed* attempt, possibly across models at different
1125
+ * rates, while `inputTokens` / `outputTokens` / the cache buckets describe only the
1126
+ * attempt that produced the content. So such a turn can carry cost with no volume
1127
+ * behind it to explain the difference — a $/token column or a "cost without tokens"
1128
+ * audit will see an outlier that is correct. Read `cost` as the bill and the token
1129
+ * fields as the size of the answer, not as two views of one thing.
924
1130
  */
925
1131
  cost?: number;
926
1132
  /**
@@ -1043,6 +1249,14 @@ export declare interface ChatRequestOptions {
1043
1249
  * @beta
1044
1250
  */
1045
1251
  responseSchema?: object;
1252
+ /**
1253
+ * Extended-thinking posture for this turn. **Omit** to keep each model's own default —
1254
+ * that is the third state, and it is not the same as either value. See
1255
+ * {@link ChatThinkingPolicy}.
1256
+ *
1257
+ * @beta
1258
+ */
1259
+ thinkingPolicy?: ChatThinkingPolicy;
1046
1260
  }
1047
1261
 
1048
1262
  /**
@@ -1164,6 +1378,46 @@ export declare const ChatTemperature: {
1164
1378
  readonly Maximum: 1;
1165
1379
  };
1166
1380
 
1381
+ /**
1382
+ * Whether the model should reason before answering, resolved per turn.
1383
+ * **Provider-neutral intent, provider-specific effect**, and — like `tool_choice` — a
1384
+ * *request*, not a guarantee: a model that cannot honour it keeps its default.
1385
+ *
1386
+ * - `'auto'` — the model decides how much to think (Anthropic `thinking: {type:'adaptive'}`,
1387
+ * Gemini `thinkingConfig.thinkingBudget: -1`). On a model whose default is thinking-*off*
1388
+ * this turns it **on**, which is the point: it is the opt-in for models where adaptive
1389
+ * thinking is available but not default.
1390
+ * - `'off'` — no reasoning tokens (Anthropic `thinking: {type:'disabled'}`, Gemini
1391
+ * `thinkingBudget: 0`). The lever for turns that only step a known sequence, where reasoning
1392
+ * is billed as output at the full candidate rate to decide something already determined.
1393
+ *
1394
+ * **Choose it per turn, not per model call.** A tool-use loop is one assistant turn, and
1395
+ * Anthropic requires a single thinking mode for its duration; toggling part-way through does
1396
+ * not error but silently disables thinking for that request, strips blocks that would leave the
1397
+ * turn structure invalid, and invalidates the prompt cache. The driver pins whatever it resolves
1398
+ * on a turn's first model call for the rest of that turn.
1399
+ *
1400
+ * Deliberately has **no token-budget member**. Gemini accepts a numeric `thinkingBudget`, but
1401
+ * Anthropic *removed* `budget_tokens` and returns a 400 for it on Sonnet 5 and every Opus 5-era
1402
+ * model — so a budget field would be unimplementable on half the supported fleet and would exist
1403
+ * only to be ignored. `'low' | 'high'` can be added later if wanted: Anthropic `effort` and
1404
+ * Gemini `thinkingLevel` do line up.
1405
+ *
1406
+ * **Omitting this is a distinct third state** — each model keeps its own default posture
1407
+ * (see `anthropicThinking`), which is *not* uniformly `'auto'` or `'off'`. So a resolver
1408
+ * returning `undefined` on some turns leaves those turns exactly as they were before this
1409
+ * option existed, rather than silently re-pricing them.
1410
+ *
1411
+ * Not every model can honour every value, and the two providers hit the same wall at the top
1412
+ * of their ranges: Anthropic Fable 5 and Gemini 2.5 Pro both think unconditionally and reject
1413
+ * (or ignore) a request to stop. Transports **clamp** rather than forward a request that would
1414
+ * 400 — the turn runs at the model's default and a one-time warning is logged. Never assume
1415
+ * `'off'` means zero reasoning tokens were billed; read the usage back.
1416
+ *
1417
+ * @beta
1418
+ */
1419
+ export declare type ChatThinkingPolicy = 'auto' | 'off';
1420
+
1167
1421
  /**
1168
1422
  * A tool call requested by the assistant.
1169
1423
  *
@@ -1774,6 +2028,27 @@ export declare type FieldLike = string | {
1774
2028
  NAME?: string;
1775
2029
  };
1776
2030
 
2031
+ /**
2032
+ * Cached / context-cache input tokens bill at a flat ~10% of the model's normal input
2033
+ * rate — consistent across models (flash $0.30→$0.03, flash-lite $0.10→$0.01,
2034
+ * pro $1.25→$0.125). Source: https://ai.google.dev/gemini-api/docs/pricing.
2035
+ * (Explicit context caching also has an hourly storage fee; implicit caching — what the
2036
+ * transport relies on — has none, so only this read discount applies. There is no write
2037
+ * premium on this provider, which is why its savings figure is never negative.)
2038
+ *
2039
+ * @beta
2040
+ */
2041
+ export declare const GEMINI_CACHED_INPUT_MULTIPLIER = 0.1;
2042
+
2043
+ /**
2044
+ * Prompt size (tokens) at or below which the standard pricing tier applies.
2045
+ * Above it, Gemini's long-context tier kicks in. Google applies a single tier
2046
+ * to the whole request based on prompt size — it is not a marginal/blended rate.
2047
+ *
2048
+ * @beta
2049
+ */
2050
+ export declare const GEMINI_LONG_CONTEXT_THRESHOLD = 200000;
2051
+
1777
2052
  /**
1778
2053
  * Gemini server-proxy AI configuration (client calls your server; server calls Gemini).
1779
2054
  *
@@ -1818,6 +2093,30 @@ declare interface GeminiProviderConfig {
1818
2093
  criteriaInstructions?: string;
1819
2094
  }
1820
2095
 
2096
+ /**
2097
+ * Resolve the per-million-token rates for a model given the request's prompt size,
2098
+ * selecting the long-context tier for tiered models when the prompt exceeds
2099
+ * {@link GEMINI_LONG_CONTEXT_THRESHOLD}.
2100
+ *
2101
+ * @remarks
2102
+ * `promptTokens` must be the **full** prompt size, cached slice included — the tier
2103
+ * is chosen on what was sent, not on what was billed at full rate. A mostly-cached
2104
+ * long prompt is still a long prompt.
2105
+ *
2106
+ * Unlike Anthropic's, this accessor is not a pure function of the model, which is
2107
+ * why there is no single `ratesFor(model)` across both providers.
2108
+ *
2109
+ * @beta
2110
+ */
2111
+ export declare function geminiRatesFor(model: GeminiModelId, promptTokens: number): TokenRates;
2112
+
2113
+ /**
2114
+ * Cost one Gemini request.
2115
+ *
2116
+ * @beta
2117
+ */
2118
+ export declare function geminiTokenCost(model: GeminiModelId, usage: GeminiUsageRecord): TokenCost;
2119
+
1821
2120
  /**
1822
2121
  * Transport for Gemini. Calls the Gemini REST API directly when `apiKey` is
1823
2122
  * provided, otherwise falls back to a server-proxy endpoint (if `serverEndpoint`
@@ -1846,6 +2145,22 @@ export declare class GeminiTransport implements AITransport, ChatTransport, Cost
1846
2145
  * Surfaced alongside `getLifetimeCost`.
1847
2146
  */
1848
2147
  private lifetimeSavingsUsd;
2148
+ /**
2149
+ * Whether we have already told the caller their `thinkingPolicy` cannot be honoured on this
2150
+ * model. Latched, because it would otherwise fire on every turn of a tool loop.
2151
+ */
2152
+ private warnedThinkingClamped;
2153
+ /**
2154
+ * Warn once when a requested policy is silently clamped. Turning thinking off is a *cost*
2155
+ * decision, so a caller who asked for it and kept paying for reasoning tokens needs to hear
2156
+ * about it — but only once, not per turn.
2157
+ *
2158
+ * Only `'off'` can actually be denied. `'auto'` is already what an omitted budget produces on
2159
+ * every model here — dynamic thinking is the documented default — so warning that it was
2160
+ * "ignored" would be false, and latching on it would spend the one warning the genuinely
2161
+ * unhonourable `'off'` needs.
2162
+ */
2163
+ private warnIfThinkingUnclampable;
1849
2164
  constructor(config?: GeminiTransportConfig);
1850
2165
  getConfig(): {
1851
2166
  provider: 'gemini';
@@ -1953,6 +2268,31 @@ declare interface GeminiTransportConfig {
1953
2268
  serverEndpoint?: string;
1954
2269
  }
1955
2270
 
2271
+ /**
2272
+ * Token counts from a Gemini `usageMetadata` block.
2273
+ *
2274
+ * @remarks
2275
+ * Note the input semantics, which are the opposite of Anthropic's:
2276
+ * `promptTokenCount` is the **whole** prompt and already includes
2277
+ * `cachedContentTokenCount`, so uncached input is a subtraction. Adding the buckets
2278
+ * instead double-charges the cached portion.
2279
+ *
2280
+ * @beta
2281
+ */
2282
+ export declare interface GeminiUsageRecord {
2283
+ /** `usageMetadata.promptTokenCount` — the whole prompt, cached slice INCLUDED. */
2284
+ promptTokens: number;
2285
+ /** `usageMetadata.candidatesTokenCount`. */
2286
+ candidateTokens: number;
2287
+ /**
2288
+ * `usageMetadata.thoughtsTokenCount`. Reported separately from candidates but
2289
+ * billed at the candidate rate, so omitting it undercounts the bill.
2290
+ */
2291
+ thoughtTokens: number;
2292
+ /** `usageMetadata.cachedContentTokenCount` — 0 when caching was not active. */
2293
+ cachedTokens: number;
2294
+ }
2295
+
1956
2296
  /**
1957
2297
  * How the host frames an interaction widget in the chat transcript. The
1958
2298
  * assistant avatar is always shown; this controls only the bubble and the
@@ -2500,6 +2840,69 @@ export declare const SUPPORTED_ANTHROPIC_MODEL_IDS: readonly AnthropicModelId[];
2500
2840
  /** @beta */
2501
2841
  export declare const SUPPORTED_GEMINI_MODEL_IDS: readonly GeminiModelId[];
2502
2842
 
2843
+ /**
2844
+ * What one request cost, and what prompt caching saved on it.
2845
+ *
2846
+ * @beta
2847
+ */
2848
+ export declare interface TokenCost {
2849
+ /** Total USD for the request, with every cache discount and premium applied. */
2850
+ costUsd: number;
2851
+ /**
2852
+ * USD saved versus paying the full input rate for every cached token, net of any
2853
+ * cache-write premium.
2854
+ *
2855
+ * **Can be negative**, and that is not an error: a write premium is an upfront
2856
+ * cost that later reads pay back, so a cold write-heavy request legitimately
2857
+ * reports a loss. Clamping it at zero destroys the signal that a cache is being
2858
+ * written more often than it is read.
2859
+ */
2860
+ savedUsd: number;
2861
+ /**
2862
+ * The four cost buckets, summing to {@link TokenCost.costUsd}.
2863
+ *
2864
+ * Reported because a single total cannot show *why* a request was cheap: a
2865
+ * cache-heavy call moves a large number of tokens for very little money, and only
2866
+ * the split says so. Also what a per-call usage row needs, rather than the one
2867
+ * number a dashboard would have to explain.
2868
+ */
2869
+ breakdown: TokenCostBreakdown;
2870
+ }
2871
+
2872
+ /**
2873
+ * Per-bucket cost split for one request. Each field is already in USD.
2874
+ *
2875
+ * @beta
2876
+ */
2877
+ export declare interface TokenCostBreakdown {
2878
+ /** Prompt tokens billed at the full uncached input rate. */
2879
+ promptUsd: number;
2880
+ /** Prompt tokens served from the provider's cache. */
2881
+ cacheReadUsd: number;
2882
+ /**
2883
+ * Prompt tokens written to the provider's cache. Always `0` on Gemini, whose
2884
+ * implicit caching has no write charge.
2885
+ *
2886
+ * A per-request charge for *populating* a cache. Not a home for a cache's ongoing
2887
+ * storage cost, which is not per-request — see the module header.
2888
+ */
2889
+ cacheWriteUsd: number;
2890
+ /** Generated output, including any reasoning tokens billed as output. */
2891
+ candidateUsd: number;
2892
+ }
2893
+
2894
+ /**
2895
+ * USD per million tokens, split by direction. `prompt` covers input/prompt tokens,
2896
+ * `candidate` covers generated output (including any reasoning tokens the provider
2897
+ * bills as output).
2898
+ *
2899
+ * @beta
2900
+ */
2901
+ export declare interface TokenRates {
2902
+ promptPerMillion: number;
2903
+ candidatePerMillion: number;
2904
+ }
2905
+
2503
2906
  /**
2504
2907
  * Why a driver turn ended in failure — the typed taxonomy the tool loop already
2505
2908
  * records onto its debug-log timeline (`turn.error` / `turn.retry` details),
@@ -2562,6 +2965,33 @@ export declare type TurnFailureReason = 'exception' | 'malformed-function-call'
2562
2965
  */
2563
2966
  export declare const VENDOR_LABELS: Readonly<Record<Exclude<AIProviderType, 'none'>, string>>;
2564
2967
 
2968
+ /**
2969
+ * Which vendor a model id belongs to, or `undefined` when no supported provider claims it.
2970
+ *
2971
+ * @remarks
2972
+ * Derived from {@link SUPPORTED_ANTHROPIC_MODEL_IDS} and {@link SUPPORTED_GEMINI_MODEL_IDS},
2973
+ * so it cannot drift from them the way a hand-maintained map does — adding a model to an
2974
+ * allowlist is enough to teach this too.
2975
+ *
2976
+ * Exists because callers hold a **model id**, not a vendor. A usage ledger reads a message's
2977
+ * `model` and then has to pick the matching pricing function — `anthropicTokenCost` vs
2978
+ * `geminiTokenCost` — whose usage records are deliberately not interchangeable, because the
2979
+ * two providers disagree about whether the prompt figure includes the cached part. Without
2980
+ * this, every such caller writes its own model→vendor map, which is the same duplication that
2981
+ * exporting the pricing removes.
2982
+ *
2983
+ * Returns `undefined` rather than falling back to a default vendor so an unrecognised model
2984
+ * is a decision the caller has to make out loud. Silently defaulting is precisely how an
2985
+ * unlisted model gets priced at some other tier's rates — a consumer has already had Haiku
2986
+ * billed at Sonnet rates that way.
2987
+ *
2988
+ * Takes a plain `string`, not a union: the interesting callers are reading a model id back
2989
+ * off a stored message or a config file, where it is untrusted text.
2990
+ *
2991
+ * @beta
2992
+ */
2993
+ export declare function vendorOfModel(modelId: string): AIProviderType | undefined;
2994
+
2565
2995
  /**
2566
2996
  * The {@link AIProviderType} behind a vendor label, or `undefined` for a label
2567
2997
  * no vendor claims.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@genesislcap/foundation-ai",
3
3
  "description": "Genesis Foundation AI - Provider-agnostic AI configuration and shared utilities",
4
- "version": "15.11.0",
4
+ "version": "15.12.0",
5
5
  "sideEffects": false,
6
6
  "license": "SEE LICENSE IN license.txt",
7
7
  "main": "dist/esm/index.js",
@@ -52,17 +52,17 @@
52
52
  }
53
53
  },
54
54
  "devDependencies": {
55
- "@genesislcap/foundation-testing": "15.11.0",
56
- "@genesislcap/genx": "15.11.0",
57
- "@genesislcap/rollup-builder": "15.11.0",
58
- "@genesislcap/ts-builder": "15.11.0",
59
- "@genesislcap/uvu-playwright-builder": "15.11.0",
60
- "@genesislcap/vite-builder": "15.11.0",
61
- "@genesislcap/webpack-builder": "15.11.0"
55
+ "@genesislcap/foundation-testing": "15.12.0",
56
+ "@genesislcap/genx": "15.12.0",
57
+ "@genesislcap/rollup-builder": "15.12.0",
58
+ "@genesislcap/ts-builder": "15.12.0",
59
+ "@genesislcap/uvu-playwright-builder": "15.12.0",
60
+ "@genesislcap/vite-builder": "15.12.0",
61
+ "@genesislcap/webpack-builder": "15.12.0"
62
62
  },
63
63
  "dependencies": {
64
- "@genesislcap/foundation-logger": "15.11.0",
65
- "@genesislcap/foundation-utils": "15.11.0",
64
+ "@genesislcap/foundation-logger": "15.12.0",
65
+ "@genesislcap/foundation-utils": "15.12.0",
66
66
  "@microsoft/fast-foundation": "2.50.0"
67
67
  },
68
68
  "repository": {
@@ -73,5 +73,5 @@
73
73
  "publishConfig": {
74
74
  "access": "public"
75
75
  },
76
- "gitHead": "261d900881ea1ca2b613033af497ffd449868eb0"
76
+ "gitHead": "06f5ab51086c6616c6b96a7a1aebccddb3af7881"
77
77
  }