@broberg/ai-sdk 0.31.0 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-LKVCPMVI.js → chunk-DEIY7O3T.js} +11 -5
- package/dist/chunk-DEIY7O3T.js.map +1 -0
- package/dist/index.d.ts +2 -2
- package/dist/index.js +21 -10
- package/dist/index.js.map +1 -1
- package/dist/pricing.js +1 -1
- package/package.json +1 -1
- package/dist/chunk-LKVCPMVI.js.map +0 -1
|
@@ -61,18 +61,24 @@ var PRICING = {
|
|
|
61
61
|
// Verify against a real key when it lands.
|
|
62
62
|
"deepseek:deepseek-chat": { inputPer1M: 0.14, outputPer1M: 0.28, version: "2026-06-30-deepseek-direct" },
|
|
63
63
|
"deepseek:deepseek-reasoner": { inputPer1M: 0.14, outputPer1M: 0.28, version: "2026-06-30-deepseek-direct" },
|
|
64
|
+
// Cached input tokens cost 10% of the input rate — $0.03 vs $0.30 (2.5-flash) and
|
|
65
|
+
// $0.01 vs $0.10 (2.5-flash-lite), read from ai.google.dev/gemini-api/docs/pricing
|
|
66
|
+
// on 2026-08-27 rather than recalled. NB the storage fee on that page ($1/1M
|
|
67
|
+
// tokens/hour) applies to EXPLICIT context caching, where you create a CachedContent
|
|
68
|
+
// object with a TTL. We use IMPLICIT caching, which has no storage charge — so this
|
|
69
|
+
// table is not silently under-billing.
|
|
64
70
|
// Google Gemini (direct). Provider key is "gemini" — matches the adapter's
|
|
65
71
|
// usage.provider + the override.provider callers pass. (Image-gen models are
|
|
66
72
|
// priced per-image in the adapter, not here.)
|
|
67
|
-
"gemini:gemini-2.5-flash": { inputPer1M: 0.3, outputPer1M: 2.5, version:
|
|
73
|
+
"gemini:gemini-2.5-flash": { inputPer1M: 0.3, cacheReadPer1M: 0.03, outputPer1M: 2.5, version: "2026-08-27-ai.google.dev" },
|
|
68
74
|
// flash-lite is the default `video` tier (F019) — cheap native video understanding.
|
|
69
|
-
"gemini:gemini-2.5-flash-lite": { inputPer1M: 0.1, outputPer1M: 0.4, version: "2026-
|
|
75
|
+
"gemini:gemini-2.5-flash-lite": { inputPer1M: 0.1, cacheReadPer1M: 0.01, outputPer1M: 0.4, version: "2026-08-27-ai.google.dev" },
|
|
70
76
|
// Vertex AI (F038) — the EU-resident route to the SAME Gemini models, so Google's
|
|
71
77
|
// published Gemini token prices apply. Listed separately because cost lookups key on
|
|
72
78
|
// `provider:model`: without these rows an EU vision/video call would silently log
|
|
73
79
|
// $0, which is worse than no tracking (a confident wrong number).
|
|
74
|
-
"vertex:gemini-2.5-flash": { inputPer1M: 0.3, outputPer1M: 2.5, version:
|
|
75
|
-
"vertex:gemini-2.5-flash-lite": { inputPer1M: 0.1, outputPer1M: 0.4, version: "2026-
|
|
80
|
+
"vertex:gemini-2.5-flash": { inputPer1M: 0.3, cacheReadPer1M: 0.03, outputPer1M: 2.5, version: "2026-08-27-ai.google.dev" },
|
|
81
|
+
"vertex:gemini-2.5-flash-lite": { inputPer1M: 0.1, cacheReadPer1M: 0.01, outputPer1M: 0.4, version: "2026-08-27-ai.google.dev" },
|
|
76
82
|
// Cached prompt tokens bill at 10% of the input rate (F039, measured 2026-08-27:
|
|
77
83
|
// an 8,810-token prefix reported 8,784 cached on the second call WITH a
|
|
78
84
|
// prompt_cache_key, and 0 without one at every size up to 57k).
|
|
@@ -114,4 +120,4 @@ export {
|
|
|
114
120
|
PRICING,
|
|
115
121
|
getPrice
|
|
116
122
|
};
|
|
117
|
-
//# sourceMappingURL=chunk-
|
|
123
|
+
//# sourceMappingURL=chunk-DEIY7O3T.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/cost/pricing.ts"],"sourcesContent":["// Versioned per-(provider, model) pricing. F3.6 populates the table + adds tests\n// + MiniMax coverage. F3.1 ships the type + lookup with an empty table, so\n// computeCost returns 0 for every model until F3.6 lands (calls still complete).\nexport interface PricingEntry {\n /** USD per 1M input tokens. */\n inputPer1M: number;\n /** USD per 1M output tokens. */\n outputPer1M: number;\n /** USD per 1M cache-read tokens (falls back to input rate if unset). */\n cacheReadPer1M?: number;\n /** USD per 1M cache-write/creation tokens (falls back to input rate if unset). */\n cacheWritePer1M?: number;\n /** Pricing snapshot version (date or tag) so stale entries are detectable. */\n version: string;\n}\n\n// USD per 1M tokens. Anthropic cache multipliers follow the standard model:\n// cache-read ≈ 0.1× input, cache-write ≈ 1.25× input. Verified against the\n// pricing tables in cms (packages/cms-ai/src/providers) + trail (model-lab).\n// MiniMax M2.7 is an estimate pending confirmation against OpenRouter's live\n// price page — flagged in its version string.\nconst V = \"2026-06-02\";\n// Mistral prices come straight from mistral.ai/pricing (per Christian's CD report).\nconst MS = \"2026-06-04-mistral.ai\";\n\n/** Keyed `${provider}:${model}`. Exported so the catalogue-research job (F014)\n * can enumerate every priced entry and diff it against the live provider lists. */\nexport const PRICING: Record<string, PricingEntry> = {\n // Anthropic (direct API). DEFAULT_TIER_MAP: fast/cheap=haiku, smart/vision=sonnet, powerful=opus.\n \"anthropic:claude-haiku-4-5\": {\n inputPer1M: 0.8,\n outputPer1M: 4.0,\n cacheReadPer1M: 0.08,\n cacheWritePer1M: 1.0,\n version: V,\n },\n \"anthropic:claude-sonnet-4-6\": {\n inputPer1M: 3.0,\n outputPer1M: 15.0,\n cacheReadPer1M: 0.3,\n cacheWritePer1M: 3.75,\n version: V,\n },\n \"anthropic:claude-opus-4-8\": {\n inputPer1M: 15.0,\n outputPer1M: 75.0,\n cacheReadPer1M: 1.5,\n cacheWritePer1M: 18.75,\n version: V,\n },\n\n // OpenAI. embedding default tier = text-embedding-3-small (no output tokens).\n \"openai:text-embedding-3-small\": { inputPer1M: 0.02, outputPer1M: 0, version: V },\n \"openai:text-embedding-3-large\": { inputPer1M: 0.13, outputPer1M: 0, version: V },\n \"openai:gpt-4o\": { inputPer1M: 2.5, outputPer1M: 10.0, version: V },\n \"openai:gpt-4o-mini\": { inputPer1M: 0.15, outputPer1M: 0.6, version: V },\n // Whisper is priced per minute, not per token — not representable here; transcribe\n // (F5.6) computes its own cost. Listed as 0 so token-based compute never charges it.\n \"openai:whisper-1\": { inputPer1M: 0, outputPer1M: 0, version: V },\n\n // OpenRouter (meta-router — model slugs include the upstream vendor). Slugs use\n // dots (claude-sonnet-4.6) to match OpenRouter's live ids; the dashed forms\n // never matched a real call. Caught by the F014 catalogue research.\n \"openrouter:anthropic/claude-sonnet-4.6\": { inputPer1M: 3.0, outputPer1M: 15.0, version: V },\n // OpenRouter ground-truth $1/$5 — a markup over Anthropic-direct's $0.8/$4\n // (the `anthropic:` entry above). Was masked while the slug used dashes.\n \"openrouter:anthropic/claude-haiku-4.5\": { inputPer1M: 1.0, outputPer1M: 5.0, version: \"2026-06-04\" },\n \"openrouter:google/gemini-2.5-flash\": { inputPer1M: 0.3, outputPer1M: 2.5, version: V },\n // Ground-truth from OpenRouter /api/v1/models (was a 0.3 estimate; now 0.279).\n \"openrouter:minimax/minimax-m2.7\": {\n inputPer1M: 0.279,\n outputPer1M: 1.2,\n version: \"2026-06-04\",\n },\n // DeepSeek V4 (CN-hosted — NOT GDPR-safe; non-personal-data workloads only).\n // On 2026-05-22 DeepSeek made the \"75% off\" promo the permanent official price.\n // V4-Pro $0.435/$0.87 is ~34x cheaper than GPT-5.5 on output; flash is cheaper\n // still. Numbers match OpenRouter /api/v1/models 1:1 (no router markup). A strong\n // cheap route for fleet background work once `claude -p` is API-billed (15 Jun).\n \"openrouter:deepseek/deepseek-v4-pro\": { inputPer1M: 0.435, outputPer1M: 0.87, version: \"2026-05-22-deepseek-official\" },\n \"openrouter:deepseek/deepseek-v4-flash\": { inputPer1M: 0.0983, outputPer1M: 0.1966, version: \"2026-05-22-deepseek-official\" },\n // DeepSeek DIRECT API (provider \"deepseek\", F030 non-PII secondary). Rates from\n // api-docs.deepseek.com 2026-06-30 ($0.14/$0.28 per 1M; both map to deepseek-v4-flash).\n // `deepseek-chat` (non-thinking) + `deepseek-reasoner` (thinking) DEPRECATE 2026-07-24.\n // (The bare `deepseek-v4-flash` basename is already priced via the openrouter entry\n // above — kept distinct here to avoid a basename collision in the F027 pricing-API.)\n // Verify against a real key when it lands.\n \"deepseek:deepseek-chat\": { inputPer1M: 0.14, outputPer1M: 0.28, version: \"2026-06-30-deepseek-direct\" },\n \"deepseek:deepseek-reasoner\": { inputPer1M: 0.14, outputPer1M: 0.28, version: \"2026-06-30-deepseek-direct\" },\n\n // Cached input tokens cost 10% of the input rate — $0.03 vs $0.30 (2.5-flash) and\n // $0.01 vs $0.10 (2.5-flash-lite), read from ai.google.dev/gemini-api/docs/pricing\n // on 2026-08-27 rather than recalled. NB the storage fee on that page ($1/1M\n // tokens/hour) applies to EXPLICIT context caching, where you create a CachedContent\n // object with a TTL. We use IMPLICIT caching, which has no storage charge — so this\n // table is not silently under-billing.\n // Google Gemini (direct). Provider key is \"gemini\" — matches the adapter's\n // usage.provider + the override.provider callers pass. (Image-gen models are\n // priced per-image in the adapter, not here.)\n \"gemini:gemini-2.5-flash\": { inputPer1M: 0.3, cacheReadPer1M: 0.03, outputPer1M: 2.5, version: \"2026-08-27-ai.google.dev\" },\n // flash-lite is the default `video` tier (F019) — cheap native video understanding.\n \"gemini:gemini-2.5-flash-lite\": { inputPer1M: 0.1, cacheReadPer1M: 0.01, outputPer1M: 0.4, version: \"2026-08-27-ai.google.dev\" },\n\n // Vertex AI (F038) — the EU-resident route to the SAME Gemini models, so Google's\n // published Gemini token prices apply. Listed separately because cost lookups key on\n // `provider:model`: without these rows an EU vision/video call would silently log\n // $0, which is worse than no tracking (a confident wrong number).\n \"vertex:gemini-2.5-flash\": { inputPer1M: 0.3, cacheReadPer1M: 0.03, outputPer1M: 2.5, version: \"2026-08-27-ai.google.dev\" },\n \"vertex:gemini-2.5-flash-lite\": { inputPer1M: 0.1, cacheReadPer1M: 0.01, outputPer1M: 0.4, version: \"2026-08-27-ai.google.dev\" },\n\n // Cached prompt tokens bill at 10% of the input rate (F039, measured 2026-08-27:\n // an 8,810-token prefix reported 8,784 cached on the second call WITH a\n // prompt_cache_key, and 0 without one at every size up to 57k).\n // Mistral (direct, La Plateforme). Official prices from mistral.ai/pricing\n // (2026-06-04, per Christian's CD report). EU/Paris-hosted — the designated\n // GDPR-safe provider for client/personal-data workloads (see F015). NB:\n // medium-3.5 is the premium \"Vibe\" coding tier ($1.5/$7.5); Large 3 ($0.5/$1.5)\n // is the cheaper frontier general-purpose model despite the higher number.\n \"mistral:mistral-large-latest\": { inputPer1M: 0.5, cacheReadPer1M: 0.05, outputPer1M: 1.5, version: MS },\n \"mistral:mistral-large-2512\": { inputPer1M: 0.5, cacheReadPer1M: 0.05, outputPer1M: 1.5, version: MS },\n \"mistral:mistral-medium-latest\": { inputPer1M: 1.5, cacheReadPer1M: 0.15, outputPer1M: 7.5, version: MS },\n \"mistral:mistral-medium-3.5\": { inputPer1M: 1.5, cacheReadPer1M: 0.15, outputPer1M: 7.5, version: MS },\n \"mistral:mistral-medium-3\": { inputPer1M: 0.4, cacheReadPer1M: 0.04, outputPer1M: 2.0, version: \"2026-06-04-or-xref\" },\n \"mistral:mistral-small-latest\": { inputPer1M: 0.1, cacheReadPer1M: 0.01, outputPer1M: 0.3, version: MS },\n \"mistral:mistral-small-2603\": { inputPer1M: 0.1, cacheReadPer1M: 0.01, outputPer1M: 0.3, version: MS },\n \"mistral:ministral-3b-latest\": { inputPer1M: 0.1, cacheReadPer1M: 0.01, outputPer1M: 0.1, version: MS },\n \"mistral:ministral-8b-latest\": { inputPer1M: 0.15, cacheReadPer1M: 0.015, outputPer1M: 0.15, version: MS },\n \"mistral:ministral-14b-latest\": { inputPer1M: 0.2, cacheReadPer1M: 0.02, outputPer1M: 0.2, version: MS },\n \"mistral:magistral-medium-latest\": { inputPer1M: 2.0, cacheReadPer1M: 0.2, outputPer1M: 5.0, version: MS },\n \"mistral:magistral-small-latest\": { inputPer1M: 0.5, cacheReadPer1M: 0.05, outputPer1M: 1.5, version: MS },\n \"mistral:devstral-latest\": { inputPer1M: 0.4, cacheReadPer1M: 0.04, outputPer1M: 2.0, version: MS },\n \"mistral:codestral-latest\": { inputPer1M: 0.3, cacheReadPer1M: 0.03, outputPer1M: 0.9, version: MS },\n \"mistral:open-mistral-nemo\": { inputPer1M: 0.15, cacheReadPer1M: 0.015, outputPer1M: 0.15, version: MS },\n // Moderation (F016.4) — per input token; output 0. (OCR is per-page in the adapter.)\n \"mistral:mistral-moderation-latest\": { inputPer1M: 0.1, cacheReadPer1M: 0.01, outputPer1M: 0, version: MS },\n // Embeddings (F016.5) — per input token.\n \"mistral:mistral-embed\": { inputPer1M: 0.1, cacheReadPer1M: 0.01, outputPer1M: 0, version: MS },\n \"mistral:codestral-embed\": { inputPer1M: 0.15, cacheReadPer1M: 0.015, outputPer1M: 0, version: MS },\n};\n\nexport function getPrice(provider: string, model: string): PricingEntry | undefined {\n const exact = PRICING[`${provider}:${model}`];\n if (exact) return exact;\n // Providers ship dated model snapshots, e.g. \"claude-haiku-4-5-20251001\".\n // Strip a trailing -YYYYMMDD and retry the base lookup so a dated variant\n // prices the same as its base model instead of falling through to 0 — a real\n // paid call must never be logged as $0 (F012). Covers openrouter slugs too.\n const base = model.replace(/-\\d{8}$/, \"\");\n if (base !== model) return PRICING[`${provider}:${base}`];\n return undefined;\n}\n"],"mappings":";AAqBA,IAAM,IAAI;AAEV,IAAM,KAAK;AAIJ,IAAM,UAAwC;AAAA;AAAA,EAEnD,8BAA8B;AAAA,IAC5B,YAAY;AAAA,IACZ,aAAa;AAAA,IACb,gBAAgB;AAAA,IAChB,iBAAiB;AAAA,IACjB,SAAS;AAAA,EACX;AAAA,EACA,+BAA+B;AAAA,IAC7B,YAAY;AAAA,IACZ,aAAa;AAAA,IACb,gBAAgB;AAAA,IAChB,iBAAiB;AAAA,IACjB,SAAS;AAAA,EACX;AAAA,EACA,6BAA6B;AAAA,IAC3B,YAAY;AAAA,IACZ,aAAa;AAAA,IACb,gBAAgB;AAAA,IAChB,iBAAiB;AAAA,IACjB,SAAS;AAAA,EACX;AAAA;AAAA,EAGA,iCAAiC,EAAE,YAAY,MAAM,aAAa,GAAG,SAAS,EAAE;AAAA,EAChF,iCAAiC,EAAE,YAAY,MAAM,aAAa,GAAG,SAAS,EAAE;AAAA,EAChF,iBAAiB,EAAE,YAAY,KAAK,aAAa,IAAM,SAAS,EAAE;AAAA,EAClE,sBAAsB,EAAE,YAAY,MAAM,aAAa,KAAK,SAAS,EAAE;AAAA;AAAA;AAAA,EAGvE,oBAAoB,EAAE,YAAY,GAAG,aAAa,GAAG,SAAS,EAAE;AAAA;AAAA;AAAA;AAAA,EAKhE,0CAA0C,EAAE,YAAY,GAAK,aAAa,IAAM,SAAS,EAAE;AAAA;AAAA;AAAA,EAG3F,yCAAyC,EAAE,YAAY,GAAK,aAAa,GAAK,SAAS,aAAa;AAAA,EACpG,sCAAsC,EAAE,YAAY,KAAK,aAAa,KAAK,SAAS,EAAE;AAAA;AAAA,EAEtF,mCAAmC;AAAA,IACjC,YAAY;AAAA,IACZ,aAAa;AAAA,IACb,SAAS;AAAA,EACX;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAMA,uCAAuC,EAAE,YAAY,OAAO,aAAa,MAAM,SAAS,+BAA+B;AAAA,EACvH,yCAAyC,EAAE,YAAY,QAAQ,aAAa,QAAQ,SAAS,+BAA+B;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAO5H,0BAA0B,EAAE,YAAY,MAAM,aAAa,MAAM,SAAS,6BAA6B;AAAA,EACvG,8BAA8B,EAAE,YAAY,MAAM,aAAa,MAAM,SAAS,6BAA6B;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAW3G,2BAA2B,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,2BAA2B;AAAA;AAAA,EAE1H,gCAAgC,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,2BAA2B;AAAA;AAAA;AAAA;AAAA;AAAA,EAM/H,2BAA2B,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,2BAA2B;AAAA,EAC1H,gCAAgC,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,2BAA2B;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAU/H,gCAAgC,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,GAAG;AAAA,EACvG,8BAA8B,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,GAAG;AAAA,EACrG,iCAAiC,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,GAAG;AAAA,EACxG,8BAA8B,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,GAAG;AAAA,EACrG,4BAA4B,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,GAAK,SAAS,qBAAqB;AAAA,EACrH,gCAAgC,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,GAAG;AAAA,EACvG,8BAA8B,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,GAAG;AAAA,EACrG,+BAA+B,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,GAAG;AAAA,EACtG,+BAA+B,EAAE,YAAY,MAAM,gBAAgB,OAAO,aAAa,MAAM,SAAS,GAAG;AAAA,EACzG,gCAAgC,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,GAAG;AAAA,EACvG,mCAAmC,EAAE,YAAY,GAAK,gBAAgB,KAAK,aAAa,GAAK,SAAS,GAAG;AAAA,EACzG,kCAAkC,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,GAAG;AAAA,EACzG,2BAA2B,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,GAAK,SAAS,GAAG;AAAA,EAClG,4BAA4B,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,KAAK,SAAS,GAAG;AAAA,EACnG,6BAA6B,EAAE,YAAY,MAAM,gBAAgB,OAAO,aAAa,MAAM,SAAS,GAAG;AAAA;AAAA,EAEvG,qCAAqC,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,GAAG,SAAS,GAAG;AAAA;AAAA,EAE1G,yBAAyB,EAAE,YAAY,KAAK,gBAAgB,MAAM,aAAa,GAAG,SAAS,GAAG;AAAA,EAC9F,2BAA2B,EAAE,YAAY,MAAM,gBAAgB,OAAO,aAAa,GAAG,SAAS,GAAG;AACpG;AAEO,SAAS,SAAS,UAAkB,OAAyC;AAClF,QAAM,QAAQ,QAAQ,GAAG,QAAQ,IAAI,KAAK,EAAE;AAC5C,MAAI,MAAO,QAAO;AAKlB,QAAM,OAAO,MAAM,QAAQ,WAAW,EAAE;AACxC,MAAI,SAAS,MAAO,QAAO,QAAQ,GAAG,QAAQ,IAAI,IAAI,EAAE;AACxD,SAAO;AACT;","names":[]}
|
package/dist/index.d.ts
CHANGED
|
@@ -2103,8 +2103,8 @@ declare const falStubAdapter: ProviderAdapter;
|
|
|
2103
2103
|
* wires the live adapters. */
|
|
2104
2104
|
declare const stubProviders: Record<string, ProviderAdapter>;
|
|
2105
2105
|
|
|
2106
|
-
declare const VERSION: "0.
|
|
2107
|
-
declare const SDK_TAG: "@broberg/ai-sdk@0.
|
|
2106
|
+
declare const VERSION: "0.32.0";
|
|
2107
|
+
declare const SDK_TAG: "@broberg/ai-sdk@0.32.0";
|
|
2108
2108
|
|
|
2109
2109
|
/** Built-in defaults. Every entry is overridable via AiConfig.defaults or a
|
|
2110
2110
|
* per-call override.
|
package/dist/index.js
CHANGED
|
@@ -10,7 +10,7 @@ import {
|
|
|
10
10
|
} from "./chunk-V2PD522L.js";
|
|
11
11
|
import {
|
|
12
12
|
getPrice
|
|
13
|
-
} from "./chunk-
|
|
13
|
+
} from "./chunk-DEIY7O3T.js";
|
|
14
14
|
|
|
15
15
|
// src/transport/http.ts
|
|
16
16
|
async function httpTransport(req) {
|
|
@@ -807,6 +807,12 @@ var GEMINI_IMAGE_PRICE_PER_IMAGE = {
|
|
|
807
807
|
"gemini-3-pro-image-preview": 0.134
|
|
808
808
|
// was $0.039 — wrong (that's the flash price); pro is $0.134
|
|
809
809
|
};
|
|
810
|
+
function splitCached(meta) {
|
|
811
|
+
const prompt = meta?.promptTokenCount ?? 0;
|
|
812
|
+
const cached = meta?.cachedContentTokenCount;
|
|
813
|
+
if (cached === void 0) return { inputTokens: prompt };
|
|
814
|
+
return { inputTokens: Math.max(0, prompt - cached), cacheReadTokens: cached };
|
|
815
|
+
}
|
|
810
816
|
function partsFrom(content) {
|
|
811
817
|
if (typeof content === "string") return [{ text: content }];
|
|
812
818
|
return content.map((p) => {
|
|
@@ -881,7 +887,7 @@ function geminiAdapter(config = {}) {
|
|
|
881
887
|
model: req.spec.model,
|
|
882
888
|
transport: "http",
|
|
883
889
|
capability: "chat",
|
|
884
|
-
|
|
890
|
+
...splitCached(data.usageMetadata),
|
|
885
891
|
outputTokens: data.usageMetadata?.candidatesTokenCount ?? 0
|
|
886
892
|
});
|
|
887
893
|
const result = { text, usage };
|
|
@@ -903,6 +909,7 @@ function geminiAdapter(config = {}) {
|
|
|
903
909
|
const toolCalls = [];
|
|
904
910
|
let inputTokens = 0;
|
|
905
911
|
let outputTokens = 0;
|
|
912
|
+
let cacheReadTokens;
|
|
906
913
|
let finishReason = null;
|
|
907
914
|
for await (const data of stream) {
|
|
908
915
|
let chunk;
|
|
@@ -921,7 +928,9 @@ function geminiAdapter(config = {}) {
|
|
|
921
928
|
}
|
|
922
929
|
if (candidate?.finishReason) finishReason = candidate.finishReason;
|
|
923
930
|
if (chunk.usageMetadata) {
|
|
924
|
-
|
|
931
|
+
const split = splitCached(chunk.usageMetadata);
|
|
932
|
+
inputTokens = chunk.usageMetadata.promptTokenCount === void 0 ? inputTokens : split.inputTokens;
|
|
933
|
+
if (split.cacheReadTokens !== void 0) cacheReadTokens = split.cacheReadTokens;
|
|
925
934
|
outputTokens = chunk.usageMetadata.candidatesTokenCount ?? outputTokens;
|
|
926
935
|
}
|
|
927
936
|
}
|
|
@@ -934,7 +943,8 @@ function geminiAdapter(config = {}) {
|
|
|
934
943
|
transport: "http",
|
|
935
944
|
capability: "chat",
|
|
936
945
|
inputTokens,
|
|
937
|
-
outputTokens
|
|
946
|
+
outputTokens,
|
|
947
|
+
...cacheReadTokens === void 0 ? {} : { cacheReadTokens }
|
|
938
948
|
});
|
|
939
949
|
yield { type: "usage", costUsd: usage.costUsd, model: usage.model, usage };
|
|
940
950
|
yield {
|
|
@@ -978,7 +988,7 @@ function geminiAdapter(config = {}) {
|
|
|
978
988
|
model: req.spec.model,
|
|
979
989
|
transport: "http",
|
|
980
990
|
capability: "image",
|
|
981
|
-
|
|
991
|
+
...splitCached(data.usageMetadata),
|
|
982
992
|
outputTokens: data.usageMetadata?.candidatesTokenCount ?? 0
|
|
983
993
|
});
|
|
984
994
|
usage.costUsd = config.pricePerImage ?? GEMINI_IMAGE_PRICE_PER_IMAGE[req.spec.model] ?? 0;
|
|
@@ -1739,7 +1749,7 @@ function vertexAdapter(config = {}) {
|
|
|
1739
1749
|
}
|
|
1740
1750
|
const data = await res.json();
|
|
1741
1751
|
const text = (data.candidates?.[0]?.content?.parts ?? []).map((p) => p.text ?? "").join("").trim();
|
|
1742
|
-
const inputTokens = data.usageMetadata
|
|
1752
|
+
const { inputTokens, cacheReadTokens } = splitCached(data.usageMetadata);
|
|
1743
1753
|
const outputTokens = data.usageMetadata?.candidatesTokenCount ?? 0;
|
|
1744
1754
|
const usage = freshUsage({
|
|
1745
1755
|
provider: "vertex",
|
|
@@ -1747,9 +1757,10 @@ function vertexAdapter(config = {}) {
|
|
|
1747
1757
|
transport: "http",
|
|
1748
1758
|
capability: "vision",
|
|
1749
1759
|
inputTokens,
|
|
1750
|
-
outputTokens
|
|
1760
|
+
outputTokens,
|
|
1761
|
+
...cacheReadTokens === void 0 ? {} : { cacheReadTokens }
|
|
1751
1762
|
});
|
|
1752
|
-
usage.costUsd = computeCost("vertex", req.spec.model, inputTokens, outputTokens);
|
|
1763
|
+
usage.costUsd = computeCost("vertex", req.spec.model, inputTokens, outputTokens, cacheReadTokens ?? 0);
|
|
1753
1764
|
return { text, usage };
|
|
1754
1765
|
}
|
|
1755
1766
|
return { name: "vertex", animate, vision };
|
|
@@ -2811,8 +2822,8 @@ var aiConfigSchema = z.object({
|
|
|
2811
2822
|
});
|
|
2812
2823
|
|
|
2813
2824
|
// src/version.ts
|
|
2814
|
-
var VERSION = "0.
|
|
2815
|
-
var SDK_TAG = "@broberg/ai-sdk@0.
|
|
2825
|
+
var VERSION = "0.32.0";
|
|
2826
|
+
var SDK_TAG = "@broberg/ai-sdk@0.32.0";
|
|
2816
2827
|
|
|
2817
2828
|
// src/cost/sinks/upmetrics.ts
|
|
2818
2829
|
function upmetricsSink(config) {
|