smoltalk 0.8.3 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -9
- package/dist/classes/ToolCall.js +18 -10
- package/dist/classes/message/ToolMessage.js +13 -10
- package/dist/clients/google.d.ts +2 -0
- package/dist/clients/google.js +124 -2
- package/dist/models.d.ts +58 -0
- package/dist/models.js +60 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -8,15 +8,6 @@ Smoltalk exposes a common API to different LLM providers, with built-in cost tra
|
|
|
8
8
|
pnpm install smoltalk
|
|
9
9
|
```
|
|
10
10
|
|
|
11
|
-
> **Upgrading to 0.6.x?** The flat API-key/host fields on `SmolConfig` have
|
|
12
|
-
> been removed in favor of nested `apiKey` and `baseUrl` maps. Migration:
|
|
13
|
-
> ```diff
|
|
14
|
-
> -{ openAiApiKey: "sk-...", googleApiKey: "...", ollamaHost: "http://..." }
|
|
15
|
-
> +{ apiKey: { openAi: "sk-...", google: "..." }, baseUrl: { ollama: "http://..." } }
|
|
16
|
-
> ```
|
|
17
|
-
> Env-var fallbacks are unchanged (`OPENAI_API_KEY`, `GEMINI_API_KEY`,
|
|
18
|
-
> `ANTHROPIC_API_KEY`, `OLLAMA_HOST`).
|
|
19
|
-
|
|
20
11
|
## Hello world example
|
|
21
12
|
|
|
22
13
|
```typescript
|
|
@@ -335,6 +326,80 @@ a `hostedTools` catalog (`getHostedTools()`); the published file is kept current
|
|
|
335
326
|
by a daily CI job that translates [models.dev](https://models.dev) into
|
|
336
327
|
smoltalk's shape.
|
|
337
328
|
|
|
329
|
+
## Custom models & pricing
|
|
330
|
+
|
|
331
|
+
If you use a model that isn't in smoltalk's baked-in catalog (a self-hosted
|
|
332
|
+
model, a brand-new release, an OpenAI-compatible endpoint), smoltalk has no
|
|
333
|
+
pricing for it and the `cost` field is simply omitted from the result — nothing
|
|
334
|
+
errors, you just get `usage` without `cost`. Teach it the price and cost
|
|
335
|
+
tracking starts working.
|
|
336
|
+
|
|
337
|
+
**One model — `registerTextModel` (recommended).** Register once at startup:
|
|
338
|
+
|
|
339
|
+
```ts
|
|
340
|
+
import { registerTextModel, textSync, userMessage } from "smoltalk";
|
|
341
|
+
|
|
342
|
+
registerTextModel({
|
|
343
|
+
modelName: "my-model",
|
|
344
|
+
provider: "openai-compat", // must match the provider you call with (see below)
|
|
345
|
+
inputTokenCost: 0.5, // USD per 1M input tokens
|
|
346
|
+
outputTokenCost: 1.5, // USD per 1M output tokens
|
|
347
|
+
cachedInputTokenCost: 0.05, // optional
|
|
348
|
+
cacheCreationInputTokenCost: 0.625, // optional
|
|
349
|
+
maxInputTokens: 128000, // required by the type, even if you only want pricing
|
|
350
|
+
maxOutputTokens: 8192,
|
|
351
|
+
});
|
|
352
|
+
|
|
353
|
+
const messages = [userMessage("hello")];
|
|
354
|
+
const res = await textSync({
|
|
355
|
+
model: "my-model",
|
|
356
|
+
provider: "openai-compat",
|
|
357
|
+
baseUrl: { openAiCompat: "https://my-endpoint/v1" },
|
|
358
|
+
messages,
|
|
359
|
+
});
|
|
360
|
+
// res.value.cost is now populated from the rates above.
|
|
361
|
+
```
|
|
362
|
+
|
|
363
|
+
**Per-call only — `config.modelData`.** When you can't register globally (e.g.
|
|
364
|
+
per-tenant rates), pass a minimal blob for a single call. It layers over the
|
|
365
|
+
baseline exactly like a refresh blob:
|
|
366
|
+
|
|
367
|
+
```ts
|
|
368
|
+
import { textSync, userMessage, type ModelDataBlob } from "smoltalk";
|
|
369
|
+
|
|
370
|
+
const modelData: ModelDataBlob = {
|
|
371
|
+
schemaVersion: 1,
|
|
372
|
+
generatedAt: new Date().toISOString(),
|
|
373
|
+
hostedTools: [],
|
|
374
|
+
models: [
|
|
375
|
+
{
|
|
376
|
+
type: "text",
|
|
377
|
+
modelName: "my-model",
|
|
378
|
+
provider: "openai-compat",
|
|
379
|
+
maxInputTokens: 128000,
|
|
380
|
+
maxOutputTokens: 8192,
|
|
381
|
+
inputTokenCost: 0.5,
|
|
382
|
+
outputTokenCost: 1.5,
|
|
383
|
+
},
|
|
384
|
+
],
|
|
385
|
+
};
|
|
386
|
+
|
|
387
|
+
const messages = [userMessage("hello")];
|
|
388
|
+
await textSync({ model: "my-model", provider: "openai-compat", messages, modelData });
|
|
389
|
+
```
|
|
390
|
+
|
|
391
|
+
**The provider must match.** The registry is keyed by `provider:modelName`, so
|
|
392
|
+
the `provider` you register (or put in the blob) has to equal the `provider` you
|
|
393
|
+
pass at call time. Registering `my-model` under `"openai-compat"` but calling it
|
|
394
|
+
with `provider: "openrouter"` looks up a different key, finds no price, and
|
|
395
|
+
silently drops the `cost` field. When in doubt, register under the same provider
|
|
396
|
+
string you call with.
|
|
397
|
+
|
|
398
|
+
Overriding an entry that *is* in the catalog works the same way — merges are
|
|
399
|
+
field-by-field (see "Refreshing model data" above), so registering just
|
|
400
|
+
`inputTokenCost` / `outputTokenCost` for a known model updates only those fields
|
|
401
|
+
and leaves its limits and capabilities intact.
|
|
402
|
+
|
|
338
403
|
## Hosted tools catalog
|
|
339
404
|
|
|
340
405
|
Each cloud provider offers server-side "hosted" tools (web search, code
|
package/dist/classes/ToolCall.js
CHANGED
|
@@ -79,17 +79,25 @@ export class ToolCall {
|
|
|
79
79
|
};
|
|
80
80
|
}
|
|
81
81
|
toGoogle() {
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
args: this.arguments,
|
|
86
|
-
},
|
|
87
|
-
// Gemini 3 requires the original thought signature echoed back on the
|
|
88
|
-
// function-call part; omitting it fails validation during tool use.
|
|
89
|
-
...(this._thoughtSignature !== undefined && {
|
|
90
|
-
thoughtSignature: this._thoughtSignature,
|
|
91
|
-
}),
|
|
82
|
+
const functionCall = {
|
|
83
|
+
name: this.name,
|
|
84
|
+
args: this.arguments,
|
|
92
85
|
};
|
|
86
|
+
// Echo the id when we have one: the Gemini API pairs a functionResponse
|
|
87
|
+
// back to its functionCall by id when present. Omit it when empty — current
|
|
88
|
+
// Gemini 3 preview models issue no ids, and an empty id is not a valid key.
|
|
89
|
+
if (this._id !== "") {
|
|
90
|
+
functionCall.id = this._id;
|
|
91
|
+
}
|
|
92
|
+
const result = {
|
|
93
|
+
functionCall,
|
|
94
|
+
};
|
|
95
|
+
// Gemini 3 requires the original thought signature echoed back on the
|
|
96
|
+
// function-call part; omitting it fails validation during tool use.
|
|
97
|
+
if (this._thoughtSignature !== undefined) {
|
|
98
|
+
result.thoughtSignature = this._thoughtSignature;
|
|
99
|
+
}
|
|
100
|
+
return result;
|
|
93
101
|
}
|
|
94
102
|
toOpenAIResponseInputItem() {
|
|
95
103
|
return {
|
|
@@ -91,18 +91,21 @@ export class ToolMessage extends BaseMessage {
|
|
|
91
91
|
};
|
|
92
92
|
}
|
|
93
93
|
toGoogleMessage() {
|
|
94
|
+
const functionResponse = {
|
|
95
|
+
name: this.name,
|
|
96
|
+
response: {
|
|
97
|
+
result: this.content,
|
|
98
|
+
},
|
|
99
|
+
};
|
|
100
|
+
// Echo the id so Gemini can pair this response to its call by id — the
|
|
101
|
+
// documented matching mechanism. Only when non-empty: current Gemini 3
|
|
102
|
+
// preview models issue no ids, and an empty id is not a valid key.
|
|
103
|
+
if (this.tool_call_id !== "") {
|
|
104
|
+
functionResponse.id = this.tool_call_id;
|
|
105
|
+
}
|
|
94
106
|
return {
|
|
95
107
|
role: "user",
|
|
96
|
-
parts: [
|
|
97
|
-
{
|
|
98
|
-
functionResponse: {
|
|
99
|
-
name: this.name,
|
|
100
|
-
response: {
|
|
101
|
-
result: this.content,
|
|
102
|
-
},
|
|
103
|
-
},
|
|
104
|
-
},
|
|
105
|
-
],
|
|
108
|
+
parts: [{ functionResponse }],
|
|
106
109
|
};
|
|
107
110
|
}
|
|
108
111
|
toOllamaMessage() {
|
package/dist/clients/google.d.ts
CHANGED
|
@@ -3,6 +3,7 @@ import { PromptResult, Result, SmolClient, SmolConfig, StreamChunk } from "../ty
|
|
|
3
3
|
import { BaseClient } from "./baseClient.js";
|
|
4
4
|
import { ModelName } from "../models.js";
|
|
5
5
|
import { HostedToolResult } from "../types.js";
|
|
6
|
+
import type { Message } from "../classes/message/index.js";
|
|
6
7
|
export type SmolGoogleConfig = SmolConfig;
|
|
7
8
|
export declare function googleWebSearchEntries(hostedTools?: string[]): any[];
|
|
8
9
|
/**
|
|
@@ -20,6 +21,7 @@ export declare function googleWebSearchEntries(hostedTools?: string[]): any[];
|
|
|
20
21
|
* See egonSchiele/agency-lang#495.
|
|
21
22
|
*/
|
|
22
23
|
export declare function geminiSupportsToolCirculation(model: string): boolean;
|
|
24
|
+
export declare function reorderToolResultsForGemini(messages: Message[]): Message[];
|
|
23
25
|
export declare function parseGoogleHostedTools(result: any, provider: string, model: string): HostedToolResult[];
|
|
24
26
|
type GeneratedRequest = {
|
|
25
27
|
contents: Content[];
|
package/dist/clients/google.js
CHANGED
|
@@ -39,6 +39,118 @@ export function geminiSupportsToolCirculation(model) {
|
|
|
39
39
|
return true;
|
|
40
40
|
return parseInt(m[1], 10) >= 3;
|
|
41
41
|
}
|
|
42
|
+
// Reorder each round's tool results to match the order of the calls that
|
|
43
|
+
// produced them. Two documented Gemini behaviors make this necessary:
|
|
44
|
+
// - A function call carries an optional `id`, and a response is paired back to
|
|
45
|
+
// its call by echoing that id.
|
|
46
|
+
// https://ai.google.dev/gemini-api/docs/function-calling
|
|
47
|
+
// - Parts must be returned in the order received (responses in call order:
|
|
48
|
+
// FC1,FC2 -> FR1,FR2), and the thought signature rides ONLY the first
|
|
49
|
+
// function call of a parallel batch; omitting it 400s on Gemini 3. So part
|
|
50
|
+
// order is load-bearing independent of ids.
|
|
51
|
+
// https://ai.google.dev/gemini-api/docs/generate-content/thought-signatures
|
|
52
|
+
// Current Gemini 3 preview models are observed to emit no function-call ids, so
|
|
53
|
+
// on them pairing is purely positional. A caller whose results arrive in
|
|
54
|
+
// completion order (rather than call order) would otherwise feed tool A's answer
|
|
55
|
+
// to tool B and break thought-signature validation.
|
|
56
|
+
//
|
|
57
|
+
// For every assistant message that carries toolCalls, the run of ToolMessages
|
|
58
|
+
// immediately following it is reordered. The run ends at the first non-tool
|
|
59
|
+
// message: any results the caller interleaved AFTER a non-tool message are left
|
|
60
|
+
// where they are — moving messages across an interloper is repair, not reorder.
|
|
61
|
+
// 1. By id, when both the call and a response have non-empty ids (order-free,
|
|
62
|
+
// takes global priority so an id match is never stolen by a name match).
|
|
63
|
+
// 2. By name + occurrence otherwise: the k-th response named X answers the
|
|
64
|
+
// k-th call named X.
|
|
65
|
+
// NEVER drop a message: a response matching no call (or any surplus) is kept at
|
|
66
|
+
// the end of the run in its original relative order. A reorder that lost a
|
|
67
|
+
// message would turn a mispairing bug into a missing-result bug, which is worse.
|
|
68
|
+
export function reorderToolResultsForGemini(messages) {
|
|
69
|
+
const out = [];
|
|
70
|
+
let i = 0;
|
|
71
|
+
while (i < messages.length) {
|
|
72
|
+
const msg = messages[i];
|
|
73
|
+
out.push(msg);
|
|
74
|
+
const toolCalls = msg.role === "assistant" ? msg.toolCalls : undefined;
|
|
75
|
+
if (!toolCalls || toolCalls.length === 0) {
|
|
76
|
+
i += 1;
|
|
77
|
+
continue;
|
|
78
|
+
}
|
|
79
|
+
// Collect the contiguous run of tool results that answers this round.
|
|
80
|
+
let j = i + 1;
|
|
81
|
+
const run = [];
|
|
82
|
+
while (j < messages.length && messages[j].role === "tool") {
|
|
83
|
+
run.push(messages[j]);
|
|
84
|
+
j += 1;
|
|
85
|
+
}
|
|
86
|
+
if (run.length === 0) {
|
|
87
|
+
i += 1;
|
|
88
|
+
continue;
|
|
89
|
+
}
|
|
90
|
+
out.push(...orderRunToMatchCalls(toolCalls, run));
|
|
91
|
+
i = j;
|
|
92
|
+
}
|
|
93
|
+
return out;
|
|
94
|
+
}
|
|
95
|
+
function orderRunToMatchCalls(toolCalls, run) {
|
|
96
|
+
const used = new Array(run.length).fill(false);
|
|
97
|
+
// assignment[callIndex] = index into `run`, or -1 if that call has no result.
|
|
98
|
+
const assignment = new Array(toolCalls.length).fill(-1);
|
|
99
|
+
function runId(k) {
|
|
100
|
+
const m = run[k];
|
|
101
|
+
if (m.role === "tool") {
|
|
102
|
+
return m.tool_call_id;
|
|
103
|
+
}
|
|
104
|
+
return "";
|
|
105
|
+
}
|
|
106
|
+
function runName(k) {
|
|
107
|
+
const m = run[k];
|
|
108
|
+
if (m.role === "tool") {
|
|
109
|
+
return m.name;
|
|
110
|
+
}
|
|
111
|
+
return "";
|
|
112
|
+
}
|
|
113
|
+
// Pass 1: id pairing (both sides non-empty), global priority.
|
|
114
|
+
toolCalls.forEach((call, ci) => {
|
|
115
|
+
if (call.id === "")
|
|
116
|
+
return;
|
|
117
|
+
const ri = run.findIndex((_r, k) => !used[k] && runId(k) !== "" && runId(k) === call.id);
|
|
118
|
+
if (ri !== -1) {
|
|
119
|
+
assignment[ci] = ri;
|
|
120
|
+
used[ri] = true;
|
|
121
|
+
}
|
|
122
|
+
});
|
|
123
|
+
// Pass 2: name + occurrence, for calls still unmatched. findIndex takes the
|
|
124
|
+
// first unused same-name response, so the k-th call named X pairs with the
|
|
125
|
+
// k-th response named X.
|
|
126
|
+
//
|
|
127
|
+
// Garbage-in caveat: if a call and its only same-name response carry DIFFERENT
|
|
128
|
+
// non-empty ids (one side lost or mangled its id), pass 1 misses and pass 2
|
|
129
|
+
// pairs them by name — emitting functionCall.id != functionResponse.id in that
|
|
130
|
+
// slot, which an id-pairing model would see as a contradiction. The input was
|
|
131
|
+
// already inconsistent; we pair positionally rather than drop the result. Not
|
|
132
|
+
// a reorder bug.
|
|
133
|
+
toolCalls.forEach((call, ci) => {
|
|
134
|
+
if (assignment[ci] !== -1)
|
|
135
|
+
return;
|
|
136
|
+
const ri = run.findIndex((_r, k) => !used[k] && runName(k) === call.name);
|
|
137
|
+
if (ri !== -1) {
|
|
138
|
+
assignment[ci] = ri;
|
|
139
|
+
used[ri] = true;
|
|
140
|
+
}
|
|
141
|
+
});
|
|
142
|
+
const ordered = [];
|
|
143
|
+
for (const ri of assignment) {
|
|
144
|
+
if (ri !== -1)
|
|
145
|
+
ordered.push(run[ri]);
|
|
146
|
+
}
|
|
147
|
+
// Never drop: append any unmatched/surplus responses in original order.
|
|
148
|
+
for (let k = 0; k < run.length; k++) {
|
|
149
|
+
if (!used[k])
|
|
150
|
+
ordered.push(run[k]);
|
|
151
|
+
}
|
|
152
|
+
return ordered;
|
|
153
|
+
}
|
|
42
154
|
export function parseGoogleHostedTools(result, provider, model) {
|
|
43
155
|
const queries = [];
|
|
44
156
|
const sources = [];
|
|
@@ -136,7 +248,12 @@ export class SmolGoogle extends BaseClient {
|
|
|
136
248
|
}
|
|
137
249
|
return true;
|
|
138
250
|
});
|
|
139
|
-
|
|
251
|
+
// Normalize tool-result ordering before conversion: Gemini pairs each
|
|
252
|
+
// functionResponse to its functionCall (by id on 3.5+, strictly by position
|
|
253
|
+
// on the Gemini 3 family), so results must leave in call order regardless of
|
|
254
|
+
// the order the caller supplied them. See reorderToolResultsForGemini.
|
|
255
|
+
const orderedMessages = reorderToolResultsForGemini(contentMessages);
|
|
256
|
+
const messages = orderedMessages.map((msg) => msg.toGoogleMessage());
|
|
140
257
|
const tools = (config.tools || []).map((tool) => {
|
|
141
258
|
return zodToGoogleTool(tool.name, tool.schema, {
|
|
142
259
|
description: tool.description,
|
|
@@ -341,7 +458,12 @@ export class SmolGoogle extends BaseClient {
|
|
|
341
458
|
const functionCall = part.functionCall;
|
|
342
459
|
// Gemini 3 rides the thought signature on the same part as the
|
|
343
460
|
// function call; capture it so it can be echoed back during tool use.
|
|
344
|
-
toolCalls.push(
|
|
461
|
+
toolCalls.push(
|
|
462
|
+
// Keep functionCall.id so id-based pairing can round-trip. Do NOT
|
|
463
|
+
// fall back to the name (as the streaming path does for its Map
|
|
464
|
+
// key): two parallel calls to the same tool would share a fake id
|
|
465
|
+
// and re-create the pairing bug at the id layer.
|
|
466
|
+
new ToolCall(functionCall.id || "", functionCall.name, functionCall.args, {
|
|
345
467
|
thoughtSignature: part.thoughtSignature,
|
|
346
468
|
}));
|
|
347
469
|
}
|
package/dist/models.d.ts
CHANGED
|
@@ -1228,6 +1228,35 @@ export declare const textModels: readonly [{
|
|
|
1228
1228
|
readonly costUnit: "characters";
|
|
1229
1229
|
readonly disabled: true;
|
|
1230
1230
|
readonly provider: "google";
|
|
1231
|
+
}, {
|
|
1232
|
+
readonly type: "text";
|
|
1233
|
+
readonly modelName: "claude-fable-5";
|
|
1234
|
+
readonly description: "Anthropic's most capable widely released model, for the most demanding reasoning and long-horizon agentic work. Thinking is always on (cannot be disabled); the raw chain of thought is never returned. Requires 30-day data retention (not available under zero data retention). 1M context window, 128K max output.";
|
|
1235
|
+
readonly maxInputTokens: 1000000;
|
|
1236
|
+
readonly maxOutputTokens: 128000;
|
|
1237
|
+
readonly inputTokenCost: 10;
|
|
1238
|
+
readonly cachedInputTokenCost: 1;
|
|
1239
|
+
readonly cacheCreationInputTokenCost: 12.5;
|
|
1240
|
+
readonly outputTokenCost: 50;
|
|
1241
|
+
readonly reasoning: {
|
|
1242
|
+
readonly thinkingStyle: "adaptive";
|
|
1243
|
+
readonly levels: readonly ["low", "medium", "high", "xhigh", "max"];
|
|
1244
|
+
readonly defaultLevel: "high";
|
|
1245
|
+
readonly canDisable: false;
|
|
1246
|
+
readonly outputsThinking: true;
|
|
1247
|
+
readonly outputsSignatures: true;
|
|
1248
|
+
};
|
|
1249
|
+
readonly modalities: {
|
|
1250
|
+
readonly input: readonly ["text", "image", "pdf"];
|
|
1251
|
+
readonly output: readonly ["text"];
|
|
1252
|
+
};
|
|
1253
|
+
readonly knowledge: "2026-01";
|
|
1254
|
+
readonly releaseDate: "2026-06-09";
|
|
1255
|
+
readonly lastUpdated: "2026-06-09";
|
|
1256
|
+
readonly family: "claude-fable";
|
|
1257
|
+
readonly openWeights: false;
|
|
1258
|
+
readonly temperatureSupported: false;
|
|
1259
|
+
readonly provider: "anthropic";
|
|
1231
1260
|
}, {
|
|
1232
1261
|
readonly type: "text";
|
|
1233
1262
|
readonly modelName: "claude-opus-4-8";
|
|
@@ -1310,6 +1339,35 @@ export declare const textModels: readonly [{
|
|
|
1310
1339
|
readonly openWeights: false;
|
|
1311
1340
|
readonly temperatureSupported: true;
|
|
1312
1341
|
readonly provider: "anthropic";
|
|
1342
|
+
}, {
|
|
1343
|
+
readonly type: "text";
|
|
1344
|
+
readonly modelName: "claude-sonnet-5";
|
|
1345
|
+
readonly description: "The best combination of speed and intelligence in the Sonnet tier, with near-Opus quality on coding and agentic work. Adaptive thinking on by default; supports the full low/medium/high/xhigh/max effort range. New tokenizer (~30% more tokens for the same text vs Sonnet 4.6). Standard pricing $3/$15; introductory $2/$10 per MTok through 2026-08-31. 1M context window, 128K max output.";
|
|
1346
|
+
readonly maxInputTokens: 1000000;
|
|
1347
|
+
readonly maxOutputTokens: 128000;
|
|
1348
|
+
readonly inputTokenCost: 3;
|
|
1349
|
+
readonly cachedInputTokenCost: 0.3;
|
|
1350
|
+
readonly cacheCreationInputTokenCost: 3.75;
|
|
1351
|
+
readonly outputTokenCost: 15;
|
|
1352
|
+
readonly reasoning: {
|
|
1353
|
+
readonly thinkingStyle: "adaptive";
|
|
1354
|
+
readonly levels: readonly ["low", "medium", "high", "xhigh", "max"];
|
|
1355
|
+
readonly defaultLevel: "high";
|
|
1356
|
+
readonly canDisable: true;
|
|
1357
|
+
readonly outputsThinking: true;
|
|
1358
|
+
readonly outputsSignatures: true;
|
|
1359
|
+
};
|
|
1360
|
+
readonly modalities: {
|
|
1361
|
+
readonly input: readonly ["text", "image", "pdf"];
|
|
1362
|
+
readonly output: readonly ["text"];
|
|
1363
|
+
};
|
|
1364
|
+
readonly knowledge: "2026-01";
|
|
1365
|
+
readonly releaseDate: "2026-06-30";
|
|
1366
|
+
readonly lastUpdated: "2026-06-30";
|
|
1367
|
+
readonly family: "claude-sonnet";
|
|
1368
|
+
readonly openWeights: false;
|
|
1369
|
+
readonly temperatureSupported: false;
|
|
1370
|
+
readonly provider: "anthropic";
|
|
1313
1371
|
}, {
|
|
1314
1372
|
readonly type: "text";
|
|
1315
1373
|
readonly modelName: "claude-sonnet-4-6";
|
package/dist/models.js
CHANGED
|
@@ -1206,6 +1206,36 @@ export const textModels = [
|
|
|
1206
1206
|
disabled: true,
|
|
1207
1207
|
provider: "google",
|
|
1208
1208
|
},
|
|
1209
|
+
{
|
|
1210
|
+
type: "text",
|
|
1211
|
+
modelName: "claude-fable-5",
|
|
1212
|
+
description: "Anthropic's most capable widely released model, for the most demanding reasoning and long-horizon agentic work. Thinking is always on (cannot be disabled); the raw chain of thought is never returned. Requires 30-day data retention (not available under zero data retention). 1M context window, 128K max output.",
|
|
1213
|
+
maxInputTokens: 1000000,
|
|
1214
|
+
maxOutputTokens: 128000,
|
|
1215
|
+
inputTokenCost: 10,
|
|
1216
|
+
cachedInputTokenCost: 1,
|
|
1217
|
+
cacheCreationInputTokenCost: 12.5,
|
|
1218
|
+
outputTokenCost: 50,
|
|
1219
|
+
reasoning: {
|
|
1220
|
+
thinkingStyle: "adaptive",
|
|
1221
|
+
levels: ["low", "medium", "high", "xhigh", "max"],
|
|
1222
|
+
defaultLevel: "high",
|
|
1223
|
+
canDisable: false,
|
|
1224
|
+
outputsThinking: true,
|
|
1225
|
+
outputsSignatures: true,
|
|
1226
|
+
},
|
|
1227
|
+
modalities: {
|
|
1228
|
+
input: ["text", "image", "pdf"],
|
|
1229
|
+
output: ["text"],
|
|
1230
|
+
},
|
|
1231
|
+
knowledge: "2026-01",
|
|
1232
|
+
releaseDate: "2026-06-09",
|
|
1233
|
+
lastUpdated: "2026-06-09",
|
|
1234
|
+
family: "claude-fable",
|
|
1235
|
+
openWeights: false,
|
|
1236
|
+
temperatureSupported: false,
|
|
1237
|
+
provider: "anthropic",
|
|
1238
|
+
},
|
|
1209
1239
|
{
|
|
1210
1240
|
type: "text",
|
|
1211
1241
|
modelName: "claude-opus-4-8",
|
|
@@ -1291,6 +1321,36 @@ export const textModels = [
|
|
|
1291
1321
|
temperatureSupported: true,
|
|
1292
1322
|
provider: "anthropic",
|
|
1293
1323
|
},
|
|
1324
|
+
{
|
|
1325
|
+
type: "text",
|
|
1326
|
+
modelName: "claude-sonnet-5",
|
|
1327
|
+
description: "The best combination of speed and intelligence in the Sonnet tier, with near-Opus quality on coding and agentic work. Adaptive thinking on by default; supports the full low/medium/high/xhigh/max effort range. New tokenizer (~30% more tokens for the same text vs Sonnet 4.6). Standard pricing $3/$15; introductory $2/$10 per MTok through 2026-08-31. 1M context window, 128K max output.",
|
|
1328
|
+
maxInputTokens: 1000000,
|
|
1329
|
+
maxOutputTokens: 128000,
|
|
1330
|
+
inputTokenCost: 3,
|
|
1331
|
+
cachedInputTokenCost: 0.3,
|
|
1332
|
+
cacheCreationInputTokenCost: 3.75,
|
|
1333
|
+
outputTokenCost: 15,
|
|
1334
|
+
reasoning: {
|
|
1335
|
+
thinkingStyle: "adaptive",
|
|
1336
|
+
levels: ["low", "medium", "high", "xhigh", "max"],
|
|
1337
|
+
defaultLevel: "high",
|
|
1338
|
+
canDisable: true,
|
|
1339
|
+
outputsThinking: true,
|
|
1340
|
+
outputsSignatures: true,
|
|
1341
|
+
},
|
|
1342
|
+
modalities: {
|
|
1343
|
+
input: ["text", "image", "pdf"],
|
|
1344
|
+
output: ["text"],
|
|
1345
|
+
},
|
|
1346
|
+
knowledge: "2026-01",
|
|
1347
|
+
releaseDate: "2026-06-30",
|
|
1348
|
+
lastUpdated: "2026-06-30",
|
|
1349
|
+
family: "claude-sonnet",
|
|
1350
|
+
openWeights: false,
|
|
1351
|
+
temperatureSupported: false,
|
|
1352
|
+
provider: "anthropic",
|
|
1353
|
+
},
|
|
1294
1354
|
{
|
|
1295
1355
|
type: "text",
|
|
1296
1356
|
modelName: "claude-sonnet-4-6",
|