smoltalk 0.8.1 → 0.8.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -1
- package/dist/clients/anthropic.d.ts +10 -0
- package/dist/clients/anthropic.js +145 -12
- package/dist/clients/baseClient.d.ts +10 -0
- package/dist/clients/baseClient.js +26 -1
- package/dist/clients/google.d.ts +15 -0
- package/dist/clients/google.js +98 -23
- package/dist/clients/ollama.js +29 -12
- package/dist/clients/openai.js +29 -13
- package/dist/clients/openaiResponses.js +27 -10
- package/dist/models.d.ts +110 -2
- package/dist/models.js +114 -2
- package/dist/types/stopReason.d.ts +19 -0
- package/dist/types/stopReason.js +1 -0
- package/dist/types.d.ts +15 -1
- package/dist/types.js +4 -1
- package/dist/util/jsonSchema.d.ts +38 -0
- package/dist/util/jsonSchema.js +133 -0
- package/dist/util/stopReason.d.ts +11 -0
- package/dist/util/stopReason.js +80 -0
- package/dist/util/tool.js +4 -3
- package/package.json +1 -1
package/dist/clients/ollama.js
CHANGED
|
@@ -4,6 +4,8 @@ import { getLogger } from "../util/logger.js";
|
|
|
4
4
|
import { redactAttachments } from "../util/redact.js";
|
|
5
5
|
import { success, } from "../types.js";
|
|
6
6
|
import { zodToGoogleTool } from "../util/tool.js";
|
|
7
|
+
import { responseFormatToJsonSchema } from "../util/jsonSchema.js";
|
|
8
|
+
import { normalizeOllamaStopReason } from "../util/stopReason.js";
|
|
7
9
|
import { sanitizeAttributes } from "../util/util.js";
|
|
8
10
|
import { resolveBaseUrl } from "../util/provider.js";
|
|
9
11
|
import { BaseClient } from "./baseClient.js";
|
|
@@ -81,7 +83,7 @@ export class SmolOllama extends BaseClient {
|
|
|
81
83
|
request.tools = tools.map((t) => ({ type: "function", function: t }));
|
|
82
84
|
}
|
|
83
85
|
if (config.responseFormat) {
|
|
84
|
-
request.format = config.responseFormat
|
|
86
|
+
request.format = responseFormatToJsonSchema(config.responseFormat);
|
|
85
87
|
}
|
|
86
88
|
Object.assign(request, sanitizeAttributes(config.rawAttributes));
|
|
87
89
|
this.logger.debug("Sending request to Ollama:", JSON.stringify(redactAttachments(request), null, 2));
|
|
@@ -116,8 +118,20 @@ export class SmolOllama extends BaseClient {
|
|
|
116
118
|
}
|
|
117
119
|
// Extract usage and calculate cost
|
|
118
120
|
const { usage, cost } = this.calculateUsageAndCost(result);
|
|
121
|
+
const rawStopReason = result.done_reason ?? undefined;
|
|
119
122
|
// Return the response, updating the chat history
|
|
120
|
-
|
|
123
|
+
const promptResult = {
|
|
124
|
+
output,
|
|
125
|
+
toolCalls,
|
|
126
|
+
usage,
|
|
127
|
+
cost,
|
|
128
|
+
model: this.getModel(),
|
|
129
|
+
stopReason: normalizeOllamaStopReason(rawStopReason),
|
|
130
|
+
};
|
|
131
|
+
if (rawStopReason) {
|
|
132
|
+
promptResult.rawStopReason = rawStopReason;
|
|
133
|
+
}
|
|
134
|
+
return success(promptResult);
|
|
121
135
|
}
|
|
122
136
|
async *_textStream(config) {
|
|
123
137
|
const messages = config.messages.map((msg) => msg.toOllamaMessage());
|
|
@@ -135,7 +149,7 @@ export class SmolOllama extends BaseClient {
|
|
|
135
149
|
request.tools = tools.map((t) => ({ type: "function", function: t }));
|
|
136
150
|
}
|
|
137
151
|
if (config.responseFormat) {
|
|
138
|
-
request.format = config.responseFormat
|
|
152
|
+
request.format = responseFormatToJsonSchema(config.responseFormat);
|
|
139
153
|
}
|
|
140
154
|
Object.assign(request, sanitizeAttributes(config.rawAttributes));
|
|
141
155
|
this.logger.debug("Sending streaming request to Ollama:", JSON.stringify(redactAttachments(request), null, 2));
|
|
@@ -201,16 +215,19 @@ export class SmolOllama extends BaseClient {
|
|
|
201
215
|
toolCalls.push(toolCall);
|
|
202
216
|
yield { type: "tool_call", toolCall };
|
|
203
217
|
}
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
},
|
|
218
|
+
const rawStopReason = lastChunk?.done_reason ?? undefined;
|
|
219
|
+
const result = {
|
|
220
|
+
output: content || null,
|
|
221
|
+
toolCalls,
|
|
222
|
+
usage,
|
|
223
|
+
cost,
|
|
224
|
+
model: this.getModel(),
|
|
225
|
+
stopReason: normalizeOllamaStopReason(rawStopReason),
|
|
213
226
|
};
|
|
227
|
+
if (rawStopReason) {
|
|
228
|
+
result.rawStopReason = rawStopReason;
|
|
229
|
+
}
|
|
230
|
+
yield { type: "done", result };
|
|
214
231
|
}
|
|
215
232
|
catch (error) {
|
|
216
233
|
this.rethrowAsSmolError(error);
|
package/dist/clients/openai.js
CHANGED
|
@@ -8,6 +8,8 @@ import { BaseClient } from "./baseClient.js";
|
|
|
8
8
|
import { SmolContentPolicyError, SmolContextWindowExceededError, smolErrorForStatus, } from "../smolError.js";
|
|
9
9
|
import { extractHttpErrorFields } from "../util/httpError.js";
|
|
10
10
|
import { zodToOpenAITool } from "../util/tool.js";
|
|
11
|
+
import { responseFormatToJsonSchema } from "../util/jsonSchema.js";
|
|
12
|
+
import { normalizeOpenAIStopReason } from "../util/stopReason.js";
|
|
11
13
|
import { Model } from "../model.js";
|
|
12
14
|
export class SmolOpenAi extends BaseClient {
|
|
13
15
|
client;
|
|
@@ -120,7 +122,7 @@ export class SmolOpenAi extends BaseClient {
|
|
|
120
122
|
type: "json_schema",
|
|
121
123
|
json_schema: {
|
|
122
124
|
name: config.responseFormatOptions?.name || "response",
|
|
123
|
-
schema: config.responseFormat
|
|
125
|
+
schema: responseFormatToJsonSchema(config.responseFormat),
|
|
124
126
|
},
|
|
125
127
|
};
|
|
126
128
|
}
|
|
@@ -185,14 +187,22 @@ export class SmolOpenAi extends BaseClient {
|
|
|
185
187
|
// response headers (e.g. LiteLLM's x-litellm-response-cost).
|
|
186
188
|
const { usage, cost } = this.calculateUsageAndCost(completion.usage, rawResponse);
|
|
187
189
|
const hostedToolResults = this.parseHostedToolResults(completion, config);
|
|
188
|
-
|
|
190
|
+
const rawStopReason = completion.choices[0]?.finish_reason ?? undefined;
|
|
191
|
+
const result = {
|
|
189
192
|
output,
|
|
190
193
|
toolCalls,
|
|
191
194
|
usage,
|
|
192
195
|
cost,
|
|
193
196
|
model: this.getModel(),
|
|
194
|
-
|
|
195
|
-
}
|
|
197
|
+
stopReason: normalizeOpenAIStopReason(rawStopReason),
|
|
198
|
+
};
|
|
199
|
+
if (rawStopReason) {
|
|
200
|
+
result.rawStopReason = rawStopReason;
|
|
201
|
+
}
|
|
202
|
+
if (hostedToolResults.length > 0) {
|
|
203
|
+
result.hostedToolResults = hostedToolResults;
|
|
204
|
+
}
|
|
205
|
+
return success(result);
|
|
196
206
|
}
|
|
197
207
|
async *_textStream(config) {
|
|
198
208
|
const request = this.buildRequest(config);
|
|
@@ -214,7 +224,11 @@ export class SmolOpenAi extends BaseClient {
|
|
|
214
224
|
const toolCallsMap = new Map();
|
|
215
225
|
let usage;
|
|
216
226
|
let cost;
|
|
227
|
+
let rawStopReason;
|
|
217
228
|
for await (const chunk of completion) {
|
|
229
|
+
const chunkFinish = chunk.choices?.[0]?.finish_reason;
|
|
230
|
+
if (chunkFinish)
|
|
231
|
+
rawStopReason = chunkFinish;
|
|
218
232
|
// Extract usage from the final chunk
|
|
219
233
|
if (chunk.usage) {
|
|
220
234
|
// Header-based cost (LiteLLM) is unsupported while streaming.
|
|
@@ -266,15 +280,17 @@ export class SmolOpenAi extends BaseClient {
|
|
|
266
280
|
toolCalls.push(toolCall);
|
|
267
281
|
yield { type: "tool_call", toolCall };
|
|
268
282
|
}
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
model: this.getModel(),
|
|
277
|
-
},
|
|
283
|
+
const result = {
|
|
284
|
+
output: content || null,
|
|
285
|
+
toolCalls,
|
|
286
|
+
usage,
|
|
287
|
+
cost,
|
|
288
|
+
model: this.getModel(),
|
|
289
|
+
stopReason: normalizeOpenAIStopReason(rawStopReason),
|
|
278
290
|
};
|
|
291
|
+
if (rawStopReason) {
|
|
292
|
+
result.rawStopReason = rawStopReason;
|
|
293
|
+
}
|
|
294
|
+
yield { type: "done", result };
|
|
279
295
|
}
|
|
280
296
|
}
|
|
@@ -5,6 +5,8 @@ import { getLogger } from "../util/logger.js";
|
|
|
5
5
|
import { redactAttachments } from "../util/redact.js";
|
|
6
6
|
import { BaseClient } from "./baseClient.js";
|
|
7
7
|
import { zodToOpenAIResponsesTool } from "../util/tool.js";
|
|
8
|
+
import { responseFormatToJsonSchema } from "../util/jsonSchema.js";
|
|
9
|
+
import { normalizeOpenAIResponsesStopReason } from "../util/stopReason.js";
|
|
8
10
|
import { sanitizeAttributes } from "../util/util.js";
|
|
9
11
|
import { WEB_SEARCH, webSearchResult, applyHostedToolCost } from "../util/hostedTools.js";
|
|
10
12
|
import { Model } from "../model.js";
|
|
@@ -122,7 +124,7 @@ export class SmolOpenAiResponses extends BaseClient {
|
|
|
122
124
|
format: {
|
|
123
125
|
type: "json_schema",
|
|
124
126
|
name: config.responseFormatOptions?.name || "response",
|
|
125
|
-
schema: config.responseFormat
|
|
127
|
+
schema: responseFormatToJsonSchema(config.responseFormat),
|
|
126
128
|
},
|
|
127
129
|
};
|
|
128
130
|
}
|
|
@@ -192,13 +194,19 @@ export class SmolOpenAiResponses extends BaseClient {
|
|
|
192
194
|
const { usage, cost } = this.calculateUsageAndCost(response.usage);
|
|
193
195
|
const parsed = parseOpenAIResponsesHostedTools(response, "openai-responses");
|
|
194
196
|
const { results: hostedToolResults, cost: finalCost } = applyHostedToolCost(parsed, cost, this.getModel(), this.config.modelData);
|
|
197
|
+
const incompleteReason = response.incomplete_details?.reason;
|
|
198
|
+
const rawStopReason = incompleteReason ?? response.status ?? undefined;
|
|
195
199
|
const result = {
|
|
196
200
|
output,
|
|
197
201
|
toolCalls,
|
|
198
202
|
usage,
|
|
199
203
|
cost: finalCost,
|
|
200
204
|
model: this.getModel(),
|
|
205
|
+
stopReason: normalizeOpenAIResponsesStopReason(response.status, incompleteReason, toolCalls.length > 0),
|
|
201
206
|
};
|
|
207
|
+
if (rawStopReason) {
|
|
208
|
+
result.rawStopReason = rawStopReason;
|
|
209
|
+
}
|
|
202
210
|
if (hostedToolResults.length > 0) {
|
|
203
211
|
result.hostedToolResults = hostedToolResults;
|
|
204
212
|
}
|
|
@@ -222,7 +230,12 @@ export class SmolOpenAiResponses extends BaseClient {
|
|
|
222
230
|
const functionCalls = new Map();
|
|
223
231
|
let usage;
|
|
224
232
|
let cost;
|
|
233
|
+
let finalResponse;
|
|
225
234
|
for await (const event of stream) {
|
|
235
|
+
if (event.type === "response.completed" ||
|
|
236
|
+
event.type === "response.incomplete") {
|
|
237
|
+
finalResponse = event.response;
|
|
238
|
+
}
|
|
226
239
|
switch (event.type) {
|
|
227
240
|
case "response.output_text.delta": {
|
|
228
241
|
content += event.delta;
|
|
@@ -291,15 +304,19 @@ export class SmolOpenAiResponses extends BaseClient {
|
|
|
291
304
|
toolCalls.push(toolCall);
|
|
292
305
|
yield { type: "tool_call", toolCall };
|
|
293
306
|
}
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
307
|
+
const incompleteReason = finalResponse?.incomplete_details?.reason;
|
|
308
|
+
const rawStopReason = incompleteReason ?? finalResponse?.status ?? undefined;
|
|
309
|
+
const result = {
|
|
310
|
+
output: content || null,
|
|
311
|
+
toolCalls,
|
|
312
|
+
usage,
|
|
313
|
+
cost,
|
|
314
|
+
model: this.getModel(),
|
|
315
|
+
stopReason: normalizeOpenAIResponsesStopReason(finalResponse?.status, incompleteReason, toolCalls.length > 0),
|
|
303
316
|
};
|
|
317
|
+
if (rawStopReason) {
|
|
318
|
+
result.rawStopReason = rawStopReason;
|
|
319
|
+
}
|
|
320
|
+
yield { type: "done", result };
|
|
304
321
|
}
|
|
305
322
|
}
|
package/dist/models.d.ts
CHANGED
|
@@ -773,6 +773,108 @@ export declare const textModels: readonly [{
|
|
|
773
773
|
readonly structuredOutput: true;
|
|
774
774
|
readonly temperatureSupported: false;
|
|
775
775
|
readonly provider: "openai-responses";
|
|
776
|
+
}, {
|
|
777
|
+
readonly type: "text";
|
|
778
|
+
readonly modelName: "gpt-5.6-sol";
|
|
779
|
+
readonly description: "GPT-5.6 Sol is the flagship model of the GPT-5.6 family for the most complex coding and agentic tasks. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
|
|
780
|
+
readonly maxInputTokens: 1050000;
|
|
781
|
+
readonly maxOutputTokens: 128000;
|
|
782
|
+
readonly inputTokenCost: 5;
|
|
783
|
+
readonly cachedInputTokenCost: 0.5;
|
|
784
|
+
readonly outputTokenCost: 30;
|
|
785
|
+
readonly longContext: {
|
|
786
|
+
readonly inputTokenCost: 10;
|
|
787
|
+
readonly cachedInputTokenCost: 1;
|
|
788
|
+
readonly outputTokenCost: 45;
|
|
789
|
+
readonly thresholdTokens: 200000;
|
|
790
|
+
};
|
|
791
|
+
readonly reasoning: {
|
|
792
|
+
readonly levels: readonly ["none", "low", "medium", "high", "xhigh", "max"];
|
|
793
|
+
readonly defaultLevel: "medium";
|
|
794
|
+
readonly canDisable: true;
|
|
795
|
+
readonly outputsThinking: false;
|
|
796
|
+
readonly outputsSignatures: false;
|
|
797
|
+
};
|
|
798
|
+
readonly modalities: {
|
|
799
|
+
readonly input: readonly ["text", "image", "pdf"];
|
|
800
|
+
readonly output: readonly ["text"];
|
|
801
|
+
};
|
|
802
|
+
readonly knowledge: "2026-02-16";
|
|
803
|
+
readonly releaseDate: "2026-07-09";
|
|
804
|
+
readonly lastUpdated: "2026-07-09";
|
|
805
|
+
readonly family: "gpt";
|
|
806
|
+
readonly openWeights: false;
|
|
807
|
+
readonly structuredOutput: true;
|
|
808
|
+
readonly temperatureSupported: false;
|
|
809
|
+
readonly provider: "openai";
|
|
810
|
+
}, {
|
|
811
|
+
readonly type: "text";
|
|
812
|
+
readonly modelName: "gpt-5.6-terra";
|
|
813
|
+
readonly description: "GPT-5.6 Terra balances capability and cost — competitive with GPT-5.5 at roughly half the price. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
|
|
814
|
+
readonly maxInputTokens: 1050000;
|
|
815
|
+
readonly maxOutputTokens: 128000;
|
|
816
|
+
readonly inputTokenCost: 2.5;
|
|
817
|
+
readonly cachedInputTokenCost: 0.25;
|
|
818
|
+
readonly outputTokenCost: 15;
|
|
819
|
+
readonly longContext: {
|
|
820
|
+
readonly inputTokenCost: 5;
|
|
821
|
+
readonly cachedInputTokenCost: 0.5;
|
|
822
|
+
readonly outputTokenCost: 22.5;
|
|
823
|
+
readonly thresholdTokens: 200000;
|
|
824
|
+
};
|
|
825
|
+
readonly reasoning: {
|
|
826
|
+
readonly levels: readonly ["none", "low", "medium", "high", "xhigh", "max"];
|
|
827
|
+
readonly defaultLevel: "medium";
|
|
828
|
+
readonly canDisable: true;
|
|
829
|
+
readonly outputsThinking: false;
|
|
830
|
+
readonly outputsSignatures: false;
|
|
831
|
+
};
|
|
832
|
+
readonly modalities: {
|
|
833
|
+
readonly input: readonly ["text", "image", "pdf"];
|
|
834
|
+
readonly output: readonly ["text"];
|
|
835
|
+
};
|
|
836
|
+
readonly knowledge: "2026-02-16";
|
|
837
|
+
readonly releaseDate: "2026-07-09";
|
|
838
|
+
readonly lastUpdated: "2026-07-09";
|
|
839
|
+
readonly family: "gpt";
|
|
840
|
+
readonly openWeights: false;
|
|
841
|
+
readonly structuredOutput: true;
|
|
842
|
+
readonly temperatureSupported: false;
|
|
843
|
+
readonly provider: "openai";
|
|
844
|
+
}, {
|
|
845
|
+
readonly type: "text";
|
|
846
|
+
readonly modelName: "gpt-5.6-luna";
|
|
847
|
+
readonly description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
|
|
848
|
+
readonly maxInputTokens: 1050000;
|
|
849
|
+
readonly maxOutputTokens: 128000;
|
|
850
|
+
readonly inputTokenCost: 1;
|
|
851
|
+
readonly cachedInputTokenCost: 0.1;
|
|
852
|
+
readonly outputTokenCost: 6;
|
|
853
|
+
readonly longContext: {
|
|
854
|
+
readonly inputTokenCost: 2;
|
|
855
|
+
readonly cachedInputTokenCost: 0.2;
|
|
856
|
+
readonly outputTokenCost: 9;
|
|
857
|
+
readonly thresholdTokens: 200000;
|
|
858
|
+
};
|
|
859
|
+
readonly reasoning: {
|
|
860
|
+
readonly levels: readonly ["none", "low", "medium", "high", "xhigh", "max"];
|
|
861
|
+
readonly defaultLevel: "medium";
|
|
862
|
+
readonly canDisable: true;
|
|
863
|
+
readonly outputsThinking: false;
|
|
864
|
+
readonly outputsSignatures: false;
|
|
865
|
+
};
|
|
866
|
+
readonly modalities: {
|
|
867
|
+
readonly input: readonly ["text", "image", "pdf"];
|
|
868
|
+
readonly output: readonly ["text"];
|
|
869
|
+
};
|
|
870
|
+
readonly knowledge: "2026-02-16";
|
|
871
|
+
readonly releaseDate: "2026-07-09";
|
|
872
|
+
readonly lastUpdated: "2026-07-09";
|
|
873
|
+
readonly family: "gpt";
|
|
874
|
+
readonly openWeights: false;
|
|
875
|
+
readonly structuredOutput: true;
|
|
876
|
+
readonly temperatureSupported: false;
|
|
877
|
+
readonly provider: "openai";
|
|
776
878
|
}, {
|
|
777
879
|
readonly type: "text";
|
|
778
880
|
readonly modelName: "gemini-3.1-pro-preview";
|
|
@@ -1040,7 +1142,7 @@ export declare const textModels: readonly [{
|
|
|
1040
1142
|
}, {
|
|
1041
1143
|
readonly type: "text";
|
|
1042
1144
|
readonly modelName: "gemini-2.0-flash";
|
|
1043
|
-
readonly description: "Workhorse model for all daily tasks. Strong overall performance and supports real-time streaming Live API. 1M context window.
|
|
1145
|
+
readonly description: "Workhorse model for all daily tasks. Strong overall performance and supports real-time streaming Live API. 1M context window. RETIRED: Shut down June 1, 2026. Use gemini-2.5-flash instead.";
|
|
1044
1146
|
readonly maxInputTokens: 1048576;
|
|
1045
1147
|
readonly maxOutputTokens: 8192;
|
|
1046
1148
|
readonly inputTokenCost: 0.1;
|
|
@@ -1073,7 +1175,7 @@ export declare const textModels: readonly [{
|
|
|
1073
1175
|
}, {
|
|
1074
1176
|
readonly type: "text";
|
|
1075
1177
|
readonly modelName: "gemini-2.0-flash-lite";
|
|
1076
|
-
readonly description: "Cost effective offering to support high throughput.
|
|
1178
|
+
readonly description: "Cost effective offering to support high throughput. RETIRED: Shut down June 1, 2026. Use gemini-2.5-flash-lite instead.";
|
|
1077
1179
|
readonly maxInputTokens: 1048576;
|
|
1078
1180
|
readonly maxOutputTokens: 8192;
|
|
1079
1181
|
readonly inputTokenCost: 0.075;
|
|
@@ -1439,6 +1541,12 @@ export declare const imageModels: readonly [{
|
|
|
1439
1541
|
readonly provider: "google";
|
|
1440
1542
|
readonly description: "Fast image generation with Gemini 3.1 Flash (GA). Supports resolutions from 512px to 4096px. ~$0.045/image at 512px, $0.067 at 1K, $0.101 at 2K, $0.151 at 4K.";
|
|
1441
1543
|
readonly costPerImage: 0.067;
|
|
1544
|
+
}, {
|
|
1545
|
+
readonly type: "image";
|
|
1546
|
+
readonly modelName: "gemini-3.1-flash-lite-image";
|
|
1547
|
+
readonly provider: "google";
|
|
1548
|
+
readonly description: "aka Nano Banana 2 Lite (GA 2026-06-30). Fastest, most cost-effective Gemini image model (~4s generation). ~$0.034/image at 1K. Recommended replacement for gemini-2.5-flash-image.";
|
|
1549
|
+
readonly costPerImage: 0.034;
|
|
1442
1550
|
}];
|
|
1443
1551
|
export declare const embeddingsModels: EmbeddingsModel[];
|
|
1444
1552
|
export type TextModelName = (typeof textModels)[number]["modelName"];
|
package/dist/models.js
CHANGED
|
@@ -733,6 +733,111 @@ export const textModels = [
|
|
|
733
733
|
temperatureSupported: false,
|
|
734
734
|
provider: "openai-responses",
|
|
735
735
|
},
|
|
736
|
+
{
|
|
737
|
+
type: "text",
|
|
738
|
+
modelName: "gpt-5.6-sol",
|
|
739
|
+
description: "GPT-5.6 Sol is the flagship model of the GPT-5.6 family for the most complex coding and agentic tasks. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
|
|
740
|
+
maxInputTokens: 1050000,
|
|
741
|
+
maxOutputTokens: 128000,
|
|
742
|
+
inputTokenCost: 5,
|
|
743
|
+
cachedInputTokenCost: 0.5,
|
|
744
|
+
outputTokenCost: 30,
|
|
745
|
+
longContext: {
|
|
746
|
+
inputTokenCost: 10,
|
|
747
|
+
cachedInputTokenCost: 1,
|
|
748
|
+
outputTokenCost: 45,
|
|
749
|
+
thresholdTokens: 200000,
|
|
750
|
+
},
|
|
751
|
+
reasoning: {
|
|
752
|
+
levels: ["none", "low", "medium", "high", "xhigh", "max"],
|
|
753
|
+
defaultLevel: "medium",
|
|
754
|
+
canDisable: true,
|
|
755
|
+
outputsThinking: false,
|
|
756
|
+
outputsSignatures: false,
|
|
757
|
+
},
|
|
758
|
+
modalities: {
|
|
759
|
+
input: ["text", "image", "pdf"],
|
|
760
|
+
output: ["text"],
|
|
761
|
+
},
|
|
762
|
+
knowledge: "2026-02-16",
|
|
763
|
+
releaseDate: "2026-07-09",
|
|
764
|
+
lastUpdated: "2026-07-09",
|
|
765
|
+
family: "gpt",
|
|
766
|
+
openWeights: false,
|
|
767
|
+
structuredOutput: true,
|
|
768
|
+
temperatureSupported: false,
|
|
769
|
+
provider: "openai",
|
|
770
|
+
},
|
|
771
|
+
{
|
|
772
|
+
type: "text",
|
|
773
|
+
modelName: "gpt-5.6-terra",
|
|
774
|
+
description: "GPT-5.6 Terra balances capability and cost — competitive with GPT-5.5 at roughly half the price. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
|
|
775
|
+
maxInputTokens: 1050000,
|
|
776
|
+
maxOutputTokens: 128000,
|
|
777
|
+
inputTokenCost: 2.5,
|
|
778
|
+
cachedInputTokenCost: 0.25,
|
|
779
|
+
outputTokenCost: 15,
|
|
780
|
+
longContext: {
|
|
781
|
+
inputTokenCost: 5,
|
|
782
|
+
cachedInputTokenCost: 0.5,
|
|
783
|
+
outputTokenCost: 22.5,
|
|
784
|
+
thresholdTokens: 200000,
|
|
785
|
+
},
|
|
786
|
+
reasoning: {
|
|
787
|
+
levels: ["none", "low", "medium", "high", "xhigh", "max"],
|
|
788
|
+
defaultLevel: "medium",
|
|
789
|
+
canDisable: true,
|
|
790
|
+
outputsThinking: false,
|
|
791
|
+
outputsSignatures: false,
|
|
792
|
+
},
|
|
793
|
+
modalities: {
|
|
794
|
+
input: ["text", "image", "pdf"],
|
|
795
|
+
output: ["text"],
|
|
796
|
+
},
|
|
797
|
+
knowledge: "2026-02-16",
|
|
798
|
+
releaseDate: "2026-07-09",
|
|
799
|
+
lastUpdated: "2026-07-09",
|
|
800
|
+
family: "gpt",
|
|
801
|
+
openWeights: false,
|
|
802
|
+
structuredOutput: true,
|
|
803
|
+
temperatureSupported: false,
|
|
804
|
+
provider: "openai",
|
|
805
|
+
},
|
|
806
|
+
{
|
|
807
|
+
type: "text",
|
|
808
|
+
modelName: "gpt-5.6-luna",
|
|
809
|
+
description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
|
|
810
|
+
maxInputTokens: 1050000,
|
|
811
|
+
maxOutputTokens: 128000,
|
|
812
|
+
inputTokenCost: 1,
|
|
813
|
+
cachedInputTokenCost: 0.1,
|
|
814
|
+
outputTokenCost: 6,
|
|
815
|
+
longContext: {
|
|
816
|
+
inputTokenCost: 2,
|
|
817
|
+
cachedInputTokenCost: 0.2,
|
|
818
|
+
outputTokenCost: 9,
|
|
819
|
+
thresholdTokens: 200000,
|
|
820
|
+
},
|
|
821
|
+
reasoning: {
|
|
822
|
+
levels: ["none", "low", "medium", "high", "xhigh", "max"],
|
|
823
|
+
defaultLevel: "medium",
|
|
824
|
+
canDisable: true,
|
|
825
|
+
outputsThinking: false,
|
|
826
|
+
outputsSignatures: false,
|
|
827
|
+
},
|
|
828
|
+
modalities: {
|
|
829
|
+
input: ["text", "image", "pdf"],
|
|
830
|
+
output: ["text"],
|
|
831
|
+
},
|
|
832
|
+
knowledge: "2026-02-16",
|
|
833
|
+
releaseDate: "2026-07-09",
|
|
834
|
+
lastUpdated: "2026-07-09",
|
|
835
|
+
family: "gpt",
|
|
836
|
+
openWeights: false,
|
|
837
|
+
structuredOutput: true,
|
|
838
|
+
temperatureSupported: false,
|
|
839
|
+
provider: "openai",
|
|
840
|
+
},
|
|
736
841
|
{
|
|
737
842
|
type: "text",
|
|
738
843
|
modelName: "gemini-3.1-pro-preview",
|
|
@@ -1009,7 +1114,7 @@ export const textModels = [
|
|
|
1009
1114
|
{
|
|
1010
1115
|
type: "text",
|
|
1011
1116
|
modelName: "gemini-2.0-flash",
|
|
1012
|
-
description: "Workhorse model for all daily tasks. Strong overall performance and supports real-time streaming Live API. 1M context window.
|
|
1117
|
+
description: "Workhorse model for all daily tasks. Strong overall performance and supports real-time streaming Live API. 1M context window. RETIRED: Shut down June 1, 2026. Use gemini-2.5-flash instead.",
|
|
1013
1118
|
maxInputTokens: 1048576,
|
|
1014
1119
|
maxOutputTokens: 8192,
|
|
1015
1120
|
inputTokenCost: 0.1,
|
|
@@ -1044,7 +1149,7 @@ export const textModels = [
|
|
|
1044
1149
|
{
|
|
1045
1150
|
type: "text",
|
|
1046
1151
|
modelName: "gemini-2.0-flash-lite",
|
|
1047
|
-
description: "Cost effective offering to support high throughput.
|
|
1152
|
+
description: "Cost effective offering to support high throughput. RETIRED: Shut down June 1, 2026. Use gemini-2.5-flash-lite instead.",
|
|
1048
1153
|
maxInputTokens: 1048576,
|
|
1049
1154
|
maxOutputTokens: 8192,
|
|
1050
1155
|
inputTokenCost: 0.075,
|
|
@@ -1434,6 +1539,13 @@ export const imageModels = [
|
|
|
1434
1539
|
description: "Fast image generation with Gemini 3.1 Flash (GA). Supports resolutions from 512px to 4096px. ~$0.045/image at 512px, $0.067 at 1K, $0.101 at 2K, $0.151 at 4K.",
|
|
1435
1540
|
costPerImage: 0.067,
|
|
1436
1541
|
},
|
|
1542
|
+
{
|
|
1543
|
+
type: "image",
|
|
1544
|
+
modelName: "gemini-3.1-flash-lite-image",
|
|
1545
|
+
provider: "google",
|
|
1546
|
+
description: "aka Nano Banana 2 Lite (GA 2026-06-30). Fastest, most cost-effective Gemini image model (~4s generation). ~$0.034/image at 1K. Recommended replacement for gemini-2.5-flash-image.",
|
|
1547
|
+
costPerImage: 0.034,
|
|
1548
|
+
},
|
|
1437
1549
|
];
|
|
1438
1550
|
export const embeddingsModels = [
|
|
1439
1551
|
{
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Normalized reason a generation turn ended, unified across providers. The
|
|
3
|
+
* untouched provider value is available separately as `PromptResult.rawStopReason`.
|
|
4
|
+
*/
|
|
5
|
+
export type StopReason =
|
|
6
|
+
/** Natural completion (OpenAI `stop`, Anthropic `end_turn`, Google `STOP`, Ollama `stop`). */
|
|
7
|
+
"stop"
|
|
8
|
+
/** Hit the max-tokens limit (`length` / `max_tokens` / `MAX_TOKENS`). */
|
|
9
|
+
| "length"
|
|
10
|
+
/** Model wants to call a tool (`tool_calls` / `tool_use`). */
|
|
11
|
+
| "tool_use"
|
|
12
|
+
/** Blocked by a safety/policy filter or refusal (`content_filter` / `refusal` / `SAFETY`). */
|
|
13
|
+
| "content_filter"
|
|
14
|
+
/** Hit a caller-supplied stop sequence (Anthropic `stop_sequence`). */
|
|
15
|
+
| "stop_sequence"
|
|
16
|
+
/** Provider paused a long-running turn (Anthropic `pause_turn`). */
|
|
17
|
+
| "pause"
|
|
18
|
+
/** Anything unmapped or unknown. */
|
|
19
|
+
| "other";
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/dist/types.d.ts
CHANGED
|
@@ -10,8 +10,10 @@ import type { ModelDataBlob } from "./modelData.js";
|
|
|
10
10
|
import { Result } from "./types/result.js";
|
|
11
11
|
import { TokenUsage } from "./types/tokenUsage.js";
|
|
12
12
|
import { CostEstimate } from "./types/costEstimate.js";
|
|
13
|
+
import { StopReason } from "./types/stopReason.js";
|
|
13
14
|
export * from "./types/costEstimate.js";
|
|
14
15
|
export * from "./types/tokenUsage.js";
|
|
16
|
+
export * from "./types/stopReason.js";
|
|
15
17
|
export type SmolConfig = {
|
|
16
18
|
/** The model to use. */
|
|
17
19
|
model: ModelName;
|
|
@@ -150,8 +152,12 @@ export type PromptResult = {
|
|
|
150
152
|
cost?: CostEstimate;
|
|
151
153
|
model?: ModelName;
|
|
152
154
|
hostedToolResults?: HostedToolResult[];
|
|
155
|
+
/** Normalized reason the turn ended, unified across providers. */
|
|
156
|
+
stopReason?: StopReason;
|
|
157
|
+
/** The untouched provider finish/stop-reason value (e.g. `end_turn`, `MAX_TOKENS`). */
|
|
158
|
+
rawStopReason?: string;
|
|
153
159
|
};
|
|
154
|
-
export declare function promptResult({ output, toolCalls, thinkingBlocks, usage, cost, model, hostedToolResults, }: Partial<PromptResult>): PromptResult;
|
|
160
|
+
export declare function promptResult({ output, toolCalls, thinkingBlocks, usage, cost, model, hostedToolResults, stopReason, rawStopReason, }: Partial<PromptResult>): PromptResult;
|
|
155
161
|
export type StreamChunk = {
|
|
156
162
|
type: "text";
|
|
157
163
|
text: string;
|
|
@@ -162,6 +168,14 @@ export type StreamChunk = {
|
|
|
162
168
|
} | {
|
|
163
169
|
type: "tool_call";
|
|
164
170
|
toolCall: ToolCall;
|
|
171
|
+
}
|
|
172
|
+
/** A provider-run web search (server-side tool). Emitted the moment a
|
|
173
|
+
* search block completes, so consumers can surface the query live. One
|
|
174
|
+
* chunk per search; a turn may emit several. The same queries also appear
|
|
175
|
+
* in the final `done` result's `hostedToolResults`. */
|
|
176
|
+
| {
|
|
177
|
+
type: "web_search";
|
|
178
|
+
query: string;
|
|
165
179
|
} | {
|
|
166
180
|
type: "done";
|
|
167
181
|
result: PromptResult;
|
package/dist/types.js
CHANGED
|
@@ -3,7 +3,8 @@ export * from "./classes/message/contentParts.js";
|
|
|
3
3
|
import z from "zod";
|
|
4
4
|
export * from "./types/costEstimate.js";
|
|
5
5
|
export * from "./types/tokenUsage.js";
|
|
6
|
-
export
|
|
6
|
+
export * from "./types/stopReason.js";
|
|
7
|
+
export function promptResult({ output, toolCalls, thinkingBlocks, usage, cost, model, hostedToolResults, stopReason, rawStopReason, }) {
|
|
7
8
|
return {
|
|
8
9
|
output: output || null,
|
|
9
10
|
toolCalls: toolCalls || [],
|
|
@@ -12,6 +13,8 @@ export function promptResult({ output, toolCalls, thinkingBlocks, usage, cost, m
|
|
|
12
13
|
cost,
|
|
13
14
|
model,
|
|
14
15
|
hostedToolResults,
|
|
16
|
+
stopReason,
|
|
17
|
+
rawStopReason,
|
|
15
18
|
};
|
|
16
19
|
}
|
|
17
20
|
export const ThinkingBlockSchema = z.object({
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* JSON Schema sanitization for structured output.
|
|
3
|
+
*
|
|
4
|
+
* Zod converts `z.any()` / `z.unknown()` to an unconstrained schema — a bare
|
|
5
|
+
* `{}` (nested), `{"$schema":…}` (top-level), or the boolean `true`. That is
|
|
6
|
+
* valid JSON Schema ("accept any value"), but every provider rejects an
|
|
7
|
+
* unconstrained node inside a structured-output or strict-tool schema, since the
|
|
8
|
+
* whole point of structured output is that it is structured. These helpers map
|
|
9
|
+
* such nodes to `{"type":"string"}` (the safe universal container) and let
|
|
10
|
+
* callers detect the whole-schema-is-`any` case so they can drop structured
|
|
11
|
+
* output entirely and return free text.
|
|
12
|
+
*/
|
|
13
|
+
/**
|
|
14
|
+
* True if `node` accepts any value: the boolean `true`, or a plain object whose
|
|
15
|
+
* every own key is a pure annotation (`{}`, `{$schema:…}`,
|
|
16
|
+
* `{$schema:…, description:…}`). `false` and any object with a structural or
|
|
17
|
+
* validation keyword are constrained.
|
|
18
|
+
*/
|
|
19
|
+
export declare function isUnconstrainedSchema(node: unknown): boolean;
|
|
20
|
+
/**
|
|
21
|
+
* Convert a Zod `responseFormat` schema to a sanitized JSON Schema for a
|
|
22
|
+
* provider's structured-output request: any nested unconstrained node becomes
|
|
23
|
+
* `{"type":"string"}`. (The whole-schema-is-`any` case is handled upstream in
|
|
24
|
+
* `BaseClient.normalizeResponseFormat`, which drops structured output entirely.)
|
|
25
|
+
*/
|
|
26
|
+
export declare function responseFormatToJsonSchema(schema: {
|
|
27
|
+
toJSONSchema: () => unknown;
|
|
28
|
+
}): object;
|
|
29
|
+
/**
|
|
30
|
+
* Returns a new JSON Schema with every unconstrained node replaced by
|
|
31
|
+
* `{"type":"string"}` (annotations preserved), recursing through all subschema
|
|
32
|
+
* positions. Idempotent — a node that already has `type` is left untouched.
|
|
33
|
+
*
|
|
34
|
+
* `additionalProperties` and `items` may be a boolean (`true`/`false`), which is
|
|
35
|
+
* a legitimate provider-accepted flag, not an any-typed value slot; booleans
|
|
36
|
+
* there are left as-is and only an object subschema is sanitized.
|
|
37
|
+
*/
|
|
38
|
+
export declare function sanitizeJsonSchema(node: unknown): unknown;
|