smoltalk 0.8.1 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,6 +4,8 @@ import { getLogger } from "../util/logger.js";
4
4
  import { redactAttachments } from "../util/redact.js";
5
5
  import { success, } from "../types.js";
6
6
  import { zodToGoogleTool } from "../util/tool.js";
7
+ import { responseFormatToJsonSchema } from "../util/jsonSchema.js";
8
+ import { normalizeOllamaStopReason } from "../util/stopReason.js";
7
9
  import { sanitizeAttributes } from "../util/util.js";
8
10
  import { resolveBaseUrl } from "../util/provider.js";
9
11
  import { BaseClient } from "./baseClient.js";
@@ -81,7 +83,7 @@ export class SmolOllama extends BaseClient {
81
83
  request.tools = tools.map((t) => ({ type: "function", function: t }));
82
84
  }
83
85
  if (config.responseFormat) {
84
- request.format = config.responseFormat.toJSONSchema();
86
+ request.format = responseFormatToJsonSchema(config.responseFormat);
85
87
  }
86
88
  Object.assign(request, sanitizeAttributes(config.rawAttributes));
87
89
  this.logger.debug("Sending request to Ollama:", JSON.stringify(redactAttachments(request), null, 2));
@@ -116,8 +118,20 @@ export class SmolOllama extends BaseClient {
116
118
  }
117
119
  // Extract usage and calculate cost
118
120
  const { usage, cost } = this.calculateUsageAndCost(result);
121
+ const rawStopReason = result.done_reason ?? undefined;
119
122
  // Return the response, updating the chat history
120
- return success({ output, toolCalls, usage, cost, model: this.getModel() });
123
+ const promptResult = {
124
+ output,
125
+ toolCalls,
126
+ usage,
127
+ cost,
128
+ model: this.getModel(),
129
+ stopReason: normalizeOllamaStopReason(rawStopReason),
130
+ };
131
+ if (rawStopReason) {
132
+ promptResult.rawStopReason = rawStopReason;
133
+ }
134
+ return success(promptResult);
121
135
  }
122
136
  async *_textStream(config) {
123
137
  const messages = config.messages.map((msg) => msg.toOllamaMessage());
@@ -135,7 +149,7 @@ export class SmolOllama extends BaseClient {
135
149
  request.tools = tools.map((t) => ({ type: "function", function: t }));
136
150
  }
137
151
  if (config.responseFormat) {
138
- request.format = config.responseFormat.toJSONSchema();
152
+ request.format = responseFormatToJsonSchema(config.responseFormat);
139
153
  }
140
154
  Object.assign(request, sanitizeAttributes(config.rawAttributes));
141
155
  this.logger.debug("Sending streaming request to Ollama:", JSON.stringify(redactAttachments(request), null, 2));
@@ -201,16 +215,19 @@ export class SmolOllama extends BaseClient {
201
215
  toolCalls.push(toolCall);
202
216
  yield { type: "tool_call", toolCall };
203
217
  }
204
- yield {
205
- type: "done",
206
- result: {
207
- output: content || null,
208
- toolCalls,
209
- usage,
210
- cost,
211
- model: this.getModel(),
212
- },
218
+ const rawStopReason = lastChunk?.done_reason ?? undefined;
219
+ const result = {
220
+ output: content || null,
221
+ toolCalls,
222
+ usage,
223
+ cost,
224
+ model: this.getModel(),
225
+ stopReason: normalizeOllamaStopReason(rawStopReason),
213
226
  };
227
+ if (rawStopReason) {
228
+ result.rawStopReason = rawStopReason;
229
+ }
230
+ yield { type: "done", result };
214
231
  }
215
232
  catch (error) {
216
233
  this.rethrowAsSmolError(error);
@@ -8,6 +8,8 @@ import { BaseClient } from "./baseClient.js";
8
8
  import { SmolContentPolicyError, SmolContextWindowExceededError, smolErrorForStatus, } from "../smolError.js";
9
9
  import { extractHttpErrorFields } from "../util/httpError.js";
10
10
  import { zodToOpenAITool } from "../util/tool.js";
11
+ import { responseFormatToJsonSchema } from "../util/jsonSchema.js";
12
+ import { normalizeOpenAIStopReason } from "../util/stopReason.js";
11
13
  import { Model } from "../model.js";
12
14
  export class SmolOpenAi extends BaseClient {
13
15
  client;
@@ -120,7 +122,7 @@ export class SmolOpenAi extends BaseClient {
120
122
  type: "json_schema",
121
123
  json_schema: {
122
124
  name: config.responseFormatOptions?.name || "response",
123
- schema: config.responseFormat.toJSONSchema(),
125
+ schema: responseFormatToJsonSchema(config.responseFormat),
124
126
  },
125
127
  };
126
128
  }
@@ -185,14 +187,22 @@ export class SmolOpenAi extends BaseClient {
185
187
  // response headers (e.g. LiteLLM's x-litellm-response-cost).
186
188
  const { usage, cost } = this.calculateUsageAndCost(completion.usage, rawResponse);
187
189
  const hostedToolResults = this.parseHostedToolResults(completion, config);
188
- return success({
190
+ const rawStopReason = completion.choices[0]?.finish_reason ?? undefined;
191
+ const result = {
189
192
  output,
190
193
  toolCalls,
191
194
  usage,
192
195
  cost,
193
196
  model: this.getModel(),
194
- ...(hostedToolResults.length > 0 ? { hostedToolResults } : {}),
195
- });
197
+ stopReason: normalizeOpenAIStopReason(rawStopReason),
198
+ };
199
+ if (rawStopReason) {
200
+ result.rawStopReason = rawStopReason;
201
+ }
202
+ if (hostedToolResults.length > 0) {
203
+ result.hostedToolResults = hostedToolResults;
204
+ }
205
+ return success(result);
196
206
  }
197
207
  async *_textStream(config) {
198
208
  const request = this.buildRequest(config);
@@ -214,7 +224,11 @@ export class SmolOpenAi extends BaseClient {
214
224
  const toolCallsMap = new Map();
215
225
  let usage;
216
226
  let cost;
227
+ let rawStopReason;
217
228
  for await (const chunk of completion) {
229
+ const chunkFinish = chunk.choices?.[0]?.finish_reason;
230
+ if (chunkFinish)
231
+ rawStopReason = chunkFinish;
218
232
  // Extract usage from the final chunk
219
233
  if (chunk.usage) {
220
234
  // Header-based cost (LiteLLM) is unsupported while streaming.
@@ -266,15 +280,17 @@ export class SmolOpenAi extends BaseClient {
266
280
  toolCalls.push(toolCall);
267
281
  yield { type: "tool_call", toolCall };
268
282
  }
269
- yield {
270
- type: "done",
271
- result: {
272
- output: content || null,
273
- toolCalls,
274
- usage,
275
- cost,
276
- model: this.getModel(),
277
- },
283
+ const result = {
284
+ output: content || null,
285
+ toolCalls,
286
+ usage,
287
+ cost,
288
+ model: this.getModel(),
289
+ stopReason: normalizeOpenAIStopReason(rawStopReason),
278
290
  };
291
+ if (rawStopReason) {
292
+ result.rawStopReason = rawStopReason;
293
+ }
294
+ yield { type: "done", result };
279
295
  }
280
296
  }
@@ -5,6 +5,8 @@ import { getLogger } from "../util/logger.js";
5
5
  import { redactAttachments } from "../util/redact.js";
6
6
  import { BaseClient } from "./baseClient.js";
7
7
  import { zodToOpenAIResponsesTool } from "../util/tool.js";
8
+ import { responseFormatToJsonSchema } from "../util/jsonSchema.js";
9
+ import { normalizeOpenAIResponsesStopReason } from "../util/stopReason.js";
8
10
  import { sanitizeAttributes } from "../util/util.js";
9
11
  import { WEB_SEARCH, webSearchResult, applyHostedToolCost } from "../util/hostedTools.js";
10
12
  import { Model } from "../model.js";
@@ -122,7 +124,7 @@ export class SmolOpenAiResponses extends BaseClient {
122
124
  format: {
123
125
  type: "json_schema",
124
126
  name: config.responseFormatOptions?.name || "response",
125
- schema: config.responseFormat.toJSONSchema(),
127
+ schema: responseFormatToJsonSchema(config.responseFormat),
126
128
  },
127
129
  };
128
130
  }
@@ -192,13 +194,19 @@ export class SmolOpenAiResponses extends BaseClient {
192
194
  const { usage, cost } = this.calculateUsageAndCost(response.usage);
193
195
  const parsed = parseOpenAIResponsesHostedTools(response, "openai-responses");
194
196
  const { results: hostedToolResults, cost: finalCost } = applyHostedToolCost(parsed, cost, this.getModel(), this.config.modelData);
197
+ const incompleteReason = response.incomplete_details?.reason;
198
+ const rawStopReason = incompleteReason ?? response.status ?? undefined;
195
199
  const result = {
196
200
  output,
197
201
  toolCalls,
198
202
  usage,
199
203
  cost: finalCost,
200
204
  model: this.getModel(),
205
+ stopReason: normalizeOpenAIResponsesStopReason(response.status, incompleteReason, toolCalls.length > 0),
201
206
  };
207
+ if (rawStopReason) {
208
+ result.rawStopReason = rawStopReason;
209
+ }
202
210
  if (hostedToolResults.length > 0) {
203
211
  result.hostedToolResults = hostedToolResults;
204
212
  }
@@ -222,7 +230,12 @@ export class SmolOpenAiResponses extends BaseClient {
222
230
  const functionCalls = new Map();
223
231
  let usage;
224
232
  let cost;
233
+ let finalResponse;
225
234
  for await (const event of stream) {
235
+ if (event.type === "response.completed" ||
236
+ event.type === "response.incomplete") {
237
+ finalResponse = event.response;
238
+ }
226
239
  switch (event.type) {
227
240
  case "response.output_text.delta": {
228
241
  content += event.delta;
@@ -291,15 +304,19 @@ export class SmolOpenAiResponses extends BaseClient {
291
304
  toolCalls.push(toolCall);
292
305
  yield { type: "tool_call", toolCall };
293
306
  }
294
- yield {
295
- type: "done",
296
- result: {
297
- output: content || null,
298
- toolCalls,
299
- usage,
300
- cost,
301
- model: this.getModel(),
302
- },
307
+ const incompleteReason = finalResponse?.incomplete_details?.reason;
308
+ const rawStopReason = incompleteReason ?? finalResponse?.status ?? undefined;
309
+ const result = {
310
+ output: content || null,
311
+ toolCalls,
312
+ usage,
313
+ cost,
314
+ model: this.getModel(),
315
+ stopReason: normalizeOpenAIResponsesStopReason(finalResponse?.status, incompleteReason, toolCalls.length > 0),
303
316
  };
317
+ if (rawStopReason) {
318
+ result.rawStopReason = rawStopReason;
319
+ }
320
+ yield { type: "done", result };
304
321
  }
305
322
  }
package/dist/models.d.ts CHANGED
@@ -773,6 +773,108 @@ export declare const textModels: readonly [{
773
773
  readonly structuredOutput: true;
774
774
  readonly temperatureSupported: false;
775
775
  readonly provider: "openai-responses";
776
+ }, {
777
+ readonly type: "text";
778
+ readonly modelName: "gpt-5.6-sol";
779
+ readonly description: "GPT-5.6 Sol is the flagship model of the GPT-5.6 family for the most complex coding and agentic tasks. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
780
+ readonly maxInputTokens: 1050000;
781
+ readonly maxOutputTokens: 128000;
782
+ readonly inputTokenCost: 5;
783
+ readonly cachedInputTokenCost: 0.5;
784
+ readonly outputTokenCost: 30;
785
+ readonly longContext: {
786
+ readonly inputTokenCost: 10;
787
+ readonly cachedInputTokenCost: 1;
788
+ readonly outputTokenCost: 45;
789
+ readonly thresholdTokens: 200000;
790
+ };
791
+ readonly reasoning: {
792
+ readonly levels: readonly ["none", "low", "medium", "high", "xhigh", "max"];
793
+ readonly defaultLevel: "medium";
794
+ readonly canDisable: true;
795
+ readonly outputsThinking: false;
796
+ readonly outputsSignatures: false;
797
+ };
798
+ readonly modalities: {
799
+ readonly input: readonly ["text", "image", "pdf"];
800
+ readonly output: readonly ["text"];
801
+ };
802
+ readonly knowledge: "2026-02-16";
803
+ readonly releaseDate: "2026-07-09";
804
+ readonly lastUpdated: "2026-07-09";
805
+ readonly family: "gpt";
806
+ readonly openWeights: false;
807
+ readonly structuredOutput: true;
808
+ readonly temperatureSupported: false;
809
+ readonly provider: "openai";
810
+ }, {
811
+ readonly type: "text";
812
+ readonly modelName: "gpt-5.6-terra";
813
+ readonly description: "GPT-5.6 Terra balances capability and cost — competitive with GPT-5.5 at roughly half the price. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
814
+ readonly maxInputTokens: 1050000;
815
+ readonly maxOutputTokens: 128000;
816
+ readonly inputTokenCost: 2.5;
817
+ readonly cachedInputTokenCost: 0.25;
818
+ readonly outputTokenCost: 15;
819
+ readonly longContext: {
820
+ readonly inputTokenCost: 5;
821
+ readonly cachedInputTokenCost: 0.5;
822
+ readonly outputTokenCost: 22.5;
823
+ readonly thresholdTokens: 200000;
824
+ };
825
+ readonly reasoning: {
826
+ readonly levels: readonly ["none", "low", "medium", "high", "xhigh", "max"];
827
+ readonly defaultLevel: "medium";
828
+ readonly canDisable: true;
829
+ readonly outputsThinking: false;
830
+ readonly outputsSignatures: false;
831
+ };
832
+ readonly modalities: {
833
+ readonly input: readonly ["text", "image", "pdf"];
834
+ readonly output: readonly ["text"];
835
+ };
836
+ readonly knowledge: "2026-02-16";
837
+ readonly releaseDate: "2026-07-09";
838
+ readonly lastUpdated: "2026-07-09";
839
+ readonly family: "gpt";
840
+ readonly openWeights: false;
841
+ readonly structuredOutput: true;
842
+ readonly temperatureSupported: false;
843
+ readonly provider: "openai";
844
+ }, {
845
+ readonly type: "text";
846
+ readonly modelName: "gpt-5.6-luna";
847
+ readonly description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
848
+ readonly maxInputTokens: 1050000;
849
+ readonly maxOutputTokens: 128000;
850
+ readonly inputTokenCost: 1;
851
+ readonly cachedInputTokenCost: 0.1;
852
+ readonly outputTokenCost: 6;
853
+ readonly longContext: {
854
+ readonly inputTokenCost: 2;
855
+ readonly cachedInputTokenCost: 0.2;
856
+ readonly outputTokenCost: 9;
857
+ readonly thresholdTokens: 200000;
858
+ };
859
+ readonly reasoning: {
860
+ readonly levels: readonly ["none", "low", "medium", "high", "xhigh", "max"];
861
+ readonly defaultLevel: "medium";
862
+ readonly canDisable: true;
863
+ readonly outputsThinking: false;
864
+ readonly outputsSignatures: false;
865
+ };
866
+ readonly modalities: {
867
+ readonly input: readonly ["text", "image", "pdf"];
868
+ readonly output: readonly ["text"];
869
+ };
870
+ readonly knowledge: "2026-02-16";
871
+ readonly releaseDate: "2026-07-09";
872
+ readonly lastUpdated: "2026-07-09";
873
+ readonly family: "gpt";
874
+ readonly openWeights: false;
875
+ readonly structuredOutput: true;
876
+ readonly temperatureSupported: false;
877
+ readonly provider: "openai";
776
878
  }, {
777
879
  readonly type: "text";
778
880
  readonly modelName: "gemini-3.1-pro-preview";
@@ -1040,7 +1142,7 @@ export declare const textModels: readonly [{
1040
1142
  }, {
1041
1143
  readonly type: "text";
1042
1144
  readonly modelName: "gemini-2.0-flash";
1043
- readonly description: "Workhorse model for all daily tasks. Strong overall performance and supports real-time streaming Live API. 1M context window. DEPRECATED: Will be shut down on March 31, 2026.";
1145
+ readonly description: "Workhorse model for all daily tasks. Strong overall performance and supports real-time streaming Live API. 1M context window. RETIRED: Shut down June 1, 2026. Use gemini-2.5-flash instead.";
1044
1146
  readonly maxInputTokens: 1048576;
1045
1147
  readonly maxOutputTokens: 8192;
1046
1148
  readonly inputTokenCost: 0.1;
@@ -1073,7 +1175,7 @@ export declare const textModels: readonly [{
1073
1175
  }, {
1074
1176
  readonly type: "text";
1075
1177
  readonly modelName: "gemini-2.0-flash-lite";
1076
- readonly description: "Cost effective offering to support high throughput. DEPRECATED: Will be shut down on March 31, 2026. Use gemini-2.5-flash-lite instead.";
1178
+ readonly description: "Cost effective offering to support high throughput. RETIRED: Shut down June 1, 2026. Use gemini-2.5-flash-lite instead.";
1077
1179
  readonly maxInputTokens: 1048576;
1078
1180
  readonly maxOutputTokens: 8192;
1079
1181
  readonly inputTokenCost: 0.075;
@@ -1439,6 +1541,12 @@ export declare const imageModels: readonly [{
1439
1541
  readonly provider: "google";
1440
1542
  readonly description: "Fast image generation with Gemini 3.1 Flash (GA). Supports resolutions from 512px to 4096px. ~$0.045/image at 512px, $0.067 at 1K, $0.101 at 2K, $0.151 at 4K.";
1441
1543
  readonly costPerImage: 0.067;
1544
+ }, {
1545
+ readonly type: "image";
1546
+ readonly modelName: "gemini-3.1-flash-lite-image";
1547
+ readonly provider: "google";
1548
+ readonly description: "aka Nano Banana 2 Lite (GA 2026-06-30). Fastest, most cost-effective Gemini image model (~4s generation). ~$0.034/image at 1K. Recommended replacement for gemini-2.5-flash-image.";
1549
+ readonly costPerImage: 0.034;
1442
1550
  }];
1443
1551
  export declare const embeddingsModels: EmbeddingsModel[];
1444
1552
  export type TextModelName = (typeof textModels)[number]["modelName"];
package/dist/models.js CHANGED
@@ -733,6 +733,111 @@ export const textModels = [
733
733
  temperatureSupported: false,
734
734
  provider: "openai-responses",
735
735
  },
736
+ {
737
+ type: "text",
738
+ modelName: "gpt-5.6-sol",
739
+ description: "GPT-5.6 Sol is the flagship model of the GPT-5.6 family for the most complex coding and agentic tasks. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
740
+ maxInputTokens: 1050000,
741
+ maxOutputTokens: 128000,
742
+ inputTokenCost: 5,
743
+ cachedInputTokenCost: 0.5,
744
+ outputTokenCost: 30,
745
+ longContext: {
746
+ inputTokenCost: 10,
747
+ cachedInputTokenCost: 1,
748
+ outputTokenCost: 45,
749
+ thresholdTokens: 200000,
750
+ },
751
+ reasoning: {
752
+ levels: ["none", "low", "medium", "high", "xhigh", "max"],
753
+ defaultLevel: "medium",
754
+ canDisable: true,
755
+ outputsThinking: false,
756
+ outputsSignatures: false,
757
+ },
758
+ modalities: {
759
+ input: ["text", "image", "pdf"],
760
+ output: ["text"],
761
+ },
762
+ knowledge: "2026-02-16",
763
+ releaseDate: "2026-07-09",
764
+ lastUpdated: "2026-07-09",
765
+ family: "gpt",
766
+ openWeights: false,
767
+ structuredOutput: true,
768
+ temperatureSupported: false,
769
+ provider: "openai",
770
+ },
771
+ {
772
+ type: "text",
773
+ modelName: "gpt-5.6-terra",
774
+ description: "GPT-5.6 Terra balances capability and cost — competitive with GPT-5.5 at roughly half the price. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
775
+ maxInputTokens: 1050000,
776
+ maxOutputTokens: 128000,
777
+ inputTokenCost: 2.5,
778
+ cachedInputTokenCost: 0.25,
779
+ outputTokenCost: 15,
780
+ longContext: {
781
+ inputTokenCost: 5,
782
+ cachedInputTokenCost: 0.5,
783
+ outputTokenCost: 22.5,
784
+ thresholdTokens: 200000,
785
+ },
786
+ reasoning: {
787
+ levels: ["none", "low", "medium", "high", "xhigh", "max"],
788
+ defaultLevel: "medium",
789
+ canDisable: true,
790
+ outputsThinking: false,
791
+ outputsSignatures: false,
792
+ },
793
+ modalities: {
794
+ input: ["text", "image", "pdf"],
795
+ output: ["text"],
796
+ },
797
+ knowledge: "2026-02-16",
798
+ releaseDate: "2026-07-09",
799
+ lastUpdated: "2026-07-09",
800
+ family: "gpt",
801
+ openWeights: false,
802
+ structuredOutput: true,
803
+ temperatureSupported: false,
804
+ provider: "openai",
805
+ },
806
+ {
807
+ type: "text",
808
+ modelName: "gpt-5.6-luna",
809
+ description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
810
+ maxInputTokens: 1050000,
811
+ maxOutputTokens: 128000,
812
+ inputTokenCost: 1,
813
+ cachedInputTokenCost: 0.1,
814
+ outputTokenCost: 6,
815
+ longContext: {
816
+ inputTokenCost: 2,
817
+ cachedInputTokenCost: 0.2,
818
+ outputTokenCost: 9,
819
+ thresholdTokens: 200000,
820
+ },
821
+ reasoning: {
822
+ levels: ["none", "low", "medium", "high", "xhigh", "max"],
823
+ defaultLevel: "medium",
824
+ canDisable: true,
825
+ outputsThinking: false,
826
+ outputsSignatures: false,
827
+ },
828
+ modalities: {
829
+ input: ["text", "image", "pdf"],
830
+ output: ["text"],
831
+ },
832
+ knowledge: "2026-02-16",
833
+ releaseDate: "2026-07-09",
834
+ lastUpdated: "2026-07-09",
835
+ family: "gpt",
836
+ openWeights: false,
837
+ structuredOutput: true,
838
+ temperatureSupported: false,
839
+ provider: "openai",
840
+ },
736
841
  {
737
842
  type: "text",
738
843
  modelName: "gemini-3.1-pro-preview",
@@ -1009,7 +1114,7 @@ export const textModels = [
1009
1114
  {
1010
1115
  type: "text",
1011
1116
  modelName: "gemini-2.0-flash",
1012
- description: "Workhorse model for all daily tasks. Strong overall performance and supports real-time streaming Live API. 1M context window. DEPRECATED: Will be shut down on March 31, 2026.",
1117
+ description: "Workhorse model for all daily tasks. Strong overall performance and supports real-time streaming Live API. 1M context window. RETIRED: Shut down June 1, 2026. Use gemini-2.5-flash instead.",
1013
1118
  maxInputTokens: 1048576,
1014
1119
  maxOutputTokens: 8192,
1015
1120
  inputTokenCost: 0.1,
@@ -1044,7 +1149,7 @@ export const textModels = [
1044
1149
  {
1045
1150
  type: "text",
1046
1151
  modelName: "gemini-2.0-flash-lite",
1047
- description: "Cost effective offering to support high throughput. DEPRECATED: Will be shut down on March 31, 2026. Use gemini-2.5-flash-lite instead.",
1152
+ description: "Cost effective offering to support high throughput. RETIRED: Shut down June 1, 2026. Use gemini-2.5-flash-lite instead.",
1048
1153
  maxInputTokens: 1048576,
1049
1154
  maxOutputTokens: 8192,
1050
1155
  inputTokenCost: 0.075,
@@ -1434,6 +1539,13 @@ export const imageModels = [
1434
1539
  description: "Fast image generation with Gemini 3.1 Flash (GA). Supports resolutions from 512px to 4096px. ~$0.045/image at 512px, $0.067 at 1K, $0.101 at 2K, $0.151 at 4K.",
1435
1540
  costPerImage: 0.067,
1436
1541
  },
1542
+ {
1543
+ type: "image",
1544
+ modelName: "gemini-3.1-flash-lite-image",
1545
+ provider: "google",
1546
+ description: "aka Nano Banana 2 Lite (GA 2026-06-30). Fastest, most cost-effective Gemini image model (~4s generation). ~$0.034/image at 1K. Recommended replacement for gemini-2.5-flash-image.",
1547
+ costPerImage: 0.034,
1548
+ },
1437
1549
  ];
1438
1550
  export const embeddingsModels = [
1439
1551
  {
@@ -0,0 +1,19 @@
1
+ /**
2
+ * Normalized reason a generation turn ended, unified across providers. The
3
+ * untouched provider value is available separately as `PromptResult.rawStopReason`.
4
+ */
5
+ export type StopReason =
6
+ /** Natural completion (OpenAI `stop`, Anthropic `end_turn`, Google `STOP`, Ollama `stop`). */
7
+ "stop"
8
+ /** Hit the max-tokens limit (`length` / `max_tokens` / `MAX_TOKENS`). */
9
+ | "length"
10
+ /** Model wants to call a tool (`tool_calls` / `tool_use`). */
11
+ | "tool_use"
12
+ /** Blocked by a safety/policy filter or refusal (`content_filter` / `refusal` / `SAFETY`). */
13
+ | "content_filter"
14
+ /** Hit a caller-supplied stop sequence (Anthropic `stop_sequence`). */
15
+ | "stop_sequence"
16
+ /** Provider paused a long-running turn (Anthropic `pause_turn`). */
17
+ | "pause"
18
+ /** Anything unmapped or unknown. */
19
+ | "other";
@@ -0,0 +1 @@
1
+ export {};
package/dist/types.d.ts CHANGED
@@ -10,8 +10,10 @@ import type { ModelDataBlob } from "./modelData.js";
10
10
  import { Result } from "./types/result.js";
11
11
  import { TokenUsage } from "./types/tokenUsage.js";
12
12
  import { CostEstimate } from "./types/costEstimate.js";
13
+ import { StopReason } from "./types/stopReason.js";
13
14
  export * from "./types/costEstimate.js";
14
15
  export * from "./types/tokenUsage.js";
16
+ export * from "./types/stopReason.js";
15
17
  export type SmolConfig = {
16
18
  /** The model to use. */
17
19
  model: ModelName;
@@ -150,8 +152,12 @@ export type PromptResult = {
150
152
  cost?: CostEstimate;
151
153
  model?: ModelName;
152
154
  hostedToolResults?: HostedToolResult[];
155
+ /** Normalized reason the turn ended, unified across providers. */
156
+ stopReason?: StopReason;
157
+ /** The untouched provider finish/stop-reason value (e.g. `end_turn`, `MAX_TOKENS`). */
158
+ rawStopReason?: string;
153
159
  };
154
- export declare function promptResult({ output, toolCalls, thinkingBlocks, usage, cost, model, hostedToolResults, }: Partial<PromptResult>): PromptResult;
160
+ export declare function promptResult({ output, toolCalls, thinkingBlocks, usage, cost, model, hostedToolResults, stopReason, rawStopReason, }: Partial<PromptResult>): PromptResult;
155
161
  export type StreamChunk = {
156
162
  type: "text";
157
163
  text: string;
@@ -162,6 +168,14 @@ export type StreamChunk = {
162
168
  } | {
163
169
  type: "tool_call";
164
170
  toolCall: ToolCall;
171
+ }
172
+ /** A provider-run web search (server-side tool). Emitted the moment a
173
+ * search block completes, so consumers can surface the query live. One
174
+ * chunk per search; a turn may emit several. The same queries also appear
175
+ * in the final `done` result's `hostedToolResults`. */
176
+ | {
177
+ type: "web_search";
178
+ query: string;
165
179
  } | {
166
180
  type: "done";
167
181
  result: PromptResult;
package/dist/types.js CHANGED
@@ -3,7 +3,8 @@ export * from "./classes/message/contentParts.js";
3
3
  import z from "zod";
4
4
  export * from "./types/costEstimate.js";
5
5
  export * from "./types/tokenUsage.js";
6
- export function promptResult({ output, toolCalls, thinkingBlocks, usage, cost, model, hostedToolResults, }) {
6
+ export * from "./types/stopReason.js";
7
+ export function promptResult({ output, toolCalls, thinkingBlocks, usage, cost, model, hostedToolResults, stopReason, rawStopReason, }) {
7
8
  return {
8
9
  output: output || null,
9
10
  toolCalls: toolCalls || [],
@@ -12,6 +13,8 @@ export function promptResult({ output, toolCalls, thinkingBlocks, usage, cost, m
12
13
  cost,
13
14
  model,
14
15
  hostedToolResults,
16
+ stopReason,
17
+ rawStopReason,
15
18
  };
16
19
  }
17
20
  export const ThinkingBlockSchema = z.object({
@@ -0,0 +1,38 @@
1
+ /**
2
+ * JSON Schema sanitization for structured output.
3
+ *
4
+ * Zod converts `z.any()` / `z.unknown()` to an unconstrained schema — a bare
5
+ * `{}` (nested), `{"$schema":…}` (top-level), or the boolean `true`. That is
6
+ * valid JSON Schema ("accept any value"), but every provider rejects an
7
+ * unconstrained node inside a structured-output or strict-tool schema, since the
8
+ * whole point of structured output is that it is structured. These helpers map
9
+ * such nodes to `{"type":"string"}` (the safe universal container) and let
10
+ * callers detect the whole-schema-is-`any` case so they can drop structured
11
+ * output entirely and return free text.
12
+ */
13
+ /**
14
+ * True if `node` accepts any value: the boolean `true`, or a plain object whose
15
+ * every own key is a pure annotation (`{}`, `{$schema:…}`,
16
+ * `{$schema:…, description:…}`). `false` and any object with a structural or
17
+ * validation keyword are constrained.
18
+ */
19
+ export declare function isUnconstrainedSchema(node: unknown): boolean;
20
+ /**
21
+ * Convert a Zod `responseFormat` schema to a sanitized JSON Schema for a
22
+ * provider's structured-output request: any nested unconstrained node becomes
23
+ * `{"type":"string"}`. (The whole-schema-is-`any` case is handled upstream in
24
+ * `BaseClient.normalizeResponseFormat`, which drops structured output entirely.)
25
+ */
26
+ export declare function responseFormatToJsonSchema(schema: {
27
+ toJSONSchema: () => unknown;
28
+ }): object;
29
+ /**
30
+ * Returns a new JSON Schema with every unconstrained node replaced by
31
+ * `{"type":"string"}` (annotations preserved), recursing through all subschema
32
+ * positions. Idempotent — a node that already has `type` is left untouched.
33
+ *
34
+ * `additionalProperties` and `items` may be a boolean (`true`/`false`), which is
35
+ * a legitimate provider-accepted flag, not an any-typed value slot; booleans
36
+ * there are left as-is and only an object subschema is sanitized.
37
+ */
38
+ export declare function sanitizeJsonSchema(node: unknown): unknown;