smoltalk 0.8.4 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +145 -18
- package/dist/classes/ToolCall.js +18 -10
- package/dist/classes/message/AssistantMessage.d.ts +2 -0
- package/dist/classes/message/ToolMessage.js +13 -10
- package/dist/classes/message/UserMessage.d.ts +21 -0
- package/dist/classes/message/UserMessage.js +3 -0
- package/dist/classes/message/contentParts.d.ts +71 -2
- package/dist/classes/message/contentParts.js +6 -0
- package/dist/classes/message/index.d.ts +5 -2
- package/dist/classes/message/index.js +7 -0
- package/dist/classes/message/renderers/AnthropicRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/AnthropicRenderer.js +3 -0
- package/dist/classes/message/renderers/GoogleRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/GoogleRenderer.js +3 -0
- package/dist/classes/message/renderers/JSONRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/JSONRenderer.js +4 -0
- package/dist/classes/message/renderers/OpenAIChatRenderer.d.ts +8 -1
- package/dist/classes/message/renderers/OpenAIChatRenderer.js +18 -0
- package/dist/classes/message/renderers/OpenAIResponsesRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/OpenAIResponsesRenderer.js +3 -0
- package/dist/classes/message/renderers/PartRenderer.d.ts +3 -2
- package/dist/classes/message/renderers/PartRenderer.js +3 -0
- package/dist/client.js +1 -0
- package/dist/clients/anthropic.js +1 -1
- package/dist/clients/baseClient.d.ts +13 -1
- package/dist/clients/baseClient.js +36 -7
- package/dist/clients/google.d.ts +2 -0
- package/dist/clients/google.js +125 -3
- package/dist/clients/ollama.js +1 -1
- package/dist/clients/openai.d.ts +2 -1
- package/dist/clients/openai.js +15 -3
- package/dist/clients/openaiCompat.d.ts +2 -0
- package/dist/clients/openaiCompat.js +5 -0
- package/dist/clients/openaiResponses.js +1 -1
- package/dist/clients/resolveAttachments.d.ts +8 -4
- package/dist/clients/resolveAttachments.js +101 -50
- package/dist/embed.d.ts +4 -0
- package/dist/files.d.ts +1 -1
- package/dist/files.js +1 -1
- package/dist/image/google.js +2 -2
- package/dist/image/openai.js +3 -3
- package/dist/image.d.ts +1 -1
- package/dist/index.d.ts +10 -2
- package/dist/index.js +7 -1
- package/dist/model.d.ts +15 -4
- package/dist/model.js +48 -7
- package/dist/models.d.ts +143 -19
- package/dist/models.js +137 -30
- package/dist/speech/baseSpeechClient.d.ts +31 -0
- package/dist/speech/baseSpeechClient.js +98 -0
- package/dist/speech/openai.d.ts +6 -0
- package/dist/speech/openai.js +39 -0
- package/dist/speech.d.ts +40 -0
- package/dist/speech.js +57 -0
- package/dist/transcription/baseTranscriptionClient.d.ts +31 -0
- package/dist/transcription/baseTranscriptionClient.js +107 -0
- package/dist/transcription/openai.d.ts +6 -0
- package/dist/transcription/openai.js +59 -0
- package/dist/transcription.d.ts +51 -0
- package/dist/transcription.js +58 -0
- package/dist/types/tokenUsage.d.ts +4 -0
- package/dist/types/tokenUsage.js +4 -0
- package/dist/types.d.ts +3 -0
- package/dist/util/attachments.d.ts +1 -1
- package/dist/util/audioMime.d.ts +9 -0
- package/dist/util/audioMime.js +33 -0
- package/dist/util/{imageRef.d.ts → blobRef.d.ts} +9 -9
- package/dist/util/{imageRef.js → blobRef.js} +6 -13
- package/dist/util/mime.d.ts +21 -0
- package/dist/util/mime.js +52 -0
- package/dist/util/modalities.d.ts +6 -2
- package/dist/util/modalities.js +13 -15
- package/dist/util/provider.d.ts +2 -0
- package/dist/util/provider.js +1 -1
- package/package.json +1 -1
package/dist/client.js
CHANGED
|
@@ -204,7 +204,7 @@ export class SmolAnthropic extends BaseClient {
|
|
|
204
204
|
}
|
|
205
205
|
this.client = new Anthropic({ apiKey });
|
|
206
206
|
this.logger = getLogger();
|
|
207
|
-
this.model = new Model(config.model,
|
|
207
|
+
this.model = new Model(config.model, config.provider, config.modelData);
|
|
208
208
|
}
|
|
209
209
|
getModel() {
|
|
210
210
|
return this.model.getModel();
|
|
@@ -1,5 +1,11 @@
|
|
|
1
1
|
import { StatelogClient } from "../statelogClient.js";
|
|
2
2
|
import { PromptResult, Result, SmolClient, SmolConfig, StreamChunk } from "../types.js";
|
|
3
|
+
export type ClientAttachmentCapabilities = {
|
|
4
|
+
/** Non-audio attachment modalities this client's serializers can render. */
|
|
5
|
+
inputModalities: readonly ("image" | "pdf")[];
|
|
6
|
+
/** Audio containers (by primary extension) accepted inline; empty = no audio. */
|
|
7
|
+
audioFormats: readonly string[];
|
|
8
|
+
};
|
|
3
9
|
export declare class BaseClient implements SmolClient {
|
|
4
10
|
protected config: SmolConfig;
|
|
5
11
|
protected statelogClient?: StatelogClient;
|
|
@@ -14,7 +20,13 @@ export declare class BaseClient implements SmolClient {
|
|
|
14
20
|
}): Promise<Result<PromptResult>>;
|
|
15
21
|
checkMessageLimit(promptConfig: SmolConfig): Result<PromptResult> | null;
|
|
16
22
|
/**
|
|
17
|
-
*
|
|
23
|
+
* What this client can accept as attachments. Subclasses override to declare
|
|
24
|
+
* more (or fewer). Checked against the messages, alongside the model's own
|
|
25
|
+
* declared modalities, before any serialization runs.
|
|
26
|
+
*/
|
|
27
|
+
protected attachmentCapabilities(): ClientAttachmentCapabilities;
|
|
28
|
+
/**
|
|
29
|
+
* Gate on input modalities and resolve any image/PDF/audio attachment refs
|
|
18
30
|
* (path/url/bytes → base64) before the synchronous serializers run. Returns
|
|
19
31
|
* the (possibly rewritten) config on success, or a Failure to surface. Shared
|
|
20
32
|
* by textSync and textStream so the two paths can't diverge.
|
|
@@ -6,11 +6,21 @@ import { stripCodeFence } from "../util/util.js";
|
|
|
6
6
|
import { success, failure, } from "../types.js";
|
|
7
7
|
import { validateHostedTools } from "../util/hostedTools.js";
|
|
8
8
|
import { resolveMessageAttachments, messagesHaveAttachments, DEFAULT_MAX_ATTACHMENT_BYTES } from "./resolveAttachments.js";
|
|
9
|
-
import {
|
|
9
|
+
import { neededInputModalities, MODALITIES_REQUIRING_DECLARATION, } from "../util/modalities.js";
|
|
10
|
+
import { modelSupportsInputModality } from "../models.js";
|
|
10
11
|
import { resolveProvider } from "../util/provider.js";
|
|
11
12
|
import { isUnconstrainedSchema } from "../util/jsonSchema.js";
|
|
12
13
|
import { z } from "zod";
|
|
13
14
|
const DEFAULT_NUM_RETRIES = 2;
|
|
15
|
+
function clientSupportsAttachment(capabilities, modality) {
|
|
16
|
+
if (modality === "audio") {
|
|
17
|
+
return capabilities.audioFormats.length > 0;
|
|
18
|
+
}
|
|
19
|
+
if (modality === "image" || modality === "pdf") {
|
|
20
|
+
return capabilities.inputModalities.includes(modality);
|
|
21
|
+
}
|
|
22
|
+
return false;
|
|
23
|
+
}
|
|
14
24
|
export class BaseClient {
|
|
15
25
|
config;
|
|
16
26
|
statelogClient;
|
|
@@ -57,24 +67,43 @@ export class BaseClient {
|
|
|
57
67
|
return null;
|
|
58
68
|
}
|
|
59
69
|
/**
|
|
60
|
-
*
|
|
70
|
+
* What this client can accept as attachments. Subclasses override to declare
|
|
71
|
+
* more (or fewer). Checked against the messages, alongside the model's own
|
|
72
|
+
* declared modalities, before any serialization runs.
|
|
73
|
+
*/
|
|
74
|
+
attachmentCapabilities() {
|
|
75
|
+
return { inputModalities: ["image", "pdf"], audioFormats: [] };
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Gate on input modalities and resolve any image/PDF/audio attachment refs
|
|
61
79
|
* (path/url/bytes → base64) before the synchronous serializers run. Returns
|
|
62
80
|
* the (possibly rewritten) config on success, or a Failure to surface. Shared
|
|
63
81
|
* by textSync and textStream so the two paths can't diverge.
|
|
64
82
|
*/
|
|
65
83
|
async prepareAttachments(config) {
|
|
66
|
-
const
|
|
67
|
-
if (
|
|
68
|
-
return modalityResult;
|
|
69
|
-
}
|
|
70
|
-
if (!messagesHaveAttachments(config.messages)) {
|
|
84
|
+
const needed = neededInputModalities(config.messages);
|
|
85
|
+
if (needed.length === 0 && !messagesHaveAttachments(config.messages)) {
|
|
71
86
|
return success(config);
|
|
72
87
|
}
|
|
73
88
|
const provider = resolveProvider(config.model, config.provider, config.modelData);
|
|
89
|
+
const capabilities = this.attachmentCapabilities();
|
|
90
|
+
for (const modality of needed) {
|
|
91
|
+
if (!clientSupportsAttachment(capabilities, modality)) {
|
|
92
|
+
return failure(`${modality[0].toUpperCase()}${modality.slice(1)} input is not supported by the "${provider}" provider.`);
|
|
93
|
+
}
|
|
94
|
+
const supported = modelSupportsInputModality(config.model, modality, config.modelData, provider);
|
|
95
|
+
if (supported === false) {
|
|
96
|
+
return failure(`Model ${config.model} does not support ${modality} input.`);
|
|
97
|
+
}
|
|
98
|
+
if (supported === undefined && MODALITIES_REQUIRING_DECLARATION.has(modality)) {
|
|
99
|
+
return failure(`Model ${config.model} does not support ${modality} input.`);
|
|
100
|
+
}
|
|
101
|
+
}
|
|
74
102
|
const maxBytes = config.attachments?.maxBytes ?? DEFAULT_MAX_ATTACHMENT_BYTES;
|
|
75
103
|
const resolved = await resolveMessageAttachments(config.messages, {
|
|
76
104
|
provider,
|
|
77
105
|
maxBytes,
|
|
106
|
+
audioFormats: capabilities.audioFormats,
|
|
78
107
|
});
|
|
79
108
|
if (!resolved.success) {
|
|
80
109
|
return resolved;
|
package/dist/clients/google.d.ts
CHANGED
|
@@ -3,6 +3,7 @@ import { PromptResult, Result, SmolClient, SmolConfig, StreamChunk } from "../ty
|
|
|
3
3
|
import { BaseClient } from "./baseClient.js";
|
|
4
4
|
import { ModelName } from "../models.js";
|
|
5
5
|
import { HostedToolResult } from "../types.js";
|
|
6
|
+
import type { Message } from "../classes/message/index.js";
|
|
6
7
|
export type SmolGoogleConfig = SmolConfig;
|
|
7
8
|
export declare function googleWebSearchEntries(hostedTools?: string[]): any[];
|
|
8
9
|
/**
|
|
@@ -20,6 +21,7 @@ export declare function googleWebSearchEntries(hostedTools?: string[]): any[];
|
|
|
20
21
|
* See egonSchiele/agency-lang#495.
|
|
21
22
|
*/
|
|
22
23
|
export declare function geminiSupportsToolCirculation(model: string): boolean;
|
|
24
|
+
export declare function reorderToolResultsForGemini(messages: Message[]): Message[];
|
|
23
25
|
export declare function parseGoogleHostedTools(result: any, provider: string, model: string): HostedToolResult[];
|
|
24
26
|
type GeneratedRequest = {
|
|
25
27
|
contents: Content[];
|
package/dist/clients/google.js
CHANGED
|
@@ -39,6 +39,118 @@ export function geminiSupportsToolCirculation(model) {
|
|
|
39
39
|
return true;
|
|
40
40
|
return parseInt(m[1], 10) >= 3;
|
|
41
41
|
}
|
|
42
|
+
// Reorder each round's tool results to match the order of the calls that
|
|
43
|
+
// produced them. Two documented Gemini behaviors make this necessary:
|
|
44
|
+
// - A function call carries an optional `id`, and a response is paired back to
|
|
45
|
+
// its call by echoing that id.
|
|
46
|
+
// https://ai.google.dev/gemini-api/docs/function-calling
|
|
47
|
+
// - Parts must be returned in the order received (responses in call order:
|
|
48
|
+
// FC1,FC2 -> FR1,FR2), and the thought signature rides ONLY the first
|
|
49
|
+
// function call of a parallel batch; omitting it 400s on Gemini 3. So part
|
|
50
|
+
// order is load-bearing independent of ids.
|
|
51
|
+
// https://ai.google.dev/gemini-api/docs/generate-content/thought-signatures
|
|
52
|
+
// Current Gemini 3 preview models are observed to emit no function-call ids, so
|
|
53
|
+
// on them pairing is purely positional. A caller whose results arrive in
|
|
54
|
+
// completion order (rather than call order) would otherwise feed tool A's answer
|
|
55
|
+
// to tool B and break thought-signature validation.
|
|
56
|
+
//
|
|
57
|
+
// For every assistant message that carries toolCalls, the run of ToolMessages
|
|
58
|
+
// immediately following it is reordered. The run ends at the first non-tool
|
|
59
|
+
// message: any results the caller interleaved AFTER a non-tool message are left
|
|
60
|
+
// where they are — moving messages across an interloper is repair, not reorder.
|
|
61
|
+
// 1. By id, when both the call and a response have non-empty ids (order-free,
|
|
62
|
+
// takes global priority so an id match is never stolen by a name match).
|
|
63
|
+
// 2. By name + occurrence otherwise: the k-th response named X answers the
|
|
64
|
+
// k-th call named X.
|
|
65
|
+
// NEVER drop a message: a response matching no call (or any surplus) is kept at
|
|
66
|
+
// the end of the run in its original relative order. A reorder that lost a
|
|
67
|
+
// message would turn a mispairing bug into a missing-result bug, which is worse.
|
|
68
|
+
export function reorderToolResultsForGemini(messages) {
|
|
69
|
+
const out = [];
|
|
70
|
+
let i = 0;
|
|
71
|
+
while (i < messages.length) {
|
|
72
|
+
const msg = messages[i];
|
|
73
|
+
out.push(msg);
|
|
74
|
+
const toolCalls = msg.role === "assistant" ? msg.toolCalls : undefined;
|
|
75
|
+
if (!toolCalls || toolCalls.length === 0) {
|
|
76
|
+
i += 1;
|
|
77
|
+
continue;
|
|
78
|
+
}
|
|
79
|
+
// Collect the contiguous run of tool results that answers this round.
|
|
80
|
+
let j = i + 1;
|
|
81
|
+
const run = [];
|
|
82
|
+
while (j < messages.length && messages[j].role === "tool") {
|
|
83
|
+
run.push(messages[j]);
|
|
84
|
+
j += 1;
|
|
85
|
+
}
|
|
86
|
+
if (run.length === 0) {
|
|
87
|
+
i += 1;
|
|
88
|
+
continue;
|
|
89
|
+
}
|
|
90
|
+
out.push(...orderRunToMatchCalls(toolCalls, run));
|
|
91
|
+
i = j;
|
|
92
|
+
}
|
|
93
|
+
return out;
|
|
94
|
+
}
|
|
95
|
+
function orderRunToMatchCalls(toolCalls, run) {
|
|
96
|
+
const used = new Array(run.length).fill(false);
|
|
97
|
+
// assignment[callIndex] = index into `run`, or -1 if that call has no result.
|
|
98
|
+
const assignment = new Array(toolCalls.length).fill(-1);
|
|
99
|
+
function runId(k) {
|
|
100
|
+
const m = run[k];
|
|
101
|
+
if (m.role === "tool") {
|
|
102
|
+
return m.tool_call_id;
|
|
103
|
+
}
|
|
104
|
+
return "";
|
|
105
|
+
}
|
|
106
|
+
function runName(k) {
|
|
107
|
+
const m = run[k];
|
|
108
|
+
if (m.role === "tool") {
|
|
109
|
+
return m.name;
|
|
110
|
+
}
|
|
111
|
+
return "";
|
|
112
|
+
}
|
|
113
|
+
// Pass 1: id pairing (both sides non-empty), global priority.
|
|
114
|
+
toolCalls.forEach((call, ci) => {
|
|
115
|
+
if (call.id === "")
|
|
116
|
+
return;
|
|
117
|
+
const ri = run.findIndex((_r, k) => !used[k] && runId(k) !== "" && runId(k) === call.id);
|
|
118
|
+
if (ri !== -1) {
|
|
119
|
+
assignment[ci] = ri;
|
|
120
|
+
used[ri] = true;
|
|
121
|
+
}
|
|
122
|
+
});
|
|
123
|
+
// Pass 2: name + occurrence, for calls still unmatched. findIndex takes the
|
|
124
|
+
// first unused same-name response, so the k-th call named X pairs with the
|
|
125
|
+
// k-th response named X.
|
|
126
|
+
//
|
|
127
|
+
// Garbage-in caveat: if a call and its only same-name response carry DIFFERENT
|
|
128
|
+
// non-empty ids (one side lost or mangled its id), pass 1 misses and pass 2
|
|
129
|
+
// pairs them by name — emitting functionCall.id != functionResponse.id in that
|
|
130
|
+
// slot, which an id-pairing model would see as a contradiction. The input was
|
|
131
|
+
// already inconsistent; we pair positionally rather than drop the result. Not
|
|
132
|
+
// a reorder bug.
|
|
133
|
+
toolCalls.forEach((call, ci) => {
|
|
134
|
+
if (assignment[ci] !== -1)
|
|
135
|
+
return;
|
|
136
|
+
const ri = run.findIndex((_r, k) => !used[k] && runName(k) === call.name);
|
|
137
|
+
if (ri !== -1) {
|
|
138
|
+
assignment[ci] = ri;
|
|
139
|
+
used[ri] = true;
|
|
140
|
+
}
|
|
141
|
+
});
|
|
142
|
+
const ordered = [];
|
|
143
|
+
for (const ri of assignment) {
|
|
144
|
+
if (ri !== -1)
|
|
145
|
+
ordered.push(run[ri]);
|
|
146
|
+
}
|
|
147
|
+
// Never drop: append any unmatched/surplus responses in original order.
|
|
148
|
+
for (let k = 0; k < run.length; k++) {
|
|
149
|
+
if (!used[k])
|
|
150
|
+
ordered.push(run[k]);
|
|
151
|
+
}
|
|
152
|
+
return ordered;
|
|
153
|
+
}
|
|
42
154
|
export function parseGoogleHostedTools(result, provider, model) {
|
|
43
155
|
const queries = [];
|
|
44
156
|
const sources = [];
|
|
@@ -97,7 +209,7 @@ export class SmolGoogle extends BaseClient {
|
|
|
97
209
|
}
|
|
98
210
|
this.client = new GoogleGenAI({ apiKey });
|
|
99
211
|
this.logger = getLogger();
|
|
100
|
-
this.model = new Model(config.model,
|
|
212
|
+
this.model = new Model(config.model, config.provider, config.modelData);
|
|
101
213
|
}
|
|
102
214
|
getClient() {
|
|
103
215
|
return this.client;
|
|
@@ -136,7 +248,12 @@ export class SmolGoogle extends BaseClient {
|
|
|
136
248
|
}
|
|
137
249
|
return true;
|
|
138
250
|
});
|
|
139
|
-
|
|
251
|
+
// Normalize tool-result ordering before conversion: Gemini pairs each
|
|
252
|
+
// functionResponse to its functionCall (by id on 3.5+, strictly by position
|
|
253
|
+
// on the Gemini 3 family), so results must leave in call order regardless of
|
|
254
|
+
// the order the caller supplied them. See reorderToolResultsForGemini.
|
|
255
|
+
const orderedMessages = reorderToolResultsForGemini(contentMessages);
|
|
256
|
+
const messages = orderedMessages.map((msg) => msg.toGoogleMessage());
|
|
140
257
|
const tools = (config.tools || []).map((tool) => {
|
|
141
258
|
return zodToGoogleTool(tool.name, tool.schema, {
|
|
142
259
|
description: tool.description,
|
|
@@ -341,7 +458,12 @@ export class SmolGoogle extends BaseClient {
|
|
|
341
458
|
const functionCall = part.functionCall;
|
|
342
459
|
// Gemini 3 rides the thought signature on the same part as the
|
|
343
460
|
// function call; capture it so it can be echoed back during tool use.
|
|
344
|
-
toolCalls.push(
|
|
461
|
+
toolCalls.push(
|
|
462
|
+
// Keep functionCall.id so id-based pairing can round-trip. Do NOT
|
|
463
|
+
// fall back to the name (as the streaming path does for its Map
|
|
464
|
+
// key): two parallel calls to the same tool would share a fake id
|
|
465
|
+
// and re-create the pairing bug at the id layer.
|
|
466
|
+
new ToolCall(functionCall.id || "", functionCall.name, functionCall.args, {
|
|
345
467
|
thoughtSignature: part.thoughtSignature,
|
|
346
468
|
}));
|
|
347
469
|
}
|
package/dist/clients/ollama.js
CHANGED
|
@@ -20,7 +20,7 @@ export class SmolOllama extends BaseClient {
|
|
|
20
20
|
constructor(config) {
|
|
21
21
|
super(config);
|
|
22
22
|
this.logger = getLogger();
|
|
23
|
-
this.model = new Model(config.model,
|
|
23
|
+
this.model = new Model(config.model, config.provider, config.modelData);
|
|
24
24
|
const apiKey = config.apiKey?.ollama;
|
|
25
25
|
if (apiKey) {
|
|
26
26
|
this.client = new Ollama({
|
package/dist/clients/openai.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import OpenAI from "openai";
|
|
2
2
|
import { PromptResult, Result, SmolClient, SmolConfig, StreamChunk, HostedToolResult } from "../types.js";
|
|
3
3
|
import { EgonLog } from "../util/logger.js";
|
|
4
|
-
import { BaseClient } from "./baseClient.js";
|
|
4
|
+
import { BaseClient, type ClientAttachmentCapabilities } from "./baseClient.js";
|
|
5
5
|
import { ModelName } from "../models.js";
|
|
6
6
|
import { Model } from "../model.js";
|
|
7
7
|
import { CostEstimate, TokenUsage } from "../types.js";
|
|
@@ -11,6 +11,7 @@ export declare class SmolOpenAi extends BaseClient implements SmolClient {
|
|
|
11
11
|
protected logger: EgonLog;
|
|
12
12
|
protected model: Model;
|
|
13
13
|
constructor(config: SmolOpenAiConfig);
|
|
14
|
+
protected attachmentCapabilities(): ClientAttachmentCapabilities;
|
|
14
15
|
/**
|
|
15
16
|
* Build the `new OpenAI({...})` options. Subclasses override to inject a
|
|
16
17
|
* different baseURL or to pull the key from a different config field.
|
package/dist/clients/openai.js
CHANGED
|
@@ -20,7 +20,11 @@ export class SmolOpenAi extends BaseClient {
|
|
|
20
20
|
const options = this.resolveClientOptions(config);
|
|
21
21
|
this.client = new OpenAI(options);
|
|
22
22
|
this.logger = getLogger();
|
|
23
|
-
this.model = new Model(config.model,
|
|
23
|
+
this.model = new Model(config.model, config.provider, config.modelData);
|
|
24
|
+
}
|
|
25
|
+
attachmentCapabilities() {
|
|
26
|
+
// Chat Completions input_audio accepts inline mp3/wav only.
|
|
27
|
+
return { inputModalities: ["image", "pdf"], audioFormats: ["mp3", "wav"] };
|
|
24
28
|
}
|
|
25
29
|
/**
|
|
26
30
|
* Build the `new OpenAI({...})` options. Subclasses override to inject a
|
|
@@ -72,14 +76,22 @@ export class SmolOpenAi extends BaseClient {
|
|
|
72
76
|
let cost;
|
|
73
77
|
if (usageData) {
|
|
74
78
|
const cached = usageData.prompt_tokens_details?.cached_tokens ?? 0;
|
|
79
|
+
const audioIn = usageData.prompt_tokens_details?.audio_tokens ?? 0;
|
|
80
|
+
const audioOut = usageData.completion_tokens_details?.audio_tokens ?? 0;
|
|
75
81
|
usage = {
|
|
76
|
-
inputTokens: Math.max(0, (usageData.prompt_tokens || 0) - cached),
|
|
77
|
-
outputTokens: usageData.completion_tokens || 0,
|
|
82
|
+
inputTokens: Math.max(0, (usageData.prompt_tokens || 0) - cached - audioIn),
|
|
83
|
+
outputTokens: Math.max(0, (usageData.completion_tokens || 0) - audioOut),
|
|
78
84
|
totalTokens: usageData.total_tokens,
|
|
79
85
|
};
|
|
80
86
|
if (cached > 0) {
|
|
81
87
|
usage.cachedInputTokens = cached;
|
|
82
88
|
}
|
|
89
|
+
if (audioIn > 0) {
|
|
90
|
+
usage.inputAudioTokens = audioIn;
|
|
91
|
+
}
|
|
92
|
+
if (audioOut > 0) {
|
|
93
|
+
usage.outputAudioTokens = audioOut;
|
|
94
|
+
}
|
|
83
95
|
// Prefer provider-supplied cost when available (e.g. OpenRouter
|
|
84
96
|
// usage.cost, DeepInfra usage.estimated_cost, LiteLLM header).
|
|
85
97
|
// Fall back to the smoltalk model-registry calculation.
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { SmolOpenAi } from "./openai.js";
|
|
2
|
+
import type { ClientAttachmentCapabilities } from "./baseClient.js";
|
|
2
3
|
import type { SmolConfig } from "../types.js";
|
|
3
4
|
/**
|
|
4
5
|
* Generic OpenAI-compatible client. Use when pointing smoltalk at any
|
|
@@ -14,6 +15,7 @@ import type { SmolConfig } from "../types.js";
|
|
|
14
15
|
* cost stays undefined — that's expected for arbitrary backends).
|
|
15
16
|
*/
|
|
16
17
|
export declare class SmolOpenAiCompat extends SmolOpenAi {
|
|
18
|
+
protected attachmentCapabilities(): ClientAttachmentCapabilities;
|
|
17
19
|
protected resolveClientOptions(config: SmolConfig): {
|
|
18
20
|
apiKey: string;
|
|
19
21
|
baseURL: string;
|
|
@@ -14,6 +14,11 @@ import { resolveApiKey, resolveBaseUrl } from "../util/provider.js";
|
|
|
14
14
|
* cost stays undefined — that's expected for arbitrary backends).
|
|
15
15
|
*/
|
|
16
16
|
export class SmolOpenAiCompat extends SmolOpenAi {
|
|
17
|
+
// Compat endpoints speak the Chat Completions wire format but do not get
|
|
18
|
+
// OpenAI's input_audio handling — declare no audio support.
|
|
19
|
+
attachmentCapabilities() {
|
|
20
|
+
return { inputModalities: ["image", "pdf"], audioFormats: [] };
|
|
21
|
+
}
|
|
17
22
|
resolveClientOptions(config) {
|
|
18
23
|
const apiKey = resolveApiKey("openai-compat", config);
|
|
19
24
|
const baseURL = resolveBaseUrl("openai-compat", config);
|
|
@@ -62,7 +62,7 @@ export class SmolOpenAiResponses extends BaseClient {
|
|
|
62
62
|
}
|
|
63
63
|
this.client = new OpenAI({ apiKey });
|
|
64
64
|
this.logger = getLogger();
|
|
65
|
-
this.model = new Model(config.model,
|
|
65
|
+
this.model = new Model(config.model, config.provider, config.modelData);
|
|
66
66
|
}
|
|
67
67
|
getClient() {
|
|
68
68
|
return this.client;
|
|
@@ -1,11 +1,15 @@
|
|
|
1
1
|
import { Message } from "../classes/message/index.js";
|
|
2
2
|
import { Result } from "../types.js";
|
|
3
3
|
export declare const DEFAULT_MAX_ATTACHMENT_BYTES: number;
|
|
4
|
+
type ResolveOptions = {
|
|
5
|
+
provider: string;
|
|
6
|
+
maxBytes: number;
|
|
7
|
+
/** Audio containers (by primary extension) the target client accepts inline. */
|
|
8
|
+
audioFormats: readonly string[];
|
|
9
|
+
};
|
|
4
10
|
/** Whether any user message carries an image/file attachment part. */
|
|
5
11
|
export declare function messagesHaveAttachments(messages: Message[]): boolean;
|
|
6
12
|
/** Whether `provider` accepts a remote URL directly for this part type. */
|
|
7
13
|
export declare function acceptsRemoteUrl(provider: string, partType: "image" | "file"): boolean;
|
|
8
|
-
export declare function resolveMessageAttachments(messages: Message[], options:
|
|
9
|
-
|
|
10
|
-
maxBytes: number;
|
|
11
|
-
}): Promise<Result<Message[]>>;
|
|
14
|
+
export declare function resolveMessageAttachments(messages: Message[], options: ResolveOptions): Promise<Result<Message[]>>;
|
|
15
|
+
export {};
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { UserMessage } from "../classes/message/index.js";
|
|
2
|
-
import {
|
|
2
|
+
import { normalizeBlob } from "../util/blobRef.js";
|
|
3
3
|
import { fileFamily } from "../util/attachments.js";
|
|
4
|
+
import { audioFormatForMime } from "../util/mime.js";
|
|
4
5
|
import { success, failure } from "../types.js";
|
|
5
6
|
export const DEFAULT_MAX_ATTACHMENT_BYTES = 20 * 1024 * 1024;
|
|
6
7
|
const URL_IMAGE_PROVIDERS = new Set([
|
|
@@ -24,7 +25,7 @@ export function messagesHaveAttachments(messages) {
|
|
|
24
25
|
continue;
|
|
25
26
|
}
|
|
26
27
|
for (const part of parts) {
|
|
27
|
-
if (part.type === "image" || part.type === "file") {
|
|
28
|
+
if (part.type === "image" || part.type === "file" || part.type === "audio") {
|
|
28
29
|
return true;
|
|
29
30
|
}
|
|
30
31
|
}
|
|
@@ -38,6 +39,100 @@ export function acceptsRemoteUrl(provider, partType) {
|
|
|
38
39
|
}
|
|
39
40
|
return URL_PDF_PROVIDERS.has(provider);
|
|
40
41
|
}
|
|
42
|
+
/** Load a ref to inline base64, gated to `allowed` MIME prefixes. Throws on failure. */
|
|
43
|
+
async function toBase64Source(source, allowed, maxBytes) {
|
|
44
|
+
const { data, mimeType } = await normalizeBlob(source, {
|
|
45
|
+
allowedMimePrefixes: allowed,
|
|
46
|
+
maxBytes,
|
|
47
|
+
});
|
|
48
|
+
return { kind: "base64", base64: Buffer.from(data).toString("base64"), mimeType };
|
|
49
|
+
}
|
|
50
|
+
/** Error message when a providerFile ref targets the wrong provider family, else null. */
|
|
51
|
+
function providerFileError(fileProvider, targetProvider) {
|
|
52
|
+
const family = fileFamily(targetProvider);
|
|
53
|
+
if (family === null || fileProvider !== family) {
|
|
54
|
+
return (`Attachment references a "${fileProvider}" file, but this call targets provider ` +
|
|
55
|
+
`"${targetProvider}" (file family ${family ?? "none"}).`);
|
|
56
|
+
}
|
|
57
|
+
return null;
|
|
58
|
+
}
|
|
59
|
+
// Audio has no providerFile/URL passthrough: Chat input_audio requires
|
|
60
|
+
// inline base64, so every audio source is normalized here.
|
|
61
|
+
async function resolveAudioPart(part, options) {
|
|
62
|
+
try {
|
|
63
|
+
const source = await toBase64Source(part.source, ["audio/"], options.maxBytes);
|
|
64
|
+
const audioFormat = audioFormatForMime(source.mimeType);
|
|
65
|
+
if (audioFormat === null || !options.audioFormats.includes(audioFormat.extension)) {
|
|
66
|
+
return failure(`Audio input for provider "${options.provider}" supports only ` +
|
|
67
|
+
`${options.audioFormats.join(", ")}; got "${source.mimeType}".`);
|
|
68
|
+
}
|
|
69
|
+
const resolved = { type: "audio", source };
|
|
70
|
+
if (part.filename !== undefined) {
|
|
71
|
+
resolved.filename = part.filename;
|
|
72
|
+
}
|
|
73
|
+
return success(resolved);
|
|
74
|
+
}
|
|
75
|
+
catch (err) {
|
|
76
|
+
return failure(`Failed to load audio attachment: ${err.message}`);
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
async function resolveImagePart(part, options) {
|
|
80
|
+
// Provider file references are validated and passed through (no download/cap).
|
|
81
|
+
if (part.source.kind === "providerFile") {
|
|
82
|
+
const mismatch = providerFileError(part.source.provider, options.provider);
|
|
83
|
+
if (mismatch !== null) {
|
|
84
|
+
return failure(mismatch);
|
|
85
|
+
}
|
|
86
|
+
if (options.provider === "openai") {
|
|
87
|
+
return failure("An image file reference requires the openai-responses provider (OpenAI Chat Completions has no image-by-file_id form).");
|
|
88
|
+
}
|
|
89
|
+
return success(part);
|
|
90
|
+
}
|
|
91
|
+
// Passthrough: keep a url ref when the target provider accepts a remote URL.
|
|
92
|
+
if (part.source.kind === "url" && acceptsRemoteUrl(options.provider, "image")) {
|
|
93
|
+
return success(part);
|
|
94
|
+
}
|
|
95
|
+
try {
|
|
96
|
+
const source = await toBase64Source(part.source, ["image/"], options.maxBytes);
|
|
97
|
+
return success({ type: "image", source });
|
|
98
|
+
}
|
|
99
|
+
catch (err) {
|
|
100
|
+
return failure(`Failed to load image attachment: ${err.message}`);
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
async function resolveFilePart(part, options) {
|
|
104
|
+
// Provider file references are validated and passed through (no download/cap).
|
|
105
|
+
if (part.source.kind === "providerFile") {
|
|
106
|
+
const mismatch = providerFileError(part.source.provider, options.provider);
|
|
107
|
+
if (mismatch !== null) {
|
|
108
|
+
return failure(mismatch);
|
|
109
|
+
}
|
|
110
|
+
return success(part);
|
|
111
|
+
}
|
|
112
|
+
// Passthrough: keep a url ref when the target provider accepts a remote URL.
|
|
113
|
+
if (part.source.kind === "url" && acceptsRemoteUrl(options.provider, "file")) {
|
|
114
|
+
return success(part);
|
|
115
|
+
}
|
|
116
|
+
try {
|
|
117
|
+
const source = await toBase64Source(part.source, ["application/pdf"], options.maxBytes);
|
|
118
|
+
return success({ type: "file", source, filename: part.filename });
|
|
119
|
+
}
|
|
120
|
+
catch (err) {
|
|
121
|
+
return failure(`Failed to load file attachment: ${err.message}`);
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
async function resolveUserPart(part, options) {
|
|
125
|
+
if (part.type === "text") {
|
|
126
|
+
return success(part);
|
|
127
|
+
}
|
|
128
|
+
if (part.type === "audio") {
|
|
129
|
+
return resolveAudioPart(part, options);
|
|
130
|
+
}
|
|
131
|
+
if (part.type === "image") {
|
|
132
|
+
return resolveImagePart(part, options);
|
|
133
|
+
}
|
|
134
|
+
return resolveFilePart(part, options);
|
|
135
|
+
}
|
|
41
136
|
export async function resolveMessageAttachments(messages, options) {
|
|
42
137
|
const out = [];
|
|
43
138
|
for (const msg of messages) {
|
|
@@ -52,55 +147,11 @@ export async function resolveMessageAttachments(messages, options) {
|
|
|
52
147
|
}
|
|
53
148
|
const resolvedParts = [];
|
|
54
149
|
for (const part of parts) {
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
}
|
|
59
|
-
// Provider file references are validated and passed through (no download/cap).
|
|
60
|
-
if (part.source.kind === "providerFile") {
|
|
61
|
-
const family = fileFamily(options.provider);
|
|
62
|
-
if (family === null || part.source.provider !== family) {
|
|
63
|
-
return failure(`Attachment references a "${part.source.provider}" file, but this call targets provider ` +
|
|
64
|
-
`"${options.provider}" (file family ${family ?? "none"}).`);
|
|
65
|
-
}
|
|
66
|
-
if (part.type === "image" && options.provider === "openai") {
|
|
67
|
-
return failure("An image file reference requires the openai-responses provider (OpenAI Chat Completions has no image-by-file_id form).");
|
|
68
|
-
}
|
|
69
|
-
resolvedParts.push(part);
|
|
70
|
-
continue;
|
|
71
|
-
}
|
|
72
|
-
// Passthrough: keep a url ref when the target provider accepts a remote URL.
|
|
73
|
-
if (part.source.kind === "url" && acceptsRemoteUrl(options.provider, part.type)) {
|
|
74
|
-
resolvedParts.push(part);
|
|
75
|
-
continue;
|
|
76
|
-
}
|
|
77
|
-
let allowed;
|
|
78
|
-
if (part.type === "image") {
|
|
79
|
-
allowed = ["image/"];
|
|
80
|
-
}
|
|
81
|
-
else {
|
|
82
|
-
allowed = ["application/pdf"];
|
|
83
|
-
}
|
|
84
|
-
try {
|
|
85
|
-
const { data, mimeType } = await normalizeImageRef(part.source, {
|
|
86
|
-
allowedMimePrefixes: allowed,
|
|
87
|
-
maxBytes: options.maxBytes,
|
|
88
|
-
});
|
|
89
|
-
const source = {
|
|
90
|
-
kind: "base64",
|
|
91
|
-
base64: Buffer.from(data).toString("base64"),
|
|
92
|
-
mimeType,
|
|
93
|
-
};
|
|
94
|
-
if (part.type === "image") {
|
|
95
|
-
resolvedParts.push({ type: "image", source });
|
|
96
|
-
}
|
|
97
|
-
else {
|
|
98
|
-
resolvedParts.push({ type: "file", source, filename: part.filename });
|
|
99
|
-
}
|
|
100
|
-
}
|
|
101
|
-
catch (err) {
|
|
102
|
-
return failure(`Failed to load ${part.type} attachment: ${err.message}`);
|
|
150
|
+
const resolved = await resolveUserPart(part, options);
|
|
151
|
+
if (!resolved.success) {
|
|
152
|
+
return resolved;
|
|
103
153
|
}
|
|
154
|
+
resolvedParts.push(resolved.value);
|
|
104
155
|
}
|
|
105
156
|
out.push(new UserMessage(resolvedParts, { name: msg.name, rawData: msg.rawData }));
|
|
106
157
|
}
|
package/dist/embed.d.ts
CHANGED
|
@@ -16,6 +16,8 @@ export type EmbedConfig = {
|
|
|
16
16
|
deepInfra?: string;
|
|
17
17
|
liteLlm?: string;
|
|
18
18
|
openAiCompat?: string;
|
|
19
|
+
/** Arbitrary provider names, for keys targeting a custom-registered provider. */
|
|
20
|
+
[provider: string]: string | undefined;
|
|
19
21
|
};
|
|
20
22
|
/** Custom base URLs, nested by provider. */
|
|
21
23
|
baseUrl?: {
|
|
@@ -23,6 +25,8 @@ export type EmbedConfig = {
|
|
|
23
25
|
deepInfra?: string;
|
|
24
26
|
liteLlm?: string;
|
|
25
27
|
openAiCompat?: string;
|
|
28
|
+
/** Arbitrary provider names, for URLs targeting a custom-registered provider. */
|
|
29
|
+
[provider: string]: string | undefined;
|
|
26
30
|
};
|
|
27
31
|
metadata?: Record<string, unknown>;
|
|
28
32
|
modelData?: ModelDataBlob;
|
package/dist/files.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { Result } from "./types/result.js";
|
|
2
2
|
import type { SmolConfig } from "./types.js";
|
|
3
3
|
import { ProviderFileRef } from "./classes/message/contentParts.js";
|
|
4
|
-
import { BlobRef } from "./util/
|
|
4
|
+
import { BlobRef } from "./util/blobRef.js";
|
|
5
5
|
/** Default cap on a resolved upload's size. Callers can raise it via opts.maxBytes. */
|
|
6
6
|
export declare const DEFAULT_UPLOAD_BYTES: number;
|
|
7
7
|
/**
|
package/dist/files.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { failure } from "./types/result.js";
|
|
2
|
-
import { loadBlob } from "./util/
|
|
2
|
+
import { loadBlob } from "./util/blobRef.js";
|
|
3
3
|
import { fileFamily } from "./util/attachments.js";
|
|
4
4
|
import { resolveApiKey } from "./util/provider.js";
|
|
5
5
|
import { openaiFileProvider } from "./files/openai.js";
|
package/dist/image/google.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { GoogleGenAI } from "@google/genai";
|
|
2
2
|
import { success, failure } from "../types/result.js";
|
|
3
3
|
import { getModel, isImageModel } from "../models.js";
|
|
4
|
-
import {
|
|
4
|
+
import { normalizeBlob } from "../util/blobRef.js";
|
|
5
5
|
import { COST_DECIMAL_PLACES, round } from "../util/util.js";
|
|
6
6
|
export async function googleImage(input, config, apiKey) {
|
|
7
7
|
try {
|
|
@@ -9,7 +9,7 @@ export async function googleImage(input, config, apiKey) {
|
|
|
9
9
|
const client = new GoogleGenAI({ apiKey });
|
|
10
10
|
const parts = [{ text: normalized.prompt }];
|
|
11
11
|
if (normalized.images && normalized.images.length > 0) {
|
|
12
|
-
const normalizedImages = await Promise.all(normalized.images.map((ref) =>
|
|
12
|
+
const normalizedImages = await Promise.all(normalized.images.map((ref) => normalizeBlob(ref)));
|
|
13
13
|
for (const img of normalizedImages) {
|
|
14
14
|
parts.push({
|
|
15
15
|
inlineData: {
|