@ai-sdk/baseten 0.0.30 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,43 @@
1
1
  # @ai-sdk/baseten
2
2
 
3
+ ## 0.1.1
4
+
5
+ ### Patch Changes
6
+
7
+ - Updated dependencies [0e51b7b]
8
+ - @ai-sdk/provider-utils@3.0.32
9
+ - @ai-sdk/openai-compatible@1.0.48
10
+
11
+ ## 0.1.0
12
+
13
+ ### Minor Changes
14
+
15
+ - f74f8b4: Make the native performance client opt-in for embeddings.
16
+
17
+ `@basetenlabs/performance-client` is no longer a dependency. It is a NAPI addon — 16 platform binary packages, ~5-16 MB installed — that could not load in edge runtimes and whose platform binaries bundlers could not resolve, and it was imported at module top level, so every consumer paid for it even though only embeddings use it.
18
+
19
+ Embeddings now go over plain HTTP to the deployment's OpenAI-compatible endpoint, which is what Baseten Embeddings Inference serves with no additional settings. To keep the native client's client-side batching and request hedging, install it yourself and pass the constructor:
20
+
21
+ ```ts
22
+ import { createBaseten } from "@ai-sdk/baseten";
23
+ import { PerformanceClient } from "@basetenlabs/performance-client";
24
+
25
+ const baseten = createBaseten({
26
+ modelURL,
27
+ performanceClient: PerformanceClient,
28
+ });
29
+ ```
30
+
31
+ The default path now supports things the previous implementation silently dropped: `abortSignal`, per-call `headers`, the `dimensions` and `user` provider options, and the provider's `fetch` option — `createBaseten({ fetch })` previously had no effect on embeddings. Response headers and warnings are now real rather than empty.
32
+
33
+ `usage.tokens` now comes from `prompt_tokens` rather than `total_tokens`, matching the `EmbeddingModelV2` contract ("we only have input tokens for embeddings") and the other providers. The values are normally identical for embeddings.
34
+
35
+ One behaviour change to be aware of: each request now sends at most 128 values. `embedMany` splits and parallelises above that, so only a direct `doEmbed` call with more than 128 values is affected — it throws `TooManyEmbeddingValuesForCallError`. The opt-in native path is unchanged and still receives everything in one call.
36
+
37
+ Separately, report token usage for streamed chat completions. The provider never set `includeUsage`, so `stream_options.include_usage` was omitted from requests and OpenAI-compatible servers returned no usage at all for streams — `streamText` reported `inputTokens`/`outputTokens`/`totalTokens` as `undefined` while `generateText` on the same model reported them correctly. This affected both the Model APIs and dedicated-deployment paths.
38
+
39
+ Also parse the error envelope dedicated deployments return. Baseten sends two different shapes: the Model APIs send `error` as a bare string (`{"error":"please check the model you provided"}`), while a dedicated deployment passes through its server's OpenAI-shaped `{"error":{"message":…,"code":…,"param":…,"type":…}}` object. The schema only accepted the string, so the object failed to parse and the message degraded to the HTTP reason phrase — a real `The model \`x\` does not exist.`surfaced as`Not Found`, or as the empty string over HTTP/2, which has no reason phrase. The schema now accepts both. This affects embeddings especially, since they require a `modelURL` and so always talk to a dedicated deployment.
40
+
3
41
  ## 0.0.30
4
42
 
5
43
  ### Patch Changes
package/dist/index.d.mts CHANGED
@@ -6,9 +6,30 @@ type BasetenChatModelId = 'deepseek-ai/DeepSeek-R1-0528' | 'deepseek-ai/DeepSeek
6
6
 
7
7
  type BasetenEmbeddingModelId = string & {};
8
8
 
9
+ /**
10
+ * The part of `@basetenlabs/performance-client` we use, declared structurally to
11
+ * keep that native addon out of our dependency and type graph.
12
+ */
13
+ type BasetenPerformanceClient = {
14
+ embed(input: string[], model: string): Promise<{
15
+ data: {
16
+ embedding: number[];
17
+ }[];
18
+ usage?: {
19
+ total_tokens?: number;
20
+ };
21
+ }>;
22
+ };
23
+ type BasetenPerformanceClientConstructor = new (baseUrl: string, apiKey?: string) => BasetenPerformanceClient;
9
24
  type BasetenErrorData = z.infer<typeof basetenErrorSchema>;
10
25
  declare const basetenErrorSchema: z.ZodObject<{
11
- error: z.ZodString;
26
+ error: z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
27
+ message: z.ZodString;
28
+ object: z.ZodOptional<z.ZodNullable<z.ZodString>>;
29
+ type: z.ZodOptional<z.ZodNullable<z.ZodString>>;
30
+ param: z.ZodOptional<z.ZodNullable<z.ZodAny>>;
31
+ code: z.ZodOptional<z.ZodNullable<z.ZodUnion<readonly [z.ZodString, z.ZodNumber]>>>;
32
+ }, z.core.$strip>]>;
12
33
  }, z.core.$strip>;
13
34
  interface BasetenProviderSettings {
14
35
  /**
@@ -34,6 +55,23 @@ interface BasetenProviderSettings {
34
55
  * or to provide a custom fetch implementation for e.g. testing.
35
56
  */
36
57
  fetch?: FetchFunction;
58
+ /**
59
+ * Opt in to Baseten's native performance client for embeddings, for
60
+ * client-side batching and request hedging. Pass the `PerformanceClient`
61
+ * constructor from `@basetenlabs/performance-client`, which you install
62
+ * yourself:
63
+ *
64
+ * ```ts
65
+ * import { PerformanceClient } from '@basetenlabs/performance-client';
66
+ *
67
+ * const baseten = createBaseten({ modelURL, performanceClient: PerformanceClient });
68
+ * ```
69
+ *
70
+ * When omitted, embeddings go over plain HTTP to Baseten's OpenAI-compatible
71
+ * endpoint — the default, since this NAPI addon cannot load in edge runtimes
72
+ * and bundlers cannot resolve its platform binaries.
73
+ */
74
+ performanceClient?: BasetenPerformanceClientConstructor;
37
75
  }
38
76
  interface BasetenProvider extends ProviderV2 {
39
77
  /**
@@ -58,4 +96,4 @@ declare const baseten: BasetenProvider;
58
96
 
59
97
  declare const VERSION: string;
60
98
 
61
- export { type BasetenChatModelId, type BasetenErrorData, type BasetenProvider, type BasetenProviderSettings, VERSION, baseten, createBaseten };
99
+ export { type BasetenChatModelId, type BasetenErrorData, type BasetenPerformanceClient, type BasetenPerformanceClientConstructor, type BasetenProvider, type BasetenProviderSettings, VERSION, baseten, createBaseten };
package/dist/index.d.ts CHANGED
@@ -6,9 +6,30 @@ type BasetenChatModelId = 'deepseek-ai/DeepSeek-R1-0528' | 'deepseek-ai/DeepSeek
6
6
 
7
7
  type BasetenEmbeddingModelId = string & {};
8
8
 
9
+ /**
10
+ * The part of `@basetenlabs/performance-client` we use, declared structurally to
11
+ * keep that native addon out of our dependency and type graph.
12
+ */
13
+ type BasetenPerformanceClient = {
14
+ embed(input: string[], model: string): Promise<{
15
+ data: {
16
+ embedding: number[];
17
+ }[];
18
+ usage?: {
19
+ total_tokens?: number;
20
+ };
21
+ }>;
22
+ };
23
+ type BasetenPerformanceClientConstructor = new (baseUrl: string, apiKey?: string) => BasetenPerformanceClient;
9
24
  type BasetenErrorData = z.infer<typeof basetenErrorSchema>;
10
25
  declare const basetenErrorSchema: z.ZodObject<{
11
- error: z.ZodString;
26
+ error: z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
27
+ message: z.ZodString;
28
+ object: z.ZodOptional<z.ZodNullable<z.ZodString>>;
29
+ type: z.ZodOptional<z.ZodNullable<z.ZodString>>;
30
+ param: z.ZodOptional<z.ZodNullable<z.ZodAny>>;
31
+ code: z.ZodOptional<z.ZodNullable<z.ZodUnion<readonly [z.ZodString, z.ZodNumber]>>>;
32
+ }, z.core.$strip>]>;
12
33
  }, z.core.$strip>;
13
34
  interface BasetenProviderSettings {
14
35
  /**
@@ -34,6 +55,23 @@ interface BasetenProviderSettings {
34
55
  * or to provide a custom fetch implementation for e.g. testing.
35
56
  */
36
57
  fetch?: FetchFunction;
58
+ /**
59
+ * Opt in to Baseten's native performance client for embeddings, for
60
+ * client-side batching and request hedging. Pass the `PerformanceClient`
61
+ * constructor from `@basetenlabs/performance-client`, which you install
62
+ * yourself:
63
+ *
64
+ * ```ts
65
+ * import { PerformanceClient } from '@basetenlabs/performance-client';
66
+ *
67
+ * const baseten = createBaseten({ modelURL, performanceClient: PerformanceClient });
68
+ * ```
69
+ *
70
+ * When omitted, embeddings go over plain HTTP to Baseten's OpenAI-compatible
71
+ * endpoint — the default, since this NAPI addon cannot load in edge runtimes
72
+ * and bundlers cannot resolve its platform binaries.
73
+ */
74
+ performanceClient?: BasetenPerformanceClientConstructor;
37
75
  }
38
76
  interface BasetenProvider extends ProviderV2 {
39
77
  /**
@@ -58,4 +96,4 @@ declare const baseten: BasetenProvider;
58
96
 
59
97
  declare const VERSION: string;
60
98
 
61
- export { type BasetenChatModelId, type BasetenErrorData, type BasetenProvider, type BasetenProviderSettings, VERSION, baseten, createBaseten };
99
+ export { type BasetenChatModelId, type BasetenErrorData, type BasetenPerformanceClient, type BasetenPerformanceClientConstructor, type BasetenProvider, type BasetenProviderSettings, VERSION, baseten, createBaseten };
package/dist/index.js CHANGED
@@ -31,18 +31,27 @@ var import_openai_compatible = require("@ai-sdk/openai-compatible");
31
31
  var import_provider = require("@ai-sdk/provider");
32
32
  var import_provider_utils = require("@ai-sdk/provider-utils");
33
33
  var import_v4 = require("zod/v4");
34
- var import_performance_client = require("@basetenlabs/performance-client");
35
34
 
36
35
  // src/version.ts
37
- var VERSION = true ? "0.0.30" : "0.0.0-test";
36
+ var VERSION = true ? "0.1.1" : "0.0.0-test";
38
37
 
39
38
  // src/baseten-provider.ts
39
+ var MAX_EMBEDDINGS_PER_CALL = 128;
40
40
  var basetenErrorSchema = import_v4.z.object({
41
- error: import_v4.z.string()
41
+ error: import_v4.z.union([
42
+ import_v4.z.string(),
43
+ import_v4.z.object({
44
+ message: import_v4.z.string(),
45
+ object: import_v4.z.string().nullish(),
46
+ type: import_v4.z.string().nullish(),
47
+ param: import_v4.z.any().nullish(),
48
+ code: import_v4.z.union([import_v4.z.string(), import_v4.z.number()]).nullish()
49
+ })
50
+ ])
42
51
  });
43
52
  var basetenErrorStructure = {
44
53
  errorSchema: basetenErrorSchema,
45
- errorToMessage: (data) => data.error
54
+ errorToMessage: (data) => typeof data.error === "string" ? data.error : data.error.message
46
55
  };
47
56
  var defaultBaseURL = "https://inference.baseten.co/v1";
48
57
  function createBaseten(options = {}) {
@@ -77,7 +86,9 @@ function createBaseten(options = {}) {
77
86
  if (isOpenAICompatible) {
78
87
  return new import_openai_compatible.OpenAICompatibleChatLanguageModel(modelId != null ? modelId : "placeholder", {
79
88
  ...getCommonModelConfig("chat", customURL),
80
- errorStructure: basetenErrorStructure
89
+ errorStructure: basetenErrorStructure,
90
+ // Or stream_options.include_usage is omitted and streams report no usage.
91
+ includeUsage: true
81
92
  });
82
93
  } else if (customURL.includes("/predict")) {
83
94
  throw new Error(
@@ -87,7 +98,8 @@ function createBaseten(options = {}) {
87
98
  }
88
99
  return new import_openai_compatible.OpenAICompatibleChatLanguageModel(modelId != null ? modelId : "chat", {
89
100
  ...getCommonModelConfig("chat"),
90
- errorStructure: basetenErrorStructure
101
+ errorStructure: basetenErrorStructure,
102
+ includeUsage: true
91
103
  });
92
104
  };
93
105
  const createTextEmbeddingModel = (modelId) => {
@@ -97,46 +109,49 @@ function createBaseten(options = {}) {
97
109
  "No model URL provided for embeddings. Please set modelURL option for embeddings."
98
110
  );
99
111
  }
100
- const isOpenAICompatible = customURL.includes("/sync");
101
- if (isOpenAICompatible) {
102
- const model = new import_openai_compatible.OpenAICompatibleEmbeddingModel(
103
- modelId != null ? modelId : "embeddings",
104
- {
105
- ...getCommonModelConfig("embedding", customURL),
106
- errorStructure: basetenErrorStructure
107
- }
108
- );
109
- const performanceClientURL = customURL.replace("/sync/v1", "/sync");
110
- const performanceClient = new import_performance_client.PerformanceClient(
111
- performanceClientURL,
112
- (0, import_provider_utils.loadApiKey)({
113
- apiKey: options.apiKey,
114
- environmentVariableName: "BASETEN_API_KEY",
115
- description: "Baseten API key"
116
- })
117
- );
118
- model.doEmbed = async (params) => {
119
- if (!params.values || !Array.isArray(params.values)) {
120
- throw new Error("params.values must be an array of strings");
121
- }
122
- const response = await performanceClient.embed(
123
- params.values,
124
- modelId != null ? modelId : "embeddings"
125
- // model_id is for Model APIs, we don't use it here for dedicated
126
- );
127
- const embeddings = response.data.map((item) => item.embedding);
128
- return {
129
- embeddings,
130
- usage: response.usage ? { tokens: response.usage.total_tokens } : void 0,
131
- response: { headers: {}, body: response }
132
- };
133
- };
134
- return model;
135
- } else {
112
+ if (!customURL.includes("/sync")) {
136
113
  throw new Error(
137
114
  "Not supported. You must use a /sync or /sync/v1 endpoint for embeddings."
138
115
  );
139
116
  }
117
+ const model = new import_openai_compatible.OpenAICompatibleEmbeddingModel(modelId != null ? modelId : "embeddings", {
118
+ ...getCommonModelConfig("embedding", customURL),
119
+ errorStructure: basetenErrorStructure,
120
+ // Over HTTP, cap each request and let `embedMany` split and parallelise.
121
+ // The native client does its own batching, so let it take everything at
122
+ // once — `embedMany` treats Infinity as "one call".
123
+ maxEmbeddingsPerCall: options.performanceClient ? Number.POSITIVE_INFINITY : MAX_EMBEDDINGS_PER_CALL
124
+ });
125
+ if (!options.performanceClient) {
126
+ return model;
127
+ }
128
+ const performanceClient = new options.performanceClient(
129
+ customURL.replace("/sync/v1", "/sync"),
130
+ (0, import_provider_utils.loadApiKey)({
131
+ apiKey: options.apiKey,
132
+ environmentVariableName: "BASETEN_API_KEY",
133
+ description: "Baseten API key"
134
+ })
135
+ );
136
+ model.doEmbed = async (params) => {
137
+ var _a2;
138
+ if (!params.values || !Array.isArray(params.values)) {
139
+ throw new Error("params.values must be an array of strings");
140
+ }
141
+ const response = await performanceClient.embed(
142
+ params.values,
143
+ // model_id is for Model APIs; dedicated deployments ignore it.
144
+ modelId != null ? modelId : "embeddings"
145
+ );
146
+ return {
147
+ embeddings: response.data.map((item) => item.embedding),
148
+ // The native client types its response as `any`; only report usage when
149
+ // a token count is actually present rather than `{ tokens: undefined }`.
150
+ usage: typeof ((_a2 = response.usage) == null ? void 0 : _a2.total_tokens) === "number" ? { tokens: response.usage.total_tokens } : void 0,
151
+ response: { headers: {}, body: response }
152
+ };
153
+ };
154
+ return model;
140
155
  };
141
156
  const provider = (modelId) => createChatModel(modelId);
142
157
  provider.chatModel = createChatModel;
package/dist/index.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"sources":["../src/index.ts","../src/baseten-provider.ts","../src/version.ts"],"sourcesContent":["export type { BasetenChatModelId } from './baseten-chat-options';\nexport { baseten, createBaseten } from './baseten-provider';\nexport type {\n BasetenProvider,\n BasetenProviderSettings,\n BasetenErrorData,\n} from './baseten-provider';\nexport { VERSION } from './version';\n","import {\n type ProviderErrorStructure,\n OpenAICompatibleChatLanguageModel,\n OpenAICompatibleEmbeddingModel,\n} from '@ai-sdk/openai-compatible';\nimport {\n type EmbeddingModelV2,\n type LanguageModelV2,\n type ProviderV2,\n NoSuchModelError,\n} from '@ai-sdk/provider';\nimport {\n type FetchFunction,\n loadApiKey,\n withoutTrailingSlash,\n withUserAgentSuffix,\n} from '@ai-sdk/provider-utils';\nimport { z } from 'zod/v4';\nimport type { BasetenChatModelId } from './baseten-chat-options';\nimport type { BasetenEmbeddingModelId } from './baseten-embedding-options';\nimport { PerformanceClient } from '@basetenlabs/performance-client';\nimport { VERSION } from './version';\n\nexport type BasetenErrorData = z.infer<typeof basetenErrorSchema>;\n\nconst basetenErrorSchema = z.object({\n error: z.string(),\n});\n\nconst basetenErrorStructure: ProviderErrorStructure<BasetenErrorData> = {\n errorSchema: basetenErrorSchema,\n errorToMessage: data => data.error,\n};\n\nexport interface BasetenProviderSettings {\n /**\n * Baseten API key. Default value is taken from the `BASETEN_API_KEY`\n * environment variable.\n */\n apiKey?: string;\n\n /**\n * Base URL for the Model APIs. Default: 'https://inference.baseten.co/v1'\n */\n baseURL?: string;\n\n /**\n * Model URL for custom models (chat or embeddings).\n * If not supplied, the default Model APIs will be used.\n */\n modelURL?: string;\n /**\n * Custom headers to include in the requests.\n */\n headers?: Record<string, string>;\n\n /**\n * Custom fetch implementation. You can use it as a middleware to intercept requests,\n * or to provide a custom fetch implementation for e.g. testing.\n */\n fetch?: FetchFunction;\n}\n\nexport interface BasetenProvider extends ProviderV2 {\n /**\nCreates a chat model for text generation.\n*/\n (modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a chat model for text generation.\n*/\n chatModel(modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a language model for text generation. Alias for chatModel.\n*/\n languageModel(modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a text embedding model for text generation.\n*/\n textEmbeddingModel(\n modelId?: BasetenEmbeddingModelId,\n ): EmbeddingModelV2<string>;\n}\n\n// by default, we use the Model APIs\nconst defaultBaseURL = 'https://inference.baseten.co/v1';\n\nexport function createBaseten(\n options: BasetenProviderSettings = {},\n): BasetenProvider {\n const baseURL = withoutTrailingSlash(options.baseURL ?? defaultBaseURL);\n const getHeaders = () =>\n withUserAgentSuffix(\n {\n Authorization: `Bearer ${loadApiKey({\n apiKey: options.apiKey,\n environmentVariableName: 'BASETEN_API_KEY',\n description: 'Baseten API key',\n })}`,\n ...options.headers,\n },\n `ai-sdk/baseten/${VERSION}`,\n );\n\n interface CommonModelConfig {\n provider: string;\n url: ({ path }: { path: string }) => string;\n headers: () => Record<string, string>;\n fetch?: FetchFunction;\n }\n\n const getCommonModelConfig = (\n modelType: string,\n customURL?: string,\n ): CommonModelConfig => ({\n provider: `baseten.${modelType}`,\n url: ({ path }) => {\n // For embeddings with /sync URLs (but not /sync/v1), we need to add /v1\n if (\n modelType === 'embedding' &&\n customURL?.includes('/sync') &&\n !customURL?.includes('/sync/v1')\n ) {\n return `${customURL}/v1${path}`;\n }\n return `${customURL || baseURL}${path}`;\n },\n headers: getHeaders,\n fetch: options.fetch,\n });\n\n const createChatModel = (modelId?: BasetenChatModelId) => {\n // Use modelURL if provided, otherwise use default Model APIs\n const customURL = options.modelURL;\n\n if (customURL) {\n // Check if this is a /sync/v1 endpoint (OpenAI-compatible) or /predict endpoint (custom)\n const isOpenAICompatible = customURL.includes('/sync/v1');\n\n if (isOpenAICompatible) {\n // For /sync/v1 endpoints, use standard OpenAI-compatible format\n return new OpenAICompatibleChatLanguageModel(modelId ?? 'placeholder', {\n ...getCommonModelConfig('chat', customURL),\n errorStructure: basetenErrorStructure,\n });\n } else if (customURL.includes('/predict')) {\n throw new Error(\n 'Not supported. You must use a /sync/v1 endpoint for chat models.',\n );\n }\n }\n\n // Use default OpenAI-compatible format for Model APIs\n return new OpenAICompatibleChatLanguageModel(modelId ?? 'chat', {\n ...getCommonModelConfig('chat'),\n errorStructure: basetenErrorStructure,\n });\n };\n\n const createTextEmbeddingModel = (modelId?: BasetenEmbeddingModelId) => {\n // Use modelURL if provided\n const customURL = options.modelURL;\n if (!customURL) {\n throw new Error(\n 'No model URL provided for embeddings. Please set modelURL option for embeddings.',\n );\n }\n\n // Check if this is a /sync or /sync/v1 endpoint (OpenAI-compatible)\n // We support both /sync and /sync/v1, stripping /v1 before passing to Performance Client, as Performance Client adds /v1 itself\n const isOpenAICompatible = customURL.includes('/sync');\n\n if (isOpenAICompatible) {\n // Create the model using OpenAICompatibleEmbeddingModel and override doEmbed\n const model = new OpenAICompatibleEmbeddingModel(\n modelId ?? 'embeddings',\n {\n ...getCommonModelConfig('embedding', customURL),\n errorStructure: basetenErrorStructure,\n },\n );\n\n // Strip /v1 from URL if present before passing to Performance Client to avoid double /v1\n const performanceClientURL = customURL.replace('/sync/v1', '/sync');\n\n // Initialize the B10 Performance Client once for reuse\n const performanceClient = new PerformanceClient(\n performanceClientURL,\n loadApiKey({\n apiKey: options.apiKey,\n environmentVariableName: 'BASETEN_API_KEY',\n description: 'Baseten API key',\n }),\n );\n\n // Override the doEmbed method to use the pre-created Performance Client\n model.doEmbed = async params => {\n if (!params.values || !Array.isArray(params.values)) {\n throw new Error('params.values must be an array of strings');\n }\n\n // Performance Client handles batching internally, so we don't need to limit in 128 here\n const response = await performanceClient.embed(\n params.values,\n modelId ?? 'embeddings', // model_id is for Model APIs, we don't use it here for dedicated\n );\n // Transform the response to match the expected format\n const embeddings = response.data.map((item: any) => item.embedding);\n\n return {\n embeddings: embeddings,\n usage: response.usage\n ? { tokens: response.usage.total_tokens }\n : undefined,\n response: { headers: {}, body: response },\n };\n };\n\n return model;\n } else {\n throw new Error(\n 'Not supported. You must use a /sync or /sync/v1 endpoint for embeddings.',\n );\n }\n };\n\n const provider = (modelId?: BasetenChatModelId) => createChatModel(modelId);\n provider.chatModel = createChatModel;\n provider.languageModel = createChatModel;\n provider.imageModel = (modelId: string) => {\n throw new NoSuchModelError({ modelId, modelType: 'imageModel' });\n };\n provider.textEmbeddingModel = createTextEmbeddingModel;\n return provider;\n}\n\nexport const baseten = createBaseten();\n","// Version string of this package injected at build time.\ndeclare const __PACKAGE_VERSION__: string | undefined;\nexport const VERSION: string =\n typeof __PACKAGE_VERSION__ !== 'undefined'\n ? __PACKAGE_VERSION__\n : '0.0.0-test';\n"],"mappings":";;;;;;;;;;;;;;;;;;;;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;;;ACAA,+BAIO;AACP,sBAKO;AACP,4BAKO;AACP,gBAAkB;AAGlB,gCAAkC;;;AClB3B,IAAM,UACX,OACI,WACA;;;ADoBN,IAAM,qBAAqB,YAAE,OAAO;AAAA,EAClC,OAAO,YAAE,OAAO;AAClB,CAAC;AAED,IAAM,wBAAkE;AAAA,EACtE,aAAa;AAAA,EACb,gBAAgB,UAAQ,KAAK;AAC/B;AAwDA,IAAM,iBAAiB;AAEhB,SAAS,cACd,UAAmC,CAAC,GACnB;AA5FnB;AA6FE,QAAM,cAAU,6CAAqB,aAAQ,YAAR,YAAmB,cAAc;AACtE,QAAM,aAAa,UACjB;AAAA,IACE;AAAA,MACE,eAAe,cAAU,kCAAW;AAAA,QAClC,QAAQ,QAAQ;AAAA,QAChB,yBAAyB;AAAA,QACzB,aAAa;AAAA,MACf,CAAC,CAAC;AAAA,MACF,GAAG,QAAQ;AAAA,IACb;AAAA,IACA,kBAAkB,OAAO;AAAA,EAC3B;AASF,QAAM,uBAAuB,CAC3B,WACA,eACuB;AAAA,IACvB,UAAU,WAAW,SAAS;AAAA,IAC9B,KAAK,CAAC,EAAE,KAAK,MAAM;AAEjB,UACE,cAAc,gBACd,uCAAW,SAAS,aACpB,EAAC,uCAAW,SAAS,cACrB;AACA,eAAO,GAAG,SAAS,MAAM,IAAI;AAAA,MAC/B;AACA,aAAO,GAAG,aAAa,OAAO,GAAG,IAAI;AAAA,IACvC;AAAA,IACA,SAAS;AAAA,IACT,OAAO,QAAQ;AAAA,EACjB;AAEA,QAAM,kBAAkB,CAAC,YAAiC;AAExD,UAAM,YAAY,QAAQ;AAE1B,QAAI,WAAW;AAEb,YAAM,qBAAqB,UAAU,SAAS,UAAU;AAExD,UAAI,oBAAoB;AAEtB,eAAO,IAAI,2DAAkC,4BAAW,eAAe;AAAA,UACrE,GAAG,qBAAqB,QAAQ,SAAS;AAAA,UACzC,gBAAgB;AAAA,QAClB,CAAC;AAAA,MACH,WAAW,UAAU,SAAS,UAAU,GAAG;AACzC,cAAM,IAAI;AAAA,UACR;AAAA,QACF;AAAA,MACF;AAAA,IACF;AAGA,WAAO,IAAI,2DAAkC,4BAAW,QAAQ;AAAA,MAC9D,GAAG,qBAAqB,MAAM;AAAA,MAC9B,gBAAgB;AAAA,IAClB,CAAC;AAAA,EACH;AAEA,QAAM,2BAA2B,CAAC,YAAsC;AAEtE,UAAM,YAAY,QAAQ;AAC1B,QAAI,CAAC,WAAW;AACd,YAAM,IAAI;AAAA,QACR;AAAA,MACF;AAAA,IACF;AAIA,UAAM,qBAAqB,UAAU,SAAS,OAAO;AAErD,QAAI,oBAAoB;AAEtB,YAAM,QAAQ,IAAI;AAAA,QAChB,4BAAW;AAAA,QACX;AAAA,UACE,GAAG,qBAAqB,aAAa,SAAS;AAAA,UAC9C,gBAAgB;AAAA,QAClB;AAAA,MACF;AAGA,YAAM,uBAAuB,UAAU,QAAQ,YAAY,OAAO;AAGlE,YAAM,oBAAoB,IAAI;AAAA,QAC5B;AAAA,YACA,kCAAW;AAAA,UACT,QAAQ,QAAQ;AAAA,UAChB,yBAAyB;AAAA,UACzB,aAAa;AAAA,QACf,CAAC;AAAA,MACH;AAGA,YAAM,UAAU,OAAM,WAAU;AAC9B,YAAI,CAAC,OAAO,UAAU,CAAC,MAAM,QAAQ,OAAO,MAAM,GAAG;AACnD,gBAAM,IAAI,MAAM,2CAA2C;AAAA,QAC7D;AAGA,cAAM,WAAW,MAAM,kBAAkB;AAAA,UACvC,OAAO;AAAA,UACP,4BAAW;AAAA;AAAA,QACb;AAEA,cAAM,aAAa,SAAS,KAAK,IAAI,CAAC,SAAc,KAAK,SAAS;AAElE,eAAO;AAAA,UACL;AAAA,UACA,OAAO,SAAS,QACZ,EAAE,QAAQ,SAAS,MAAM,aAAa,IACtC;AAAA,UACJ,UAAU,EAAE,SAAS,CAAC,GAAG,MAAM,SAAS;AAAA,QAC1C;AAAA,MACF;AAEA,aAAO;AAAA,IACT,OAAO;AACL,YAAM,IAAI;AAAA,QACR;AAAA,MACF;AAAA,IACF;AAAA,EACF;AAEA,QAAM,WAAW,CAAC,YAAiC,gBAAgB,OAAO;AAC1E,WAAS,YAAY;AACrB,WAAS,gBAAgB;AACzB,WAAS,aAAa,CAAC,YAAoB;AACzC,UAAM,IAAI,iCAAiB,EAAE,SAAS,WAAW,aAAa,CAAC;AAAA,EACjE;AACA,WAAS,qBAAqB;AAC9B,SAAO;AACT;AAEO,IAAM,UAAU,cAAc;","names":[]}
1
+ {"version":3,"sources":["../src/index.ts","../src/baseten-provider.ts","../src/version.ts"],"sourcesContent":["export type { BasetenChatModelId } from './baseten-chat-options';\nexport { baseten, createBaseten } from './baseten-provider';\nexport type {\n BasetenProvider,\n BasetenProviderSettings,\n BasetenErrorData,\n BasetenPerformanceClient,\n BasetenPerformanceClientConstructor,\n} from './baseten-provider';\nexport { VERSION } from './version';\n","import {\n type ProviderErrorStructure,\n OpenAICompatibleChatLanguageModel,\n OpenAICompatibleEmbeddingModel,\n} from '@ai-sdk/openai-compatible';\nimport {\n type EmbeddingModelV2,\n type LanguageModelV2,\n type ProviderV2,\n NoSuchModelError,\n} from '@ai-sdk/provider';\nimport {\n type FetchFunction,\n loadApiKey,\n withoutTrailingSlash,\n withUserAgentSuffix,\n} from '@ai-sdk/provider-utils';\nimport { z } from 'zod/v4';\nimport type { BasetenChatModelId } from './baseten-chat-options';\nimport type { BasetenEmbeddingModelId } from './baseten-embedding-options';\nimport { VERSION } from './version';\n\n/**\n * Baseten's per-request embedding input limit: larger batches are rejected with\n * `413 batch size N > maximum allowed batch size 128`. It is also the native\n * performance client's own default `batchSize`. `embedMany` splits larger\n * inputs into chunks of this size and runs them in parallel.\n */\nconst MAX_EMBEDDINGS_PER_CALL = 128;\n\n/**\n * The part of `@basetenlabs/performance-client` we use, declared structurally to\n * keep that native addon out of our dependency and type graph.\n */\nexport type BasetenPerformanceClient = {\n embed(\n input: string[],\n model: string,\n ): Promise<{\n data: { embedding: number[] }[];\n usage?: { total_tokens?: number };\n }>;\n};\n\nexport type BasetenPerformanceClientConstructor = new (\n baseUrl: string,\n apiKey?: string,\n) => BasetenPerformanceClient;\n\nexport type BasetenErrorData = z.infer<typeof basetenErrorSchema>;\n\n// Baseten returns two different envelopes. The Model APIs send a bare string\n// (`{\"error\":\"please check the model you provided\"}`), while dedicated\n// deployments pass through their server's OpenAI-shaped object. Parsing only\n// the string form left dedicated-deployment errors falling back to the HTTP\n// reason phrase — \"Not Found\", or nothing at all over HTTP/2.\nconst basetenErrorSchema = z.object({\n error: z.union([\n z.string(),\n z.object({\n message: z.string(),\n object: z.string().nullish(),\n type: z.string().nullish(),\n param: z.any().nullish(),\n code: z.union([z.string(), z.number()]).nullish(),\n }),\n ]),\n});\n\nconst basetenErrorStructure: ProviderErrorStructure<BasetenErrorData> = {\n errorSchema: basetenErrorSchema,\n errorToMessage: data =>\n typeof data.error === 'string' ? data.error : data.error.message,\n};\n\nexport interface BasetenProviderSettings {\n /**\n * Baseten API key. Default value is taken from the `BASETEN_API_KEY`\n * environment variable.\n */\n apiKey?: string;\n\n /**\n * Base URL for the Model APIs. Default: 'https://inference.baseten.co/v1'\n */\n baseURL?: string;\n\n /**\n * Model URL for custom models (chat or embeddings).\n * If not supplied, the default Model APIs will be used.\n */\n modelURL?: string;\n /**\n * Custom headers to include in the requests.\n */\n headers?: Record<string, string>;\n\n /**\n * Custom fetch implementation. You can use it as a middleware to intercept requests,\n * or to provide a custom fetch implementation for e.g. testing.\n */\n fetch?: FetchFunction;\n\n /**\n * Opt in to Baseten's native performance client for embeddings, for\n * client-side batching and request hedging. Pass the `PerformanceClient`\n * constructor from `@basetenlabs/performance-client`, which you install\n * yourself:\n *\n * ```ts\n * import { PerformanceClient } from '@basetenlabs/performance-client';\n *\n * const baseten = createBaseten({ modelURL, performanceClient: PerformanceClient });\n * ```\n *\n * When omitted, embeddings go over plain HTTP to Baseten's OpenAI-compatible\n * endpoint — the default, since this NAPI addon cannot load in edge runtimes\n * and bundlers cannot resolve its platform binaries.\n */\n performanceClient?: BasetenPerformanceClientConstructor;\n}\n\nexport interface BasetenProvider extends ProviderV2 {\n /**\nCreates a chat model for text generation.\n*/\n (modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a chat model for text generation.\n*/\n chatModel(modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a language model for text generation. Alias for chatModel.\n*/\n languageModel(modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a text embedding model for text generation.\n*/\n textEmbeddingModel(\n modelId?: BasetenEmbeddingModelId,\n ): EmbeddingModelV2<string>;\n}\n\n// by default, we use the Model APIs\nconst defaultBaseURL = 'https://inference.baseten.co/v1';\n\nexport function createBaseten(\n options: BasetenProviderSettings = {},\n): BasetenProvider {\n const baseURL = withoutTrailingSlash(options.baseURL ?? defaultBaseURL);\n const getHeaders = () =>\n withUserAgentSuffix(\n {\n Authorization: `Bearer ${loadApiKey({\n apiKey: options.apiKey,\n environmentVariableName: 'BASETEN_API_KEY',\n description: 'Baseten API key',\n })}`,\n ...options.headers,\n },\n `ai-sdk/baseten/${VERSION}`,\n );\n\n interface CommonModelConfig {\n provider: string;\n url: ({ path }: { path: string }) => string;\n headers: () => Record<string, string>;\n fetch?: FetchFunction;\n }\n\n const getCommonModelConfig = (\n modelType: string,\n customURL?: string,\n ): CommonModelConfig => ({\n provider: `baseten.${modelType}`,\n url: ({ path }) => {\n // For embeddings with /sync URLs (but not /sync/v1), we need to add /v1\n if (\n modelType === 'embedding' &&\n customURL?.includes('/sync') &&\n !customURL?.includes('/sync/v1')\n ) {\n return `${customURL}/v1${path}`;\n }\n return `${customURL || baseURL}${path}`;\n },\n headers: getHeaders,\n fetch: options.fetch,\n });\n\n const createChatModel = (modelId?: BasetenChatModelId) => {\n // Use modelURL if provided, otherwise use default Model APIs\n const customURL = options.modelURL;\n\n if (customURL) {\n // Check if this is a /sync/v1 endpoint (OpenAI-compatible) or /predict endpoint (custom)\n const isOpenAICompatible = customURL.includes('/sync/v1');\n\n if (isOpenAICompatible) {\n // For /sync/v1 endpoints, use standard OpenAI-compatible format\n return new OpenAICompatibleChatLanguageModel(modelId ?? 'placeholder', {\n ...getCommonModelConfig('chat', customURL),\n errorStructure: basetenErrorStructure,\n // Or stream_options.include_usage is omitted and streams report no usage.\n includeUsage: true,\n });\n } else if (customURL.includes('/predict')) {\n throw new Error(\n 'Not supported. You must use a /sync/v1 endpoint for chat models.',\n );\n }\n }\n\n // Use default OpenAI-compatible format for Model APIs\n return new OpenAICompatibleChatLanguageModel(modelId ?? 'chat', {\n ...getCommonModelConfig('chat'),\n errorStructure: basetenErrorStructure,\n includeUsage: true,\n });\n };\n\n const createTextEmbeddingModel = (modelId?: BasetenEmbeddingModelId) => {\n // Use modelURL if provided\n const customURL = options.modelURL;\n if (!customURL) {\n throw new Error(\n 'No model URL provided for embeddings. Please set modelURL option for embeddings.',\n );\n }\n\n if (!customURL.includes('/sync')) {\n throw new Error(\n 'Not supported. You must use a /sync or /sync/v1 endpoint for embeddings.',\n );\n }\n\n // BEI embedding deployments are OpenAI-compatible with no extra settings, so\n // plain HTTP is the default and needs no override.\n const model = new OpenAICompatibleEmbeddingModel(modelId ?? 'embeddings', {\n ...getCommonModelConfig('embedding', customURL),\n errorStructure: basetenErrorStructure,\n // Over HTTP, cap each request and let `embedMany` split and parallelise.\n // The native client does its own batching, so let it take everything at\n // once — `embedMany` treats Infinity as \"one call\".\n maxEmbeddingsPerCall: options.performanceClient\n ? Number.POSITIVE_INFINITY\n : MAX_EMBEDDINGS_PER_CALL,\n });\n\n if (!options.performanceClient) {\n return model;\n }\n\n // Opted in to the native client. It appends /v1 itself, so hand it the bare\n // /sync form.\n const performanceClient = new options.performanceClient(\n customURL.replace('/sync/v1', '/sync'),\n loadApiKey({\n apiKey: options.apiKey,\n environmentVariableName: 'BASETEN_API_KEY',\n description: 'Baseten API key',\n }),\n );\n\n model.doEmbed = async params => {\n if (!params.values || !Array.isArray(params.values)) {\n throw new Error('params.values must be an array of strings');\n }\n\n const response = await performanceClient.embed(\n params.values,\n // model_id is for Model APIs; dedicated deployments ignore it.\n modelId ?? 'embeddings',\n );\n\n return {\n embeddings: response.data.map(item => item.embedding),\n // The native client types its response as `any`; only report usage when\n // a token count is actually present rather than `{ tokens: undefined }`.\n usage:\n typeof response.usage?.total_tokens === 'number'\n ? { tokens: response.usage.total_tokens }\n : undefined,\n response: { headers: {}, body: response },\n };\n };\n\n return model;\n };\n\n const provider = (modelId?: BasetenChatModelId) => createChatModel(modelId);\n provider.chatModel = createChatModel;\n provider.languageModel = createChatModel;\n provider.imageModel = (modelId: string) => {\n throw new NoSuchModelError({ modelId, modelType: 'imageModel' });\n };\n provider.textEmbeddingModel = createTextEmbeddingModel;\n return provider;\n}\n\nexport const baseten = createBaseten();\n","// Version string of this package injected at build time.\ndeclare const __PACKAGE_VERSION__: string | undefined;\nexport const VERSION: string =\n typeof __PACKAGE_VERSION__ !== 'undefined'\n ? __PACKAGE_VERSION__\n : '0.0.0-test';\n"],"mappings":";;;;;;;;;;;;;;;;;;;;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;;;ACAA,+BAIO;AACP,sBAKO;AACP,4BAKO;AACP,gBAAkB;;;ACfX,IAAM,UACX,OACI,UACA;;;ADuBN,IAAM,0BAA0B;AA4BhC,IAAM,qBAAqB,YAAE,OAAO;AAAA,EAClC,OAAO,YAAE,MAAM;AAAA,IACb,YAAE,OAAO;AAAA,IACT,YAAE,OAAO;AAAA,MACP,SAAS,YAAE,OAAO;AAAA,MAClB,QAAQ,YAAE,OAAO,EAAE,QAAQ;AAAA,MAC3B,MAAM,YAAE,OAAO,EAAE,QAAQ;AAAA,MACzB,OAAO,YAAE,IAAI,EAAE,QAAQ;AAAA,MACvB,MAAM,YAAE,MAAM,CAAC,YAAE,OAAO,GAAG,YAAE,OAAO,CAAC,CAAC,EAAE,QAAQ;AAAA,IAClD,CAAC;AAAA,EACH,CAAC;AACH,CAAC;AAED,IAAM,wBAAkE;AAAA,EACtE,aAAa;AAAA,EACb,gBAAgB,UACd,OAAO,KAAK,UAAU,WAAW,KAAK,QAAQ,KAAK,MAAM;AAC7D;AA0EA,IAAM,iBAAiB;AAEhB,SAAS,cACd,UAAmC,CAAC,GACnB;AAvJnB;AAwJE,QAAM,cAAU,6CAAqB,aAAQ,YAAR,YAAmB,cAAc;AACtE,QAAM,aAAa,UACjB;AAAA,IACE;AAAA,MACE,eAAe,cAAU,kCAAW;AAAA,QAClC,QAAQ,QAAQ;AAAA,QAChB,yBAAyB;AAAA,QACzB,aAAa;AAAA,MACf,CAAC,CAAC;AAAA,MACF,GAAG,QAAQ;AAAA,IACb;AAAA,IACA,kBAAkB,OAAO;AAAA,EAC3B;AASF,QAAM,uBAAuB,CAC3B,WACA,eACuB;AAAA,IACvB,UAAU,WAAW,SAAS;AAAA,IAC9B,KAAK,CAAC,EAAE,KAAK,MAAM;AAEjB,UACE,cAAc,gBACd,uCAAW,SAAS,aACpB,EAAC,uCAAW,SAAS,cACrB;AACA,eAAO,GAAG,SAAS,MAAM,IAAI;AAAA,MAC/B;AACA,aAAO,GAAG,aAAa,OAAO,GAAG,IAAI;AAAA,IACvC;AAAA,IACA,SAAS;AAAA,IACT,OAAO,QAAQ;AAAA,EACjB;AAEA,QAAM,kBAAkB,CAAC,YAAiC;AAExD,UAAM,YAAY,QAAQ;AAE1B,QAAI,WAAW;AAEb,YAAM,qBAAqB,UAAU,SAAS,UAAU;AAExD,UAAI,oBAAoB;AAEtB,eAAO,IAAI,2DAAkC,4BAAW,eAAe;AAAA,UACrE,GAAG,qBAAqB,QAAQ,SAAS;AAAA,UACzC,gBAAgB;AAAA;AAAA,UAEhB,cAAc;AAAA,QAChB,CAAC;AAAA,MACH,WAAW,UAAU,SAAS,UAAU,GAAG;AACzC,cAAM,IAAI;AAAA,UACR;AAAA,QACF;AAAA,MACF;AAAA,IACF;AAGA,WAAO,IAAI,2DAAkC,4BAAW,QAAQ;AAAA,MAC9D,GAAG,qBAAqB,MAAM;AAAA,MAC9B,gBAAgB;AAAA,MAChB,cAAc;AAAA,IAChB,CAAC;AAAA,EACH;AAEA,QAAM,2BAA2B,CAAC,YAAsC;AAEtE,UAAM,YAAY,QAAQ;AAC1B,QAAI,CAAC,WAAW;AACd,YAAM,IAAI;AAAA,QACR;AAAA,MACF;AAAA,IACF;AAEA,QAAI,CAAC,UAAU,SAAS,OAAO,GAAG;AAChC,YAAM,IAAI;AAAA,QACR;AAAA,MACF;AAAA,IACF;AAIA,UAAM,QAAQ,IAAI,wDAA+B,4BAAW,cAAc;AAAA,MACxE,GAAG,qBAAqB,aAAa,SAAS;AAAA,MAC9C,gBAAgB;AAAA;AAAA;AAAA;AAAA,MAIhB,sBAAsB,QAAQ,oBAC1B,OAAO,oBACP;AAAA,IACN,CAAC;AAED,QAAI,CAAC,QAAQ,mBAAmB;AAC9B,aAAO;AAAA,IACT;AAIA,UAAM,oBAAoB,IAAI,QAAQ;AAAA,MACpC,UAAU,QAAQ,YAAY,OAAO;AAAA,UACrC,kCAAW;AAAA,QACT,QAAQ,QAAQ;AAAA,QAChB,yBAAyB;AAAA,QACzB,aAAa;AAAA,MACf,CAAC;AAAA,IACH;AAEA,UAAM,UAAU,OAAM,WAAU;AA3QpC,UAAAA;AA4QM,UAAI,CAAC,OAAO,UAAU,CAAC,MAAM,QAAQ,OAAO,MAAM,GAAG;AACnD,cAAM,IAAI,MAAM,2CAA2C;AAAA,MAC7D;AAEA,YAAM,WAAW,MAAM,kBAAkB;AAAA,QACvC,OAAO;AAAA;AAAA,QAEP,4BAAW;AAAA,MACb;AAEA,aAAO;AAAA,QACL,YAAY,SAAS,KAAK,IAAI,UAAQ,KAAK,SAAS;AAAA;AAAA;AAAA,QAGpD,OACE,SAAOA,MAAA,SAAS,UAAT,gBAAAA,IAAgB,kBAAiB,WACpC,EAAE,QAAQ,SAAS,MAAM,aAAa,IACtC;AAAA,QACN,UAAU,EAAE,SAAS,CAAC,GAAG,MAAM,SAAS;AAAA,MAC1C;AAAA,IACF;AAEA,WAAO;AAAA,EACT;AAEA,QAAM,WAAW,CAAC,YAAiC,gBAAgB,OAAO;AAC1E,WAAS,YAAY;AACrB,WAAS,gBAAgB;AACzB,WAAS,aAAa,CAAC,YAAoB;AACzC,UAAM,IAAI,iCAAiB,EAAE,SAAS,WAAW,aAAa,CAAC;AAAA,EACjE;AACA,WAAS,qBAAqB;AAC9B,SAAO;AACT;AAEO,IAAM,UAAU,cAAc;","names":["_a"]}
package/dist/index.mjs CHANGED
@@ -12,18 +12,27 @@ import {
12
12
  withUserAgentSuffix
13
13
  } from "@ai-sdk/provider-utils";
14
14
  import { z } from "zod/v4";
15
- import { PerformanceClient } from "@basetenlabs/performance-client";
16
15
 
17
16
  // src/version.ts
18
- var VERSION = true ? "0.0.30" : "0.0.0-test";
17
+ var VERSION = true ? "0.1.1" : "0.0.0-test";
19
18
 
20
19
  // src/baseten-provider.ts
20
+ var MAX_EMBEDDINGS_PER_CALL = 128;
21
21
  var basetenErrorSchema = z.object({
22
- error: z.string()
22
+ error: z.union([
23
+ z.string(),
24
+ z.object({
25
+ message: z.string(),
26
+ object: z.string().nullish(),
27
+ type: z.string().nullish(),
28
+ param: z.any().nullish(),
29
+ code: z.union([z.string(), z.number()]).nullish()
30
+ })
31
+ ])
23
32
  });
24
33
  var basetenErrorStructure = {
25
34
  errorSchema: basetenErrorSchema,
26
- errorToMessage: (data) => data.error
35
+ errorToMessage: (data) => typeof data.error === "string" ? data.error : data.error.message
27
36
  };
28
37
  var defaultBaseURL = "https://inference.baseten.co/v1";
29
38
  function createBaseten(options = {}) {
@@ -58,7 +67,9 @@ function createBaseten(options = {}) {
58
67
  if (isOpenAICompatible) {
59
68
  return new OpenAICompatibleChatLanguageModel(modelId != null ? modelId : "placeholder", {
60
69
  ...getCommonModelConfig("chat", customURL),
61
- errorStructure: basetenErrorStructure
70
+ errorStructure: basetenErrorStructure,
71
+ // Or stream_options.include_usage is omitted and streams report no usage.
72
+ includeUsage: true
62
73
  });
63
74
  } else if (customURL.includes("/predict")) {
64
75
  throw new Error(
@@ -68,7 +79,8 @@ function createBaseten(options = {}) {
68
79
  }
69
80
  return new OpenAICompatibleChatLanguageModel(modelId != null ? modelId : "chat", {
70
81
  ...getCommonModelConfig("chat"),
71
- errorStructure: basetenErrorStructure
82
+ errorStructure: basetenErrorStructure,
83
+ includeUsage: true
72
84
  });
73
85
  };
74
86
  const createTextEmbeddingModel = (modelId) => {
@@ -78,46 +90,49 @@ function createBaseten(options = {}) {
78
90
  "No model URL provided for embeddings. Please set modelURL option for embeddings."
79
91
  );
80
92
  }
81
- const isOpenAICompatible = customURL.includes("/sync");
82
- if (isOpenAICompatible) {
83
- const model = new OpenAICompatibleEmbeddingModel(
84
- modelId != null ? modelId : "embeddings",
85
- {
86
- ...getCommonModelConfig("embedding", customURL),
87
- errorStructure: basetenErrorStructure
88
- }
89
- );
90
- const performanceClientURL = customURL.replace("/sync/v1", "/sync");
91
- const performanceClient = new PerformanceClient(
92
- performanceClientURL,
93
- loadApiKey({
94
- apiKey: options.apiKey,
95
- environmentVariableName: "BASETEN_API_KEY",
96
- description: "Baseten API key"
97
- })
98
- );
99
- model.doEmbed = async (params) => {
100
- if (!params.values || !Array.isArray(params.values)) {
101
- throw new Error("params.values must be an array of strings");
102
- }
103
- const response = await performanceClient.embed(
104
- params.values,
105
- modelId != null ? modelId : "embeddings"
106
- // model_id is for Model APIs, we don't use it here for dedicated
107
- );
108
- const embeddings = response.data.map((item) => item.embedding);
109
- return {
110
- embeddings,
111
- usage: response.usage ? { tokens: response.usage.total_tokens } : void 0,
112
- response: { headers: {}, body: response }
113
- };
114
- };
115
- return model;
116
- } else {
93
+ if (!customURL.includes("/sync")) {
117
94
  throw new Error(
118
95
  "Not supported. You must use a /sync or /sync/v1 endpoint for embeddings."
119
96
  );
120
97
  }
98
+ const model = new OpenAICompatibleEmbeddingModel(modelId != null ? modelId : "embeddings", {
99
+ ...getCommonModelConfig("embedding", customURL),
100
+ errorStructure: basetenErrorStructure,
101
+ // Over HTTP, cap each request and let `embedMany` split and parallelise.
102
+ // The native client does its own batching, so let it take everything at
103
+ // once — `embedMany` treats Infinity as "one call".
104
+ maxEmbeddingsPerCall: options.performanceClient ? Number.POSITIVE_INFINITY : MAX_EMBEDDINGS_PER_CALL
105
+ });
106
+ if (!options.performanceClient) {
107
+ return model;
108
+ }
109
+ const performanceClient = new options.performanceClient(
110
+ customURL.replace("/sync/v1", "/sync"),
111
+ loadApiKey({
112
+ apiKey: options.apiKey,
113
+ environmentVariableName: "BASETEN_API_KEY",
114
+ description: "Baseten API key"
115
+ })
116
+ );
117
+ model.doEmbed = async (params) => {
118
+ var _a2;
119
+ if (!params.values || !Array.isArray(params.values)) {
120
+ throw new Error("params.values must be an array of strings");
121
+ }
122
+ const response = await performanceClient.embed(
123
+ params.values,
124
+ // model_id is for Model APIs; dedicated deployments ignore it.
125
+ modelId != null ? modelId : "embeddings"
126
+ );
127
+ return {
128
+ embeddings: response.data.map((item) => item.embedding),
129
+ // The native client types its response as `any`; only report usage when
130
+ // a token count is actually present rather than `{ tokens: undefined }`.
131
+ usage: typeof ((_a2 = response.usage) == null ? void 0 : _a2.total_tokens) === "number" ? { tokens: response.usage.total_tokens } : void 0,
132
+ response: { headers: {}, body: response }
133
+ };
134
+ };
135
+ return model;
121
136
  };
122
137
  const provider = (modelId) => createChatModel(modelId);
123
138
  provider.chatModel = createChatModel;
@@ -1 +1 @@
1
- {"version":3,"sources":["../src/baseten-provider.ts","../src/version.ts"],"sourcesContent":["import {\n type ProviderErrorStructure,\n OpenAICompatibleChatLanguageModel,\n OpenAICompatibleEmbeddingModel,\n} from '@ai-sdk/openai-compatible';\nimport {\n type EmbeddingModelV2,\n type LanguageModelV2,\n type ProviderV2,\n NoSuchModelError,\n} from '@ai-sdk/provider';\nimport {\n type FetchFunction,\n loadApiKey,\n withoutTrailingSlash,\n withUserAgentSuffix,\n} from '@ai-sdk/provider-utils';\nimport { z } from 'zod/v4';\nimport type { BasetenChatModelId } from './baseten-chat-options';\nimport type { BasetenEmbeddingModelId } from './baseten-embedding-options';\nimport { PerformanceClient } from '@basetenlabs/performance-client';\nimport { VERSION } from './version';\n\nexport type BasetenErrorData = z.infer<typeof basetenErrorSchema>;\n\nconst basetenErrorSchema = z.object({\n error: z.string(),\n});\n\nconst basetenErrorStructure: ProviderErrorStructure<BasetenErrorData> = {\n errorSchema: basetenErrorSchema,\n errorToMessage: data => data.error,\n};\n\nexport interface BasetenProviderSettings {\n /**\n * Baseten API key. Default value is taken from the `BASETEN_API_KEY`\n * environment variable.\n */\n apiKey?: string;\n\n /**\n * Base URL for the Model APIs. Default: 'https://inference.baseten.co/v1'\n */\n baseURL?: string;\n\n /**\n * Model URL for custom models (chat or embeddings).\n * If not supplied, the default Model APIs will be used.\n */\n modelURL?: string;\n /**\n * Custom headers to include in the requests.\n */\n headers?: Record<string, string>;\n\n /**\n * Custom fetch implementation. You can use it as a middleware to intercept requests,\n * or to provide a custom fetch implementation for e.g. testing.\n */\n fetch?: FetchFunction;\n}\n\nexport interface BasetenProvider extends ProviderV2 {\n /**\nCreates a chat model for text generation.\n*/\n (modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a chat model for text generation.\n*/\n chatModel(modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a language model for text generation. Alias for chatModel.\n*/\n languageModel(modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a text embedding model for text generation.\n*/\n textEmbeddingModel(\n modelId?: BasetenEmbeddingModelId,\n ): EmbeddingModelV2<string>;\n}\n\n// by default, we use the Model APIs\nconst defaultBaseURL = 'https://inference.baseten.co/v1';\n\nexport function createBaseten(\n options: BasetenProviderSettings = {},\n): BasetenProvider {\n const baseURL = withoutTrailingSlash(options.baseURL ?? defaultBaseURL);\n const getHeaders = () =>\n withUserAgentSuffix(\n {\n Authorization: `Bearer ${loadApiKey({\n apiKey: options.apiKey,\n environmentVariableName: 'BASETEN_API_KEY',\n description: 'Baseten API key',\n })}`,\n ...options.headers,\n },\n `ai-sdk/baseten/${VERSION}`,\n );\n\n interface CommonModelConfig {\n provider: string;\n url: ({ path }: { path: string }) => string;\n headers: () => Record<string, string>;\n fetch?: FetchFunction;\n }\n\n const getCommonModelConfig = (\n modelType: string,\n customURL?: string,\n ): CommonModelConfig => ({\n provider: `baseten.${modelType}`,\n url: ({ path }) => {\n // For embeddings with /sync URLs (but not /sync/v1), we need to add /v1\n if (\n modelType === 'embedding' &&\n customURL?.includes('/sync') &&\n !customURL?.includes('/sync/v1')\n ) {\n return `${customURL}/v1${path}`;\n }\n return `${customURL || baseURL}${path}`;\n },\n headers: getHeaders,\n fetch: options.fetch,\n });\n\n const createChatModel = (modelId?: BasetenChatModelId) => {\n // Use modelURL if provided, otherwise use default Model APIs\n const customURL = options.modelURL;\n\n if (customURL) {\n // Check if this is a /sync/v1 endpoint (OpenAI-compatible) or /predict endpoint (custom)\n const isOpenAICompatible = customURL.includes('/sync/v1');\n\n if (isOpenAICompatible) {\n // For /sync/v1 endpoints, use standard OpenAI-compatible format\n return new OpenAICompatibleChatLanguageModel(modelId ?? 'placeholder', {\n ...getCommonModelConfig('chat', customURL),\n errorStructure: basetenErrorStructure,\n });\n } else if (customURL.includes('/predict')) {\n throw new Error(\n 'Not supported. You must use a /sync/v1 endpoint for chat models.',\n );\n }\n }\n\n // Use default OpenAI-compatible format for Model APIs\n return new OpenAICompatibleChatLanguageModel(modelId ?? 'chat', {\n ...getCommonModelConfig('chat'),\n errorStructure: basetenErrorStructure,\n });\n };\n\n const createTextEmbeddingModel = (modelId?: BasetenEmbeddingModelId) => {\n // Use modelURL if provided\n const customURL = options.modelURL;\n if (!customURL) {\n throw new Error(\n 'No model URL provided for embeddings. Please set modelURL option for embeddings.',\n );\n }\n\n // Check if this is a /sync or /sync/v1 endpoint (OpenAI-compatible)\n // We support both /sync and /sync/v1, stripping /v1 before passing to Performance Client, as Performance Client adds /v1 itself\n const isOpenAICompatible = customURL.includes('/sync');\n\n if (isOpenAICompatible) {\n // Create the model using OpenAICompatibleEmbeddingModel and override doEmbed\n const model = new OpenAICompatibleEmbeddingModel(\n modelId ?? 'embeddings',\n {\n ...getCommonModelConfig('embedding', customURL),\n errorStructure: basetenErrorStructure,\n },\n );\n\n // Strip /v1 from URL if present before passing to Performance Client to avoid double /v1\n const performanceClientURL = customURL.replace('/sync/v1', '/sync');\n\n // Initialize the B10 Performance Client once for reuse\n const performanceClient = new PerformanceClient(\n performanceClientURL,\n loadApiKey({\n apiKey: options.apiKey,\n environmentVariableName: 'BASETEN_API_KEY',\n description: 'Baseten API key',\n }),\n );\n\n // Override the doEmbed method to use the pre-created Performance Client\n model.doEmbed = async params => {\n if (!params.values || !Array.isArray(params.values)) {\n throw new Error('params.values must be an array of strings');\n }\n\n // Performance Client handles batching internally, so we don't need to limit in 128 here\n const response = await performanceClient.embed(\n params.values,\n modelId ?? 'embeddings', // model_id is for Model APIs, we don't use it here for dedicated\n );\n // Transform the response to match the expected format\n const embeddings = response.data.map((item: any) => item.embedding);\n\n return {\n embeddings: embeddings,\n usage: response.usage\n ? { tokens: response.usage.total_tokens }\n : undefined,\n response: { headers: {}, body: response },\n };\n };\n\n return model;\n } else {\n throw new Error(\n 'Not supported. You must use a /sync or /sync/v1 endpoint for embeddings.',\n );\n }\n };\n\n const provider = (modelId?: BasetenChatModelId) => createChatModel(modelId);\n provider.chatModel = createChatModel;\n provider.languageModel = createChatModel;\n provider.imageModel = (modelId: string) => {\n throw new NoSuchModelError({ modelId, modelType: 'imageModel' });\n };\n provider.textEmbeddingModel = createTextEmbeddingModel;\n return provider;\n}\n\nexport const baseten = createBaseten();\n","// Version string of this package injected at build time.\ndeclare const __PACKAGE_VERSION__: string | undefined;\nexport const VERSION: string =\n typeof __PACKAGE_VERSION__ !== 'undefined'\n ? __PACKAGE_VERSION__\n : '0.0.0-test';\n"],"mappings":";AAAA;AAAA,EAEE;AAAA,EACA;AAAA,OACK;AACP;AAAA,EAIE;AAAA,OACK;AACP;AAAA,EAEE;AAAA,EACA;AAAA,EACA;AAAA,OACK;AACP,SAAS,SAAS;AAGlB,SAAS,yBAAyB;;;AClB3B,IAAM,UACX,OACI,WACA;;;ADoBN,IAAM,qBAAqB,EAAE,OAAO;AAAA,EAClC,OAAO,EAAE,OAAO;AAClB,CAAC;AAED,IAAM,wBAAkE;AAAA,EACtE,aAAa;AAAA,EACb,gBAAgB,UAAQ,KAAK;AAC/B;AAwDA,IAAM,iBAAiB;AAEhB,SAAS,cACd,UAAmC,CAAC,GACnB;AA5FnB;AA6FE,QAAM,UAAU,sBAAqB,aAAQ,YAAR,YAAmB,cAAc;AACtE,QAAM,aAAa,MACjB;AAAA,IACE;AAAA,MACE,eAAe,UAAU,WAAW;AAAA,QAClC,QAAQ,QAAQ;AAAA,QAChB,yBAAyB;AAAA,QACzB,aAAa;AAAA,MACf,CAAC,CAAC;AAAA,MACF,GAAG,QAAQ;AAAA,IACb;AAAA,IACA,kBAAkB,OAAO;AAAA,EAC3B;AASF,QAAM,uBAAuB,CAC3B,WACA,eACuB;AAAA,IACvB,UAAU,WAAW,SAAS;AAAA,IAC9B,KAAK,CAAC,EAAE,KAAK,MAAM;AAEjB,UACE,cAAc,gBACd,uCAAW,SAAS,aACpB,EAAC,uCAAW,SAAS,cACrB;AACA,eAAO,GAAG,SAAS,MAAM,IAAI;AAAA,MAC/B;AACA,aAAO,GAAG,aAAa,OAAO,GAAG,IAAI;AAAA,IACvC;AAAA,IACA,SAAS;AAAA,IACT,OAAO,QAAQ;AAAA,EACjB;AAEA,QAAM,kBAAkB,CAAC,YAAiC;AAExD,UAAM,YAAY,QAAQ;AAE1B,QAAI,WAAW;AAEb,YAAM,qBAAqB,UAAU,SAAS,UAAU;AAExD,UAAI,oBAAoB;AAEtB,eAAO,IAAI,kCAAkC,4BAAW,eAAe;AAAA,UACrE,GAAG,qBAAqB,QAAQ,SAAS;AAAA,UACzC,gBAAgB;AAAA,QAClB,CAAC;AAAA,MACH,WAAW,UAAU,SAAS,UAAU,GAAG;AACzC,cAAM,IAAI;AAAA,UACR;AAAA,QACF;AAAA,MACF;AAAA,IACF;AAGA,WAAO,IAAI,kCAAkC,4BAAW,QAAQ;AAAA,MAC9D,GAAG,qBAAqB,MAAM;AAAA,MAC9B,gBAAgB;AAAA,IAClB,CAAC;AAAA,EACH;AAEA,QAAM,2BAA2B,CAAC,YAAsC;AAEtE,UAAM,YAAY,QAAQ;AAC1B,QAAI,CAAC,WAAW;AACd,YAAM,IAAI;AAAA,QACR;AAAA,MACF;AAAA,IACF;AAIA,UAAM,qBAAqB,UAAU,SAAS,OAAO;AAErD,QAAI,oBAAoB;AAEtB,YAAM,QAAQ,IAAI;AAAA,QAChB,4BAAW;AAAA,QACX;AAAA,UACE,GAAG,qBAAqB,aAAa,SAAS;AAAA,UAC9C,gBAAgB;AAAA,QAClB;AAAA,MACF;AAGA,YAAM,uBAAuB,UAAU,QAAQ,YAAY,OAAO;AAGlE,YAAM,oBAAoB,IAAI;AAAA,QAC5B;AAAA,QACA,WAAW;AAAA,UACT,QAAQ,QAAQ;AAAA,UAChB,yBAAyB;AAAA,UACzB,aAAa;AAAA,QACf,CAAC;AAAA,MACH;AAGA,YAAM,UAAU,OAAM,WAAU;AAC9B,YAAI,CAAC,OAAO,UAAU,CAAC,MAAM,QAAQ,OAAO,MAAM,GAAG;AACnD,gBAAM,IAAI,MAAM,2CAA2C;AAAA,QAC7D;AAGA,cAAM,WAAW,MAAM,kBAAkB;AAAA,UACvC,OAAO;AAAA,UACP,4BAAW;AAAA;AAAA,QACb;AAEA,cAAM,aAAa,SAAS,KAAK,IAAI,CAAC,SAAc,KAAK,SAAS;AAElE,eAAO;AAAA,UACL;AAAA,UACA,OAAO,SAAS,QACZ,EAAE,QAAQ,SAAS,MAAM,aAAa,IACtC;AAAA,UACJ,UAAU,EAAE,SAAS,CAAC,GAAG,MAAM,SAAS;AAAA,QAC1C;AAAA,MACF;AAEA,aAAO;AAAA,IACT,OAAO;AACL,YAAM,IAAI;AAAA,QACR;AAAA,MACF;AAAA,IACF;AAAA,EACF;AAEA,QAAM,WAAW,CAAC,YAAiC,gBAAgB,OAAO;AAC1E,WAAS,YAAY;AACrB,WAAS,gBAAgB;AACzB,WAAS,aAAa,CAAC,YAAoB;AACzC,UAAM,IAAI,iBAAiB,EAAE,SAAS,WAAW,aAAa,CAAC;AAAA,EACjE;AACA,WAAS,qBAAqB;AAC9B,SAAO;AACT;AAEO,IAAM,UAAU,cAAc;","names":[]}
1
+ {"version":3,"sources":["../src/baseten-provider.ts","../src/version.ts"],"sourcesContent":["import {\n type ProviderErrorStructure,\n OpenAICompatibleChatLanguageModel,\n OpenAICompatibleEmbeddingModel,\n} from '@ai-sdk/openai-compatible';\nimport {\n type EmbeddingModelV2,\n type LanguageModelV2,\n type ProviderV2,\n NoSuchModelError,\n} from '@ai-sdk/provider';\nimport {\n type FetchFunction,\n loadApiKey,\n withoutTrailingSlash,\n withUserAgentSuffix,\n} from '@ai-sdk/provider-utils';\nimport { z } from 'zod/v4';\nimport type { BasetenChatModelId } from './baseten-chat-options';\nimport type { BasetenEmbeddingModelId } from './baseten-embedding-options';\nimport { VERSION } from './version';\n\n/**\n * Baseten's per-request embedding input limit: larger batches are rejected with\n * `413 batch size N > maximum allowed batch size 128`. It is also the native\n * performance client's own default `batchSize`. `embedMany` splits larger\n * inputs into chunks of this size and runs them in parallel.\n */\nconst MAX_EMBEDDINGS_PER_CALL = 128;\n\n/**\n * The part of `@basetenlabs/performance-client` we use, declared structurally to\n * keep that native addon out of our dependency and type graph.\n */\nexport type BasetenPerformanceClient = {\n embed(\n input: string[],\n model: string,\n ): Promise<{\n data: { embedding: number[] }[];\n usage?: { total_tokens?: number };\n }>;\n};\n\nexport type BasetenPerformanceClientConstructor = new (\n baseUrl: string,\n apiKey?: string,\n) => BasetenPerformanceClient;\n\nexport type BasetenErrorData = z.infer<typeof basetenErrorSchema>;\n\n// Baseten returns two different envelopes. The Model APIs send a bare string\n// (`{\"error\":\"please check the model you provided\"}`), while dedicated\n// deployments pass through their server's OpenAI-shaped object. Parsing only\n// the string form left dedicated-deployment errors falling back to the HTTP\n// reason phrase — \"Not Found\", or nothing at all over HTTP/2.\nconst basetenErrorSchema = z.object({\n error: z.union([\n z.string(),\n z.object({\n message: z.string(),\n object: z.string().nullish(),\n type: z.string().nullish(),\n param: z.any().nullish(),\n code: z.union([z.string(), z.number()]).nullish(),\n }),\n ]),\n});\n\nconst basetenErrorStructure: ProviderErrorStructure<BasetenErrorData> = {\n errorSchema: basetenErrorSchema,\n errorToMessage: data =>\n typeof data.error === 'string' ? data.error : data.error.message,\n};\n\nexport interface BasetenProviderSettings {\n /**\n * Baseten API key. Default value is taken from the `BASETEN_API_KEY`\n * environment variable.\n */\n apiKey?: string;\n\n /**\n * Base URL for the Model APIs. Default: 'https://inference.baseten.co/v1'\n */\n baseURL?: string;\n\n /**\n * Model URL for custom models (chat or embeddings).\n * If not supplied, the default Model APIs will be used.\n */\n modelURL?: string;\n /**\n * Custom headers to include in the requests.\n */\n headers?: Record<string, string>;\n\n /**\n * Custom fetch implementation. You can use it as a middleware to intercept requests,\n * or to provide a custom fetch implementation for e.g. testing.\n */\n fetch?: FetchFunction;\n\n /**\n * Opt in to Baseten's native performance client for embeddings, for\n * client-side batching and request hedging. Pass the `PerformanceClient`\n * constructor from `@basetenlabs/performance-client`, which you install\n * yourself:\n *\n * ```ts\n * import { PerformanceClient } from '@basetenlabs/performance-client';\n *\n * const baseten = createBaseten({ modelURL, performanceClient: PerformanceClient });\n * ```\n *\n * When omitted, embeddings go over plain HTTP to Baseten's OpenAI-compatible\n * endpoint — the default, since this NAPI addon cannot load in edge runtimes\n * and bundlers cannot resolve its platform binaries.\n */\n performanceClient?: BasetenPerformanceClientConstructor;\n}\n\nexport interface BasetenProvider extends ProviderV2 {\n /**\nCreates a chat model for text generation.\n*/\n (modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a chat model for text generation.\n*/\n chatModel(modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a language model for text generation. Alias for chatModel.\n*/\n languageModel(modelId?: BasetenChatModelId): LanguageModelV2;\n\n /**\nCreates a text embedding model for text generation.\n*/\n textEmbeddingModel(\n modelId?: BasetenEmbeddingModelId,\n ): EmbeddingModelV2<string>;\n}\n\n// by default, we use the Model APIs\nconst defaultBaseURL = 'https://inference.baseten.co/v1';\n\nexport function createBaseten(\n options: BasetenProviderSettings = {},\n): BasetenProvider {\n const baseURL = withoutTrailingSlash(options.baseURL ?? defaultBaseURL);\n const getHeaders = () =>\n withUserAgentSuffix(\n {\n Authorization: `Bearer ${loadApiKey({\n apiKey: options.apiKey,\n environmentVariableName: 'BASETEN_API_KEY',\n description: 'Baseten API key',\n })}`,\n ...options.headers,\n },\n `ai-sdk/baseten/${VERSION}`,\n );\n\n interface CommonModelConfig {\n provider: string;\n url: ({ path }: { path: string }) => string;\n headers: () => Record<string, string>;\n fetch?: FetchFunction;\n }\n\n const getCommonModelConfig = (\n modelType: string,\n customURL?: string,\n ): CommonModelConfig => ({\n provider: `baseten.${modelType}`,\n url: ({ path }) => {\n // For embeddings with /sync URLs (but not /sync/v1), we need to add /v1\n if (\n modelType === 'embedding' &&\n customURL?.includes('/sync') &&\n !customURL?.includes('/sync/v1')\n ) {\n return `${customURL}/v1${path}`;\n }\n return `${customURL || baseURL}${path}`;\n },\n headers: getHeaders,\n fetch: options.fetch,\n });\n\n const createChatModel = (modelId?: BasetenChatModelId) => {\n // Use modelURL if provided, otherwise use default Model APIs\n const customURL = options.modelURL;\n\n if (customURL) {\n // Check if this is a /sync/v1 endpoint (OpenAI-compatible) or /predict endpoint (custom)\n const isOpenAICompatible = customURL.includes('/sync/v1');\n\n if (isOpenAICompatible) {\n // For /sync/v1 endpoints, use standard OpenAI-compatible format\n return new OpenAICompatibleChatLanguageModel(modelId ?? 'placeholder', {\n ...getCommonModelConfig('chat', customURL),\n errorStructure: basetenErrorStructure,\n // Or stream_options.include_usage is omitted and streams report no usage.\n includeUsage: true,\n });\n } else if (customURL.includes('/predict')) {\n throw new Error(\n 'Not supported. You must use a /sync/v1 endpoint for chat models.',\n );\n }\n }\n\n // Use default OpenAI-compatible format for Model APIs\n return new OpenAICompatibleChatLanguageModel(modelId ?? 'chat', {\n ...getCommonModelConfig('chat'),\n errorStructure: basetenErrorStructure,\n includeUsage: true,\n });\n };\n\n const createTextEmbeddingModel = (modelId?: BasetenEmbeddingModelId) => {\n // Use modelURL if provided\n const customURL = options.modelURL;\n if (!customURL) {\n throw new Error(\n 'No model URL provided for embeddings. Please set modelURL option for embeddings.',\n );\n }\n\n if (!customURL.includes('/sync')) {\n throw new Error(\n 'Not supported. You must use a /sync or /sync/v1 endpoint for embeddings.',\n );\n }\n\n // BEI embedding deployments are OpenAI-compatible with no extra settings, so\n // plain HTTP is the default and needs no override.\n const model = new OpenAICompatibleEmbeddingModel(modelId ?? 'embeddings', {\n ...getCommonModelConfig('embedding', customURL),\n errorStructure: basetenErrorStructure,\n // Over HTTP, cap each request and let `embedMany` split and parallelise.\n // The native client does its own batching, so let it take everything at\n // once — `embedMany` treats Infinity as \"one call\".\n maxEmbeddingsPerCall: options.performanceClient\n ? Number.POSITIVE_INFINITY\n : MAX_EMBEDDINGS_PER_CALL,\n });\n\n if (!options.performanceClient) {\n return model;\n }\n\n // Opted in to the native client. It appends /v1 itself, so hand it the bare\n // /sync form.\n const performanceClient = new options.performanceClient(\n customURL.replace('/sync/v1', '/sync'),\n loadApiKey({\n apiKey: options.apiKey,\n environmentVariableName: 'BASETEN_API_KEY',\n description: 'Baseten API key',\n }),\n );\n\n model.doEmbed = async params => {\n if (!params.values || !Array.isArray(params.values)) {\n throw new Error('params.values must be an array of strings');\n }\n\n const response = await performanceClient.embed(\n params.values,\n // model_id is for Model APIs; dedicated deployments ignore it.\n modelId ?? 'embeddings',\n );\n\n return {\n embeddings: response.data.map(item => item.embedding),\n // The native client types its response as `any`; only report usage when\n // a token count is actually present rather than `{ tokens: undefined }`.\n usage:\n typeof response.usage?.total_tokens === 'number'\n ? { tokens: response.usage.total_tokens }\n : undefined,\n response: { headers: {}, body: response },\n };\n };\n\n return model;\n };\n\n const provider = (modelId?: BasetenChatModelId) => createChatModel(modelId);\n provider.chatModel = createChatModel;\n provider.languageModel = createChatModel;\n provider.imageModel = (modelId: string) => {\n throw new NoSuchModelError({ modelId, modelType: 'imageModel' });\n };\n provider.textEmbeddingModel = createTextEmbeddingModel;\n return provider;\n}\n\nexport const baseten = createBaseten();\n","// Version string of this package injected at build time.\ndeclare const __PACKAGE_VERSION__: string | undefined;\nexport const VERSION: string =\n typeof __PACKAGE_VERSION__ !== 'undefined'\n ? __PACKAGE_VERSION__\n : '0.0.0-test';\n"],"mappings":";AAAA;AAAA,EAEE;AAAA,EACA;AAAA,OACK;AACP;AAAA,EAIE;AAAA,OACK;AACP;AAAA,EAEE;AAAA,EACA;AAAA,EACA;AAAA,OACK;AACP,SAAS,SAAS;;;ACfX,IAAM,UACX,OACI,UACA;;;ADuBN,IAAM,0BAA0B;AA4BhC,IAAM,qBAAqB,EAAE,OAAO;AAAA,EAClC,OAAO,EAAE,MAAM;AAAA,IACb,EAAE,OAAO;AAAA,IACT,EAAE,OAAO;AAAA,MACP,SAAS,EAAE,OAAO;AAAA,MAClB,QAAQ,EAAE,OAAO,EAAE,QAAQ;AAAA,MAC3B,MAAM,EAAE,OAAO,EAAE,QAAQ;AAAA,MACzB,OAAO,EAAE,IAAI,EAAE,QAAQ;AAAA,MACvB,MAAM,EAAE,MAAM,CAAC,EAAE,OAAO,GAAG,EAAE,OAAO,CAAC,CAAC,EAAE,QAAQ;AAAA,IAClD,CAAC;AAAA,EACH,CAAC;AACH,CAAC;AAED,IAAM,wBAAkE;AAAA,EACtE,aAAa;AAAA,EACb,gBAAgB,UACd,OAAO,KAAK,UAAU,WAAW,KAAK,QAAQ,KAAK,MAAM;AAC7D;AA0EA,IAAM,iBAAiB;AAEhB,SAAS,cACd,UAAmC,CAAC,GACnB;AAvJnB;AAwJE,QAAM,UAAU,sBAAqB,aAAQ,YAAR,YAAmB,cAAc;AACtE,QAAM,aAAa,MACjB;AAAA,IACE;AAAA,MACE,eAAe,UAAU,WAAW;AAAA,QAClC,QAAQ,QAAQ;AAAA,QAChB,yBAAyB;AAAA,QACzB,aAAa;AAAA,MACf,CAAC,CAAC;AAAA,MACF,GAAG,QAAQ;AAAA,IACb;AAAA,IACA,kBAAkB,OAAO;AAAA,EAC3B;AASF,QAAM,uBAAuB,CAC3B,WACA,eACuB;AAAA,IACvB,UAAU,WAAW,SAAS;AAAA,IAC9B,KAAK,CAAC,EAAE,KAAK,MAAM;AAEjB,UACE,cAAc,gBACd,uCAAW,SAAS,aACpB,EAAC,uCAAW,SAAS,cACrB;AACA,eAAO,GAAG,SAAS,MAAM,IAAI;AAAA,MAC/B;AACA,aAAO,GAAG,aAAa,OAAO,GAAG,IAAI;AAAA,IACvC;AAAA,IACA,SAAS;AAAA,IACT,OAAO,QAAQ;AAAA,EACjB;AAEA,QAAM,kBAAkB,CAAC,YAAiC;AAExD,UAAM,YAAY,QAAQ;AAE1B,QAAI,WAAW;AAEb,YAAM,qBAAqB,UAAU,SAAS,UAAU;AAExD,UAAI,oBAAoB;AAEtB,eAAO,IAAI,kCAAkC,4BAAW,eAAe;AAAA,UACrE,GAAG,qBAAqB,QAAQ,SAAS;AAAA,UACzC,gBAAgB;AAAA;AAAA,UAEhB,cAAc;AAAA,QAChB,CAAC;AAAA,MACH,WAAW,UAAU,SAAS,UAAU,GAAG;AACzC,cAAM,IAAI;AAAA,UACR;AAAA,QACF;AAAA,MACF;AAAA,IACF;AAGA,WAAO,IAAI,kCAAkC,4BAAW,QAAQ;AAAA,MAC9D,GAAG,qBAAqB,MAAM;AAAA,MAC9B,gBAAgB;AAAA,MAChB,cAAc;AAAA,IAChB,CAAC;AAAA,EACH;AAEA,QAAM,2BAA2B,CAAC,YAAsC;AAEtE,UAAM,YAAY,QAAQ;AAC1B,QAAI,CAAC,WAAW;AACd,YAAM,IAAI;AAAA,QACR;AAAA,MACF;AAAA,IACF;AAEA,QAAI,CAAC,UAAU,SAAS,OAAO,GAAG;AAChC,YAAM,IAAI;AAAA,QACR;AAAA,MACF;AAAA,IACF;AAIA,UAAM,QAAQ,IAAI,+BAA+B,4BAAW,cAAc;AAAA,MACxE,GAAG,qBAAqB,aAAa,SAAS;AAAA,MAC9C,gBAAgB;AAAA;AAAA;AAAA;AAAA,MAIhB,sBAAsB,QAAQ,oBAC1B,OAAO,oBACP;AAAA,IACN,CAAC;AAED,QAAI,CAAC,QAAQ,mBAAmB;AAC9B,aAAO;AAAA,IACT;AAIA,UAAM,oBAAoB,IAAI,QAAQ;AAAA,MACpC,UAAU,QAAQ,YAAY,OAAO;AAAA,MACrC,WAAW;AAAA,QACT,QAAQ,QAAQ;AAAA,QAChB,yBAAyB;AAAA,QACzB,aAAa;AAAA,MACf,CAAC;AAAA,IACH;AAEA,UAAM,UAAU,OAAM,WAAU;AA3QpC,UAAAA;AA4QM,UAAI,CAAC,OAAO,UAAU,CAAC,MAAM,QAAQ,OAAO,MAAM,GAAG;AACnD,cAAM,IAAI,MAAM,2CAA2C;AAAA,MAC7D;AAEA,YAAM,WAAW,MAAM,kBAAkB;AAAA,QACvC,OAAO;AAAA;AAAA,QAEP,4BAAW;AAAA,MACb;AAEA,aAAO;AAAA,QACL,YAAY,SAAS,KAAK,IAAI,UAAQ,KAAK,SAAS;AAAA;AAAA;AAAA,QAGpD,OACE,SAAOA,MAAA,SAAS,UAAT,gBAAAA,IAAgB,kBAAiB,WACpC,EAAE,QAAQ,SAAS,MAAM,aAAa,IACtC;AAAA,QACN,UAAU,EAAE,SAAS,CAAC,GAAG,MAAM,SAAS;AAAA,MAC1C;AAAA,IACF;AAEA,WAAO;AAAA,EACT;AAEA,QAAM,WAAW,CAAC,YAAiC,gBAAgB,OAAO;AAC1E,WAAS,YAAY;AACrB,WAAS,gBAAgB;AACzB,WAAS,aAAa,CAAC,YAAoB;AACzC,UAAM,IAAI,iBAAiB,EAAE,SAAS,WAAW,aAAa,CAAC;AAAA,EACjE;AACA,WAAS,qBAAqB;AAC9B,SAAO;AACT;AAEO,IAAM,UAAU,cAAc;","names":["_a"]}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ai-sdk/baseten",
3
- "version": "0.0.30",
3
+ "version": "0.1.1",
4
4
  "license": "Apache-2.0",
5
5
  "sideEffects": false,
6
6
  "main": "./dist/index.js",
@@ -19,16 +19,16 @@
19
19
  }
20
20
  },
21
21
  "dependencies": {
22
- "@basetenlabs/performance-client": "^0.0.10",
23
- "@ai-sdk/openai-compatible": "1.0.47",
22
+ "@ai-sdk/openai-compatible": "1.0.48",
24
23
  "@ai-sdk/provider": "2.0.3",
25
- "@ai-sdk/provider-utils": "3.0.31"
24
+ "@ai-sdk/provider-utils": "3.0.32"
26
25
  },
27
26
  "devDependencies": {
28
27
  "@types/node": "20.17.24",
29
28
  "tsup": "^8",
30
29
  "typescript": "5.8.3",
31
30
  "zod": "3.25.76",
31
+ "@ai-sdk/test-server": "0.0.4",
32
32
  "@vercel/ai-tsconfig": "0.0.0"
33
33
  },
34
34
  "peerDependencies": {