@opencode/ai 2.0.15 → 2.0.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (224) hide show
  1. package/README.md +441 -99
  2. package/dist/ai-client.d.ts +8 -0
  3. package/dist/ai-client.js +12 -0
  4. package/dist/experimental/evaluation-client.d.ts +3 -3
  5. package/dist/experimental/evaluation-client.js +1 -1
  6. package/dist/experimental/evaluation.js +1 -1
  7. package/dist/generation.d.ts +39 -29
  8. package/dist/generation.js +62 -36
  9. package/dist/image-client.d.ts +59 -17
  10. package/dist/image-client.js +16 -24
  11. package/dist/image.d.ts +394 -55
  12. package/dist/image.js +48 -57
  13. package/dist/index.d.ts +15 -2
  14. package/dist/index.js +10 -0
  15. package/dist/llm.d.ts +7 -5
  16. package/dist/llm.js +10 -4
  17. package/dist/media-client.d.ts +30 -0
  18. package/dist/media-client.js +51 -0
  19. package/dist/media-model.d.ts +43 -0
  20. package/dist/media-model.js +47 -0
  21. package/dist/media.d.ts +10 -9
  22. package/dist/media.js +11 -12
  23. package/dist/promise.d.ts +473 -15
  24. package/dist/promise.js +71 -13
  25. package/dist/protocols/alibaba-chat.d.ts +12 -0
  26. package/dist/protocols/alibaba-chat.js +4 -1
  27. package/dist/protocols/alibaba-messages.d.ts +1 -1
  28. package/dist/protocols/alibaba-messages.js +6 -4
  29. package/dist/protocols/alibaba-responses.d.ts +2 -2
  30. package/dist/protocols/anthropic-messages.d.ts +34 -34
  31. package/dist/protocols/anthropic-messages.js +14 -9
  32. package/dist/protocols/assemblyai-transcription.d.ts +41 -0
  33. package/dist/protocols/assemblyai-transcription.js +136 -0
  34. package/dist/protocols/bedrock-converse.d.ts +7 -0
  35. package/dist/protocols/bedrock-converse.js +39 -19
  36. package/dist/protocols/bfl-images.d.ts +38 -0
  37. package/dist/protocols/bfl-images.js +153 -0
  38. package/dist/protocols/cartesia-speech.d.ts +143 -0
  39. package/dist/protocols/cartesia-speech.js +120 -0
  40. package/dist/protocols/deepgram-speech.d.ts +135 -0
  41. package/dist/protocols/deepgram-speech.js +91 -0
  42. package/dist/protocols/deepgram-transcription.d.ts +26 -0
  43. package/dist/protocols/deepgram-transcription.js +125 -0
  44. package/dist/protocols/elevenlabs-speech.d.ts +138 -0
  45. package/dist/protocols/elevenlabs-speech.js +111 -0
  46. package/dist/protocols/fal-images.d.ts +25 -0
  47. package/dist/protocols/fal-images.js +104 -0
  48. package/dist/protocols/fal-video.d.ts +29 -0
  49. package/dist/protocols/fal-video.js +73 -0
  50. package/dist/protocols/gemini.d.ts +22 -22
  51. package/dist/protocols/gemini.js +19 -36
  52. package/dist/protocols/google-images.d.ts +3 -3
  53. package/dist/protocols/google-images.js +12 -34
  54. package/dist/protocols/google-speech.d.ts +146 -0
  55. package/dist/protocols/google-speech.js +88 -0
  56. package/dist/protocols/google-transcription.d.ts +174 -0
  57. package/dist/protocols/google-transcription.js +126 -0
  58. package/dist/protocols/google-video.d.ts +26 -0
  59. package/dist/protocols/google-video.js +142 -0
  60. package/dist/protocols/meta-images.d.ts +3 -4
  61. package/dist/protocols/meta-images.js +13 -29
  62. package/dist/protocols/meta-messages.d.ts +5 -5
  63. package/dist/protocols/meta-responses.d.ts +4 -4
  64. package/dist/protocols/meta-responses.js +6 -4
  65. package/dist/protocols/open-responses.d.ts +10 -9
  66. package/dist/protocols/open-responses.js +4 -9
  67. package/dist/protocols/openai-chat.d.ts +84 -0
  68. package/dist/protocols/openai-chat.js +28 -17
  69. package/dist/protocols/openai-compatible-chat.d.ts +12 -0
  70. package/dist/protocols/openai-compatible-responses.d.ts +2 -2
  71. package/dist/protocols/openai-images.d.ts +130 -7
  72. package/dist/protocols/openai-images.js +143 -91
  73. package/dist/protocols/openai-responses.d.ts +20 -20
  74. package/dist/protocols/openai-responses.js +31 -31
  75. package/dist/protocols/openai-speech.d.ts +136 -0
  76. package/dist/protocols/openai-speech.js +97 -0
  77. package/dist/protocols/openai-transcription.d.ts +211 -0
  78. package/dist/protocols/openai-transcription.js +202 -0
  79. package/dist/protocols/replicate-images.d.ts +28 -0
  80. package/dist/protocols/replicate-images.js +127 -0
  81. package/dist/protocols/runway-video.d.ts +38 -0
  82. package/dist/protocols/runway-video.js +140 -0
  83. package/dist/protocols/shared.d.ts +21 -17
  84. package/dist/protocols/shared.js +27 -36
  85. package/dist/protocols/stability-images.d.ts +39 -0
  86. package/dist/protocols/stability-images.js +136 -0
  87. package/dist/protocols/utils/fal-queue.d.ts +26 -0
  88. package/dist/protocols/utils/fal-queue.js +67 -0
  89. package/dist/protocols/utils/gemini-generate-content.d.ts +65 -0
  90. package/dist/protocols/utils/gemini-generate-content.js +65 -0
  91. package/dist/protocols/utils/gemini-json-schema.d.ts +3 -0
  92. package/dist/protocols/utils/gemini-json-schema.js +76 -0
  93. package/dist/protocols/utils/media-input.d.ts +23 -1
  94. package/dist/protocols/utils/media-input.js +40 -0
  95. package/dist/protocols/utils/responses-checkpoint.js +3 -7
  96. package/dist/protocols/utils/responses-compaction.d.ts +3 -1
  97. package/dist/protocols/utils/responses-compaction.js +16 -3
  98. package/dist/protocols/utils/speech-stream.d.ts +49 -0
  99. package/dist/protocols/utils/speech-stream.js +64 -0
  100. package/dist/protocols/utils/tool-schema.d.ts +2 -2
  101. package/dist/protocols/utils/tool-schema.js +62 -19
  102. package/dist/protocols/xai-images.d.ts +4 -4
  103. package/dist/protocols/xai-images.js +14 -40
  104. package/dist/protocols/xai-responses.d.ts +2 -2
  105. package/dist/protocols/xai-responses.js +1 -1
  106. package/dist/protocols/xai-video.d.ts +34 -0
  107. package/dist/protocols/xai-video.js +141 -0
  108. package/dist/protocols/zai-chat.d.ts +13 -1
  109. package/dist/protocols/zai-images.d.ts +2 -2
  110. package/dist/protocols/zai-images.js +11 -14
  111. package/dist/protocols/zai-messages.d.ts +1 -1
  112. package/dist/provider-error.js +10 -1
  113. package/dist/providers/alibaba.d.ts +15 -3
  114. package/dist/providers/amazon-bedrock-mantle.d.ts +14 -2
  115. package/dist/providers/amazon-bedrock.d.ts +2 -0
  116. package/dist/providers/amazon-bedrock.js +1 -0
  117. package/dist/providers/anthropic-compatible.d.ts +5 -5
  118. package/dist/providers/anthropic.d.ts +5 -5
  119. package/dist/providers/assemblyai.d.ts +25 -0
  120. package/dist/providers/assemblyai.js +24 -0
  121. package/dist/providers/azure.d.ts +20 -8
  122. package/dist/providers/azure.js +2 -2
  123. package/dist/providers/baseten.d.ts +24 -0
  124. package/dist/providers/black-forest-labs.d.ts +25 -0
  125. package/dist/providers/black-forest-labs.js +23 -0
  126. package/dist/providers/cartesia.d.ts +24 -0
  127. package/dist/providers/cartesia.js +17 -0
  128. package/dist/providers/cerebras.d.ts +24 -0
  129. package/dist/providers/cloudflare-ai-gateway.d.ts +42 -18
  130. package/dist/providers/cloudflare-workers-ai.d.ts +24 -0
  131. package/dist/providers/deepgram.d.ts +29 -0
  132. package/dist/providers/deepgram.js +25 -0
  133. package/dist/providers/deepinfra.d.ts +24 -0
  134. package/dist/providers/deepseek.d.ts +24 -0
  135. package/dist/providers/elevenlabs.d.ts +24 -0
  136. package/dist/providers/elevenlabs.js +23 -0
  137. package/dist/providers/fal.d.ts +29 -0
  138. package/dist/providers/fal.js +26 -0
  139. package/dist/providers/fireworks.d.ts +24 -0
  140. package/dist/providers/google-vertex-chat.d.ts +12 -0
  141. package/dist/providers/google-vertex-messages.d.ts +5 -5
  142. package/dist/providers/google-vertex-responses.d.ts +2 -2
  143. package/dist/providers/google-vertex.d.ts +6 -6
  144. package/dist/providers/google.d.ts +21 -6
  145. package/dist/providers/google.js +13 -9
  146. package/dist/providers/groq.d.ts +24 -0
  147. package/dist/providers/index.d.ts +9 -0
  148. package/dist/providers/index.js +9 -0
  149. package/dist/providers/meta.d.ts +22 -10
  150. package/dist/providers/meta.js +4 -8
  151. package/dist/providers/minimax.d.ts +19 -7
  152. package/dist/providers/moonshot.d.ts +19 -7
  153. package/dist/providers/moonshot.js +3 -3
  154. package/dist/providers/openai-compatible-responses.d.ts +2 -2
  155. package/dist/providers/openai-compatible.d.ts +12 -0
  156. package/dist/providers/openai-options.d.ts +3 -9
  157. package/dist/providers/openai-options.js +4 -7
  158. package/dist/providers/openai.d.ts +33 -12
  159. package/dist/providers/openai.js +17 -9
  160. package/dist/providers/opencode-zen.js +1 -1
  161. package/dist/providers/openrouter.d.ts +54 -7
  162. package/dist/providers/openrouter.js +10 -5
  163. package/dist/providers/replicate.d.ts +25 -0
  164. package/dist/providers/replicate.js +17 -0
  165. package/dist/providers/runway.d.ts +24 -0
  166. package/dist/providers/runway.js +17 -0
  167. package/dist/providers/stability.d.ts +28 -0
  168. package/dist/providers/stability.js +18 -0
  169. package/dist/providers/togetherai.d.ts +24 -0
  170. package/dist/providers/typesafe-ai.js +1 -1
  171. package/dist/providers/vercel-ai-gateway.js +1 -1
  172. package/dist/providers/xai.d.ts +17 -0
  173. package/dist/providers/xai.js +7 -9
  174. package/dist/providers/zai-coding-plan.d.ts +16 -4
  175. package/dist/providers/zai.d.ts +13 -1
  176. package/dist/providers/zai.js +4 -8
  177. package/dist/route/auth.d.ts +5 -2
  178. package/dist/route/auth.js +20 -13
  179. package/dist/route/client.d.ts +9 -7
  180. package/dist/route/client.js +6 -8
  181. package/dist/route/endpoint.d.ts +1 -0
  182. package/dist/route/endpoint.js +2 -2
  183. package/dist/route/executor-service.d.ts +4 -2
  184. package/dist/route/executor-service.js +2 -1
  185. package/dist/route/executor.d.ts +3 -1
  186. package/dist/route/executor.js +7 -0
  187. package/dist/route/framing.d.ts +15 -2
  188. package/dist/route/framing.js +53 -4
  189. package/dist/route/index.d.ts +1 -1
  190. package/dist/route/media-protocol.d.ts +145 -18
  191. package/dist/route/media-protocol.js +96 -28
  192. package/dist/route/media.d.ts +54 -9
  193. package/dist/route/media.js +194 -38
  194. package/dist/route/protocol.d.ts +3 -1
  195. package/dist/schema/events.d.ts +0 -6
  196. package/dist/schema/messages.d.ts +0 -3
  197. package/dist/schema/options.d.ts +10 -6
  198. package/dist/schema/options.js +10 -5
  199. package/dist/speech-client.d.ts +66 -0
  200. package/dist/speech-client.js +21 -0
  201. package/dist/speech.d.ts +1307 -0
  202. package/dist/speech.js +124 -0
  203. package/dist/testing.d.ts +2 -2
  204. package/dist/transcription-client.d.ts +82 -0
  205. package/dist/transcription-client.js +13 -0
  206. package/dist/transcription.d.ts +1492 -0
  207. package/dist/transcription.js +127 -0
  208. package/dist/utils/bytes.d.ts +1 -0
  209. package/dist/utils/bytes.js +10 -0
  210. package/dist/utils/json.d.ts +4 -0
  211. package/dist/utils/json.js +4 -0
  212. package/dist/utils/media-type.d.ts +3 -1
  213. package/dist/utils/media-type.js +25 -2
  214. package/dist/video-client.d.ts +59 -0
  215. package/dist/video-client.js +20 -0
  216. package/dist/video.d.ts +1351 -0
  217. package/dist/video.js +118 -0
  218. package/package.json +3 -3
  219. package/dist/protocols/utils/gemini-tool-schema.d.ts +0 -2
  220. package/dist/protocols/utils/gemini-tool-schema.js +0 -103
  221. package/dist/protocols/utils/meta-image.d.ts +0 -2
  222. package/dist/protocols/utils/meta-image.js +0 -13
  223. package/dist/protocols/utils/openai-image.d.ts +0 -5
  224. package/dist/protocols/utils/openai-image.js +0 -18
@@ -11,8 +11,8 @@ import { BedrockCache } from "./utils/bedrock-cache.js";
11
11
  import { BedrockMedia } from "./utils/bedrock-media.js";
12
12
  import { Lifecycle } from "./utils/lifecycle.js";
13
13
  import { MistralToolID } from "./utils/mistral-tool-id.js";
14
- import { ToolSchemaProjection } from "./utils/tool-schema.js";
15
14
  import { ToolStream } from "./utils/tool-stream.js";
15
+ import { concatBytes } from "../utils/bytes.js";
16
16
  const ADAPTER = "bedrock-converse";
17
17
  // =============================================================================
18
18
  // Request Body Schema
@@ -155,17 +155,17 @@ const BedrockEvent = Schema.Struct({
155
155
  // =============================================================================
156
156
  // Request Lowering
157
157
  // =============================================================================
158
- const lowerToolSpec = (tool, inputSchema) => ({
158
+ const lowerToolSpec = (tool) => ({
159
159
  toolSpec: {
160
160
  name: tool.name,
161
161
  ...(tool.description.trim().length > 0 ? { description: tool.description } : {}),
162
- inputSchema: { json: inputSchema },
162
+ inputSchema: { json: tool.inputSchema },
163
163
  },
164
164
  });
165
- const lowerTools = (compatibility, breakpoints, tools) => {
165
+ const lowerTools = (breakpoints, tools) => {
166
166
  const result = [];
167
167
  for (const tool of tools) {
168
- result.push(lowerToolSpec(tool, ToolSchemaProjection.modelCompatibility(tool.inputSchema, compatibility)));
168
+ result.push(lowerToolSpec(tool));
169
169
  const cachePoint = BedrockCache.block(breakpoints, tool.cache);
170
170
  if (cachePoint)
171
171
  result.push(cachePoint);
@@ -337,10 +337,32 @@ const lowerSystem = (breakpoints, system) => {
337
337
  .flatMap((part) => textWithCache(breakpoints, part.text, part.cache));
338
338
  return content.length === 0 ? undefined : content;
339
339
  };
340
+ // Nova 2 rejects `maxTokens` at high reasoning effort, where its output can exceed the field's maximum. Other models
341
+ // that take `reasoningConfig`, such as Grok on Bedrock, accept it.
342
+ const isNova2 = (model) => /\bamazon\.nova-2-/.test(model.id);
343
+ const isHighReasoningEffort = Schema.is(Schema.Struct({
344
+ additionalModelRequestFields: Schema.Struct({
345
+ reasoningConfig: Schema.Struct({ maxReasoningEffort: Schema.Literal("high") }),
346
+ }),
347
+ }));
348
+ const Options = Schema.Struct({
349
+ thinking: Schema.optional(Schema.Struct({ type: Schema.Literal("enabled"), budgetTokens: Schema.Number })),
350
+ });
351
+ const decodeOptions = ProviderShared.validateWith(Schema.decodeUnknownEffect(Options));
352
+ // Claude on Bedrock requires the thinking budget below `maxTokens`, with a minimum of 1,024.
353
+ const MIN_THINKING_BUDGET = 1_024;
340
354
  const fromRequest = Effect.fn("BedrockConverse.fromRequest")(function* (request) {
341
355
  const toolChoice = request.toolChoice ? yield* lowerToolChoice(request.toolChoice) : undefined;
342
356
  const flattened = ProviderShared.flattenToolRequest(request);
343
357
  const generation = request.generation;
358
+ const options = yield* decodeOptions(request.providerOptions ?? {});
359
+ const maxTokens = isNova2(request.model) && isHighReasoningEffort(request.http?.body) ? undefined : generation?.maxTokens;
360
+ const thinking = options.thinking === undefined
361
+ ? undefined
362
+ : {
363
+ type: "enabled",
364
+ budget_tokens: ProviderShared.fitThinkingBudget(options.thinking.budgetTokens, maxTokens, MIN_THINKING_BUDGET),
365
+ };
344
366
  // Bedrock-Claude shares Anthropic's 4-breakpoint cap. Spend the budget in
345
367
  // tools → system → messages order to favour the highest-impact prefixes.
346
368
  const breakpoints = BedrockCache.breakpoints(request.model.id);
@@ -348,7 +370,7 @@ const fromRequest = Effect.fn("BedrockConverse.fromRequest")(function* (request)
348
370
  if (flattened.tools.length === 0)
349
371
  return undefined;
350
372
  return {
351
- tools: lowerTools(request.model.compatibility?.toolSchema, breakpoints, flattened.tools),
373
+ tools: lowerTools(breakpoints, flattened.tools),
352
374
  // Converse has no native "none". Keep definitions stable for prompt
353
375
  // caching and omit only the unsupported choice.
354
376
  toolChoice,
@@ -360,13 +382,13 @@ const fromRequest = Effect.fn("BedrockConverse.fromRequest")(function* (request)
360
382
  yield* Effect.logWarning(`Bedrock Converse: dropped ${breakpoints.dropped} cache breakpoint(s); the API allows at most ${BedrockCache.BEDROCK_BREAKPOINT_CAP} per request.`);
361
383
  }
362
384
  const inferenceConfig = (() => {
363
- if (generation?.maxTokens === undefined &&
385
+ if (maxTokens === undefined &&
364
386
  generation?.temperature === undefined &&
365
387
  generation?.topP === undefined &&
366
388
  (generation?.stop === undefined || generation.stop.length === 0))
367
389
  return undefined;
368
390
  return {
369
- maxTokens: generation?.maxTokens,
391
+ maxTokens,
370
392
  temperature: generation?.temperature,
371
393
  topP: generation?.topP,
372
394
  stopSequences: generation?.stop,
@@ -378,9 +400,14 @@ const fromRequest = Effect.fn("BedrockConverse.fromRequest")(function* (request)
378
400
  system,
379
401
  inferenceConfig,
380
402
  toolConfig,
381
- // Converse's base inferenceConfig has no topK; Anthropic/Nova accept it
382
- // as a model-specific field, so it goes through additionalModelRequestFields.
383
- additionalModelRequestFields: generation?.topK === undefined ? undefined : { top_k: generation.topK },
403
+ // Converse's base inferenceConfig has no topK or thinking; Anthropic/Nova accept them
404
+ // as model-specific fields, so they go through additionalModelRequestFields.
405
+ additionalModelRequestFields: generation?.topK === undefined && thinking === undefined
406
+ ? undefined
407
+ : {
408
+ ...(generation?.topK === undefined ? {} : { top_k: generation.topK }),
409
+ ...(thinking === undefined ? {} : { thinking }),
410
+ },
384
411
  };
385
412
  });
386
413
  // =============================================================================
@@ -413,14 +440,7 @@ const mapUsage = (usage, providerMetadataKey) => {
413
440
  providerMetadata: { [providerMetadataKey]: usage },
414
441
  });
415
442
  };
416
- const encodeRedactedContent = (chunks) => {
417
- const bytes = new Uint8Array(chunks.reduce((total, chunk) => total + chunk.length, 0));
418
- chunks.reduce((offset, chunk) => {
419
- bytes.set(chunk, offset);
420
- return offset + chunk.length;
421
- }, 0);
422
- return Encoding.encodeBase64(bytes);
423
- };
443
+ const encodeRedactedContent = (chunks) => Encoding.encodeBase64(concatBytes(chunks));
424
444
  const step = (state, event) => Effect.gen(function* () {
425
445
  if (event.contentBlockStart?.start?.toolUse) {
426
446
  const index = event.contentBlockStart.contentBlockIndex;
@@ -0,0 +1,38 @@
1
+ import { Schema } from "effect";
2
+ import { ImageModel, ImageResponse, type ImageRequestFor } from "../image.js";
3
+ import { MediaProtocol } from "../route/media-protocol.js";
4
+ import { MediaRoute } from "../route/media.js";
5
+ export declare const DEFAULT_BASE_URL = "https://api.bfl.ai";
6
+ export type BlackForestLabsImageOptions = {
7
+ readonly safety_tolerance?: number;
8
+ readonly prompt_upsampling?: boolean;
9
+ readonly disable_pup?: boolean;
10
+ readonly raw?: boolean;
11
+ readonly guidance?: number;
12
+ readonly steps?: number;
13
+ } & Record<string, unknown>;
14
+ export type Request = ImageRequestFor<BlackForestLabsImageOptions>;
15
+ /**
16
+ * Regional clusters answer on different hosts, so the returned `polling_url` is followed verbatim. BFL reports the
17
+ * credit cost on submit, so it rides on the token; it is optional so tokens persisted before it existed still decode.
18
+ */
19
+ export declare const Token: Schema.Struct<{
20
+ readonly id: Schema.String;
21
+ readonly pollingURL: Schema.String;
22
+ readonly cost: Schema.optionalKey<Schema.Number>;
23
+ }>;
24
+ export type Token = Schema.Schema.Type<typeof Token>;
25
+ export declare const protocol: MediaProtocol.Queued<Request, ImageResponse, {
26
+ readonly id: string;
27
+ readonly pollingURL: string;
28
+ readonly cost?: number | undefined;
29
+ }>;
30
+ export declare const model: (input: MediaRoute.ModelInput) => ImageModel<BlackForestLabsImageOptions>;
31
+ export declare const BlackForestLabsImages: {
32
+ readonly protocol: MediaProtocol.Queued<Request, ImageResponse, {
33
+ readonly id: string;
34
+ readonly pollingURL: string;
35
+ readonly cost?: number | undefined;
36
+ }>;
37
+ readonly model: (input: MediaRoute.ModelInput) => ImageModel<BlackForestLabsImageOptions>;
38
+ };
@@ -0,0 +1,153 @@
1
+ import { Effect, Schema } from "effect";
2
+ import { ImageModel, ImageResponse } from "../image.js";
3
+ import { Media } from "../media.js";
4
+ import { MediaProtocol } from "../route/media-protocol.js";
5
+ import { MediaRoute } from "../route/media.js";
6
+ import { mergeJsonRecords } from "../schema/index.js";
7
+ import { ProviderShared, optionalNull } from "./shared.js";
8
+ import { MediaInput } from "./utils/media-input.js";
9
+ const route = MediaProtocol.identity({ id: "bfl-images", name: "Black Forest Labs", provider: "black-forest-labs" });
10
+ export const DEFAULT_BASE_URL = "https://api.bfl.ai";
11
+ // ---------------------------------------------------------------------------
12
+ // 2. Token and response schemas
13
+ // ---------------------------------------------------------------------------
14
+ /**
15
+ * Regional clusters answer on different hosts, so the returned `polling_url` is followed verbatim. BFL reports the
16
+ * credit cost on submit, so it rides on the token; it is optional so tokens persisted before it existed still decode.
17
+ */
18
+ export const Token = Schema.Struct({
19
+ id: Schema.String,
20
+ pollingURL: Schema.String,
21
+ cost: Schema.optionalKey(Schema.Number),
22
+ });
23
+ const StartResponse = Schema.Struct({
24
+ id: Schema.String,
25
+ polling_url: Schema.String,
26
+ cost: optionalNull(Schema.Number),
27
+ });
28
+ const Result = Schema.Struct({
29
+ id: Schema.String,
30
+ status: Schema.String,
31
+ result: optionalNull(Schema.StructWithRest(Schema.Struct({ sample: Schema.String, seed: optionalNull(Schema.Number), prompt: optionalNull(Schema.String) }), [Schema.Record(Schema.String, Schema.Unknown)])),
32
+ cost: optionalNull(Schema.Number),
33
+ });
34
+ const STATUS = {
35
+ Pending: "running",
36
+ Reasoning: "running",
37
+ Generating: "running",
38
+ Ready: "completed",
39
+ Error: "failed",
40
+ // Moderation is terminal; `decodeResult` reports it as a content-policy failure.
41
+ "Content Moderated": "failed",
42
+ "Request Moderated": "failed",
43
+ "Task not found": "expired",
44
+ };
45
+ const isModerated = (status) => status === "Content Moderated" || status === "Request Moderated";
46
+ const capabilities = (model) => {
47
+ if (model.startsWith("flux-pro-1.0-fill"))
48
+ return { sizing: "none", imageField: "image", maxImages: 1, mask: true };
49
+ if (model.startsWith("flux-pro-1.0-expand"))
50
+ return { sizing: "none", imageField: "image", maxImages: 1, mask: false };
51
+ if (model.startsWith("flux-kontext"))
52
+ return { sizing: "aspectRatio", imageField: "input_image", maxImages: 4, mask: false };
53
+ if (model.startsWith("flux-pro-1.1-ultra"))
54
+ return { sizing: "aspectRatio", imageField: "image_prompt", maxImages: 1, mask: false };
55
+ if (model.startsWith("flux-pro-1.1") || model.startsWith("flux-dev"))
56
+ return { sizing: "dimensions", imageField: "image_prompt", maxImages: 1, mask: false };
57
+ if (model.startsWith("flux-2-klein"))
58
+ return { sizing: "dimensions", imageField: "input_image", maxImages: 4, mask: false };
59
+ return { sizing: "dimensions", imageField: "input_image", maxImages: 8, mask: false };
60
+ };
61
+ const validate = (request, model) => {
62
+ const id = request.model.id;
63
+ const images = request.images?.length ?? 0;
64
+ if (request.n !== undefined && request.n > 1)
65
+ return Effect.fail(route.unsupported("media.n", `${id} generates one image per request; call it once per image`));
66
+ if (request.size !== undefined && model.sizing !== "dimensions")
67
+ return Effect.fail(route.unsupported("media.size", `${id} does not take size (width and height)`));
68
+ if (request.aspectRatio !== undefined && model.sizing !== "aspectRatio")
69
+ return Effect.fail(route.unsupported("media.aspectRatio", `${id} does not take aspectRatio`));
70
+ if (images > model.maxImages)
71
+ return Effect.fail(route.unsupported("media.images", `${id} takes at most ${model.maxImages} images`));
72
+ if (request.mask !== undefined && !model.mask)
73
+ return Effect.fail(route.unsupported("media.mask", `${id} does not inpaint; use flux-pro-1.0-fill`));
74
+ return Effect.void;
75
+ };
76
+ const imageInput = (asset) => {
77
+ const value = asset.inline()?.base64 ?? ProviderShared.mediaUrl(asset);
78
+ if (value === undefined)
79
+ return Effect.fail(ProviderShared.invalidRequest(`${route.name} accepts inline images or https URLs`));
80
+ return Effect.succeed(value);
81
+ };
82
+ const fromRequest = Effect.fn("BlackForestLabsImages.fromRequest")(function* (request) {
83
+ const model = capabilities(request.model.id);
84
+ yield* validate(request, model);
85
+ const images = yield* Effect.forEach(request.images ?? [], imageInput);
86
+ const fields = images.map((image, index) => [
87
+ index === 0 ? model.imageField : `${model.imageField}_${index + 1}`,
88
+ image,
89
+ ]);
90
+ return MediaProtocol.json(mergeJsonRecords({
91
+ prompt: request.prompt,
92
+ ...(request.size === undefined ? {} : MediaInput.dimensions(request.size)),
93
+ aspect_ratio: request.aspectRatio,
94
+ seed: request.seed,
95
+ output_format: request.format,
96
+ mask: request.mask === undefined ? undefined : yield* imageInput(request.mask),
97
+ ...Object.fromEntries(fields),
98
+ }, request.providerOptions, request.http?.body) ?? {});
99
+ });
100
+ // ---------------------------------------------------------------------------
101
+ // 6. Response decoding
102
+ // ---------------------------------------------------------------------------
103
+ const decodeStart = route.decodeStarted(StartResponse, (value) => ({
104
+ token: {
105
+ id: value.id,
106
+ pollingURL: value.polling_url,
107
+ ...(value.cost === undefined || value.cost === null ? {} : { cost: value.cost }),
108
+ },
109
+ snapshot: { id: value.id, status: "queued" },
110
+ }));
111
+ const decodeDocument = route.decodeJson(Result);
112
+ const decodeStatus = Effect.fn("BlackForestLabsImages.decodeStatus")(function* (response, context) {
113
+ const output = yield* decodeDocument(response);
114
+ return { id: context.token.id, status: yield* MediaProtocol.status(STATUS, output.value.status, output) };
115
+ });
116
+ const decodeResult = Effect.fn("BlackForestLabsImages.decodeResult")(function* (response, context) {
117
+ const output = yield* decodeDocument(response);
118
+ const document = output.value;
119
+ const status = yield* MediaProtocol.status(STATUS, document.status, output);
120
+ if (isModerated(document.status))
121
+ return yield* output.contentPolicy(`${route.name} moderated the generation`);
122
+ if (status === "failed" || status === "expired")
123
+ return yield* output.ended(status, `${route.name} generation ${context.token.id} ended with ${document.status}`);
124
+ if (status !== "completed")
125
+ return yield* output.pending(context.token.id);
126
+ if (document.result === undefined || document.result === null)
127
+ return yield* output.invalid(`${route.name} generation ${context.token.id} has no result`);
128
+ const { sample, seed, prompt, ...rest } = document.result;
129
+ // A settled `cost` on the result supersedes the submit-time cost carried on the token.
130
+ const cost = document.cost ?? context.token.cost;
131
+ return new ImageResponse({
132
+ // `sample` is a signed URL that expires 10 minutes after the result is ready, so it is downloaded now.
133
+ images: [yield* context.materialize(Media.url(sample))],
134
+ usage: cost === undefined ? undefined : { type: "credits", credits: cost },
135
+ providerMetadata: {
136
+ bfl: { id: context.token.id, seed: seed ?? undefined, prompt: prompt ?? undefined, ...rest },
137
+ },
138
+ });
139
+ });
140
+ // ---------------------------------------------------------------------------
141
+ // 7. Protocol and route
142
+ // ---------------------------------------------------------------------------
143
+ export const protocol = MediaProtocol.queued(route, {
144
+ token: Token,
145
+ start: { body: { from: fromRequest }, decode: decodeStart },
146
+ status: { path: (token) => token.pollingURL, decode: decodeStatus },
147
+ result: { path: (token) => token.pollingURL, decode: decodeResult },
148
+ });
149
+ export const model = (input) => ImageModel.fromRoute({ protocol, baseURL: DEFAULT_BASE_URL, path: ({ request }) => `/v1/${request.model.id}` }, input);
150
+ export const BlackForestLabsImages = {
151
+ protocol,
152
+ model,
153
+ };
@@ -0,0 +1,143 @@
1
+ import { MediaProtocol } from "../route/media-protocol.js";
2
+ import { MediaRoute } from "../route/media.js";
3
+ import { type OpenString } from "../schema/index.js";
4
+ import { SpeechModel, type SpeechRequestFor } from "../speech.js";
5
+ import { SpeechStream } from "./utils/speech-stream.js";
6
+ export declare const DEFAULT_BASE_URL = "https://api.cartesia.ai";
7
+ export declare const API_VERSION = "2026-08-14";
8
+ export declare const BYTES_PATH = "/tts/bytes";
9
+ export declare const SSE_PATH = "/tts/sse";
10
+ export type CartesiaEncoding = SpeechStream.PcmEncoding;
11
+ export type CartesiaSpeechOptions = {
12
+ readonly sampleRate?: 8000 | 16000 | 22050 | 24000 | 44100 | 48000;
13
+ readonly bitRate?: 32000 | 64000 | 96000 | 128000 | 192000;
14
+ readonly encoding?: CartesiaEncoding;
15
+ readonly generation_config?: {
16
+ readonly volume?: number;
17
+ readonly emotion?: OpenString<"neutral" | "calm" | "angry" | "content" | "sad" | "scared">;
18
+ };
19
+ readonly pronunciation_dict_id?: string;
20
+ } & Record<string, unknown>;
21
+ export type Request = SpeechRequestFor<CartesiaSpeechOptions>;
22
+ interface State extends SpeechStream.Audio {
23
+ readonly done: boolean;
24
+ }
25
+ export declare const protocol: MediaProtocol.Streamed<Request, {
26
+ readonly id: string;
27
+ readonly type: "generation-queued";
28
+ readonly position?: number | undefined;
29
+ } | {
30
+ readonly id: string;
31
+ readonly type: "generation-progress";
32
+ readonly progress?: number | undefined;
33
+ } | {
34
+ readonly type: "audio-delta";
35
+ readonly chunk: Uint8Array<ArrayBufferLike>;
36
+ } | {
37
+ readonly type: "timestamps";
38
+ readonly items: readonly {
39
+ readonly text: string;
40
+ readonly startSeconds: number;
41
+ readonly endSeconds: number;
42
+ }[];
43
+ } | {
44
+ readonly type: "finish";
45
+ readonly audio: import("../media.js").Asset;
46
+ readonly providerMetadata?: {
47
+ readonly [x: string]: {
48
+ readonly [x: string]: unknown;
49
+ };
50
+ } | undefined;
51
+ readonly usage?: {
52
+ readonly type: "tokens";
53
+ readonly input?: number | undefined;
54
+ readonly output?: number | undefined;
55
+ readonly total?: number | undefined;
56
+ readonly details?: {
57
+ readonly [x: string]: unknown;
58
+ } | undefined;
59
+ } | {
60
+ readonly type: "seconds";
61
+ readonly seconds: number;
62
+ } | {
63
+ readonly type: "characters";
64
+ readonly characters: number;
65
+ } | {
66
+ readonly type: "credits";
67
+ readonly credits: number;
68
+ } | {
69
+ readonly type: "compute";
70
+ readonly seconds: number;
71
+ } | undefined;
72
+ readonly notices?: readonly {
73
+ readonly type: "other" | "moderated" | "filtered";
74
+ readonly message: string;
75
+ readonly providerMetadata?: {
76
+ readonly [x: string]: {
77
+ readonly [x: string]: unknown;
78
+ };
79
+ } | undefined;
80
+ }[] | undefined;
81
+ }, string | Uint8Array<ArrayBufferLike>, State>;
82
+ export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<CartesiaSpeechOptions>;
83
+ export declare const CartesiaSpeech: {
84
+ readonly protocol: MediaProtocol.Streamed<Request, {
85
+ readonly id: string;
86
+ readonly type: "generation-queued";
87
+ readonly position?: number | undefined;
88
+ } | {
89
+ readonly id: string;
90
+ readonly type: "generation-progress";
91
+ readonly progress?: number | undefined;
92
+ } | {
93
+ readonly type: "audio-delta";
94
+ readonly chunk: Uint8Array<ArrayBufferLike>;
95
+ } | {
96
+ readonly type: "timestamps";
97
+ readonly items: readonly {
98
+ readonly text: string;
99
+ readonly startSeconds: number;
100
+ readonly endSeconds: number;
101
+ }[];
102
+ } | {
103
+ readonly type: "finish";
104
+ readonly audio: import("../media.js").Asset;
105
+ readonly providerMetadata?: {
106
+ readonly [x: string]: {
107
+ readonly [x: string]: unknown;
108
+ };
109
+ } | undefined;
110
+ readonly usage?: {
111
+ readonly type: "tokens";
112
+ readonly input?: number | undefined;
113
+ readonly output?: number | undefined;
114
+ readonly total?: number | undefined;
115
+ readonly details?: {
116
+ readonly [x: string]: unknown;
117
+ } | undefined;
118
+ } | {
119
+ readonly type: "seconds";
120
+ readonly seconds: number;
121
+ } | {
122
+ readonly type: "characters";
123
+ readonly characters: number;
124
+ } | {
125
+ readonly type: "credits";
126
+ readonly credits: number;
127
+ } | {
128
+ readonly type: "compute";
129
+ readonly seconds: number;
130
+ } | undefined;
131
+ readonly notices?: readonly {
132
+ readonly type: "other" | "moderated" | "filtered";
133
+ readonly message: string;
134
+ readonly providerMetadata?: {
135
+ readonly [x: string]: {
136
+ readonly [x: string]: unknown;
137
+ };
138
+ } | undefined;
139
+ }[] | undefined;
140
+ }, string | Uint8Array<ArrayBufferLike>, State>;
141
+ readonly model: (input: MediaRoute.ModelInput) => SpeechModel<CartesiaSpeechOptions>;
142
+ };
143
+ export {};
@@ -0,0 +1,120 @@
1
+ import { Effect, Schema } from "effect";
2
+ import { classifyProviderFailure } from "../provider-error.js";
3
+ import { Framing } from "../route/framing.js";
4
+ import { MediaProtocol } from "../route/media-protocol.js";
5
+ import { MediaRoute } from "../route/media.js";
6
+ import { AIError, mergeJsonRecords } from "../schema/index.js";
7
+ import { SpeechModel } from "../speech.js";
8
+ import { ProviderShared, optionalNull } from "./shared.js";
9
+ import { SpeechStream } from "./utils/speech-stream.js";
10
+ const route = MediaProtocol.identity({ id: "cartesia-speech", name: "Cartesia", provider: "cartesia" });
11
+ export const DEFAULT_BASE_URL = "https://api.cartesia.ai";
12
+ export const API_VERSION = "2026-08-14";
13
+ export const BYTES_PATH = "/tts/bytes";
14
+ export const SSE_PATH = "/tts/sse";
15
+ const DEFAULT_SAMPLE_RATE = 44100;
16
+ const DEFAULT_BIT_RATE = 128000;
17
+ // ---------------------------------------------------------------------------
18
+ // 3. Streaming event schema
19
+ // ---------------------------------------------------------------------------
20
+ /** `phoneme_timestamps` and future record types are ignored. */
21
+ const SseEvent = Schema.Struct({
22
+ type: Schema.String,
23
+ data: Schema.optional(Schema.Uint8ArrayFromBase64),
24
+ word_timestamps: Schema.optional(Schema.Struct({
25
+ words: Schema.Array(Schema.String),
26
+ start: Schema.Array(Schema.Number),
27
+ end: Schema.Array(Schema.Number),
28
+ })),
29
+ status_code: Schema.optional(Schema.Number),
30
+ title: Schema.optional(Schema.String),
31
+ message: Schema.optional(Schema.String),
32
+ error_code: optionalNull(Schema.String),
33
+ });
34
+ const decodeEvent = route.decodeFrame(SseEvent);
35
+ // ---------------------------------------------------------------------------
36
+ // 5. Request body construction
37
+ // ---------------------------------------------------------------------------
38
+ /** Timestamps exist only on the SSE endpoint, so a `generate` that asks for them collects an SSE stream. */
39
+ const usesSse = (request) => request.mode === "stream" || request.timestamps === true;
40
+ const CONTAINERS = { pcm: "raw", wav: "wav", mp3: "mp3" };
41
+ const outputFormat = Effect.fn("CartesiaSpeech.outputFormat")(function* (request) {
42
+ const sse = usesSse(request);
43
+ const format = request.format ?? (sse ? "pcm" : "mp3");
44
+ const container = CONTAINERS[format];
45
+ if (container === undefined)
46
+ return yield* route.unsupported("media.format", `${route.name} supports the pcm, wav, and mp3 formats, not "${format}"`);
47
+ if (sse && container !== "raw")
48
+ return yield* route.unsupported("media.format", `${route.name} streams and timestamps only raw PCM; request format "pcm" instead of "${format}"`);
49
+ const sampleRate = request.providerOptions?.sampleRate ?? DEFAULT_SAMPLE_RATE;
50
+ if (container === "mp3")
51
+ return { container, sample_rate: sampleRate, bit_rate: request.providerOptions?.bitRate ?? DEFAULT_BIT_RATE };
52
+ return { container, encoding: request.providerOptions?.encoding ?? "pcm_s16le", sample_rate: sampleRate };
53
+ });
54
+ const fromRequest = Effect.fn("CartesiaSpeech.fromRequest")(function* (request) {
55
+ const voice = SpeechStream.voiceID(request.voice);
56
+ if (voice === undefined)
57
+ return yield* ProviderShared.invalidRequest(`${route.name} requires a voice id; pass it as \`voice\``);
58
+ const { sampleRate: _sampleRate, bitRate: _bitRate, encoding: _encoding, ...native } = request.providerOptions ?? {};
59
+ return MediaProtocol.json(mergeJsonRecords({
60
+ model_id: request.model.id,
61
+ transcript: request.text,
62
+ voice,
63
+ output_format: yield* outputFormat(request),
64
+ language: request.language,
65
+ generation_config: request.speed === undefined ? undefined : { speed: request.speed },
66
+ add_timestamps: request.timestamps === true ? true : undefined,
67
+ }, native, request.http?.body) ?? {});
68
+ });
69
+ // ---------------------------------------------------------------------------
70
+ // 6. Stream parsing
71
+ // ---------------------------------------------------------------------------
72
+ const onEvent = Effect.fn("CartesiaSpeech.onEvent")(function* (state, frame) {
73
+ const event = yield* decodeEvent(frame);
74
+ if (event.type === "chunk" && event.data !== undefined)
75
+ return SpeechStream.delta(state, event.data);
76
+ if (event.type === "timestamps" && event.word_timestamps !== undefined) {
77
+ const words = event.word_timestamps;
78
+ return [state, SpeechStream.timestamps(words.words, words.start, words.end)];
79
+ }
80
+ if (event.type === "done")
81
+ return [{ ...state, done: true }, []];
82
+ if (event.type === "error")
83
+ return yield* new AIError({
84
+ reason: classifyProviderFailure({
85
+ message: `${route.name} stream failed${event.title === undefined ? "" : ` (${event.title})`}: ${event.message ?? "unknown error"}`,
86
+ status: event.status_code,
87
+ rawBody: frame,
88
+ }),
89
+ });
90
+ return [state, []];
91
+ });
92
+ const finish = Effect.fn("CartesiaSpeech.finish")(function* (state, context) {
93
+ if (usesSse(context.request) && !state.done)
94
+ return yield* route.incomplete();
95
+ const format = yield* outputFormat(context.request);
96
+ return yield* SpeechStream.finish(route, state, format.container === "raw"
97
+ ? SpeechStream.pcm(format.encoding, format.sample_rate)
98
+ : SpeechStream.container(format.container, format.sample_rate));
99
+ });
100
+ // ---------------------------------------------------------------------------
101
+ // 7. Protocol and route
102
+ // ---------------------------------------------------------------------------
103
+ export const protocol = MediaProtocol.stream(route, {
104
+ unsupported: ["instructions"],
105
+ body: { from: fromRequest },
106
+ frames: (bytes, context) => (usesSse(context.request) ? Framing.sse.frame(bytes) : bytes),
107
+ initial: () => ({ chunks: [], done: false }),
108
+ step: SpeechStream.step(onEvent),
109
+ finish,
110
+ });
111
+ export const model = (input) => SpeechModel.fromRoute({
112
+ protocol,
113
+ baseURL: DEFAULT_BASE_URL,
114
+ headers: { "Cartesia-Version": API_VERSION },
115
+ path: ({ request }) => (usesSse(request) ? SSE_PATH : BYTES_PATH),
116
+ }, input);
117
+ export const CartesiaSpeech = {
118
+ protocol,
119
+ model,
120
+ };