@opencode/ai 2.0.15 → 2.0.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (224) hide show
  1. package/README.md +441 -99
  2. package/dist/ai-client.d.ts +8 -0
  3. package/dist/ai-client.js +12 -0
  4. package/dist/experimental/evaluation-client.d.ts +3 -3
  5. package/dist/experimental/evaluation-client.js +1 -1
  6. package/dist/experimental/evaluation.js +1 -1
  7. package/dist/generation.d.ts +39 -29
  8. package/dist/generation.js +62 -36
  9. package/dist/image-client.d.ts +59 -17
  10. package/dist/image-client.js +16 -24
  11. package/dist/image.d.ts +394 -55
  12. package/dist/image.js +48 -57
  13. package/dist/index.d.ts +15 -2
  14. package/dist/index.js +10 -0
  15. package/dist/llm.d.ts +7 -5
  16. package/dist/llm.js +10 -4
  17. package/dist/media-client.d.ts +30 -0
  18. package/dist/media-client.js +51 -0
  19. package/dist/media-model.d.ts +43 -0
  20. package/dist/media-model.js +47 -0
  21. package/dist/media.d.ts +10 -9
  22. package/dist/media.js +11 -12
  23. package/dist/promise.d.ts +473 -15
  24. package/dist/promise.js +71 -13
  25. package/dist/protocols/alibaba-chat.d.ts +12 -0
  26. package/dist/protocols/alibaba-chat.js +4 -1
  27. package/dist/protocols/alibaba-messages.d.ts +1 -1
  28. package/dist/protocols/alibaba-messages.js +6 -4
  29. package/dist/protocols/alibaba-responses.d.ts +2 -2
  30. package/dist/protocols/anthropic-messages.d.ts +34 -34
  31. package/dist/protocols/anthropic-messages.js +14 -9
  32. package/dist/protocols/assemblyai-transcription.d.ts +41 -0
  33. package/dist/protocols/assemblyai-transcription.js +136 -0
  34. package/dist/protocols/bedrock-converse.d.ts +7 -0
  35. package/dist/protocols/bedrock-converse.js +39 -19
  36. package/dist/protocols/bfl-images.d.ts +38 -0
  37. package/dist/protocols/bfl-images.js +153 -0
  38. package/dist/protocols/cartesia-speech.d.ts +143 -0
  39. package/dist/protocols/cartesia-speech.js +120 -0
  40. package/dist/protocols/deepgram-speech.d.ts +135 -0
  41. package/dist/protocols/deepgram-speech.js +91 -0
  42. package/dist/protocols/deepgram-transcription.d.ts +26 -0
  43. package/dist/protocols/deepgram-transcription.js +125 -0
  44. package/dist/protocols/elevenlabs-speech.d.ts +138 -0
  45. package/dist/protocols/elevenlabs-speech.js +111 -0
  46. package/dist/protocols/fal-images.d.ts +25 -0
  47. package/dist/protocols/fal-images.js +104 -0
  48. package/dist/protocols/fal-video.d.ts +29 -0
  49. package/dist/protocols/fal-video.js +73 -0
  50. package/dist/protocols/gemini.d.ts +22 -22
  51. package/dist/protocols/gemini.js +19 -36
  52. package/dist/protocols/google-images.d.ts +3 -3
  53. package/dist/protocols/google-images.js +12 -34
  54. package/dist/protocols/google-speech.d.ts +146 -0
  55. package/dist/protocols/google-speech.js +88 -0
  56. package/dist/protocols/google-transcription.d.ts +174 -0
  57. package/dist/protocols/google-transcription.js +126 -0
  58. package/dist/protocols/google-video.d.ts +26 -0
  59. package/dist/protocols/google-video.js +142 -0
  60. package/dist/protocols/meta-images.d.ts +3 -4
  61. package/dist/protocols/meta-images.js +13 -29
  62. package/dist/protocols/meta-messages.d.ts +5 -5
  63. package/dist/protocols/meta-responses.d.ts +4 -4
  64. package/dist/protocols/meta-responses.js +6 -4
  65. package/dist/protocols/open-responses.d.ts +10 -9
  66. package/dist/protocols/open-responses.js +4 -9
  67. package/dist/protocols/openai-chat.d.ts +84 -0
  68. package/dist/protocols/openai-chat.js +28 -17
  69. package/dist/protocols/openai-compatible-chat.d.ts +12 -0
  70. package/dist/protocols/openai-compatible-responses.d.ts +2 -2
  71. package/dist/protocols/openai-images.d.ts +130 -7
  72. package/dist/protocols/openai-images.js +143 -91
  73. package/dist/protocols/openai-responses.d.ts +20 -20
  74. package/dist/protocols/openai-responses.js +31 -31
  75. package/dist/protocols/openai-speech.d.ts +136 -0
  76. package/dist/protocols/openai-speech.js +97 -0
  77. package/dist/protocols/openai-transcription.d.ts +211 -0
  78. package/dist/protocols/openai-transcription.js +202 -0
  79. package/dist/protocols/replicate-images.d.ts +28 -0
  80. package/dist/protocols/replicate-images.js +127 -0
  81. package/dist/protocols/runway-video.d.ts +38 -0
  82. package/dist/protocols/runway-video.js +140 -0
  83. package/dist/protocols/shared.d.ts +21 -17
  84. package/dist/protocols/shared.js +27 -36
  85. package/dist/protocols/stability-images.d.ts +39 -0
  86. package/dist/protocols/stability-images.js +136 -0
  87. package/dist/protocols/utils/fal-queue.d.ts +26 -0
  88. package/dist/protocols/utils/fal-queue.js +67 -0
  89. package/dist/protocols/utils/gemini-generate-content.d.ts +65 -0
  90. package/dist/protocols/utils/gemini-generate-content.js +65 -0
  91. package/dist/protocols/utils/gemini-json-schema.d.ts +3 -0
  92. package/dist/protocols/utils/gemini-json-schema.js +76 -0
  93. package/dist/protocols/utils/media-input.d.ts +23 -1
  94. package/dist/protocols/utils/media-input.js +40 -0
  95. package/dist/protocols/utils/responses-checkpoint.js +3 -7
  96. package/dist/protocols/utils/responses-compaction.d.ts +3 -1
  97. package/dist/protocols/utils/responses-compaction.js +16 -3
  98. package/dist/protocols/utils/speech-stream.d.ts +49 -0
  99. package/dist/protocols/utils/speech-stream.js +64 -0
  100. package/dist/protocols/utils/tool-schema.d.ts +2 -2
  101. package/dist/protocols/utils/tool-schema.js +62 -19
  102. package/dist/protocols/xai-images.d.ts +4 -4
  103. package/dist/protocols/xai-images.js +14 -40
  104. package/dist/protocols/xai-responses.d.ts +2 -2
  105. package/dist/protocols/xai-responses.js +1 -1
  106. package/dist/protocols/xai-video.d.ts +34 -0
  107. package/dist/protocols/xai-video.js +141 -0
  108. package/dist/protocols/zai-chat.d.ts +13 -1
  109. package/dist/protocols/zai-images.d.ts +2 -2
  110. package/dist/protocols/zai-images.js +11 -14
  111. package/dist/protocols/zai-messages.d.ts +1 -1
  112. package/dist/provider-error.js +10 -1
  113. package/dist/providers/alibaba.d.ts +15 -3
  114. package/dist/providers/amazon-bedrock-mantle.d.ts +14 -2
  115. package/dist/providers/amazon-bedrock.d.ts +2 -0
  116. package/dist/providers/amazon-bedrock.js +1 -0
  117. package/dist/providers/anthropic-compatible.d.ts +5 -5
  118. package/dist/providers/anthropic.d.ts +5 -5
  119. package/dist/providers/assemblyai.d.ts +25 -0
  120. package/dist/providers/assemblyai.js +24 -0
  121. package/dist/providers/azure.d.ts +20 -8
  122. package/dist/providers/azure.js +2 -2
  123. package/dist/providers/baseten.d.ts +24 -0
  124. package/dist/providers/black-forest-labs.d.ts +25 -0
  125. package/dist/providers/black-forest-labs.js +23 -0
  126. package/dist/providers/cartesia.d.ts +24 -0
  127. package/dist/providers/cartesia.js +17 -0
  128. package/dist/providers/cerebras.d.ts +24 -0
  129. package/dist/providers/cloudflare-ai-gateway.d.ts +42 -18
  130. package/dist/providers/cloudflare-workers-ai.d.ts +24 -0
  131. package/dist/providers/deepgram.d.ts +29 -0
  132. package/dist/providers/deepgram.js +25 -0
  133. package/dist/providers/deepinfra.d.ts +24 -0
  134. package/dist/providers/deepseek.d.ts +24 -0
  135. package/dist/providers/elevenlabs.d.ts +24 -0
  136. package/dist/providers/elevenlabs.js +23 -0
  137. package/dist/providers/fal.d.ts +29 -0
  138. package/dist/providers/fal.js +26 -0
  139. package/dist/providers/fireworks.d.ts +24 -0
  140. package/dist/providers/google-vertex-chat.d.ts +12 -0
  141. package/dist/providers/google-vertex-messages.d.ts +5 -5
  142. package/dist/providers/google-vertex-responses.d.ts +2 -2
  143. package/dist/providers/google-vertex.d.ts +6 -6
  144. package/dist/providers/google.d.ts +21 -6
  145. package/dist/providers/google.js +13 -9
  146. package/dist/providers/groq.d.ts +24 -0
  147. package/dist/providers/index.d.ts +9 -0
  148. package/dist/providers/index.js +9 -0
  149. package/dist/providers/meta.d.ts +22 -10
  150. package/dist/providers/meta.js +4 -8
  151. package/dist/providers/minimax.d.ts +19 -7
  152. package/dist/providers/moonshot.d.ts +19 -7
  153. package/dist/providers/moonshot.js +3 -3
  154. package/dist/providers/openai-compatible-responses.d.ts +2 -2
  155. package/dist/providers/openai-compatible.d.ts +12 -0
  156. package/dist/providers/openai-options.d.ts +3 -9
  157. package/dist/providers/openai-options.js +4 -7
  158. package/dist/providers/openai.d.ts +33 -12
  159. package/dist/providers/openai.js +17 -9
  160. package/dist/providers/opencode-zen.js +1 -1
  161. package/dist/providers/openrouter.d.ts +54 -7
  162. package/dist/providers/openrouter.js +10 -5
  163. package/dist/providers/replicate.d.ts +25 -0
  164. package/dist/providers/replicate.js +17 -0
  165. package/dist/providers/runway.d.ts +24 -0
  166. package/dist/providers/runway.js +17 -0
  167. package/dist/providers/stability.d.ts +28 -0
  168. package/dist/providers/stability.js +18 -0
  169. package/dist/providers/togetherai.d.ts +24 -0
  170. package/dist/providers/typesafe-ai.js +1 -1
  171. package/dist/providers/vercel-ai-gateway.js +1 -1
  172. package/dist/providers/xai.d.ts +17 -0
  173. package/dist/providers/xai.js +7 -9
  174. package/dist/providers/zai-coding-plan.d.ts +16 -4
  175. package/dist/providers/zai.d.ts +13 -1
  176. package/dist/providers/zai.js +4 -8
  177. package/dist/route/auth.d.ts +5 -2
  178. package/dist/route/auth.js +20 -13
  179. package/dist/route/client.d.ts +9 -7
  180. package/dist/route/client.js +6 -8
  181. package/dist/route/endpoint.d.ts +1 -0
  182. package/dist/route/endpoint.js +2 -2
  183. package/dist/route/executor-service.d.ts +4 -2
  184. package/dist/route/executor-service.js +2 -1
  185. package/dist/route/executor.d.ts +3 -1
  186. package/dist/route/executor.js +7 -0
  187. package/dist/route/framing.d.ts +15 -2
  188. package/dist/route/framing.js +53 -4
  189. package/dist/route/index.d.ts +1 -1
  190. package/dist/route/media-protocol.d.ts +145 -18
  191. package/dist/route/media-protocol.js +96 -28
  192. package/dist/route/media.d.ts +54 -9
  193. package/dist/route/media.js +194 -38
  194. package/dist/route/protocol.d.ts +3 -1
  195. package/dist/schema/events.d.ts +0 -6
  196. package/dist/schema/messages.d.ts +0 -3
  197. package/dist/schema/options.d.ts +10 -6
  198. package/dist/schema/options.js +10 -5
  199. package/dist/speech-client.d.ts +66 -0
  200. package/dist/speech-client.js +21 -0
  201. package/dist/speech.d.ts +1307 -0
  202. package/dist/speech.js +124 -0
  203. package/dist/testing.d.ts +2 -2
  204. package/dist/transcription-client.d.ts +82 -0
  205. package/dist/transcription-client.js +13 -0
  206. package/dist/transcription.d.ts +1492 -0
  207. package/dist/transcription.js +127 -0
  208. package/dist/utils/bytes.d.ts +1 -0
  209. package/dist/utils/bytes.js +10 -0
  210. package/dist/utils/json.d.ts +4 -0
  211. package/dist/utils/json.js +4 -0
  212. package/dist/utils/media-type.d.ts +3 -1
  213. package/dist/utils/media-type.js +25 -2
  214. package/dist/video-client.d.ts +59 -0
  215. package/dist/video-client.js +20 -0
  216. package/dist/video.d.ts +1351 -0
  217. package/dist/video.js +118 -0
  218. package/package.json +3 -3
  219. package/dist/protocols/utils/gemini-tool-schema.d.ts +0 -2
  220. package/dist/protocols/utils/gemini-tool-schema.js +0 -103
  221. package/dist/protocols/utils/meta-image.d.ts +0 -2
  222. package/dist/protocols/utils/meta-image.js +0 -13
  223. package/dist/protocols/utils/openai-image.d.ts +0 -5
  224. package/dist/protocols/utils/openai-image.js +0 -18
@@ -0,0 +1,135 @@
1
+ import { MediaProtocol } from "../route/media-protocol.js";
2
+ import { MediaRoute } from "../route/media.js";
3
+ import { type OpenString } from "../schema/index.js";
4
+ import { SpeechModel, type SpeechRequestFor } from "../speech.js";
5
+ import { SpeechStream } from "./utils/speech-stream.js";
6
+ export declare const DEFAULT_BASE_URL = "https://api.deepgram.com";
7
+ export declare const PATH = "/v1/speak";
8
+ export type DeepgramEncoding = OpenString<"linear16" | "mulaw" | "alaw" | "mp3" | "opus" | "flac" | "aac">;
9
+ export type DeepgramSpeechOptions = {
10
+ readonly encoding?: DeepgramEncoding;
11
+ readonly container?: OpenString<"wav" | "ogg" | "none">;
12
+ readonly sampleRate?: number;
13
+ readonly bitRate?: number;
14
+ readonly mip_opt_out?: boolean;
15
+ readonly tag?: string;
16
+ } & Record<string, unknown>;
17
+ export type Request = SpeechRequestFor<DeepgramSpeechOptions>;
18
+ export declare const protocol: MediaProtocol.Streamed<Request, {
19
+ readonly id: string;
20
+ readonly type: "generation-queued";
21
+ readonly position?: number | undefined;
22
+ } | {
23
+ readonly id: string;
24
+ readonly type: "generation-progress";
25
+ readonly progress?: number | undefined;
26
+ } | {
27
+ readonly type: "audio-delta";
28
+ readonly chunk: Uint8Array<ArrayBufferLike>;
29
+ } | {
30
+ readonly type: "timestamps";
31
+ readonly items: readonly {
32
+ readonly text: string;
33
+ readonly startSeconds: number;
34
+ readonly endSeconds: number;
35
+ }[];
36
+ } | {
37
+ readonly type: "finish";
38
+ readonly audio: import("../media.js").Asset;
39
+ readonly providerMetadata?: {
40
+ readonly [x: string]: {
41
+ readonly [x: string]: unknown;
42
+ };
43
+ } | undefined;
44
+ readonly usage?: {
45
+ readonly type: "tokens";
46
+ readonly input?: number | undefined;
47
+ readonly output?: number | undefined;
48
+ readonly total?: number | undefined;
49
+ readonly details?: {
50
+ readonly [x: string]: unknown;
51
+ } | undefined;
52
+ } | {
53
+ readonly type: "seconds";
54
+ readonly seconds: number;
55
+ } | {
56
+ readonly type: "characters";
57
+ readonly characters: number;
58
+ } | {
59
+ readonly type: "credits";
60
+ readonly credits: number;
61
+ } | {
62
+ readonly type: "compute";
63
+ readonly seconds: number;
64
+ } | undefined;
65
+ readonly notices?: readonly {
66
+ readonly type: "other" | "moderated" | "filtered";
67
+ readonly message: string;
68
+ readonly providerMetadata?: {
69
+ readonly [x: string]: {
70
+ readonly [x: string]: unknown;
71
+ };
72
+ } | undefined;
73
+ }[] | undefined;
74
+ }, Uint8Array<ArrayBufferLike>, SpeechStream.Audio>;
75
+ export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<DeepgramSpeechOptions>;
76
+ export declare const DeepgramSpeech: {
77
+ readonly protocol: MediaProtocol.Streamed<Request, {
78
+ readonly id: string;
79
+ readonly type: "generation-queued";
80
+ readonly position?: number | undefined;
81
+ } | {
82
+ readonly id: string;
83
+ readonly type: "generation-progress";
84
+ readonly progress?: number | undefined;
85
+ } | {
86
+ readonly type: "audio-delta";
87
+ readonly chunk: Uint8Array<ArrayBufferLike>;
88
+ } | {
89
+ readonly type: "timestamps";
90
+ readonly items: readonly {
91
+ readonly text: string;
92
+ readonly startSeconds: number;
93
+ readonly endSeconds: number;
94
+ }[];
95
+ } | {
96
+ readonly type: "finish";
97
+ readonly audio: import("../media.js").Asset;
98
+ readonly providerMetadata?: {
99
+ readonly [x: string]: {
100
+ readonly [x: string]: unknown;
101
+ };
102
+ } | undefined;
103
+ readonly usage?: {
104
+ readonly type: "tokens";
105
+ readonly input?: number | undefined;
106
+ readonly output?: number | undefined;
107
+ readonly total?: number | undefined;
108
+ readonly details?: {
109
+ readonly [x: string]: unknown;
110
+ } | undefined;
111
+ } | {
112
+ readonly type: "seconds";
113
+ readonly seconds: number;
114
+ } | {
115
+ readonly type: "characters";
116
+ readonly characters: number;
117
+ } | {
118
+ readonly type: "credits";
119
+ readonly credits: number;
120
+ } | {
121
+ readonly type: "compute";
122
+ readonly seconds: number;
123
+ } | undefined;
124
+ readonly notices?: readonly {
125
+ readonly type: "other" | "moderated" | "filtered";
126
+ readonly message: string;
127
+ readonly providerMetadata?: {
128
+ readonly [x: string]: {
129
+ readonly [x: string]: unknown;
130
+ };
131
+ } | undefined;
132
+ }[] | undefined;
133
+ }, Uint8Array<ArrayBufferLike>, SpeechStream.Audio>;
134
+ readonly model: (input: MediaRoute.ModelInput) => SpeechModel<DeepgramSpeechOptions>;
135
+ };
@@ -0,0 +1,91 @@
1
+ import { Effect } from "effect";
2
+ import { MediaProtocol } from "../route/media-protocol.js";
3
+ import { MediaRoute } from "../route/media.js";
4
+ import { mergeJsonRecords } from "../schema/index.js";
5
+ import { SpeechModel } from "../speech.js";
6
+ import { MediaInput } from "./utils/media-input.js";
7
+ import { SpeechStream } from "./utils/speech-stream.js";
8
+ const route = MediaProtocol.identity({ id: "deepgram-speech", name: "Deepgram", provider: "deepgram" });
9
+ export const DEFAULT_BASE_URL = "https://api.deepgram.com";
10
+ export const PATH = "/v1/speak";
11
+ // ---------------------------------------------------------------------------
12
+ // 5. Request body construction
13
+ // ---------------------------------------------------------------------------
14
+ const FORMATS = {
15
+ mp3: { encoding: "mp3" },
16
+ wav: { encoding: "linear16", container: "wav" },
17
+ pcm: { encoding: "linear16", container: "none" },
18
+ opus: { encoding: "opus" },
19
+ flac: { encoding: "flac" },
20
+ aac: { encoding: "aac" },
21
+ };
22
+ const audioFormat = (request) => {
23
+ const format = request.format === undefined ? undefined : FORMATS[request.format];
24
+ return {
25
+ encoding: request.providerOptions?.encoding ?? format?.encoding,
26
+ container: request.providerOptions?.container ?? format?.container,
27
+ };
28
+ };
29
+ const queryParameters = (request) => {
30
+ const { encoding: _encoding, container: _container, sampleRate, bitRate, ...native } = request.providerOptions ?? {};
31
+ return MediaInput.query(route.id, {
32
+ ...native,
33
+ model: request.model.id,
34
+ ...audioFormat(request),
35
+ sample_rate: sampleRate,
36
+ bit_rate: bitRate,
37
+ speed: request.speed,
38
+ });
39
+ };
40
+ const fromRequest = Effect.fn("DeepgramSpeech.fromRequest")(function* (request) {
41
+ // Not in `unsupported`: that list would also reject `timestamps: false`, which asks for nothing.
42
+ if (request.timestamps === true)
43
+ return yield* route.unsupported("media.timestamps", `${route.name} does not return timestamps`);
44
+ if (request.format !== undefined &&
45
+ FORMATS[request.format] === undefined &&
46
+ request.providerOptions?.encoding === undefined)
47
+ return yield* route.unsupported("media.format", `${route.name} has no encoding for format "${request.format}"; pass providerOptions.encoding`);
48
+ return MediaProtocol.json(mergeJsonRecords({ text: request.text }, request.http?.body) ?? {}, yield* queryParameters(request));
49
+ });
50
+ // ---------------------------------------------------------------------------
51
+ // 6. Stream parsing
52
+ // ---------------------------------------------------------------------------
53
+ const HEADERLESS_ENCODINGS = {
54
+ linear16: "pcm_s16le",
55
+ mulaw: "pcm_mulaw",
56
+ alaw: "pcm_alaw",
57
+ };
58
+ const finish = (state, context) => {
59
+ const headers = context.http.headers;
60
+ const mediaType = headers["content-type"];
61
+ const format = audioFormat(context.request);
62
+ const encoding = HEADERLESS_ENCODINGS[format.encoding ?? ""];
63
+ const requestID = headers["dg-request-id"];
64
+ const modelName = headers["dg-model-name"];
65
+ return SpeechStream.finish(route, state, {
66
+ ...(format.container === "none" && encoding !== undefined
67
+ ? SpeechStream.pcm(encoding, SpeechStream.sampleRate(mediaType), mediaType)
68
+ : // Deepgram's default encoding is MP3; WAV is a container around any encoding.
69
+ { mediaType, info: { format: format.container === "wav" ? "wav" : (format.encoding ?? "mp3") } }),
70
+ usage: SpeechStream.headerUsage("characters", headers["dg-char-count"]),
71
+ providerMetadata: requestID === undefined && modelName === undefined
72
+ ? undefined
73
+ : { deepgram: { requestId: requestID, modelName } },
74
+ });
75
+ };
76
+ // ---------------------------------------------------------------------------
77
+ // 7. Protocol and route
78
+ // ---------------------------------------------------------------------------
79
+ export const protocol = MediaProtocol.stream(route, {
80
+ unsupported: ["voice", "language", "instructions"],
81
+ body: { from: fromRequest },
82
+ frames: (bytes) => bytes,
83
+ initial: () => ({ chunks: [] }),
84
+ step: (state, frame) => Effect.succeed(SpeechStream.delta(state, frame)),
85
+ finish,
86
+ });
87
+ export const model = (input) => SpeechModel.fromRoute({ protocol, baseURL: DEFAULT_BASE_URL, path: PATH }, input);
88
+ export const DeepgramSpeech = {
89
+ protocol,
90
+ model,
91
+ };
@@ -0,0 +1,26 @@
1
+ import { MediaProtocol } from "../route/media-protocol.js";
2
+ import { MediaRoute } from "../route/media.js";
3
+ import { type OpenString } from "../schema/index.js";
4
+ import { TranscriptionModel, TranscriptionResponse, type TranscriptionRequestFor } from "../transcription.js";
5
+ export declare const DEFAULT_BASE_URL = "https://api.deepgram.com";
6
+ export declare const PATH = "/v1/listen";
7
+ export type DeepgramTranscriptionOptions = {
8
+ readonly smart_format?: boolean;
9
+ readonly punctuate?: boolean;
10
+ readonly paragraphs?: boolean;
11
+ readonly utterances?: boolean;
12
+ readonly detect_language?: boolean | ReadonlyArray<string>;
13
+ readonly keyterm?: ReadonlyArray<string>;
14
+ readonly diarize_model?: OpenString<"latest" | "v1" | "v2">;
15
+ readonly filler_words?: boolean;
16
+ readonly numerals?: boolean;
17
+ readonly mip_opt_out?: boolean;
18
+ readonly tag?: string | ReadonlyArray<string>;
19
+ } & Record<string, unknown>;
20
+ export type Request = TranscriptionRequestFor<DeepgramTranscriptionOptions>;
21
+ export declare const protocol: MediaProtocol.Inline<Request, TranscriptionResponse>;
22
+ export declare const model: (input: MediaRoute.ModelInput) => TranscriptionModel<DeepgramTranscriptionOptions>;
23
+ export declare const DeepgramTranscription: {
24
+ readonly protocol: MediaProtocol.Inline<Request, TranscriptionResponse>;
25
+ readonly model: (input: MediaRoute.ModelInput) => TranscriptionModel<DeepgramTranscriptionOptions>;
26
+ };
@@ -0,0 +1,125 @@
1
+ import { Effect, Schema } from "effect";
2
+ import { MediaProtocol } from "../route/media-protocol.js";
3
+ import { MediaRoute } from "../route/media.js";
4
+ import { mergeJsonRecords } from "../schema/index.js";
5
+ import { TranscriptionModel, TranscriptionResponse } from "../transcription.js";
6
+ import { ProviderShared } from "./shared.js";
7
+ import { MediaInput } from "./utils/media-input.js";
8
+ const route = MediaProtocol.identity({ id: "deepgram-transcription", name: "Deepgram", provider: "deepgram" });
9
+ export const DEFAULT_BASE_URL = "https://api.deepgram.com";
10
+ export const PATH = "/v1/listen";
11
+ // ---------------------------------------------------------------------------
12
+ // 2. Response schema
13
+ // ---------------------------------------------------------------------------
14
+ const Word = Schema.Struct({
15
+ word: Schema.String,
16
+ start: Schema.Number,
17
+ end: Schema.Number,
18
+ confidence: Schema.optional(Schema.Number),
19
+ speaker: Schema.optional(Schema.Number),
20
+ punctuated_word: Schema.optional(Schema.String),
21
+ });
22
+ const ListenResponse = Schema.Struct({
23
+ metadata: Schema.optional(Schema.Struct({ request_id: Schema.optional(Schema.String), duration: Schema.optional(Schema.Number) })),
24
+ results: Schema.Struct({
25
+ channels: Schema.Array(Schema.Struct({
26
+ alternatives: Schema.optional(Schema.Array(Schema.Struct({ transcript: Schema.String, words: Schema.optional(Schema.Array(Word)) }))),
27
+ detected_language: Schema.optional(Schema.String),
28
+ })),
29
+ utterances: Schema.optional(Schema.Array(Schema.Struct({
30
+ start: Schema.Number,
31
+ end: Schema.Number,
32
+ transcript: Schema.String,
33
+ speaker: Schema.optional(Schema.Number),
34
+ words: Schema.optional(Schema.Array(Word)),
35
+ }))),
36
+ }),
37
+ });
38
+ // ---------------------------------------------------------------------------
39
+ // 5. Request body construction
40
+ // ---------------------------------------------------------------------------
41
+ const query = (request) => MediaInput.query(route.id, mergeJsonRecords({
42
+ model: request.model.id,
43
+ smart_format: true,
44
+ language: request.language,
45
+ // Deepgram assumes English unless asked to detect, unlike the other routes' auto-detection.
46
+ detect_language: request.language === undefined ? true : undefined,
47
+ // `diarize=true` is deprecated in favor of choosing a diarization model.
48
+ diarize_model: request.diarize === true ? "latest" : undefined,
49
+ utterances: request.diarize === true || request.timestamps === "segment" ? true : undefined,
50
+ }, request.providerOptions) ?? {});
51
+ const fromRequest = Effect.fn("DeepgramTranscription.fromRequest")(function* (request) {
52
+ const url = ProviderShared.mediaUrl(request.audio);
53
+ if (url !== undefined)
54
+ return MediaProtocol.json(mergeJsonRecords({ url }, request.http?.body) ?? {}, yield* query(request));
55
+ if (request.http?.body !== undefined)
56
+ return yield* ProviderShared.invalidRequest(`${route.name} sends inline audio as the raw body, so http.body cannot apply`);
57
+ const audio = yield* MediaInput.inlineBytes(route.id, request.audio);
58
+ return MediaProtocol.binary(audio, request.audio.mediaType, yield* query(request));
59
+ });
60
+ // ---------------------------------------------------------------------------
61
+ // 6. Response decoding
62
+ // ---------------------------------------------------------------------------
63
+ const decodeListen = route.decodeJson(ListenResponse);
64
+ const speaker = (value) => (value === undefined ? undefined : String(value));
65
+ const wordText = (word) => word.punctuated_word ?? word.word;
66
+ // Utterances split on pauses, not speakers: the v2 diarizer labels a whole utterance with one speaker even when its
67
+ // words change speaker, so segments split each utterance at speaker changes.
68
+ const speakerTurns = (words) => words.reduce((turns, word) => {
69
+ const last = turns.at(-1);
70
+ if (last === undefined || last[0].speaker !== word.speaker)
71
+ return [...turns, [word]];
72
+ last.push(word);
73
+ return turns;
74
+ }, []);
75
+ const decodeResponse = Effect.fn("DeepgramTranscription.decodeResponse")(function* (response) {
76
+ const output = yield* decodeListen(response);
77
+ const channel = output.value.results.channels[0];
78
+ const alternative = channel?.alternatives?.[0];
79
+ if (alternative === undefined)
80
+ return yield* output.invalid(`${route.name} returned no transcript`);
81
+ const duration = output.value.metadata?.duration;
82
+ const requestID = output.value.metadata?.request_id;
83
+ return new TranscriptionResponse({
84
+ text: alternative.transcript,
85
+ segments: output.value.results.utterances?.flatMap((utterance) => utterance.words === undefined || utterance.words.length === 0
86
+ ? [
87
+ {
88
+ text: utterance.transcript,
89
+ startSeconds: utterance.start,
90
+ endSeconds: utterance.end,
91
+ speaker: speaker(utterance.speaker),
92
+ },
93
+ ]
94
+ : speakerTurns(utterance.words).map((turn) => ({
95
+ text: turn.map(wordText).join(" "),
96
+ startSeconds: turn[0].start,
97
+ endSeconds: turn[turn.length - 1].end,
98
+ speaker: speaker(turn[0].speaker),
99
+ }))),
100
+ words: alternative.words?.map((word) => ({
101
+ text: wordText(word),
102
+ startSeconds: word.start,
103
+ endSeconds: word.end,
104
+ speaker: speaker(word.speaker),
105
+ confidence: word.confidence,
106
+ })),
107
+ language: channel?.detected_language?.toLowerCase(),
108
+ durationSeconds: duration,
109
+ usage: duration === undefined ? undefined : { type: "seconds", seconds: duration },
110
+ providerMetadata: requestID === undefined ? undefined : { deepgram: { requestId: requestID } },
111
+ });
112
+ });
113
+ // ---------------------------------------------------------------------------
114
+ // 7. Protocol and route
115
+ // ---------------------------------------------------------------------------
116
+ export const protocol = MediaProtocol.inline(route, {
117
+ unsupported: ["prompt", "speakers"],
118
+ body: { from: fromRequest },
119
+ response: { decode: decodeResponse },
120
+ });
121
+ export const model = (input) => TranscriptionModel.fromRoute({ protocol, baseURL: DEFAULT_BASE_URL, path: PATH }, input);
122
+ export const DeepgramTranscription = {
123
+ protocol,
124
+ model,
125
+ };
@@ -0,0 +1,138 @@
1
+ import { MediaProtocol } from "../route/media-protocol.js";
2
+ import { MediaRoute } from "../route/media.js";
3
+ import { type OpenString } from "../schema/index.js";
4
+ import { SpeechModel, type SpeechRequestFor } from "../speech.js";
5
+ import { SpeechStream } from "./utils/speech-stream.js";
6
+ export declare const DEFAULT_BASE_URL = "https://api.elevenlabs.io";
7
+ export declare const PATH = "/v1/text-to-speech";
8
+ export type ElevenLabsOutputFormat = OpenString<"mp3_22050_32" | "mp3_24000_48" | "mp3_44100_32" | "mp3_44100_64" | "mp3_44100_96" | "mp3_44100_128" | "mp3_44100_192" | "pcm_8000" | "pcm_16000" | "pcm_22050" | "pcm_24000" | "pcm_32000" | "pcm_44100" | "pcm_48000" | "wav_8000" | "wav_16000" | "wav_22050" | "wav_24000" | "wav_32000" | "wav_44100" | "wav_48000" | "ulaw_8000" | "alaw_8000" | "opus_48000_32" | "opus_48000_64" | "opus_48000_96" | "opus_48000_128" | "opus_48000_192">;
9
+ export type ElevenLabsSpeechOptions = {
10
+ readonly outputFormat?: ElevenLabsOutputFormat;
11
+ readonly voice_settings?: {
12
+ readonly stability?: number;
13
+ readonly similarity_boost?: number;
14
+ readonly style?: number;
15
+ readonly use_speaker_boost?: boolean;
16
+ };
17
+ readonly seed?: number;
18
+ readonly apply_text_normalization?: OpenString<"auto" | "on" | "off">;
19
+ } & Record<string, unknown>;
20
+ export type Request = SpeechRequestFor<ElevenLabsSpeechOptions>;
21
+ export declare const protocol: MediaProtocol.Streamed<Request, {
22
+ readonly id: string;
23
+ readonly type: "generation-queued";
24
+ readonly position?: number | undefined;
25
+ } | {
26
+ readonly id: string;
27
+ readonly type: "generation-progress";
28
+ readonly progress?: number | undefined;
29
+ } | {
30
+ readonly type: "audio-delta";
31
+ readonly chunk: Uint8Array<ArrayBufferLike>;
32
+ } | {
33
+ readonly type: "timestamps";
34
+ readonly items: readonly {
35
+ readonly text: string;
36
+ readonly startSeconds: number;
37
+ readonly endSeconds: number;
38
+ }[];
39
+ } | {
40
+ readonly type: "finish";
41
+ readonly audio: import("../media.js").Asset;
42
+ readonly providerMetadata?: {
43
+ readonly [x: string]: {
44
+ readonly [x: string]: unknown;
45
+ };
46
+ } | undefined;
47
+ readonly usage?: {
48
+ readonly type: "tokens";
49
+ readonly input?: number | undefined;
50
+ readonly output?: number | undefined;
51
+ readonly total?: number | undefined;
52
+ readonly details?: {
53
+ readonly [x: string]: unknown;
54
+ } | undefined;
55
+ } | {
56
+ readonly type: "seconds";
57
+ readonly seconds: number;
58
+ } | {
59
+ readonly type: "characters";
60
+ readonly characters: number;
61
+ } | {
62
+ readonly type: "credits";
63
+ readonly credits: number;
64
+ } | {
65
+ readonly type: "compute";
66
+ readonly seconds: number;
67
+ } | undefined;
68
+ readonly notices?: readonly {
69
+ readonly type: "other" | "moderated" | "filtered";
70
+ readonly message: string;
71
+ readonly providerMetadata?: {
72
+ readonly [x: string]: {
73
+ readonly [x: string]: unknown;
74
+ };
75
+ } | undefined;
76
+ }[] | undefined;
77
+ }, string | Uint8Array<ArrayBufferLike>, SpeechStream.Audio>;
78
+ export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<ElevenLabsSpeechOptions>;
79
+ export declare const ElevenLabsSpeech: {
80
+ readonly protocol: MediaProtocol.Streamed<Request, {
81
+ readonly id: string;
82
+ readonly type: "generation-queued";
83
+ readonly position?: number | undefined;
84
+ } | {
85
+ readonly id: string;
86
+ readonly type: "generation-progress";
87
+ readonly progress?: number | undefined;
88
+ } | {
89
+ readonly type: "audio-delta";
90
+ readonly chunk: Uint8Array<ArrayBufferLike>;
91
+ } | {
92
+ readonly type: "timestamps";
93
+ readonly items: readonly {
94
+ readonly text: string;
95
+ readonly startSeconds: number;
96
+ readonly endSeconds: number;
97
+ }[];
98
+ } | {
99
+ readonly type: "finish";
100
+ readonly audio: import("../media.js").Asset;
101
+ readonly providerMetadata?: {
102
+ readonly [x: string]: {
103
+ readonly [x: string]: unknown;
104
+ };
105
+ } | undefined;
106
+ readonly usage?: {
107
+ readonly type: "tokens";
108
+ readonly input?: number | undefined;
109
+ readonly output?: number | undefined;
110
+ readonly total?: number | undefined;
111
+ readonly details?: {
112
+ readonly [x: string]: unknown;
113
+ } | undefined;
114
+ } | {
115
+ readonly type: "seconds";
116
+ readonly seconds: number;
117
+ } | {
118
+ readonly type: "characters";
119
+ readonly characters: number;
120
+ } | {
121
+ readonly type: "credits";
122
+ readonly credits: number;
123
+ } | {
124
+ readonly type: "compute";
125
+ readonly seconds: number;
126
+ } | undefined;
127
+ readonly notices?: readonly {
128
+ readonly type: "other" | "moderated" | "filtered";
129
+ readonly message: string;
130
+ readonly providerMetadata?: {
131
+ readonly [x: string]: {
132
+ readonly [x: string]: unknown;
133
+ };
134
+ } | undefined;
135
+ }[] | undefined;
136
+ }, string | Uint8Array<ArrayBufferLike>, SpeechStream.Audio>;
137
+ readonly model: (input: MediaRoute.ModelInput) => SpeechModel<ElevenLabsSpeechOptions>;
138
+ };
@@ -0,0 +1,111 @@
1
+ import { Effect, Schema } from "effect";
2
+ import { Framing } from "../route/framing.js";
3
+ import { MediaProtocol } from "../route/media-protocol.js";
4
+ import { MediaRoute } from "../route/media.js";
5
+ import { mergeJsonRecords } from "../schema/index.js";
6
+ import { SpeechModel } from "../speech.js";
7
+ import { ProviderShared, optionalNull } from "./shared.js";
8
+ import { SpeechStream } from "./utils/speech-stream.js";
9
+ const route = MediaProtocol.identity({ id: "elevenlabs-speech", name: "ElevenLabs", provider: "elevenlabs" });
10
+ export const DEFAULT_BASE_URL = "https://api.elevenlabs.io";
11
+ export const PATH = "/v1/text-to-speech";
12
+ // ---------------------------------------------------------------------------
13
+ // 3. Streaming event schema
14
+ // ---------------------------------------------------------------------------
15
+ const Alignment = Schema.Struct({
16
+ characters: Schema.Array(Schema.String),
17
+ character_start_times_seconds: Schema.Array(Schema.Number),
18
+ character_end_times_seconds: Schema.Array(Schema.Number),
19
+ });
20
+ const TimestampedAudio = Schema.Struct({
21
+ audio_base64: Schema.Uint8ArrayFromBase64,
22
+ alignment: optionalNull(Alignment),
23
+ });
24
+ const decodeRecord = route.decodeFrame(TimestampedAudio);
25
+ // ---------------------------------------------------------------------------
26
+ // 5. Request body construction
27
+ // ---------------------------------------------------------------------------
28
+ const OUTPUT_FORMATS = {
29
+ mp3: "mp3_44100_128",
30
+ pcm: "pcm_24000",
31
+ wav: "wav_24000",
32
+ opus: "opus_48000_64",
33
+ };
34
+ /** WAV is served only by the non-streaming endpoints. */
35
+ const outputFormat = Effect.fn("ElevenLabsSpeech.outputFormat")(function* (request) {
36
+ const format = request.providerOptions?.outputFormat ?? OUTPUT_FORMATS[request.format ?? "mp3"];
37
+ if (format === undefined)
38
+ return yield* route.unsupported("media.format", `${route.name} has no default output format for "${request.format}"; pass providerOptions.outputFormat`);
39
+ if (request.mode === "stream" && format.startsWith("wav_"))
40
+ return yield* route.unsupported("media.format", `${route.name} streams mp3, pcm, opus, ulaw, and alaw but not "${format}"; use generate for WAV`);
41
+ return format;
42
+ });
43
+ const fromRequest = Effect.fn("ElevenLabsSpeech.fromRequest")(function* (request) {
44
+ if (request.voice === undefined)
45
+ return yield* ProviderShared.invalidRequest(`${route.name} requires a voice id; pass it as \`voice\``);
46
+ const { outputFormat: _outputFormat, ...native } = request.providerOptions ?? {};
47
+ return MediaProtocol.json(mergeJsonRecords({
48
+ text: request.text,
49
+ model_id: request.model.id,
50
+ language_code: request.language,
51
+ voice_settings: request.speed === undefined ? undefined : { speed: request.speed },
52
+ }, native, request.http?.body) ?? {}, { output_format: yield* outputFormat(request) });
53
+ });
54
+ const path = (request) => `${PATH}/${encodeURIComponent(SpeechStream.voiceID(request.voice) ?? "")}${request.mode === "stream" ? "/stream" : ""}${request.timestamps === true ? "/with-timestamps" : ""}`;
55
+ // ---------------------------------------------------------------------------
56
+ // 6. Stream parsing
57
+ // ---------------------------------------------------------------------------
58
+ const onRecord = Effect.fn("ElevenLabsSpeech.onRecord")(function* (state, frame) {
59
+ const record = yield* decodeRecord(frame);
60
+ const [next, events] = SpeechStream.delta(state, record.audio_base64);
61
+ const alignment = record.alignment;
62
+ if (!alignment)
63
+ return [next, events];
64
+ return [
65
+ next,
66
+ [
67
+ ...events,
68
+ ...SpeechStream.timestamps(alignment.characters, alignment.character_start_times_seconds, alignment.character_end_times_seconds),
69
+ ],
70
+ ];
71
+ });
72
+ const PCM_CODECS = {
73
+ pcm: "pcm_s16le",
74
+ ulaw: "pcm_mulaw",
75
+ alaw: "pcm_alaw",
76
+ };
77
+ const describeOutput = (format) => {
78
+ const [codec = format, rate] = format.split("_");
79
+ const sampleRate = rate === undefined ? undefined : Number(rate);
80
+ const encoding = PCM_CODECS[codec];
81
+ return encoding === undefined ? SpeechStream.container(codec, sampleRate) : SpeechStream.pcm(encoding, sampleRate);
82
+ };
83
+ const finish = Effect.fn("ElevenLabsSpeech.finish")(function* (state, context) {
84
+ const requestID = context.http.headers["request-id"];
85
+ return yield* SpeechStream.finish(route, state, {
86
+ ...describeOutput(yield* outputFormat(context.request)),
87
+ // `character-cost` is billed credits, not a character count (3 for 20 characters on `eleven_flash_v2_5`).
88
+ usage: SpeechStream.headerUsage("credits", context.http.headers["character-cost"]),
89
+ providerMetadata: requestID === undefined ? undefined : { elevenlabs: { requestId: requestID } },
90
+ });
91
+ });
92
+ // ---------------------------------------------------------------------------
93
+ // 7. Protocol and route
94
+ // ---------------------------------------------------------------------------
95
+ export const protocol = MediaProtocol.stream(route, {
96
+ unsupported: ["instructions"],
97
+ body: { from: fromRequest },
98
+ frames: (bytes, context) => {
99
+ if (context.request.timestamps !== true)
100
+ return bytes;
101
+ return context.request.mode === "stream" ? Framing.lines.frame(bytes) : Framing.document.frame(bytes);
102
+ },
103
+ initial: () => ({ chunks: [] }),
104
+ step: SpeechStream.step(onRecord),
105
+ finish,
106
+ });
107
+ export const model = (input) => SpeechModel.fromRoute({ protocol, baseURL: DEFAULT_BASE_URL, path: ({ request }) => path(request) }, input);
108
+ export const ElevenLabsSpeech = {
109
+ protocol,
110
+ model,
111
+ };
@@ -0,0 +1,25 @@
1
+ import { ImageModel, ImageResponse, type ImageRequestFor } from "../image.js";
2
+ import { MediaProtocol } from "../route/media-protocol.js";
3
+ import { MediaRoute } from "../route/media.js";
4
+ import { type OpenString } from "../schema/index.js";
5
+ export type FalImageOptions = {
6
+ readonly image_size?: OpenString<"square_hd" | "square" | "portrait_4_3" | "portrait_16_9" | "landscape_4_3" | "landscape_16_9">;
7
+ readonly enable_safety_checker?: boolean;
8
+ } & Record<string, unknown>;
9
+ export type Request = ImageRequestFor<FalImageOptions>;
10
+ export declare const protocol: MediaProtocol.Queued<Request, ImageResponse, {
11
+ readonly requestID: string;
12
+ readonly statusURL: string;
13
+ readonly responseURL: string;
14
+ readonly cancelURL: string;
15
+ }>;
16
+ export declare const model: (input: MediaRoute.ModelInput) => ImageModel<FalImageOptions>;
17
+ export declare const FalImages: {
18
+ readonly protocol: MediaProtocol.Queued<Request, ImageResponse, {
19
+ readonly requestID: string;
20
+ readonly statusURL: string;
21
+ readonly responseURL: string;
22
+ readonly cancelURL: string;
23
+ }>;
24
+ readonly model: (input: MediaRoute.ModelInput) => ImageModel<FalImageOptions>;
25
+ };