@opencode/ai 2.0.16 → 2.0.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +187 -129
- package/dist/ai-client.d.ts +8 -0
- package/dist/ai-client.js +12 -0
- package/dist/experimental/evaluation-client.d.ts +3 -3
- package/dist/experimental/evaluation-client.js +1 -1
- package/dist/experimental/evaluation.js +1 -1
- package/dist/generation.d.ts +6 -10
- package/dist/generation.js +17 -20
- package/dist/image-client.d.ts +59 -24
- package/dist/image-client.js +16 -40
- package/dist/image.d.ts +15 -27
- package/dist/image.js +4 -15
- package/dist/index.d.ts +2 -1
- package/dist/index.js +1 -0
- package/dist/llm.d.ts +7 -5
- package/dist/llm.js +10 -4
- package/dist/media-client.d.ts +30 -0
- package/dist/media-client.js +51 -0
- package/dist/media-model.d.ts +9 -10
- package/dist/media-model.js +10 -12
- package/dist/media.js +2 -2
- package/dist/promise.d.ts +68 -30
- package/dist/promise.js +52 -31
- package/dist/protocols/alibaba-chat.js +4 -1
- package/dist/protocols/alibaba-messages.d.ts +1 -1
- package/dist/protocols/alibaba-messages.js +6 -4
- package/dist/protocols/anthropic-messages.d.ts +34 -34
- package/dist/protocols/anthropic-messages.js +14 -8
- package/dist/protocols/assemblyai-transcription.d.ts +2 -1
- package/dist/protocols/assemblyai-transcription.js +14 -16
- package/dist/protocols/bedrock-converse.d.ts +7 -0
- package/dist/protocols/bedrock-converse.js +37 -11
- package/dist/protocols/bfl-images.d.ts +7 -1
- package/dist/protocols/bfl-images.js +36 -36
- package/dist/protocols/cartesia-speech.d.ts +18 -2
- package/dist/protocols/cartesia-speech.js +10 -16
- package/dist/protocols/deepgram-speech.d.ts +19 -3
- package/dist/protocols/deepgram-speech.js +11 -12
- package/dist/protocols/deepgram-transcription.d.ts +2 -1
- package/dist/protocols/deepgram-transcription.js +9 -13
- package/dist/protocols/elevenlabs-speech.d.ts +19 -3
- package/dist/protocols/elevenlabs-speech.js +9 -13
- package/dist/protocols/fal-images.d.ts +2 -1
- package/dist/protocols/fal-images.js +30 -40
- package/dist/protocols/fal-video.d.ts +2 -2
- package/dist/protocols/fal-video.js +9 -24
- package/dist/protocols/gemini.d.ts +13 -13
- package/dist/protocols/gemini.js +16 -7
- package/dist/protocols/google-images.d.ts +3 -3
- package/dist/protocols/google-images.js +11 -21
- package/dist/protocols/google-speech.d.ts +16 -0
- package/dist/protocols/google-speech.js +20 -16
- package/dist/protocols/google-transcription.d.ts +2 -1
- package/dist/protocols/google-transcription.js +8 -20
- package/dist/protocols/google-video.d.ts +2 -2
- package/dist/protocols/google-video.js +14 -30
- package/dist/protocols/meta-images.d.ts +3 -4
- package/dist/protocols/meta-images.js +12 -21
- package/dist/protocols/meta-messages.d.ts +5 -5
- package/dist/protocols/meta-responses.js +6 -4
- package/dist/protocols/open-responses.d.ts +4 -3
- package/dist/protocols/open-responses.js +4 -8
- package/dist/protocols/openai-chat.js +3 -4
- package/dist/protocols/openai-images.d.ts +7 -5
- package/dist/protocols/openai-images.js +65 -66
- package/dist/protocols/openai-responses.d.ts +10 -10
- package/dist/protocols/openai-responses.js +31 -30
- package/dist/protocols/openai-speech.d.ts +21 -1
- package/dist/protocols/openai-speech.js +11 -12
- package/dist/protocols/openai-transcription.d.ts +4 -0
- package/dist/protocols/openai-transcription.js +44 -32
- package/dist/protocols/replicate-images.js +11 -17
- package/dist/protocols/runway-video.d.ts +3 -3
- package/dist/protocols/runway-video.js +11 -17
- package/dist/protocols/shared.d.ts +11 -17
- package/dist/protocols/shared.js +11 -40
- package/dist/protocols/stability-images.d.ts +2 -1
- package/dist/protocols/stability-images.js +24 -36
- package/dist/protocols/utils/fal-queue.d.ts +1 -3
- package/dist/protocols/utils/fal-queue.js +4 -6
- package/dist/protocols/utils/media-input.d.ts +15 -1
- package/dist/protocols/utils/media-input.js +22 -0
- package/dist/protocols/utils/responses-checkpoint.js +3 -7
- package/dist/protocols/utils/responses-compaction.d.ts +3 -1
- package/dist/protocols/utils/responses-compaction.js +16 -3
- package/dist/protocols/utils/speech-stream.d.ts +3 -3
- package/dist/protocols/utils/speech-stream.js +1 -4
- package/dist/protocols/utils/tool-schema.d.ts +2 -2
- package/dist/protocols/utils/tool-schema.js +29 -9
- package/dist/protocols/xai-images.d.ts +4 -4
- package/dist/protocols/xai-images.js +14 -29
- package/dist/protocols/xai-responses.js +1 -1
- package/dist/protocols/xai-video.js +12 -18
- package/dist/protocols/zai-images.d.ts +2 -2
- package/dist/protocols/zai-images.js +11 -14
- package/dist/protocols/zai-messages.d.ts +1 -1
- package/dist/provider-error.js +7 -1
- package/dist/providers/alibaba.d.ts +1 -1
- package/dist/providers/amazon-bedrock.d.ts +2 -0
- package/dist/providers/amazon-bedrock.js +1 -0
- package/dist/providers/anthropic-compatible.d.ts +5 -5
- package/dist/providers/anthropic.d.ts +5 -5
- package/dist/providers/assemblyai.d.ts +1 -1
- package/dist/providers/assemblyai.js +5 -10
- package/dist/providers/azure.d.ts +4 -4
- package/dist/providers/azure.js +2 -2
- package/dist/providers/black-forest-labs.d.ts +1 -1
- package/dist/providers/black-forest-labs.js +5 -10
- package/dist/providers/cartesia.d.ts +1 -1
- package/dist/providers/cartesia.js +5 -10
- package/dist/providers/cloudflare-ai-gateway.d.ts +14 -14
- package/dist/providers/deepgram.d.ts +1 -1
- package/dist/providers/deepgram.js +6 -12
- package/dist/providers/elevenlabs.d.ts +1 -1
- package/dist/providers/elevenlabs.js +5 -10
- package/dist/providers/fal.d.ts +1 -1
- package/dist/providers/fal.js +5 -12
- package/dist/providers/google-vertex-messages.d.ts +5 -5
- package/dist/providers/google-vertex.d.ts +3 -3
- package/dist/providers/google.d.ts +3 -3
- package/dist/providers/google.js +7 -12
- package/dist/providers/meta.d.ts +8 -8
- package/dist/providers/meta.js +4 -8
- package/dist/providers/minimax.d.ts +5 -5
- package/dist/providers/moonshot.d.ts +5 -5
- package/dist/providers/openai-options.d.ts +3 -9
- package/dist/providers/openai-options.js +4 -7
- package/dist/providers/openai.d.ts +9 -10
- package/dist/providers/openai.js +11 -12
- package/dist/providers/opencode-zen.js +1 -1
- package/dist/providers/openrouter.d.ts +6 -7
- package/dist/providers/openrouter.js +10 -5
- package/dist/providers/replicate.d.ts +1 -1
- package/dist/providers/replicate.js +5 -10
- package/dist/providers/runway.d.ts +1 -1
- package/dist/providers/runway.js +5 -10
- package/dist/providers/stability.d.ts +1 -1
- package/dist/providers/stability.js +6 -11
- package/dist/providers/typesafe-ai.js +1 -1
- package/dist/providers/vercel-ai-gateway.js +1 -1
- package/dist/providers/xai.js +5 -10
- package/dist/providers/zai-coding-plan.d.ts +1 -1
- package/dist/providers/zai.js +4 -8
- package/dist/route/auth.d.ts +1 -1
- package/dist/route/auth.js +14 -13
- package/dist/route/client.d.ts +9 -7
- package/dist/route/client.js +6 -8
- package/dist/route/endpoint.d.ts +1 -0
- package/dist/route/endpoint.js +2 -2
- package/dist/route/executor-service.d.ts +4 -2
- package/dist/route/executor-service.js +2 -1
- package/dist/route/executor.d.ts +3 -1
- package/dist/route/executor.js +7 -0
- package/dist/route/framing.d.ts +10 -1
- package/dist/route/framing.js +45 -5
- package/dist/route/index.d.ts +1 -1
- package/dist/route/media-protocol.d.ts +49 -35
- package/dist/route/media-protocol.js +77 -65
- package/dist/route/media.d.ts +10 -16
- package/dist/route/media.js +46 -62
- package/dist/route/protocol.d.ts +3 -1
- package/dist/schema/options.d.ts +6 -3
- package/dist/schema/options.js +7 -3
- package/dist/speech-client.d.ts +62 -17
- package/dist/speech-client.js +17 -21
- package/dist/speech.d.ts +178 -21
- package/dist/speech.js +13 -8
- package/dist/testing.d.ts +2 -2
- package/dist/transcription-client.d.ts +78 -24
- package/dist/transcription-client.js +9 -40
- package/dist/transcription.d.ts +15 -27
- package/dist/transcription.js +4 -10
- package/dist/utils/json.d.ts +4 -0
- package/dist/utils/json.js +4 -0
- package/dist/utils/media-type.d.ts +2 -1
- package/dist/utils/media-type.js +3 -1
- package/dist/video-client.d.ts +55 -24
- package/dist/video-client.js +16 -36
- package/dist/video.d.ts +18 -26
- package/dist/video.js +15 -16
- package/package.json +3 -3
- package/dist/protocols/utils/meta-image.d.ts +0 -2
- package/dist/protocols/utils/meta-image.js +0 -13
- package/dist/protocols/utils/openai-image.d.ts +0 -5
- package/dist/protocols/utils/openai-image.js +0 -18
|
@@ -3,21 +3,27 @@ import { ImageModel, ImageResponse } from "../image.js";
|
|
|
3
3
|
import { Media } from "../media.js";
|
|
4
4
|
import { MediaProtocol } from "../route/media-protocol.js";
|
|
5
5
|
import { MediaRoute } from "../route/media.js";
|
|
6
|
-
import {
|
|
6
|
+
import { mergeJsonRecords } from "../schema/index.js";
|
|
7
7
|
import { ProviderShared, optionalNull } from "./shared.js";
|
|
8
8
|
import { MediaInput } from "./utils/media-input.js";
|
|
9
|
-
const
|
|
10
|
-
const NAME = "Black Forest Labs";
|
|
11
|
-
const PROVIDER = ProviderID.make("black-forest-labs");
|
|
9
|
+
const route = MediaProtocol.identity({ id: "bfl-images", name: "Black Forest Labs", provider: "black-forest-labs" });
|
|
12
10
|
export const DEFAULT_BASE_URL = "https://api.bfl.ai";
|
|
13
11
|
// ---------------------------------------------------------------------------
|
|
14
12
|
// 2. Token and response schemas
|
|
15
13
|
// ---------------------------------------------------------------------------
|
|
16
|
-
/**
|
|
17
|
-
|
|
14
|
+
/**
|
|
15
|
+
* Regional clusters answer on different hosts, so the returned `polling_url` is followed verbatim. BFL reports the
|
|
16
|
+
* credit cost on submit, so it rides on the token; it is optional so tokens persisted before it existed still decode.
|
|
17
|
+
*/
|
|
18
|
+
export const Token = Schema.Struct({
|
|
19
|
+
id: Schema.String,
|
|
20
|
+
pollingURL: Schema.String,
|
|
21
|
+
cost: Schema.optionalKey(Schema.Number),
|
|
22
|
+
});
|
|
18
23
|
const StartResponse = Schema.Struct({
|
|
19
24
|
id: Schema.String,
|
|
20
25
|
polling_url: Schema.String,
|
|
26
|
+
cost: optionalNull(Schema.Number),
|
|
21
27
|
});
|
|
22
28
|
const Result = Schema.Struct({
|
|
23
29
|
id: Schema.String,
|
|
@@ -52,31 +58,25 @@ const capabilities = (model) => {
|
|
|
52
58
|
return { sizing: "dimensions", imageField: "input_image", maxImages: 4, mask: false };
|
|
53
59
|
return { sizing: "dimensions", imageField: "input_image", maxImages: 8, mask: false };
|
|
54
60
|
};
|
|
55
|
-
const unsupported = (model, field, message) => ProviderShared.unsupportedOperation({
|
|
56
|
-
operation: `media.${field}`,
|
|
57
|
-
provider: PROVIDER,
|
|
58
|
-
route: ADAPTER,
|
|
59
|
-
message: `${model} ${message}`,
|
|
60
|
-
});
|
|
61
61
|
const validate = (request, model) => {
|
|
62
62
|
const id = request.model.id;
|
|
63
63
|
const images = request.images?.length ?? 0;
|
|
64
64
|
if (request.n !== undefined && request.n > 1)
|
|
65
|
-
return Effect.fail(unsupported(
|
|
65
|
+
return Effect.fail(route.unsupported("media.n", `${id} generates one image per request; call it once per image`));
|
|
66
66
|
if (request.size !== undefined && model.sizing !== "dimensions")
|
|
67
|
-
return Effect.fail(unsupported(
|
|
67
|
+
return Effect.fail(route.unsupported("media.size", `${id} does not take size (width and height)`));
|
|
68
68
|
if (request.aspectRatio !== undefined && model.sizing !== "aspectRatio")
|
|
69
|
-
return Effect.fail(unsupported(
|
|
69
|
+
return Effect.fail(route.unsupported("media.aspectRatio", `${id} does not take aspectRatio`));
|
|
70
70
|
if (images > model.maxImages)
|
|
71
|
-
return Effect.fail(unsupported(
|
|
71
|
+
return Effect.fail(route.unsupported("media.images", `${id} takes at most ${model.maxImages} images`));
|
|
72
72
|
if (request.mask !== undefined && !model.mask)
|
|
73
|
-
return Effect.fail(unsupported(
|
|
73
|
+
return Effect.fail(route.unsupported("media.mask", `${id} does not inpaint; use flux-pro-1.0-fill`));
|
|
74
74
|
return Effect.void;
|
|
75
75
|
};
|
|
76
76
|
const imageInput = (asset) => {
|
|
77
77
|
const value = asset.inline()?.base64 ?? ProviderShared.mediaUrl(asset);
|
|
78
78
|
if (value === undefined)
|
|
79
|
-
return Effect.fail(ProviderShared.invalidRequest(`${
|
|
79
|
+
return Effect.fail(ProviderShared.invalidRequest(`${route.name} accepts inline images or https URLs`));
|
|
80
80
|
return Effect.succeed(value);
|
|
81
81
|
};
|
|
82
82
|
const fromRequest = Effect.fn("BlackForestLabsImages.fromRequest")(function* (request) {
|
|
@@ -100,11 +100,15 @@ const fromRequest = Effect.fn("BlackForestLabsImages.fromRequest")(function* (re
|
|
|
100
100
|
// ---------------------------------------------------------------------------
|
|
101
101
|
// 6. Response decoding
|
|
102
102
|
// ---------------------------------------------------------------------------
|
|
103
|
-
const decodeStart =
|
|
104
|
-
token: {
|
|
103
|
+
const decodeStart = route.decodeStarted(StartResponse, (value) => ({
|
|
104
|
+
token: {
|
|
105
|
+
id: value.id,
|
|
106
|
+
pollingURL: value.polling_url,
|
|
107
|
+
...(value.cost === undefined || value.cost === null ? {} : { cost: value.cost }),
|
|
108
|
+
},
|
|
105
109
|
snapshot: { id: value.id, status: "queued" },
|
|
106
110
|
}));
|
|
107
|
-
const decodeDocument =
|
|
111
|
+
const decodeDocument = route.decodeJson(Result);
|
|
108
112
|
const decodeStatus = Effect.fn("BlackForestLabsImages.decodeStatus")(function* (response, context) {
|
|
109
113
|
const output = yield* decodeDocument(response);
|
|
110
114
|
return { id: context.token.id, status: yield* MediaProtocol.status(STATUS, output.value.status, output) };
|
|
@@ -114,16 +118,20 @@ const decodeResult = Effect.fn("BlackForestLabsImages.decodeResult")(function* (
|
|
|
114
118
|
const document = output.value;
|
|
115
119
|
const status = yield* MediaProtocol.status(STATUS, document.status, output);
|
|
116
120
|
if (isModerated(document.status))
|
|
117
|
-
return yield* output.contentPolicy(`${
|
|
121
|
+
return yield* output.contentPolicy(`${route.name} moderated the generation`);
|
|
118
122
|
if (status === "failed" || status === "expired")
|
|
119
|
-
return yield* output.ended(status, `${
|
|
120
|
-
if (status !== "completed"
|
|
121
|
-
return yield* output.
|
|
123
|
+
return yield* output.ended(status, `${route.name} generation ${context.token.id} ended with ${document.status}`);
|
|
124
|
+
if (status !== "completed")
|
|
125
|
+
return yield* output.pending(context.token.id);
|
|
126
|
+
if (document.result === undefined || document.result === null)
|
|
127
|
+
return yield* output.invalid(`${route.name} generation ${context.token.id} has no result`);
|
|
122
128
|
const { sample, seed, prompt, ...rest } = document.result;
|
|
129
|
+
// A settled `cost` on the result supersedes the submit-time cost carried on the token.
|
|
130
|
+
const cost = document.cost ?? context.token.cost;
|
|
123
131
|
return new ImageResponse({
|
|
124
132
|
// `sample` is a signed URL that expires 10 minutes after the result is ready, so it is downloaded now.
|
|
125
133
|
images: [yield* context.materialize(Media.url(sample))],
|
|
126
|
-
usage:
|
|
134
|
+
usage: cost === undefined ? undefined : { type: "credits", credits: cost },
|
|
127
135
|
providerMetadata: {
|
|
128
136
|
bfl: { id: context.token.id, seed: seed ?? undefined, prompt: prompt ?? undefined, ...rest },
|
|
129
137
|
},
|
|
@@ -132,21 +140,13 @@ const decodeResult = Effect.fn("BlackForestLabsImages.decodeResult")(function* (
|
|
|
132
140
|
// ---------------------------------------------------------------------------
|
|
133
141
|
// 7. Protocol and route
|
|
134
142
|
// ---------------------------------------------------------------------------
|
|
135
|
-
export const protocol = MediaProtocol.queued({
|
|
136
|
-
id: ADAPTER,
|
|
137
|
-
name: NAME,
|
|
143
|
+
export const protocol = MediaProtocol.queued(route, {
|
|
138
144
|
token: Token,
|
|
139
145
|
start: { body: { from: fromRequest }, decode: decodeStart },
|
|
140
146
|
status: { path: (token) => token.pollingURL, decode: decodeStatus },
|
|
141
147
|
result: { path: (token) => token.pollingURL, decode: decodeResult },
|
|
142
148
|
});
|
|
143
|
-
export const model = (input) => ImageModel.fromRoute({
|
|
144
|
-
id: ADAPTER,
|
|
145
|
-
provider: PROVIDER,
|
|
146
|
-
protocol,
|
|
147
|
-
baseURL: DEFAULT_BASE_URL,
|
|
148
|
-
path: ({ request }) => `/v1/${request.model.id}`,
|
|
149
|
-
}, input);
|
|
149
|
+
export const model = (input) => ImageModel.fromRoute({ protocol, baseURL: DEFAULT_BASE_URL, path: ({ request }) => `/v1/${request.model.id}` }, input);
|
|
150
150
|
export const BlackForestLabsImages = {
|
|
151
151
|
protocol,
|
|
152
152
|
model,
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { MediaProtocol } from "../route/media-protocol.js";
|
|
2
2
|
import { MediaRoute } from "../route/media.js";
|
|
3
|
+
import { type OpenString } from "../schema/index.js";
|
|
3
4
|
import { SpeechModel, type SpeechRequestFor } from "../speech.js";
|
|
4
5
|
import { SpeechStream } from "./utils/speech-stream.js";
|
|
5
6
|
export declare const DEFAULT_BASE_URL = "https://api.cartesia.ai";
|
|
6
7
|
export declare const API_VERSION = "2026-08-14";
|
|
7
8
|
export declare const BYTES_PATH = "/tts/bytes";
|
|
8
9
|
export declare const SSE_PATH = "/tts/sse";
|
|
9
|
-
export type CartesiaSpeechString<Known extends string> = Known | (string & {});
|
|
10
10
|
export type CartesiaEncoding = SpeechStream.PcmEncoding;
|
|
11
11
|
export type CartesiaSpeechOptions = {
|
|
12
12
|
readonly sampleRate?: 8000 | 16000 | 22050 | 24000 | 44100 | 48000;
|
|
@@ -14,7 +14,7 @@ export type CartesiaSpeechOptions = {
|
|
|
14
14
|
readonly encoding?: CartesiaEncoding;
|
|
15
15
|
readonly generation_config?: {
|
|
16
16
|
readonly volume?: number;
|
|
17
|
-
readonly emotion?:
|
|
17
|
+
readonly emotion?: OpenString<"neutral" | "calm" | "angry" | "content" | "sad" | "scared">;
|
|
18
18
|
};
|
|
19
19
|
readonly pronunciation_dict_id?: string;
|
|
20
20
|
} & Record<string, unknown>;
|
|
@@ -23,6 +23,14 @@ interface State extends SpeechStream.Audio {
|
|
|
23
23
|
readonly done: boolean;
|
|
24
24
|
}
|
|
25
25
|
export declare const protocol: MediaProtocol.Streamed<Request, {
|
|
26
|
+
readonly id: string;
|
|
27
|
+
readonly type: "generation-queued";
|
|
28
|
+
readonly position?: number | undefined;
|
|
29
|
+
} | {
|
|
30
|
+
readonly id: string;
|
|
31
|
+
readonly type: "generation-progress";
|
|
32
|
+
readonly progress?: number | undefined;
|
|
33
|
+
} | {
|
|
26
34
|
readonly type: "audio-delta";
|
|
27
35
|
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
28
36
|
} | {
|
|
@@ -74,6 +82,14 @@ export declare const protocol: MediaProtocol.Streamed<Request, {
|
|
|
74
82
|
export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<CartesiaSpeechOptions>;
|
|
75
83
|
export declare const CartesiaSpeech: {
|
|
76
84
|
readonly protocol: MediaProtocol.Streamed<Request, {
|
|
85
|
+
readonly id: string;
|
|
86
|
+
readonly type: "generation-queued";
|
|
87
|
+
readonly position?: number | undefined;
|
|
88
|
+
} | {
|
|
89
|
+
readonly id: string;
|
|
90
|
+
readonly type: "generation-progress";
|
|
91
|
+
readonly progress?: number | undefined;
|
|
92
|
+
} | {
|
|
77
93
|
readonly type: "audio-delta";
|
|
78
94
|
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
79
95
|
} | {
|
|
@@ -3,13 +3,11 @@ import { classifyProviderFailure } from "../provider-error.js";
|
|
|
3
3
|
import { Framing } from "../route/framing.js";
|
|
4
4
|
import { MediaProtocol } from "../route/media-protocol.js";
|
|
5
5
|
import { MediaRoute } from "../route/media.js";
|
|
6
|
-
import { AIError,
|
|
6
|
+
import { AIError, mergeJsonRecords } from "../schema/index.js";
|
|
7
7
|
import { SpeechModel } from "../speech.js";
|
|
8
8
|
import { ProviderShared, optionalNull } from "./shared.js";
|
|
9
9
|
import { SpeechStream } from "./utils/speech-stream.js";
|
|
10
|
-
const
|
|
11
|
-
const NAME = "Cartesia";
|
|
12
|
-
const PROVIDER = ProviderID.make("cartesia");
|
|
10
|
+
const route = MediaProtocol.identity({ id: "cartesia-speech", name: "Cartesia", provider: "cartesia" });
|
|
13
11
|
export const DEFAULT_BASE_URL = "https://api.cartesia.ai";
|
|
14
12
|
export const API_VERSION = "2026-08-14";
|
|
15
13
|
export const BYTES_PATH = "/tts/bytes";
|
|
@@ -33,7 +31,7 @@ const SseEvent = Schema.Struct({
|
|
|
33
31
|
message: Schema.optional(Schema.String),
|
|
34
32
|
error_code: optionalNull(Schema.String),
|
|
35
33
|
});
|
|
36
|
-
const decodeEvent =
|
|
34
|
+
const decodeEvent = route.decodeFrame(SseEvent);
|
|
37
35
|
// ---------------------------------------------------------------------------
|
|
38
36
|
// 5. Request body construction
|
|
39
37
|
// ---------------------------------------------------------------------------
|
|
@@ -45,9 +43,9 @@ const outputFormat = Effect.fn("CartesiaSpeech.outputFormat")(function* (request
|
|
|
45
43
|
const format = request.format ?? (sse ? "pcm" : "mp3");
|
|
46
44
|
const container = CONTAINERS[format];
|
|
47
45
|
if (container === undefined)
|
|
48
|
-
return yield*
|
|
46
|
+
return yield* route.unsupported("media.format", `${route.name} supports the pcm, wav, and mp3 formats, not "${format}"`);
|
|
49
47
|
if (sse && container !== "raw")
|
|
50
|
-
return yield*
|
|
48
|
+
return yield* route.unsupported("media.format", `${route.name} streams and timestamps only raw PCM; request format "pcm" instead of "${format}"`);
|
|
51
49
|
const sampleRate = request.providerOptions?.sampleRate ?? DEFAULT_SAMPLE_RATE;
|
|
52
50
|
if (container === "mp3")
|
|
53
51
|
return { container, sample_rate: sampleRate, bit_rate: request.providerOptions?.bitRate ?? DEFAULT_BIT_RATE };
|
|
@@ -56,7 +54,7 @@ const outputFormat = Effect.fn("CartesiaSpeech.outputFormat")(function* (request
|
|
|
56
54
|
const fromRequest = Effect.fn("CartesiaSpeech.fromRequest")(function* (request) {
|
|
57
55
|
const voice = SpeechStream.voiceID(request.voice);
|
|
58
56
|
if (voice === undefined)
|
|
59
|
-
return yield* ProviderShared.invalidRequest(`${
|
|
57
|
+
return yield* ProviderShared.invalidRequest(`${route.name} requires a voice id; pass it as \`voice\``);
|
|
60
58
|
const { sampleRate: _sampleRate, bitRate: _bitRate, encoding: _encoding, ...native } = request.providerOptions ?? {};
|
|
61
59
|
return MediaProtocol.json(mergeJsonRecords({
|
|
62
60
|
model_id: request.model.id,
|
|
@@ -84,7 +82,7 @@ const onEvent = Effect.fn("CartesiaSpeech.onEvent")(function* (state, frame) {
|
|
|
84
82
|
if (event.type === "error")
|
|
85
83
|
return yield* new AIError({
|
|
86
84
|
reason: classifyProviderFailure({
|
|
87
|
-
message: `${
|
|
85
|
+
message: `${route.name} stream failed${event.title === undefined ? "" : ` (${event.title})`}: ${event.message ?? "unknown error"}`,
|
|
88
86
|
status: event.status_code,
|
|
89
87
|
rawBody: frame,
|
|
90
88
|
}),
|
|
@@ -93,18 +91,16 @@ const onEvent = Effect.fn("CartesiaSpeech.onEvent")(function* (state, frame) {
|
|
|
93
91
|
});
|
|
94
92
|
const finish = Effect.fn("CartesiaSpeech.finish")(function* (state, context) {
|
|
95
93
|
if (usesSse(context.request) && !state.done)
|
|
96
|
-
return yield*
|
|
94
|
+
return yield* route.incomplete();
|
|
97
95
|
const format = yield* outputFormat(context.request);
|
|
98
|
-
return yield* SpeechStream.finish(
|
|
96
|
+
return yield* SpeechStream.finish(route, state, format.container === "raw"
|
|
99
97
|
? SpeechStream.pcm(format.encoding, format.sample_rate)
|
|
100
98
|
: SpeechStream.container(format.container, format.sample_rate));
|
|
101
99
|
});
|
|
102
100
|
// ---------------------------------------------------------------------------
|
|
103
101
|
// 7. Protocol and route
|
|
104
102
|
// ---------------------------------------------------------------------------
|
|
105
|
-
export const protocol = MediaProtocol.stream({
|
|
106
|
-
id: ADAPTER,
|
|
107
|
-
name: NAME,
|
|
103
|
+
export const protocol = MediaProtocol.stream(route, {
|
|
108
104
|
unsupported: ["instructions"],
|
|
109
105
|
body: { from: fromRequest },
|
|
110
106
|
frames: (bytes, context) => (usesSse(context.request) ? Framing.sse.frame(bytes) : bytes),
|
|
@@ -113,8 +109,6 @@ export const protocol = MediaProtocol.stream({
|
|
|
113
109
|
finish,
|
|
114
110
|
});
|
|
115
111
|
export const model = (input) => SpeechModel.fromRoute({
|
|
116
|
-
id: ADAPTER,
|
|
117
|
-
provider: PROVIDER,
|
|
118
112
|
protocol,
|
|
119
113
|
baseURL: DEFAULT_BASE_URL,
|
|
120
114
|
headers: { "Cartesia-Version": API_VERSION },
|
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { MediaProtocol } from "../route/media-protocol.js";
|
|
2
2
|
import { MediaRoute } from "../route/media.js";
|
|
3
|
+
import { type OpenString } from "../schema/index.js";
|
|
3
4
|
import { SpeechModel, type SpeechRequestFor } from "../speech.js";
|
|
4
5
|
import { SpeechStream } from "./utils/speech-stream.js";
|
|
5
6
|
export declare const DEFAULT_BASE_URL = "https://api.deepgram.com";
|
|
6
7
|
export declare const PATH = "/v1/speak";
|
|
7
|
-
export type
|
|
8
|
-
export type DeepgramEncoding = DeepgramSpeechString<"linear16" | "mulaw" | "alaw" | "mp3" | "opus" | "flac" | "aac">;
|
|
8
|
+
export type DeepgramEncoding = OpenString<"linear16" | "mulaw" | "alaw" | "mp3" | "opus" | "flac" | "aac">;
|
|
9
9
|
export type DeepgramSpeechOptions = {
|
|
10
10
|
readonly encoding?: DeepgramEncoding;
|
|
11
|
-
readonly container?:
|
|
11
|
+
readonly container?: OpenString<"wav" | "ogg" | "none">;
|
|
12
12
|
readonly sampleRate?: number;
|
|
13
13
|
readonly bitRate?: number;
|
|
14
14
|
readonly mip_opt_out?: boolean;
|
|
@@ -16,6 +16,14 @@ export type DeepgramSpeechOptions = {
|
|
|
16
16
|
} & Record<string, unknown>;
|
|
17
17
|
export type Request = SpeechRequestFor<DeepgramSpeechOptions>;
|
|
18
18
|
export declare const protocol: MediaProtocol.Streamed<Request, {
|
|
19
|
+
readonly id: string;
|
|
20
|
+
readonly type: "generation-queued";
|
|
21
|
+
readonly position?: number | undefined;
|
|
22
|
+
} | {
|
|
23
|
+
readonly id: string;
|
|
24
|
+
readonly type: "generation-progress";
|
|
25
|
+
readonly progress?: number | undefined;
|
|
26
|
+
} | {
|
|
19
27
|
readonly type: "audio-delta";
|
|
20
28
|
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
21
29
|
} | {
|
|
@@ -67,6 +75,14 @@ export declare const protocol: MediaProtocol.Streamed<Request, {
|
|
|
67
75
|
export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<DeepgramSpeechOptions>;
|
|
68
76
|
export declare const DeepgramSpeech: {
|
|
69
77
|
readonly protocol: MediaProtocol.Streamed<Request, {
|
|
78
|
+
readonly id: string;
|
|
79
|
+
readonly type: "generation-queued";
|
|
80
|
+
readonly position?: number | undefined;
|
|
81
|
+
} | {
|
|
82
|
+
readonly id: string;
|
|
83
|
+
readonly type: "generation-progress";
|
|
84
|
+
readonly progress?: number | undefined;
|
|
85
|
+
} | {
|
|
70
86
|
readonly type: "audio-delta";
|
|
71
87
|
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
72
88
|
} | {
|
|
@@ -1,13 +1,11 @@
|
|
|
1
1
|
import { Effect } from "effect";
|
|
2
2
|
import { MediaProtocol } from "../route/media-protocol.js";
|
|
3
3
|
import { MediaRoute } from "../route/media.js";
|
|
4
|
-
import {
|
|
4
|
+
import { mergeJsonRecords } from "../schema/index.js";
|
|
5
5
|
import { SpeechModel } from "../speech.js";
|
|
6
6
|
import { MediaInput } from "./utils/media-input.js";
|
|
7
7
|
import { SpeechStream } from "./utils/speech-stream.js";
|
|
8
|
-
const
|
|
9
|
-
const NAME = "Deepgram";
|
|
10
|
-
const PROVIDER = ProviderID.make("deepgram");
|
|
8
|
+
const route = MediaProtocol.identity({ id: "deepgram-speech", name: "Deepgram", provider: "deepgram" });
|
|
11
9
|
export const DEFAULT_BASE_URL = "https://api.deepgram.com";
|
|
12
10
|
export const PATH = "/v1/speak";
|
|
13
11
|
// ---------------------------------------------------------------------------
|
|
@@ -30,7 +28,7 @@ const audioFormat = (request) => {
|
|
|
30
28
|
};
|
|
31
29
|
const queryParameters = (request) => {
|
|
32
30
|
const { encoding: _encoding, container: _container, sampleRate, bitRate, ...native } = request.providerOptions ?? {};
|
|
33
|
-
return MediaInput.query(
|
|
31
|
+
return MediaInput.query(route.id, {
|
|
34
32
|
...native,
|
|
35
33
|
model: request.model.id,
|
|
36
34
|
...audioFormat(request),
|
|
@@ -40,10 +38,13 @@ const queryParameters = (request) => {
|
|
|
40
38
|
});
|
|
41
39
|
};
|
|
42
40
|
const fromRequest = Effect.fn("DeepgramSpeech.fromRequest")(function* (request) {
|
|
41
|
+
// Not in `unsupported`: that list would also reject `timestamps: false`, which asks for nothing.
|
|
42
|
+
if (request.timestamps === true)
|
|
43
|
+
return yield* route.unsupported("media.timestamps", `${route.name} does not return timestamps`);
|
|
43
44
|
if (request.format !== undefined &&
|
|
44
45
|
FORMATS[request.format] === undefined &&
|
|
45
46
|
request.providerOptions?.encoding === undefined)
|
|
46
|
-
return yield*
|
|
47
|
+
return yield* route.unsupported("media.format", `${route.name} has no encoding for format "${request.format}"; pass providerOptions.encoding`);
|
|
47
48
|
return MediaProtocol.json(mergeJsonRecords({ text: request.text }, request.http?.body) ?? {}, yield* queryParameters(request));
|
|
48
49
|
});
|
|
49
50
|
// ---------------------------------------------------------------------------
|
|
@@ -61,7 +62,7 @@ const finish = (state, context) => {
|
|
|
61
62
|
const encoding = HEADERLESS_ENCODINGS[format.encoding ?? ""];
|
|
62
63
|
const requestID = headers["dg-request-id"];
|
|
63
64
|
const modelName = headers["dg-model-name"];
|
|
64
|
-
return SpeechStream.finish(
|
|
65
|
+
return SpeechStream.finish(route, state, {
|
|
65
66
|
...(format.container === "none" && encoding !== undefined
|
|
66
67
|
? SpeechStream.pcm(encoding, SpeechStream.sampleRate(mediaType), mediaType)
|
|
67
68
|
: // Deepgram's default encoding is MP3; WAV is a container around any encoding.
|
|
@@ -75,17 +76,15 @@ const finish = (state, context) => {
|
|
|
75
76
|
// ---------------------------------------------------------------------------
|
|
76
77
|
// 7. Protocol and route
|
|
77
78
|
// ---------------------------------------------------------------------------
|
|
78
|
-
export const protocol = MediaProtocol.stream({
|
|
79
|
-
|
|
80
|
-
name: NAME,
|
|
81
|
-
unsupported: ["voice", "language", "instructions", "timestamps"],
|
|
79
|
+
export const protocol = MediaProtocol.stream(route, {
|
|
80
|
+
unsupported: ["voice", "language", "instructions"],
|
|
82
81
|
body: { from: fromRequest },
|
|
83
82
|
frames: (bytes) => bytes,
|
|
84
83
|
initial: () => ({ chunks: [] }),
|
|
85
84
|
step: (state, frame) => Effect.succeed(SpeechStream.delta(state, frame)),
|
|
86
85
|
finish,
|
|
87
86
|
});
|
|
88
|
-
export const model = (input) => SpeechModel.fromRoute({
|
|
87
|
+
export const model = (input) => SpeechModel.fromRoute({ protocol, baseURL: DEFAULT_BASE_URL, path: PATH }, input);
|
|
89
88
|
export const DeepgramSpeech = {
|
|
90
89
|
protocol,
|
|
91
90
|
model,
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { MediaProtocol } from "../route/media-protocol.js";
|
|
2
2
|
import { MediaRoute } from "../route/media.js";
|
|
3
|
+
import { type OpenString } from "../schema/index.js";
|
|
3
4
|
import { TranscriptionModel, TranscriptionResponse, type TranscriptionRequestFor } from "../transcription.js";
|
|
4
5
|
export declare const DEFAULT_BASE_URL = "https://api.deepgram.com";
|
|
5
6
|
export declare const PATH = "/v1/listen";
|
|
@@ -10,7 +11,7 @@ export type DeepgramTranscriptionOptions = {
|
|
|
10
11
|
readonly utterances?: boolean;
|
|
11
12
|
readonly detect_language?: boolean | ReadonlyArray<string>;
|
|
12
13
|
readonly keyterm?: ReadonlyArray<string>;
|
|
13
|
-
readonly diarize_model?: "latest" | "v1" | "v2"
|
|
14
|
+
readonly diarize_model?: OpenString<"latest" | "v1" | "v2">;
|
|
14
15
|
readonly filler_words?: boolean;
|
|
15
16
|
readonly numerals?: boolean;
|
|
16
17
|
readonly mip_opt_out?: boolean;
|
|
@@ -1,13 +1,11 @@
|
|
|
1
1
|
import { Effect, Schema } from "effect";
|
|
2
2
|
import { MediaProtocol } from "../route/media-protocol.js";
|
|
3
3
|
import { MediaRoute } from "../route/media.js";
|
|
4
|
-
import {
|
|
4
|
+
import { mergeJsonRecords } from "../schema/index.js";
|
|
5
5
|
import { TranscriptionModel, TranscriptionResponse } from "../transcription.js";
|
|
6
6
|
import { ProviderShared } from "./shared.js";
|
|
7
7
|
import { MediaInput } from "./utils/media-input.js";
|
|
8
|
-
const
|
|
9
|
-
const NAME = "Deepgram";
|
|
10
|
-
const PROVIDER = ProviderID.make("deepgram");
|
|
8
|
+
const route = MediaProtocol.identity({ id: "deepgram-transcription", name: "Deepgram", provider: "deepgram" });
|
|
11
9
|
export const DEFAULT_BASE_URL = "https://api.deepgram.com";
|
|
12
10
|
export const PATH = "/v1/listen";
|
|
13
11
|
// ---------------------------------------------------------------------------
|
|
@@ -40,7 +38,7 @@ const ListenResponse = Schema.Struct({
|
|
|
40
38
|
// ---------------------------------------------------------------------------
|
|
41
39
|
// 5. Request body construction
|
|
42
40
|
// ---------------------------------------------------------------------------
|
|
43
|
-
const query = (request) => MediaInput.query(
|
|
41
|
+
const query = (request) => MediaInput.query(route.id, mergeJsonRecords({
|
|
44
42
|
model: request.model.id,
|
|
45
43
|
smart_format: true,
|
|
46
44
|
language: request.language,
|
|
@@ -55,14 +53,14 @@ const fromRequest = Effect.fn("DeepgramTranscription.fromRequest")(function* (re
|
|
|
55
53
|
if (url !== undefined)
|
|
56
54
|
return MediaProtocol.json(mergeJsonRecords({ url }, request.http?.body) ?? {}, yield* query(request));
|
|
57
55
|
if (request.http?.body !== undefined)
|
|
58
|
-
return yield* ProviderShared.invalidRequest(`${
|
|
59
|
-
const audio = yield* MediaInput.inlineBytes(
|
|
56
|
+
return yield* ProviderShared.invalidRequest(`${route.name} sends inline audio as the raw body, so http.body cannot apply`);
|
|
57
|
+
const audio = yield* MediaInput.inlineBytes(route.id, request.audio);
|
|
60
58
|
return MediaProtocol.binary(audio, request.audio.mediaType, yield* query(request));
|
|
61
59
|
});
|
|
62
60
|
// ---------------------------------------------------------------------------
|
|
63
61
|
// 6. Response decoding
|
|
64
62
|
// ---------------------------------------------------------------------------
|
|
65
|
-
const decodeListen =
|
|
63
|
+
const decodeListen = route.decodeJson(ListenResponse);
|
|
66
64
|
const speaker = (value) => (value === undefined ? undefined : String(value));
|
|
67
65
|
const wordText = (word) => word.punctuated_word ?? word.word;
|
|
68
66
|
// Utterances split on pauses, not speakers: the v2 diarizer labels a whole utterance with one speaker even when its
|
|
@@ -79,7 +77,7 @@ const decodeResponse = Effect.fn("DeepgramTranscription.decodeResponse")(functio
|
|
|
79
77
|
const channel = output.value.results.channels[0];
|
|
80
78
|
const alternative = channel?.alternatives?.[0];
|
|
81
79
|
if (alternative === undefined)
|
|
82
|
-
return yield* output.invalid(`${
|
|
80
|
+
return yield* output.invalid(`${route.name} returned no transcript`);
|
|
83
81
|
const duration = output.value.metadata?.duration;
|
|
84
82
|
const requestID = output.value.metadata?.request_id;
|
|
85
83
|
return new TranscriptionResponse({
|
|
@@ -115,14 +113,12 @@ const decodeResponse = Effect.fn("DeepgramTranscription.decodeResponse")(functio
|
|
|
115
113
|
// ---------------------------------------------------------------------------
|
|
116
114
|
// 7. Protocol and route
|
|
117
115
|
// ---------------------------------------------------------------------------
|
|
118
|
-
export const protocol = MediaProtocol.inline({
|
|
119
|
-
id: ADAPTER,
|
|
120
|
-
name: NAME,
|
|
116
|
+
export const protocol = MediaProtocol.inline(route, {
|
|
121
117
|
unsupported: ["prompt", "speakers"],
|
|
122
118
|
body: { from: fromRequest },
|
|
123
119
|
response: { decode: decodeResponse },
|
|
124
120
|
});
|
|
125
|
-
export const model = (input) => TranscriptionModel.fromRoute({
|
|
121
|
+
export const model = (input) => TranscriptionModel.fromRoute({ protocol, baseURL: DEFAULT_BASE_URL, path: PATH }, input);
|
|
126
122
|
export const DeepgramTranscription = {
|
|
127
123
|
protocol,
|
|
128
124
|
model,
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
import { MediaProtocol } from "../route/media-protocol.js";
|
|
2
2
|
import { MediaRoute } from "../route/media.js";
|
|
3
|
+
import { type OpenString } from "../schema/index.js";
|
|
3
4
|
import { SpeechModel, type SpeechRequestFor } from "../speech.js";
|
|
4
5
|
import { SpeechStream } from "./utils/speech-stream.js";
|
|
5
6
|
export declare const DEFAULT_BASE_URL = "https://api.elevenlabs.io";
|
|
6
7
|
export declare const PATH = "/v1/text-to-speech";
|
|
7
|
-
export type
|
|
8
|
-
export type ElevenLabsOutputFormat = ElevenLabsSpeechString<"mp3_22050_32" | "mp3_24000_48" | "mp3_44100_32" | "mp3_44100_64" | "mp3_44100_96" | "mp3_44100_128" | "mp3_44100_192" | "pcm_8000" | "pcm_16000" | "pcm_22050" | "pcm_24000" | "pcm_32000" | "pcm_44100" | "pcm_48000" | "wav_8000" | "wav_16000" | "wav_22050" | "wav_24000" | "wav_32000" | "wav_44100" | "wav_48000" | "ulaw_8000" | "alaw_8000" | "opus_48000_32" | "opus_48000_64" | "opus_48000_96" | "opus_48000_128" | "opus_48000_192">;
|
|
8
|
+
export type ElevenLabsOutputFormat = OpenString<"mp3_22050_32" | "mp3_24000_48" | "mp3_44100_32" | "mp3_44100_64" | "mp3_44100_96" | "mp3_44100_128" | "mp3_44100_192" | "pcm_8000" | "pcm_16000" | "pcm_22050" | "pcm_24000" | "pcm_32000" | "pcm_44100" | "pcm_48000" | "wav_8000" | "wav_16000" | "wav_22050" | "wav_24000" | "wav_32000" | "wav_44100" | "wav_48000" | "ulaw_8000" | "alaw_8000" | "opus_48000_32" | "opus_48000_64" | "opus_48000_96" | "opus_48000_128" | "opus_48000_192">;
|
|
9
9
|
export type ElevenLabsSpeechOptions = {
|
|
10
10
|
readonly outputFormat?: ElevenLabsOutputFormat;
|
|
11
11
|
readonly voice_settings?: {
|
|
@@ -15,10 +15,18 @@ export type ElevenLabsSpeechOptions = {
|
|
|
15
15
|
readonly use_speaker_boost?: boolean;
|
|
16
16
|
};
|
|
17
17
|
readonly seed?: number;
|
|
18
|
-
readonly apply_text_normalization?:
|
|
18
|
+
readonly apply_text_normalization?: OpenString<"auto" | "on" | "off">;
|
|
19
19
|
} & Record<string, unknown>;
|
|
20
20
|
export type Request = SpeechRequestFor<ElevenLabsSpeechOptions>;
|
|
21
21
|
export declare const protocol: MediaProtocol.Streamed<Request, {
|
|
22
|
+
readonly id: string;
|
|
23
|
+
readonly type: "generation-queued";
|
|
24
|
+
readonly position?: number | undefined;
|
|
25
|
+
} | {
|
|
26
|
+
readonly id: string;
|
|
27
|
+
readonly type: "generation-progress";
|
|
28
|
+
readonly progress?: number | undefined;
|
|
29
|
+
} | {
|
|
22
30
|
readonly type: "audio-delta";
|
|
23
31
|
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
24
32
|
} | {
|
|
@@ -70,6 +78,14 @@ export declare const protocol: MediaProtocol.Streamed<Request, {
|
|
|
70
78
|
export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<ElevenLabsSpeechOptions>;
|
|
71
79
|
export declare const ElevenLabsSpeech: {
|
|
72
80
|
readonly protocol: MediaProtocol.Streamed<Request, {
|
|
81
|
+
readonly id: string;
|
|
82
|
+
readonly type: "generation-queued";
|
|
83
|
+
readonly position?: number | undefined;
|
|
84
|
+
} | {
|
|
85
|
+
readonly id: string;
|
|
86
|
+
readonly type: "generation-progress";
|
|
87
|
+
readonly progress?: number | undefined;
|
|
88
|
+
} | {
|
|
73
89
|
readonly type: "audio-delta";
|
|
74
90
|
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
75
91
|
} | {
|
|
@@ -2,13 +2,11 @@ import { Effect, Schema } from "effect";
|
|
|
2
2
|
import { Framing } from "../route/framing.js";
|
|
3
3
|
import { MediaProtocol } from "../route/media-protocol.js";
|
|
4
4
|
import { MediaRoute } from "../route/media.js";
|
|
5
|
-
import {
|
|
5
|
+
import { mergeJsonRecords } from "../schema/index.js";
|
|
6
6
|
import { SpeechModel } from "../speech.js";
|
|
7
7
|
import { ProviderShared, optionalNull } from "./shared.js";
|
|
8
8
|
import { SpeechStream } from "./utils/speech-stream.js";
|
|
9
|
-
const
|
|
10
|
-
const NAME = "ElevenLabs";
|
|
11
|
-
const PROVIDER = ProviderID.make("elevenlabs");
|
|
9
|
+
const route = MediaProtocol.identity({ id: "elevenlabs-speech", name: "ElevenLabs", provider: "elevenlabs" });
|
|
12
10
|
export const DEFAULT_BASE_URL = "https://api.elevenlabs.io";
|
|
13
11
|
export const PATH = "/v1/text-to-speech";
|
|
14
12
|
// ---------------------------------------------------------------------------
|
|
@@ -23,7 +21,7 @@ const TimestampedAudio = Schema.Struct({
|
|
|
23
21
|
audio_base64: Schema.Uint8ArrayFromBase64,
|
|
24
22
|
alignment: optionalNull(Alignment),
|
|
25
23
|
});
|
|
26
|
-
const decodeRecord =
|
|
24
|
+
const decodeRecord = route.decodeFrame(TimestampedAudio);
|
|
27
25
|
// ---------------------------------------------------------------------------
|
|
28
26
|
// 5. Request body construction
|
|
29
27
|
// ---------------------------------------------------------------------------
|
|
@@ -37,14 +35,14 @@ const OUTPUT_FORMATS = {
|
|
|
37
35
|
const outputFormat = Effect.fn("ElevenLabsSpeech.outputFormat")(function* (request) {
|
|
38
36
|
const format = request.providerOptions?.outputFormat ?? OUTPUT_FORMATS[request.format ?? "mp3"];
|
|
39
37
|
if (format === undefined)
|
|
40
|
-
return yield*
|
|
38
|
+
return yield* route.unsupported("media.format", `${route.name} has no default output format for "${request.format}"; pass providerOptions.outputFormat`);
|
|
41
39
|
if (request.mode === "stream" && format.startsWith("wav_"))
|
|
42
|
-
return yield*
|
|
40
|
+
return yield* route.unsupported("media.format", `${route.name} streams mp3, pcm, opus, ulaw, and alaw but not "${format}"; use generate for WAV`);
|
|
43
41
|
return format;
|
|
44
42
|
});
|
|
45
43
|
const fromRequest = Effect.fn("ElevenLabsSpeech.fromRequest")(function* (request) {
|
|
46
44
|
if (request.voice === undefined)
|
|
47
|
-
return yield* ProviderShared.invalidRequest(`${
|
|
45
|
+
return yield* ProviderShared.invalidRequest(`${route.name} requires a voice id; pass it as \`voice\``);
|
|
48
46
|
const { outputFormat: _outputFormat, ...native } = request.providerOptions ?? {};
|
|
49
47
|
return MediaProtocol.json(mergeJsonRecords({
|
|
50
48
|
text: request.text,
|
|
@@ -84,7 +82,7 @@ const describeOutput = (format) => {
|
|
|
84
82
|
};
|
|
85
83
|
const finish = Effect.fn("ElevenLabsSpeech.finish")(function* (state, context) {
|
|
86
84
|
const requestID = context.http.headers["request-id"];
|
|
87
|
-
return yield* SpeechStream.finish(
|
|
85
|
+
return yield* SpeechStream.finish(route, state, {
|
|
88
86
|
...describeOutput(yield* outputFormat(context.request)),
|
|
89
87
|
// `character-cost` is billed credits, not a character count (3 for 20 characters on `eleven_flash_v2_5`).
|
|
90
88
|
usage: SpeechStream.headerUsage("credits", context.http.headers["character-cost"]),
|
|
@@ -94,9 +92,7 @@ const finish = Effect.fn("ElevenLabsSpeech.finish")(function* (state, context) {
|
|
|
94
92
|
// ---------------------------------------------------------------------------
|
|
95
93
|
// 7. Protocol and route
|
|
96
94
|
// ---------------------------------------------------------------------------
|
|
97
|
-
export const protocol = MediaProtocol.stream({
|
|
98
|
-
id: ADAPTER,
|
|
99
|
-
name: NAME,
|
|
95
|
+
export const protocol = MediaProtocol.stream(route, {
|
|
100
96
|
unsupported: ["instructions"],
|
|
101
97
|
body: { from: fromRequest },
|
|
102
98
|
frames: (bytes, context) => {
|
|
@@ -108,7 +104,7 @@ export const protocol = MediaProtocol.stream({
|
|
|
108
104
|
step: SpeechStream.step(onRecord),
|
|
109
105
|
finish,
|
|
110
106
|
});
|
|
111
|
-
export const model = (input) => SpeechModel.fromRoute({
|
|
107
|
+
export const model = (input) => SpeechModel.fromRoute({ protocol, baseURL: DEFAULT_BASE_URL, path: ({ request }) => path(request) }, input);
|
|
112
108
|
export const ElevenLabsSpeech = {
|
|
113
109
|
protocol,
|
|
114
110
|
model,
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
import { ImageModel, ImageResponse, type ImageRequestFor } from "../image.js";
|
|
2
2
|
import { MediaProtocol } from "../route/media-protocol.js";
|
|
3
3
|
import { MediaRoute } from "../route/media.js";
|
|
4
|
+
import { type OpenString } from "../schema/index.js";
|
|
4
5
|
export type FalImageOptions = {
|
|
5
|
-
readonly image_size?: "square_hd" | "square" | "portrait_4_3" | "portrait_16_9" | "landscape_4_3" | "landscape_16_9"
|
|
6
|
+
readonly image_size?: OpenString<"square_hd" | "square" | "portrait_4_3" | "portrait_16_9" | "landscape_4_3" | "landscape_16_9">;
|
|
6
7
|
readonly enable_safety_checker?: boolean;
|
|
7
8
|
} & Record<string, unknown>;
|
|
8
9
|
export type Request = ImageRequestFor<FalImageOptions>;
|