@opencode/ai 2.0.14 → 2.0.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +399 -56
- package/dist/experimental/evaluation-client.d.ts +1 -1
- package/dist/experimental/evaluation-client.js +39 -3
- package/dist/experimental/evaluation.d.ts +4 -4
- package/dist/experimental/evaluation.js +2 -2
- package/dist/experimental/system-one.d.ts +3 -3
- package/dist/experimental/system-one.js +40 -51
- package/dist/generation.d.ts +83 -0
- package/dist/generation.js +113 -0
- package/dist/image-client.d.ts +16 -7
- package/dist/image-client.js +29 -13
- package/dist/image.d.ts +1410 -81
- package/dist/image.js +97 -67
- package/dist/index.d.ts +17 -2
- package/dist/index.js +12 -1
- package/dist/llm.d.ts +9 -1
- package/dist/media-model.d.ts +44 -0
- package/dist/media-model.js +49 -0
- package/dist/media.d.ts +213 -0
- package/dist/media.js +227 -0
- package/dist/promise.d.ts +974 -0
- package/dist/promise.js +81 -0
- package/dist/protocols/alibaba-chat.d.ts +12 -0
- package/dist/protocols/alibaba-responses.d.ts +2 -2
- package/dist/protocols/anthropic-messages.js +8 -19
- package/dist/protocols/assemblyai-transcription.d.ts +40 -0
- package/dist/protocols/assemblyai-transcription.js +138 -0
- package/dist/protocols/bedrock-converse.d.ts +4 -4
- package/dist/protocols/bedrock-converse.js +6 -17
- package/dist/protocols/bfl-images.d.ts +32 -0
- package/dist/protocols/bfl-images.js +153 -0
- package/dist/protocols/cartesia-speech.d.ts +127 -0
- package/dist/protocols/cartesia-speech.js +126 -0
- package/dist/protocols/deepgram-speech.d.ts +119 -0
- package/dist/protocols/deepgram-speech.js +92 -0
- package/dist/protocols/deepgram-transcription.d.ts +25 -0
- package/dist/protocols/deepgram-transcription.js +129 -0
- package/dist/protocols/elevenlabs-speech.d.ts +122 -0
- package/dist/protocols/elevenlabs-speech.js +115 -0
- package/dist/protocols/fal-images.d.ts +24 -0
- package/dist/protocols/fal-images.js +114 -0
- package/dist/protocols/fal-video.d.ts +29 -0
- package/dist/protocols/fal-video.js +88 -0
- package/dist/protocols/gemini.d.ts +30 -9
- package/dist/protocols/gemini.js +45 -35
- package/dist/protocols/google-images.d.ts +9 -21
- package/dist/protocols/google-images.js +158 -133
- package/dist/protocols/google-speech.d.ts +130 -0
- package/dist/protocols/google-speech.js +84 -0
- package/dist/protocols/google-transcription.d.ts +173 -0
- package/dist/protocols/google-transcription.js +138 -0
- package/dist/protocols/google-video.d.ts +26 -0
- package/dist/protocols/google-video.js +158 -0
- package/dist/protocols/meta-images.d.ts +7 -12
- package/dist/protocols/meta-images.js +85 -66
- package/dist/protocols/meta-responses.d.ts +4 -4
- package/dist/protocols/meta-responses.js +1 -1
- package/dist/protocols/mistral-chat.js +7 -6
- package/dist/protocols/open-responses.d.ts +17 -9
- package/dist/protocols/open-responses.js +24 -14
- package/dist/protocols/openai-chat.d.ts +118 -1
- package/dist/protocols/openai-chat.js +125 -44
- package/dist/protocols/openai-compatible-chat.d.ts +12 -0
- package/dist/protocols/openai-compatible-responses.d.ts +2 -2
- package/dist/protocols/openai-images.d.ts +128 -18
- package/dist/protocols/openai-images.js +177 -154
- package/dist/protocols/openai-responses.d.ts +15 -15
- package/dist/protocols/openai-responses.js +5 -6
- package/dist/protocols/openai-speech.d.ts +116 -0
- package/dist/protocols/openai-speech.js +98 -0
- package/dist/protocols/openai-transcription.d.ts +207 -0
- package/dist/protocols/openai-transcription.js +190 -0
- package/dist/protocols/replicate-images.d.ts +28 -0
- package/dist/protocols/replicate-images.js +133 -0
- package/dist/protocols/runway-video.d.ts +38 -0
- package/dist/protocols/runway-video.js +146 -0
- package/dist/protocols/shared.d.ts +27 -17
- package/dist/protocols/shared.js +52 -35
- package/dist/protocols/stability-images.d.ts +38 -0
- package/dist/protocols/stability-images.js +148 -0
- package/dist/protocols/utils/bedrock-media.d.ts +2 -3
- package/dist/protocols/utils/bedrock-media.js +4 -4
- package/dist/protocols/utils/fal-queue.d.ts +28 -0
- package/dist/protocols/utils/fal-queue.js +69 -0
- package/dist/protocols/utils/gemini-generate-content.d.ts +65 -0
- package/dist/protocols/utils/gemini-generate-content.js +65 -0
- package/dist/protocols/utils/gemini-json-schema.d.ts +3 -0
- package/dist/protocols/utils/gemini-json-schema.js +76 -0
- package/dist/protocols/utils/media-input.d.ts +18 -0
- package/dist/protocols/utils/media-input.js +35 -0
- package/dist/protocols/utils/responses-compaction.js +6 -5
- package/dist/protocols/utils/speech-stream.d.ts +49 -0
- package/dist/protocols/utils/speech-stream.js +67 -0
- package/dist/protocols/utils/tool-schema.d.ts +2 -2
- package/dist/protocols/utils/tool-schema.js +40 -17
- package/dist/protocols/utils/tool-stream.d.ts +27 -3
- package/dist/protocols/xai-images.d.ts +9 -15
- package/dist/protocols/xai-images.js +75 -84
- package/dist/protocols/xai-responses.d.ts +2 -2
- package/dist/protocols/xai-video.d.ts +34 -0
- package/dist/protocols/xai-video.js +147 -0
- package/dist/protocols/zai-chat.d.ts +13 -1
- package/dist/protocols/zai-images.d.ts +9 -13
- package/dist/protocols/zai-images.js +59 -57
- package/dist/provider-error.js +3 -0
- package/dist/providers/alibaba.d.ts +14 -2
- package/dist/providers/amazon-bedrock-mantle.d.ts +14 -2
- package/dist/providers/amazon-bedrock.d.ts +2 -2
- package/dist/providers/assemblyai.d.ts +25 -0
- package/dist/providers/assemblyai.js +29 -0
- package/dist/providers/azure.d.ts +18 -6
- package/dist/providers/baseten.d.ts +24 -0
- package/dist/providers/black-forest-labs.d.ts +25 -0
- package/dist/providers/black-forest-labs.js +28 -0
- package/dist/providers/cartesia.d.ts +24 -0
- package/dist/providers/cartesia.js +22 -0
- package/dist/providers/cerebras.d.ts +24 -0
- package/dist/providers/cerebras.js +6 -1
- package/dist/providers/cloudflare-ai-gateway.d.ts +30 -6
- package/dist/providers/cloudflare-workers-ai.d.ts +24 -0
- package/dist/providers/deepgram.d.ts +29 -0
- package/dist/providers/deepgram.js +31 -0
- package/dist/providers/deepinfra.d.ts +24 -0
- package/dist/providers/deepinfra.js +6 -1
- package/dist/providers/deepseek.d.ts +24 -0
- package/dist/providers/elevenlabs.d.ts +24 -0
- package/dist/providers/elevenlabs.js +28 -0
- package/dist/providers/fal.d.ts +29 -0
- package/dist/providers/fal.js +33 -0
- package/dist/providers/fireworks.d.ts +24 -0
- package/dist/providers/google-vertex-chat.d.ts +12 -0
- package/dist/providers/google-vertex-responses.d.ts +2 -2
- package/dist/providers/google-vertex.d.ts +10 -3
- package/dist/providers/google.d.ts +25 -3
- package/dist/providers/google.js +11 -2
- package/dist/providers/groq.d.ts +24 -0
- package/dist/providers/index.d.ts +10 -0
- package/dist/providers/index.js +10 -0
- package/dist/providers/meta.d.ts +14 -2
- package/dist/providers/minimax.d.ts +14 -2
- package/dist/providers/moonshot.d.ts +14 -2
- package/dist/providers/moonshot.js +3 -3
- package/dist/providers/openai-compatible-responses.d.ts +2 -2
- package/dist/providers/openai-compatible.d.ts +12 -0
- package/dist/providers/openai.d.ts +25 -3
- package/dist/providers/openai.js +10 -1
- package/dist/providers/openrouter.d.ts +67 -0
- package/dist/providers/openrouter.js +13 -1
- package/dist/providers/replicate.d.ts +25 -0
- package/dist/providers/replicate.js +22 -0
- package/dist/providers/runway.d.ts +24 -0
- package/dist/providers/runway.js +22 -0
- package/dist/providers/stability.d.ts +28 -0
- package/dist/providers/stability.js +23 -0
- package/dist/providers/togetherai.d.ts +24 -0
- package/dist/providers/vercel-ai-gateway.d.ts +41 -0
- package/dist/providers/vercel-ai-gateway.js +85 -0
- package/dist/providers/xai.d.ts +17 -0
- package/dist/providers/xai.js +5 -2
- package/dist/providers/zai-coding-plan.d.ts +15 -3
- package/dist/providers/zai.d.ts +13 -1
- package/dist/route/auth.d.ts +4 -1
- package/dist/route/auth.js +6 -0
- package/dist/route/client.d.ts +9 -1
- package/dist/route/endpoint.d.ts +10 -10
- package/dist/route/executor-service.d.ts +12 -0
- package/dist/route/executor-service.js +3 -0
- package/dist/route/executor.d.ts +4 -9
- package/dist/route/executor.js +3 -3
- package/dist/route/framing.d.ts +5 -1
- package/dist/route/framing.js +9 -0
- package/dist/route/index.d.ts +2 -0
- package/dist/route/index.js +2 -0
- package/dist/route/media-protocol.d.ts +158 -0
- package/dist/route/media-protocol.js +96 -0
- package/dist/route/media.d.ts +97 -0
- package/dist/route/media.js +236 -0
- package/dist/schema/errors.d.ts +13 -3
- package/dist/schema/errors.js +7 -0
- package/dist/schema/events.d.ts +557 -40
- package/dist/schema/events.js +35 -2
- package/dist/schema/messages.d.ts +95 -8
- package/dist/schema/messages.js +8 -6
- package/dist/schema/options.d.ts +6 -3
- package/dist/schema/options.js +6 -2
- package/dist/speech-client.d.ts +21 -0
- package/dist/speech-client.js +25 -0
- package/dist/speech.d.ts +1150 -0
- package/dist/speech.js +119 -0
- package/dist/testing.d.ts +72 -8
- package/dist/transcription-client.d.ts +28 -0
- package/dist/transcription-client.js +44 -0
- package/dist/transcription.d.ts +1504 -0
- package/dist/transcription.js +133 -0
- package/dist/utils/bytes.d.ts +1 -0
- package/dist/utils/bytes.js +10 -0
- package/dist/utils/media-type.d.ts +7 -0
- package/dist/utils/media-type.js +70 -0
- package/dist/utils/sanitize.js +3 -1
- package/dist/video-client.d.ts +28 -0
- package/dist/video-client.js +40 -0
- package/dist/video.d.ts +1359 -0
- package/dist/video.js +119 -0
- package/package.json +7 -3
- package/dist/protocols/utils/gemini-tool-schema.d.ts +0 -2
- package/dist/protocols/utils/gemini-tool-schema.js +0 -103
- package/dist/protocols/utils/image-input.d.ts +0 -21
- package/dist/protocols/utils/image-input.js +0 -20
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
import { Effect, Schema } from "effect";
|
|
2
|
+
import { ImageModel, ImageResponse } from "../image.js";
|
|
3
|
+
import { Media } from "../media.js";
|
|
4
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
5
|
+
import { MediaRoute } from "../route/media.js";
|
|
6
|
+
import { ProviderID, mergeJsonRecords } from "../schema/index.js";
|
|
7
|
+
import { ProviderShared, optionalNull } from "./shared.js";
|
|
8
|
+
import { MediaInput } from "./utils/media-input.js";
|
|
9
|
+
const ADAPTER = "bfl-images";
|
|
10
|
+
const NAME = "Black Forest Labs";
|
|
11
|
+
const PROVIDER = ProviderID.make("black-forest-labs");
|
|
12
|
+
export const DEFAULT_BASE_URL = "https://api.bfl.ai";
|
|
13
|
+
// ---------------------------------------------------------------------------
|
|
14
|
+
// 2. Token and response schemas
|
|
15
|
+
// ---------------------------------------------------------------------------
|
|
16
|
+
/** Regional clusters answer on different hosts, so the returned `polling_url` is followed verbatim. */
|
|
17
|
+
export const Token = Schema.Struct({ id: Schema.String, pollingURL: Schema.String });
|
|
18
|
+
const StartResponse = Schema.Struct({
|
|
19
|
+
id: Schema.String,
|
|
20
|
+
polling_url: Schema.String,
|
|
21
|
+
});
|
|
22
|
+
const Result = Schema.Struct({
|
|
23
|
+
id: Schema.String,
|
|
24
|
+
status: Schema.String,
|
|
25
|
+
result: optionalNull(Schema.StructWithRest(Schema.Struct({ sample: Schema.String, seed: optionalNull(Schema.Number), prompt: optionalNull(Schema.String) }), [Schema.Record(Schema.String, Schema.Unknown)])),
|
|
26
|
+
cost: optionalNull(Schema.Number),
|
|
27
|
+
});
|
|
28
|
+
const STATUS = {
|
|
29
|
+
Pending: "running",
|
|
30
|
+
Reasoning: "running",
|
|
31
|
+
Generating: "running",
|
|
32
|
+
Ready: "completed",
|
|
33
|
+
Error: "failed",
|
|
34
|
+
// Moderation is terminal; `decodeResult` reports it as a content-policy failure.
|
|
35
|
+
"Content Moderated": "failed",
|
|
36
|
+
"Request Moderated": "failed",
|
|
37
|
+
"Task not found": "expired",
|
|
38
|
+
};
|
|
39
|
+
const isModerated = (status) => status === "Content Moderated" || status === "Request Moderated";
|
|
40
|
+
const capabilities = (model) => {
|
|
41
|
+
if (model.startsWith("flux-pro-1.0-fill"))
|
|
42
|
+
return { sizing: "none", imageField: "image", maxImages: 1, mask: true };
|
|
43
|
+
if (model.startsWith("flux-pro-1.0-expand"))
|
|
44
|
+
return { sizing: "none", imageField: "image", maxImages: 1, mask: false };
|
|
45
|
+
if (model.startsWith("flux-kontext"))
|
|
46
|
+
return { sizing: "aspectRatio", imageField: "input_image", maxImages: 4, mask: false };
|
|
47
|
+
if (model.startsWith("flux-pro-1.1-ultra"))
|
|
48
|
+
return { sizing: "aspectRatio", imageField: "image_prompt", maxImages: 1, mask: false };
|
|
49
|
+
if (model.startsWith("flux-pro-1.1") || model.startsWith("flux-dev"))
|
|
50
|
+
return { sizing: "dimensions", imageField: "image_prompt", maxImages: 1, mask: false };
|
|
51
|
+
if (model.startsWith("flux-2-klein"))
|
|
52
|
+
return { sizing: "dimensions", imageField: "input_image", maxImages: 4, mask: false };
|
|
53
|
+
return { sizing: "dimensions", imageField: "input_image", maxImages: 8, mask: false };
|
|
54
|
+
};
|
|
55
|
+
const unsupported = (model, field, message) => ProviderShared.unsupportedOperation({
|
|
56
|
+
operation: `media.${field}`,
|
|
57
|
+
provider: PROVIDER,
|
|
58
|
+
route: ADAPTER,
|
|
59
|
+
message: `${model} ${message}`,
|
|
60
|
+
});
|
|
61
|
+
const validate = (request, model) => {
|
|
62
|
+
const id = request.model.id;
|
|
63
|
+
const images = request.images?.length ?? 0;
|
|
64
|
+
if (request.n !== undefined && request.n > 1)
|
|
65
|
+
return Effect.fail(unsupported(id, "n", "generates one image per request; call it once per image"));
|
|
66
|
+
if (request.size !== undefined && model.sizing !== "dimensions")
|
|
67
|
+
return Effect.fail(unsupported(id, "size", "does not take size (width and height)"));
|
|
68
|
+
if (request.aspectRatio !== undefined && model.sizing !== "aspectRatio")
|
|
69
|
+
return Effect.fail(unsupported(id, "aspectRatio", "does not take aspectRatio"));
|
|
70
|
+
if (images > model.maxImages)
|
|
71
|
+
return Effect.fail(unsupported(id, "images", `takes at most ${model.maxImages} images`));
|
|
72
|
+
if (request.mask !== undefined && !model.mask)
|
|
73
|
+
return Effect.fail(unsupported(id, "mask", "does not inpaint; use flux-pro-1.0-fill"));
|
|
74
|
+
return Effect.void;
|
|
75
|
+
};
|
|
76
|
+
const imageInput = (asset) => {
|
|
77
|
+
const value = asset.inline()?.base64 ?? ProviderShared.mediaUrl(asset);
|
|
78
|
+
if (value === undefined)
|
|
79
|
+
return Effect.fail(ProviderShared.invalidRequest(`${NAME} accepts inline images or https URLs`));
|
|
80
|
+
return Effect.succeed(value);
|
|
81
|
+
};
|
|
82
|
+
const fromRequest = Effect.fn("BlackForestLabsImages.fromRequest")(function* (request) {
|
|
83
|
+
const model = capabilities(request.model.id);
|
|
84
|
+
yield* validate(request, model);
|
|
85
|
+
const images = yield* Effect.forEach(request.images ?? [], imageInput);
|
|
86
|
+
const fields = images.map((image, index) => [
|
|
87
|
+
index === 0 ? model.imageField : `${model.imageField}_${index + 1}`,
|
|
88
|
+
image,
|
|
89
|
+
]);
|
|
90
|
+
return MediaProtocol.json(mergeJsonRecords({
|
|
91
|
+
prompt: request.prompt,
|
|
92
|
+
...(request.size === undefined ? {} : MediaInput.dimensions(request.size)),
|
|
93
|
+
aspect_ratio: request.aspectRatio,
|
|
94
|
+
seed: request.seed,
|
|
95
|
+
output_format: request.format,
|
|
96
|
+
mask: request.mask === undefined ? undefined : yield* imageInput(request.mask),
|
|
97
|
+
...Object.fromEntries(fields),
|
|
98
|
+
}, request.providerOptions, request.http?.body) ?? {});
|
|
99
|
+
});
|
|
100
|
+
// ---------------------------------------------------------------------------
|
|
101
|
+
// 6. Response decoding
|
|
102
|
+
// ---------------------------------------------------------------------------
|
|
103
|
+
const decodeStart = MediaProtocol.decodeStarted(ADAPTER, NAME, StartResponse, (value) => ({
|
|
104
|
+
token: { id: value.id, pollingURL: value.polling_url },
|
|
105
|
+
snapshot: { id: value.id, status: "queued" },
|
|
106
|
+
}));
|
|
107
|
+
const decodeDocument = MediaProtocol.decodeJson(ADAPTER, NAME, Result);
|
|
108
|
+
const decodeStatus = Effect.fn("BlackForestLabsImages.decodeStatus")(function* (response, context) {
|
|
109
|
+
const output = yield* decodeDocument(response);
|
|
110
|
+
return { id: context.token.id, status: yield* MediaProtocol.status(STATUS, output.value.status, output) };
|
|
111
|
+
});
|
|
112
|
+
const decodeResult = Effect.fn("BlackForestLabsImages.decodeResult")(function* (response, context) {
|
|
113
|
+
const output = yield* decodeDocument(response);
|
|
114
|
+
const document = output.value;
|
|
115
|
+
const status = yield* MediaProtocol.status(STATUS, document.status, output);
|
|
116
|
+
if (isModerated(document.status))
|
|
117
|
+
return yield* output.contentPolicy(`${NAME} moderated the generation`);
|
|
118
|
+
if (status === "failed" || status === "expired")
|
|
119
|
+
return yield* output.ended(status, `${NAME} generation ${context.token.id} ended with ${document.status}`);
|
|
120
|
+
if (status !== "completed" || document.result === undefined || document.result === null)
|
|
121
|
+
return yield* output.invalid(`${NAME} generation ${context.token.id} has no result`);
|
|
122
|
+
const { sample, seed, prompt, ...rest } = document.result;
|
|
123
|
+
return new ImageResponse({
|
|
124
|
+
// `sample` is a signed URL that expires 10 minutes after the result is ready, so it is downloaded now.
|
|
125
|
+
images: [yield* context.materialize(Media.url(sample))],
|
|
126
|
+
usage: document.cost === undefined || document.cost === null ? undefined : { type: "credits", credits: document.cost },
|
|
127
|
+
providerMetadata: {
|
|
128
|
+
bfl: { id: context.token.id, seed: seed ?? undefined, prompt: prompt ?? undefined, ...rest },
|
|
129
|
+
},
|
|
130
|
+
});
|
|
131
|
+
});
|
|
132
|
+
// ---------------------------------------------------------------------------
|
|
133
|
+
// 7. Protocol and route
|
|
134
|
+
// ---------------------------------------------------------------------------
|
|
135
|
+
export const protocol = MediaProtocol.queued({
|
|
136
|
+
id: ADAPTER,
|
|
137
|
+
name: NAME,
|
|
138
|
+
token: Token,
|
|
139
|
+
start: { body: { from: fromRequest }, decode: decodeStart },
|
|
140
|
+
status: { path: (token) => token.pollingURL, decode: decodeStatus },
|
|
141
|
+
result: { path: (token) => token.pollingURL, decode: decodeResult },
|
|
142
|
+
});
|
|
143
|
+
export const model = (input) => ImageModel.fromRoute({
|
|
144
|
+
id: ADAPTER,
|
|
145
|
+
provider: PROVIDER,
|
|
146
|
+
protocol,
|
|
147
|
+
baseURL: DEFAULT_BASE_URL,
|
|
148
|
+
path: ({ request }) => `/v1/${request.model.id}`,
|
|
149
|
+
}, input);
|
|
150
|
+
export const BlackForestLabsImages = {
|
|
151
|
+
protocol,
|
|
152
|
+
model,
|
|
153
|
+
};
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
2
|
+
import { MediaRoute } from "../route/media.js";
|
|
3
|
+
import { SpeechModel, type SpeechRequestFor } from "../speech.js";
|
|
4
|
+
import { SpeechStream } from "./utils/speech-stream.js";
|
|
5
|
+
export declare const DEFAULT_BASE_URL = "https://api.cartesia.ai";
|
|
6
|
+
export declare const API_VERSION = "2026-08-14";
|
|
7
|
+
export declare const BYTES_PATH = "/tts/bytes";
|
|
8
|
+
export declare const SSE_PATH = "/tts/sse";
|
|
9
|
+
export type CartesiaSpeechString<Known extends string> = Known | (string & {});
|
|
10
|
+
export type CartesiaEncoding = SpeechStream.PcmEncoding;
|
|
11
|
+
export type CartesiaSpeechOptions = {
|
|
12
|
+
readonly sampleRate?: 8000 | 16000 | 22050 | 24000 | 44100 | 48000;
|
|
13
|
+
readonly bitRate?: 32000 | 64000 | 96000 | 128000 | 192000;
|
|
14
|
+
readonly encoding?: CartesiaEncoding;
|
|
15
|
+
readonly generation_config?: {
|
|
16
|
+
readonly volume?: number;
|
|
17
|
+
readonly emotion?: CartesiaSpeechString<"neutral" | "calm" | "angry" | "content" | "sad" | "scared">;
|
|
18
|
+
};
|
|
19
|
+
readonly pronunciation_dict_id?: string;
|
|
20
|
+
} & Record<string, unknown>;
|
|
21
|
+
export type Request = SpeechRequestFor<CartesiaSpeechOptions>;
|
|
22
|
+
interface State extends SpeechStream.Audio {
|
|
23
|
+
readonly done: boolean;
|
|
24
|
+
}
|
|
25
|
+
export declare const protocol: MediaProtocol.Streamed<Request, {
|
|
26
|
+
readonly type: "audio-delta";
|
|
27
|
+
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
28
|
+
} | {
|
|
29
|
+
readonly type: "timestamps";
|
|
30
|
+
readonly items: readonly {
|
|
31
|
+
readonly text: string;
|
|
32
|
+
readonly startSeconds: number;
|
|
33
|
+
readonly endSeconds: number;
|
|
34
|
+
}[];
|
|
35
|
+
} | {
|
|
36
|
+
readonly type: "finish";
|
|
37
|
+
readonly audio: import("../media.js").Asset;
|
|
38
|
+
readonly providerMetadata?: {
|
|
39
|
+
readonly [x: string]: {
|
|
40
|
+
readonly [x: string]: unknown;
|
|
41
|
+
};
|
|
42
|
+
} | undefined;
|
|
43
|
+
readonly usage?: {
|
|
44
|
+
readonly type: "tokens";
|
|
45
|
+
readonly input?: number | undefined;
|
|
46
|
+
readonly output?: number | undefined;
|
|
47
|
+
readonly total?: number | undefined;
|
|
48
|
+
readonly details?: {
|
|
49
|
+
readonly [x: string]: unknown;
|
|
50
|
+
} | undefined;
|
|
51
|
+
} | {
|
|
52
|
+
readonly type: "seconds";
|
|
53
|
+
readonly seconds: number;
|
|
54
|
+
} | {
|
|
55
|
+
readonly type: "characters";
|
|
56
|
+
readonly characters: number;
|
|
57
|
+
} | {
|
|
58
|
+
readonly type: "credits";
|
|
59
|
+
readonly credits: number;
|
|
60
|
+
} | {
|
|
61
|
+
readonly type: "compute";
|
|
62
|
+
readonly seconds: number;
|
|
63
|
+
} | undefined;
|
|
64
|
+
readonly notices?: readonly {
|
|
65
|
+
readonly type: "other" | "moderated" | "filtered";
|
|
66
|
+
readonly message: string;
|
|
67
|
+
readonly providerMetadata?: {
|
|
68
|
+
readonly [x: string]: {
|
|
69
|
+
readonly [x: string]: unknown;
|
|
70
|
+
};
|
|
71
|
+
} | undefined;
|
|
72
|
+
}[] | undefined;
|
|
73
|
+
}, string | Uint8Array<ArrayBufferLike>, State>;
|
|
74
|
+
export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<CartesiaSpeechOptions>;
|
|
75
|
+
export declare const CartesiaSpeech: {
|
|
76
|
+
readonly protocol: MediaProtocol.Streamed<Request, {
|
|
77
|
+
readonly type: "audio-delta";
|
|
78
|
+
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
79
|
+
} | {
|
|
80
|
+
readonly type: "timestamps";
|
|
81
|
+
readonly items: readonly {
|
|
82
|
+
readonly text: string;
|
|
83
|
+
readonly startSeconds: number;
|
|
84
|
+
readonly endSeconds: number;
|
|
85
|
+
}[];
|
|
86
|
+
} | {
|
|
87
|
+
readonly type: "finish";
|
|
88
|
+
readonly audio: import("../media.js").Asset;
|
|
89
|
+
readonly providerMetadata?: {
|
|
90
|
+
readonly [x: string]: {
|
|
91
|
+
readonly [x: string]: unknown;
|
|
92
|
+
};
|
|
93
|
+
} | undefined;
|
|
94
|
+
readonly usage?: {
|
|
95
|
+
readonly type: "tokens";
|
|
96
|
+
readonly input?: number | undefined;
|
|
97
|
+
readonly output?: number | undefined;
|
|
98
|
+
readonly total?: number | undefined;
|
|
99
|
+
readonly details?: {
|
|
100
|
+
readonly [x: string]: unknown;
|
|
101
|
+
} | undefined;
|
|
102
|
+
} | {
|
|
103
|
+
readonly type: "seconds";
|
|
104
|
+
readonly seconds: number;
|
|
105
|
+
} | {
|
|
106
|
+
readonly type: "characters";
|
|
107
|
+
readonly characters: number;
|
|
108
|
+
} | {
|
|
109
|
+
readonly type: "credits";
|
|
110
|
+
readonly credits: number;
|
|
111
|
+
} | {
|
|
112
|
+
readonly type: "compute";
|
|
113
|
+
readonly seconds: number;
|
|
114
|
+
} | undefined;
|
|
115
|
+
readonly notices?: readonly {
|
|
116
|
+
readonly type: "other" | "moderated" | "filtered";
|
|
117
|
+
readonly message: string;
|
|
118
|
+
readonly providerMetadata?: {
|
|
119
|
+
readonly [x: string]: {
|
|
120
|
+
readonly [x: string]: unknown;
|
|
121
|
+
};
|
|
122
|
+
} | undefined;
|
|
123
|
+
}[] | undefined;
|
|
124
|
+
}, string | Uint8Array<ArrayBufferLike>, State>;
|
|
125
|
+
readonly model: (input: MediaRoute.ModelInput) => SpeechModel<CartesiaSpeechOptions>;
|
|
126
|
+
};
|
|
127
|
+
export {};
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
import { Effect, Schema } from "effect";
|
|
2
|
+
import { classifyProviderFailure } from "../provider-error.js";
|
|
3
|
+
import { Framing } from "../route/framing.js";
|
|
4
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
5
|
+
import { MediaRoute } from "../route/media.js";
|
|
6
|
+
import { AIError, ProviderID, mergeJsonRecords } from "../schema/index.js";
|
|
7
|
+
import { SpeechModel } from "../speech.js";
|
|
8
|
+
import { ProviderShared, optionalNull } from "./shared.js";
|
|
9
|
+
import { SpeechStream } from "./utils/speech-stream.js";
|
|
10
|
+
const ADAPTER = "cartesia-speech";
|
|
11
|
+
const NAME = "Cartesia";
|
|
12
|
+
const PROVIDER = ProviderID.make("cartesia");
|
|
13
|
+
export const DEFAULT_BASE_URL = "https://api.cartesia.ai";
|
|
14
|
+
export const API_VERSION = "2026-08-14";
|
|
15
|
+
export const BYTES_PATH = "/tts/bytes";
|
|
16
|
+
export const SSE_PATH = "/tts/sse";
|
|
17
|
+
const DEFAULT_SAMPLE_RATE = 44100;
|
|
18
|
+
const DEFAULT_BIT_RATE = 128000;
|
|
19
|
+
// ---------------------------------------------------------------------------
|
|
20
|
+
// 3. Streaming event schema
|
|
21
|
+
// ---------------------------------------------------------------------------
|
|
22
|
+
/** `phoneme_timestamps` and future record types are ignored. */
|
|
23
|
+
const SseEvent = Schema.Struct({
|
|
24
|
+
type: Schema.String,
|
|
25
|
+
data: Schema.optional(Schema.Uint8ArrayFromBase64),
|
|
26
|
+
word_timestamps: Schema.optional(Schema.Struct({
|
|
27
|
+
words: Schema.Array(Schema.String),
|
|
28
|
+
start: Schema.Array(Schema.Number),
|
|
29
|
+
end: Schema.Array(Schema.Number),
|
|
30
|
+
})),
|
|
31
|
+
status_code: Schema.optional(Schema.Number),
|
|
32
|
+
title: Schema.optional(Schema.String),
|
|
33
|
+
message: Schema.optional(Schema.String),
|
|
34
|
+
error_code: optionalNull(Schema.String),
|
|
35
|
+
});
|
|
36
|
+
const decodeEvent = MediaProtocol.decodeFrame(ADAPTER, NAME, SseEvent);
|
|
37
|
+
// ---------------------------------------------------------------------------
|
|
38
|
+
// 5. Request body construction
|
|
39
|
+
// ---------------------------------------------------------------------------
|
|
40
|
+
/** Timestamps exist only on the SSE endpoint, so a `generate` that asks for them collects an SSE stream. */
|
|
41
|
+
const usesSse = (request) => request.mode === "stream" || request.timestamps === true;
|
|
42
|
+
const CONTAINERS = { pcm: "raw", wav: "wav", mp3: "mp3" };
|
|
43
|
+
const outputFormat = Effect.fn("CartesiaSpeech.outputFormat")(function* (request) {
|
|
44
|
+
const sse = usesSse(request);
|
|
45
|
+
const format = request.format ?? (sse ? "pcm" : "mp3");
|
|
46
|
+
const container = CONTAINERS[format];
|
|
47
|
+
if (container === undefined)
|
|
48
|
+
return yield* SpeechStream.unsupportedFormat(PROVIDER, ADAPTER, `${NAME} supports the pcm, wav, and mp3 formats, not "${format}"`);
|
|
49
|
+
if (sse && container !== "raw")
|
|
50
|
+
return yield* SpeechStream.unsupportedFormat(PROVIDER, ADAPTER, `${NAME} streams and timestamps only raw PCM; request format "pcm" instead of "${format}"`);
|
|
51
|
+
const sampleRate = request.providerOptions?.sampleRate ?? DEFAULT_SAMPLE_RATE;
|
|
52
|
+
if (container === "mp3")
|
|
53
|
+
return { container, sample_rate: sampleRate, bit_rate: request.providerOptions?.bitRate ?? DEFAULT_BIT_RATE };
|
|
54
|
+
return { container, encoding: request.providerOptions?.encoding ?? "pcm_s16le", sample_rate: sampleRate };
|
|
55
|
+
});
|
|
56
|
+
const fromRequest = Effect.fn("CartesiaSpeech.fromRequest")(function* (request) {
|
|
57
|
+
const voice = SpeechStream.voiceID(request.voice);
|
|
58
|
+
if (voice === undefined)
|
|
59
|
+
return yield* ProviderShared.invalidRequest(`${NAME} requires a voice id; pass it as \`voice\``);
|
|
60
|
+
const { sampleRate: _sampleRate, bitRate: _bitRate, encoding: _encoding, ...native } = request.providerOptions ?? {};
|
|
61
|
+
return MediaProtocol.json(mergeJsonRecords({
|
|
62
|
+
model_id: request.model.id,
|
|
63
|
+
transcript: request.text,
|
|
64
|
+
voice,
|
|
65
|
+
output_format: yield* outputFormat(request),
|
|
66
|
+
language: request.language,
|
|
67
|
+
generation_config: request.speed === undefined ? undefined : { speed: request.speed },
|
|
68
|
+
add_timestamps: request.timestamps === true ? true : undefined,
|
|
69
|
+
}, native, request.http?.body) ?? {});
|
|
70
|
+
});
|
|
71
|
+
// ---------------------------------------------------------------------------
|
|
72
|
+
// 6. Stream parsing
|
|
73
|
+
// ---------------------------------------------------------------------------
|
|
74
|
+
const onEvent = Effect.fn("CartesiaSpeech.onEvent")(function* (state, frame) {
|
|
75
|
+
const event = yield* decodeEvent(frame);
|
|
76
|
+
if (event.type === "chunk" && event.data !== undefined)
|
|
77
|
+
return SpeechStream.delta(state, event.data);
|
|
78
|
+
if (event.type === "timestamps" && event.word_timestamps !== undefined) {
|
|
79
|
+
const words = event.word_timestamps;
|
|
80
|
+
return [state, SpeechStream.timestamps(words.words, words.start, words.end)];
|
|
81
|
+
}
|
|
82
|
+
if (event.type === "done")
|
|
83
|
+
return [{ ...state, done: true }, []];
|
|
84
|
+
if (event.type === "error")
|
|
85
|
+
return yield* new AIError({
|
|
86
|
+
reason: classifyProviderFailure({
|
|
87
|
+
message: `${NAME} stream failed${event.title === undefined ? "" : ` (${event.title})`}: ${event.message ?? "unknown error"}`,
|
|
88
|
+
status: event.status_code,
|
|
89
|
+
rawBody: frame,
|
|
90
|
+
}),
|
|
91
|
+
});
|
|
92
|
+
return [state, []];
|
|
93
|
+
});
|
|
94
|
+
const finish = Effect.fn("CartesiaSpeech.finish")(function* (state, context) {
|
|
95
|
+
if (usesSse(context.request) && !state.done)
|
|
96
|
+
return yield* MediaProtocol.incomplete(ADAPTER);
|
|
97
|
+
const format = yield* outputFormat(context.request);
|
|
98
|
+
return yield* SpeechStream.finish(ADAPTER, state, format.container === "raw"
|
|
99
|
+
? SpeechStream.pcm(format.encoding, format.sample_rate)
|
|
100
|
+
: SpeechStream.container(format.container, format.sample_rate));
|
|
101
|
+
});
|
|
102
|
+
// ---------------------------------------------------------------------------
|
|
103
|
+
// 7. Protocol and route
|
|
104
|
+
// ---------------------------------------------------------------------------
|
|
105
|
+
export const protocol = MediaProtocol.stream({
|
|
106
|
+
id: ADAPTER,
|
|
107
|
+
name: NAME,
|
|
108
|
+
unsupported: ["instructions"],
|
|
109
|
+
body: { from: fromRequest },
|
|
110
|
+
frames: (bytes, context) => (usesSse(context.request) ? Framing.sse.frame(bytes) : bytes),
|
|
111
|
+
initial: () => ({ chunks: [], done: false }),
|
|
112
|
+
step: SpeechStream.step(onEvent),
|
|
113
|
+
finish,
|
|
114
|
+
});
|
|
115
|
+
export const model = (input) => SpeechModel.fromRoute({
|
|
116
|
+
id: ADAPTER,
|
|
117
|
+
provider: PROVIDER,
|
|
118
|
+
protocol,
|
|
119
|
+
baseURL: DEFAULT_BASE_URL,
|
|
120
|
+
headers: { "Cartesia-Version": API_VERSION },
|
|
121
|
+
path: ({ request }) => (usesSse(request) ? SSE_PATH : BYTES_PATH),
|
|
122
|
+
}, input);
|
|
123
|
+
export const CartesiaSpeech = {
|
|
124
|
+
protocol,
|
|
125
|
+
model,
|
|
126
|
+
};
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
2
|
+
import { MediaRoute } from "../route/media.js";
|
|
3
|
+
import { SpeechModel, type SpeechRequestFor } from "../speech.js";
|
|
4
|
+
import { SpeechStream } from "./utils/speech-stream.js";
|
|
5
|
+
export declare const DEFAULT_BASE_URL = "https://api.deepgram.com";
|
|
6
|
+
export declare const PATH = "/v1/speak";
|
|
7
|
+
export type DeepgramSpeechString<Known extends string> = Known | (string & {});
|
|
8
|
+
export type DeepgramEncoding = DeepgramSpeechString<"linear16" | "mulaw" | "alaw" | "mp3" | "opus" | "flac" | "aac">;
|
|
9
|
+
export type DeepgramSpeechOptions = {
|
|
10
|
+
readonly encoding?: DeepgramEncoding;
|
|
11
|
+
readonly container?: DeepgramSpeechString<"wav" | "ogg" | "none">;
|
|
12
|
+
readonly sampleRate?: number;
|
|
13
|
+
readonly bitRate?: number;
|
|
14
|
+
readonly mip_opt_out?: boolean;
|
|
15
|
+
readonly tag?: string;
|
|
16
|
+
} & Record<string, unknown>;
|
|
17
|
+
export type Request = SpeechRequestFor<DeepgramSpeechOptions>;
|
|
18
|
+
export declare const protocol: MediaProtocol.Streamed<Request, {
|
|
19
|
+
readonly type: "audio-delta";
|
|
20
|
+
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
21
|
+
} | {
|
|
22
|
+
readonly type: "timestamps";
|
|
23
|
+
readonly items: readonly {
|
|
24
|
+
readonly text: string;
|
|
25
|
+
readonly startSeconds: number;
|
|
26
|
+
readonly endSeconds: number;
|
|
27
|
+
}[];
|
|
28
|
+
} | {
|
|
29
|
+
readonly type: "finish";
|
|
30
|
+
readonly audio: import("../media.js").Asset;
|
|
31
|
+
readonly providerMetadata?: {
|
|
32
|
+
readonly [x: string]: {
|
|
33
|
+
readonly [x: string]: unknown;
|
|
34
|
+
};
|
|
35
|
+
} | undefined;
|
|
36
|
+
readonly usage?: {
|
|
37
|
+
readonly type: "tokens";
|
|
38
|
+
readonly input?: number | undefined;
|
|
39
|
+
readonly output?: number | undefined;
|
|
40
|
+
readonly total?: number | undefined;
|
|
41
|
+
readonly details?: {
|
|
42
|
+
readonly [x: string]: unknown;
|
|
43
|
+
} | undefined;
|
|
44
|
+
} | {
|
|
45
|
+
readonly type: "seconds";
|
|
46
|
+
readonly seconds: number;
|
|
47
|
+
} | {
|
|
48
|
+
readonly type: "characters";
|
|
49
|
+
readonly characters: number;
|
|
50
|
+
} | {
|
|
51
|
+
readonly type: "credits";
|
|
52
|
+
readonly credits: number;
|
|
53
|
+
} | {
|
|
54
|
+
readonly type: "compute";
|
|
55
|
+
readonly seconds: number;
|
|
56
|
+
} | undefined;
|
|
57
|
+
readonly notices?: readonly {
|
|
58
|
+
readonly type: "other" | "moderated" | "filtered";
|
|
59
|
+
readonly message: string;
|
|
60
|
+
readonly providerMetadata?: {
|
|
61
|
+
readonly [x: string]: {
|
|
62
|
+
readonly [x: string]: unknown;
|
|
63
|
+
};
|
|
64
|
+
} | undefined;
|
|
65
|
+
}[] | undefined;
|
|
66
|
+
}, Uint8Array<ArrayBufferLike>, SpeechStream.Audio>;
|
|
67
|
+
export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<DeepgramSpeechOptions>;
|
|
68
|
+
export declare const DeepgramSpeech: {
|
|
69
|
+
readonly protocol: MediaProtocol.Streamed<Request, {
|
|
70
|
+
readonly type: "audio-delta";
|
|
71
|
+
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
72
|
+
} | {
|
|
73
|
+
readonly type: "timestamps";
|
|
74
|
+
readonly items: readonly {
|
|
75
|
+
readonly text: string;
|
|
76
|
+
readonly startSeconds: number;
|
|
77
|
+
readonly endSeconds: number;
|
|
78
|
+
}[];
|
|
79
|
+
} | {
|
|
80
|
+
readonly type: "finish";
|
|
81
|
+
readonly audio: import("../media.js").Asset;
|
|
82
|
+
readonly providerMetadata?: {
|
|
83
|
+
readonly [x: string]: {
|
|
84
|
+
readonly [x: string]: unknown;
|
|
85
|
+
};
|
|
86
|
+
} | undefined;
|
|
87
|
+
readonly usage?: {
|
|
88
|
+
readonly type: "tokens";
|
|
89
|
+
readonly input?: number | undefined;
|
|
90
|
+
readonly output?: number | undefined;
|
|
91
|
+
readonly total?: number | undefined;
|
|
92
|
+
readonly details?: {
|
|
93
|
+
readonly [x: string]: unknown;
|
|
94
|
+
} | undefined;
|
|
95
|
+
} | {
|
|
96
|
+
readonly type: "seconds";
|
|
97
|
+
readonly seconds: number;
|
|
98
|
+
} | {
|
|
99
|
+
readonly type: "characters";
|
|
100
|
+
readonly characters: number;
|
|
101
|
+
} | {
|
|
102
|
+
readonly type: "credits";
|
|
103
|
+
readonly credits: number;
|
|
104
|
+
} | {
|
|
105
|
+
readonly type: "compute";
|
|
106
|
+
readonly seconds: number;
|
|
107
|
+
} | undefined;
|
|
108
|
+
readonly notices?: readonly {
|
|
109
|
+
readonly type: "other" | "moderated" | "filtered";
|
|
110
|
+
readonly message: string;
|
|
111
|
+
readonly providerMetadata?: {
|
|
112
|
+
readonly [x: string]: {
|
|
113
|
+
readonly [x: string]: unknown;
|
|
114
|
+
};
|
|
115
|
+
} | undefined;
|
|
116
|
+
}[] | undefined;
|
|
117
|
+
}, Uint8Array<ArrayBufferLike>, SpeechStream.Audio>;
|
|
118
|
+
readonly model: (input: MediaRoute.ModelInput) => SpeechModel<DeepgramSpeechOptions>;
|
|
119
|
+
};
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
import { Effect } from "effect";
|
|
2
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
3
|
+
import { MediaRoute } from "../route/media.js";
|
|
4
|
+
import { ProviderID, mergeJsonRecords } from "../schema/index.js";
|
|
5
|
+
import { SpeechModel } from "../speech.js";
|
|
6
|
+
import { MediaInput } from "./utils/media-input.js";
|
|
7
|
+
import { SpeechStream } from "./utils/speech-stream.js";
|
|
8
|
+
const ADAPTER = "deepgram-speech";
|
|
9
|
+
const NAME = "Deepgram";
|
|
10
|
+
const PROVIDER = ProviderID.make("deepgram");
|
|
11
|
+
export const DEFAULT_BASE_URL = "https://api.deepgram.com";
|
|
12
|
+
export const PATH = "/v1/speak";
|
|
13
|
+
// ---------------------------------------------------------------------------
|
|
14
|
+
// 5. Request body construction
|
|
15
|
+
// ---------------------------------------------------------------------------
|
|
16
|
+
const FORMATS = {
|
|
17
|
+
mp3: { encoding: "mp3" },
|
|
18
|
+
wav: { encoding: "linear16", container: "wav" },
|
|
19
|
+
pcm: { encoding: "linear16", container: "none" },
|
|
20
|
+
opus: { encoding: "opus" },
|
|
21
|
+
flac: { encoding: "flac" },
|
|
22
|
+
aac: { encoding: "aac" },
|
|
23
|
+
};
|
|
24
|
+
const audioFormat = (request) => {
|
|
25
|
+
const format = request.format === undefined ? undefined : FORMATS[request.format];
|
|
26
|
+
return {
|
|
27
|
+
encoding: request.providerOptions?.encoding ?? format?.encoding,
|
|
28
|
+
container: request.providerOptions?.container ?? format?.container,
|
|
29
|
+
};
|
|
30
|
+
};
|
|
31
|
+
const queryParameters = (request) => {
|
|
32
|
+
const { encoding: _encoding, container: _container, sampleRate, bitRate, ...native } = request.providerOptions ?? {};
|
|
33
|
+
return MediaInput.query(ADAPTER, {
|
|
34
|
+
...native,
|
|
35
|
+
model: request.model.id,
|
|
36
|
+
...audioFormat(request),
|
|
37
|
+
sample_rate: sampleRate,
|
|
38
|
+
bit_rate: bitRate,
|
|
39
|
+
speed: request.speed,
|
|
40
|
+
});
|
|
41
|
+
};
|
|
42
|
+
const fromRequest = Effect.fn("DeepgramSpeech.fromRequest")(function* (request) {
|
|
43
|
+
if (request.format !== undefined &&
|
|
44
|
+
FORMATS[request.format] === undefined &&
|
|
45
|
+
request.providerOptions?.encoding === undefined)
|
|
46
|
+
return yield* SpeechStream.unsupportedFormat(PROVIDER, ADAPTER, `${NAME} has no encoding for format "${request.format}"; pass providerOptions.encoding`);
|
|
47
|
+
return MediaProtocol.json(mergeJsonRecords({ text: request.text }, request.http?.body) ?? {}, yield* queryParameters(request));
|
|
48
|
+
});
|
|
49
|
+
// ---------------------------------------------------------------------------
|
|
50
|
+
// 6. Stream parsing
|
|
51
|
+
// ---------------------------------------------------------------------------
|
|
52
|
+
const HEADERLESS_ENCODINGS = {
|
|
53
|
+
linear16: "pcm_s16le",
|
|
54
|
+
mulaw: "pcm_mulaw",
|
|
55
|
+
alaw: "pcm_alaw",
|
|
56
|
+
};
|
|
57
|
+
const finish = (state, context) => {
|
|
58
|
+
const headers = context.http.headers;
|
|
59
|
+
const mediaType = headers["content-type"];
|
|
60
|
+
const format = audioFormat(context.request);
|
|
61
|
+
const encoding = HEADERLESS_ENCODINGS[format.encoding ?? ""];
|
|
62
|
+
const requestID = headers["dg-request-id"];
|
|
63
|
+
const modelName = headers["dg-model-name"];
|
|
64
|
+
return SpeechStream.finish(ADAPTER, state, {
|
|
65
|
+
...(format.container === "none" && encoding !== undefined
|
|
66
|
+
? SpeechStream.pcm(encoding, SpeechStream.sampleRate(mediaType), mediaType)
|
|
67
|
+
: // Deepgram's default encoding is MP3; WAV is a container around any encoding.
|
|
68
|
+
{ mediaType, info: { format: format.container === "wav" ? "wav" : (format.encoding ?? "mp3") } }),
|
|
69
|
+
usage: SpeechStream.headerUsage("characters", headers["dg-char-count"]),
|
|
70
|
+
providerMetadata: requestID === undefined && modelName === undefined
|
|
71
|
+
? undefined
|
|
72
|
+
: { deepgram: { requestId: requestID, modelName } },
|
|
73
|
+
});
|
|
74
|
+
};
|
|
75
|
+
// ---------------------------------------------------------------------------
|
|
76
|
+
// 7. Protocol and route
|
|
77
|
+
// ---------------------------------------------------------------------------
|
|
78
|
+
export const protocol = MediaProtocol.stream({
|
|
79
|
+
id: ADAPTER,
|
|
80
|
+
name: NAME,
|
|
81
|
+
unsupported: ["voice", "language", "instructions", "timestamps"],
|
|
82
|
+
body: { from: fromRequest },
|
|
83
|
+
frames: (bytes) => bytes,
|
|
84
|
+
initial: () => ({ chunks: [] }),
|
|
85
|
+
step: (state, frame) => Effect.succeed(SpeechStream.delta(state, frame)),
|
|
86
|
+
finish,
|
|
87
|
+
});
|
|
88
|
+
export const model = (input) => SpeechModel.fromRoute({ id: ADAPTER, provider: PROVIDER, protocol, baseURL: DEFAULT_BASE_URL, path: PATH }, input);
|
|
89
|
+
export const DeepgramSpeech = {
|
|
90
|
+
protocol,
|
|
91
|
+
model,
|
|
92
|
+
};
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
2
|
+
import { MediaRoute } from "../route/media.js";
|
|
3
|
+
import { TranscriptionModel, TranscriptionResponse, type TranscriptionRequestFor } from "../transcription.js";
|
|
4
|
+
export declare const DEFAULT_BASE_URL = "https://api.deepgram.com";
|
|
5
|
+
export declare const PATH = "/v1/listen";
|
|
6
|
+
export type DeepgramTranscriptionOptions = {
|
|
7
|
+
readonly smart_format?: boolean;
|
|
8
|
+
readonly punctuate?: boolean;
|
|
9
|
+
readonly paragraphs?: boolean;
|
|
10
|
+
readonly utterances?: boolean;
|
|
11
|
+
readonly detect_language?: boolean | ReadonlyArray<string>;
|
|
12
|
+
readonly keyterm?: ReadonlyArray<string>;
|
|
13
|
+
readonly diarize_model?: "latest" | "v1" | "v2" | (string & {});
|
|
14
|
+
readonly filler_words?: boolean;
|
|
15
|
+
readonly numerals?: boolean;
|
|
16
|
+
readonly mip_opt_out?: boolean;
|
|
17
|
+
readonly tag?: string | ReadonlyArray<string>;
|
|
18
|
+
} & Record<string, unknown>;
|
|
19
|
+
export type Request = TranscriptionRequestFor<DeepgramTranscriptionOptions>;
|
|
20
|
+
export declare const protocol: MediaProtocol.Inline<Request, TranscriptionResponse>;
|
|
21
|
+
export declare const model: (input: MediaRoute.ModelInput) => TranscriptionModel<DeepgramTranscriptionOptions>;
|
|
22
|
+
export declare const DeepgramTranscription: {
|
|
23
|
+
readonly protocol: MediaProtocol.Inline<Request, TranscriptionResponse>;
|
|
24
|
+
readonly model: (input: MediaRoute.ModelInput) => TranscriptionModel<DeepgramTranscriptionOptions>;
|
|
25
|
+
};
|