@opencode/ai 2.0.15 → 2.0.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +286 -2
- package/dist/generation.d.ts +36 -22
- package/dist/generation.js +53 -24
- package/dist/image-client.d.ts +14 -7
- package/dist/image-client.js +24 -8
- package/dist/image.d.ts +398 -47
- package/dist/image.js +47 -45
- package/dist/index.d.ts +13 -1
- package/dist/index.js +9 -0
- package/dist/media-model.d.ts +44 -0
- package/dist/media-model.js +49 -0
- package/dist/media.d.ts +10 -9
- package/dist/media.js +9 -10
- package/dist/promise.d.ts +428 -8
- package/dist/promise.js +40 -3
- package/dist/protocols/alibaba-chat.d.ts +12 -0
- package/dist/protocols/alibaba-responses.d.ts +2 -2
- package/dist/protocols/anthropic-messages.js +1 -2
- package/dist/protocols/assemblyai-transcription.d.ts +40 -0
- package/dist/protocols/assemblyai-transcription.js +138 -0
- package/dist/protocols/bedrock-converse.js +5 -11
- package/dist/protocols/bfl-images.d.ts +32 -0
- package/dist/protocols/bfl-images.js +153 -0
- package/dist/protocols/cartesia-speech.d.ts +127 -0
- package/dist/protocols/cartesia-speech.js +126 -0
- package/dist/protocols/deepgram-speech.d.ts +119 -0
- package/dist/protocols/deepgram-speech.js +92 -0
- package/dist/protocols/deepgram-transcription.d.ts +25 -0
- package/dist/protocols/deepgram-transcription.js +129 -0
- package/dist/protocols/elevenlabs-speech.d.ts +122 -0
- package/dist/protocols/elevenlabs-speech.js +115 -0
- package/dist/protocols/fal-images.d.ts +24 -0
- package/dist/protocols/fal-images.js +114 -0
- package/dist/protocols/fal-video.d.ts +29 -0
- package/dist/protocols/fal-video.js +88 -0
- package/dist/protocols/gemini.d.ts +9 -9
- package/dist/protocols/gemini.js +8 -34
- package/dist/protocols/google-images.js +2 -14
- package/dist/protocols/google-speech.d.ts +130 -0
- package/dist/protocols/google-speech.js +84 -0
- package/dist/protocols/google-transcription.d.ts +173 -0
- package/dist/protocols/google-transcription.js +138 -0
- package/dist/protocols/google-video.d.ts +26 -0
- package/dist/protocols/google-video.js +158 -0
- package/dist/protocols/meta-images.js +2 -9
- package/dist/protocols/meta-responses.d.ts +4 -4
- package/dist/protocols/meta-responses.js +1 -1
- package/dist/protocols/open-responses.d.ts +6 -6
- package/dist/protocols/open-responses.js +1 -2
- package/dist/protocols/openai-chat.d.ts +84 -0
- package/dist/protocols/openai-chat.js +26 -14
- package/dist/protocols/openai-compatible-chat.d.ts +12 -0
- package/dist/protocols/openai-compatible-responses.d.ts +2 -2
- package/dist/protocols/openai-images.d.ts +124 -3
- package/dist/protocols/openai-images.js +107 -54
- package/dist/protocols/openai-responses.d.ts +15 -15
- package/dist/protocols/openai-responses.js +5 -6
- package/dist/protocols/openai-speech.d.ts +116 -0
- package/dist/protocols/openai-speech.js +98 -0
- package/dist/protocols/openai-transcription.d.ts +207 -0
- package/dist/protocols/openai-transcription.js +190 -0
- package/dist/protocols/replicate-images.d.ts +28 -0
- package/dist/protocols/replicate-images.js +133 -0
- package/dist/protocols/runway-video.d.ts +38 -0
- package/dist/protocols/runway-video.js +146 -0
- package/dist/protocols/shared.d.ts +13 -3
- package/dist/protocols/shared.js +23 -3
- package/dist/protocols/stability-images.d.ts +38 -0
- package/dist/protocols/stability-images.js +148 -0
- package/dist/protocols/utils/fal-queue.d.ts +28 -0
- package/dist/protocols/utils/fal-queue.js +69 -0
- package/dist/protocols/utils/gemini-generate-content.d.ts +65 -0
- package/dist/protocols/utils/gemini-generate-content.js +65 -0
- package/dist/protocols/utils/gemini-json-schema.d.ts +3 -0
- package/dist/protocols/utils/gemini-json-schema.js +76 -0
- package/dist/protocols/utils/media-input.d.ts +8 -0
- package/dist/protocols/utils/media-input.js +18 -0
- package/dist/protocols/utils/speech-stream.d.ts +49 -0
- package/dist/protocols/utils/speech-stream.js +67 -0
- package/dist/protocols/utils/tool-schema.d.ts +2 -2
- package/dist/protocols/utils/tool-schema.js +40 -17
- package/dist/protocols/xai-images.js +1 -12
- package/dist/protocols/xai-responses.d.ts +2 -2
- package/dist/protocols/xai-video.d.ts +34 -0
- package/dist/protocols/xai-video.js +147 -0
- package/dist/protocols/zai-chat.d.ts +13 -1
- package/dist/provider-error.js +3 -0
- package/dist/providers/alibaba.d.ts +14 -2
- package/dist/providers/amazon-bedrock-mantle.d.ts +14 -2
- package/dist/providers/assemblyai.d.ts +25 -0
- package/dist/providers/assemblyai.js +29 -0
- package/dist/providers/azure.d.ts +18 -6
- package/dist/providers/baseten.d.ts +24 -0
- package/dist/providers/black-forest-labs.d.ts +25 -0
- package/dist/providers/black-forest-labs.js +28 -0
- package/dist/providers/cartesia.d.ts +24 -0
- package/dist/providers/cartesia.js +22 -0
- package/dist/providers/cerebras.d.ts +24 -0
- package/dist/providers/cloudflare-ai-gateway.d.ts +30 -6
- package/dist/providers/cloudflare-workers-ai.d.ts +24 -0
- package/dist/providers/deepgram.d.ts +29 -0
- package/dist/providers/deepgram.js +31 -0
- package/dist/providers/deepinfra.d.ts +24 -0
- package/dist/providers/deepseek.d.ts +24 -0
- package/dist/providers/elevenlabs.d.ts +24 -0
- package/dist/providers/elevenlabs.js +28 -0
- package/dist/providers/fal.d.ts +29 -0
- package/dist/providers/fal.js +33 -0
- package/dist/providers/fireworks.d.ts +24 -0
- package/dist/providers/google-vertex-chat.d.ts +12 -0
- package/dist/providers/google-vertex-responses.d.ts +2 -2
- package/dist/providers/google-vertex.d.ts +3 -3
- package/dist/providers/google.d.ts +18 -3
- package/dist/providers/google.js +11 -2
- package/dist/providers/groq.d.ts +24 -0
- package/dist/providers/index.d.ts +9 -0
- package/dist/providers/index.js +9 -0
- package/dist/providers/meta.d.ts +14 -2
- package/dist/providers/minimax.d.ts +14 -2
- package/dist/providers/moonshot.d.ts +14 -2
- package/dist/providers/moonshot.js +3 -3
- package/dist/providers/openai-compatible-responses.d.ts +2 -2
- package/dist/providers/openai-compatible.d.ts +12 -0
- package/dist/providers/openai.d.ts +25 -3
- package/dist/providers/openai.js +10 -1
- package/dist/providers/openrouter.d.ts +48 -0
- package/dist/providers/replicate.d.ts +25 -0
- package/dist/providers/replicate.js +22 -0
- package/dist/providers/runway.d.ts +24 -0
- package/dist/providers/runway.js +22 -0
- package/dist/providers/stability.d.ts +28 -0
- package/dist/providers/stability.js +23 -0
- package/dist/providers/togetherai.d.ts +24 -0
- package/dist/providers/xai.d.ts +17 -0
- package/dist/providers/xai.js +5 -2
- package/dist/providers/zai-coding-plan.d.ts +15 -3
- package/dist/providers/zai.d.ts +13 -1
- package/dist/route/auth.d.ts +4 -1
- package/dist/route/auth.js +6 -0
- package/dist/route/framing.d.ts +5 -1
- package/dist/route/framing.js +9 -0
- package/dist/route/media-protocol.d.ts +116 -3
- package/dist/route/media-protocol.js +60 -4
- package/dist/route/media.d.ts +58 -7
- package/dist/route/media.js +201 -29
- package/dist/schema/events.d.ts +0 -6
- package/dist/schema/messages.d.ts +0 -3
- package/dist/schema/options.d.ts +4 -3
- package/dist/schema/options.js +3 -2
- package/dist/speech-client.d.ts +21 -0
- package/dist/speech-client.js +25 -0
- package/dist/speech.d.ts +1150 -0
- package/dist/speech.js +119 -0
- package/dist/transcription-client.d.ts +28 -0
- package/dist/transcription-client.js +44 -0
- package/dist/transcription.d.ts +1504 -0
- package/dist/transcription.js +133 -0
- package/dist/utils/bytes.d.ts +1 -0
- package/dist/utils/bytes.js +10 -0
- package/dist/utils/media-type.d.ts +1 -0
- package/dist/utils/media-type.js +22 -1
- package/dist/video-client.d.ts +28 -0
- package/dist/video-client.js +40 -0
- package/dist/video.d.ts +1359 -0
- package/dist/video.js +119 -0
- package/package.json +3 -3
- package/dist/protocols/utils/gemini-tool-schema.d.ts +0 -2
- package/dist/protocols/utils/gemini-tool-schema.js +0 -103
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
import { Effect, Schema } from "effect";
|
|
2
|
+
import { Media } from "../media.js";
|
|
3
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
4
|
+
import { MediaRoute } from "../route/media.js";
|
|
5
|
+
import { ProviderID, mergeJsonRecords } from "../schema/index.js";
|
|
6
|
+
import { TranscriptionModel, TranscriptionResponse } from "../transcription.js";
|
|
7
|
+
import { ProviderShared, optionalNull } from "./shared.js";
|
|
8
|
+
import { MediaInput } from "./utils/media-input.js";
|
|
9
|
+
const ADAPTER = "assemblyai-transcription";
|
|
10
|
+
const NAME = "AssemblyAI";
|
|
11
|
+
const PROVIDER = ProviderID.make("assemblyai");
|
|
12
|
+
export const DEFAULT_BASE_URL = "https://api.assemblyai.com";
|
|
13
|
+
export const PATH = "/v2/transcript";
|
|
14
|
+
export const UPLOAD_PATH = "/v2/upload";
|
|
15
|
+
// ---------------------------------------------------------------------------
|
|
16
|
+
// 2. Token and response schemas
|
|
17
|
+
// ---------------------------------------------------------------------------
|
|
18
|
+
export const Token = Schema.Struct({ transcriptID: Schema.String });
|
|
19
|
+
const Upload = Schema.Struct({ upload_url: Schema.String });
|
|
20
|
+
/** Word and utterance times are milliseconds. */
|
|
21
|
+
const Transcript = Schema.Struct({
|
|
22
|
+
id: Schema.String,
|
|
23
|
+
status: Schema.String,
|
|
24
|
+
text: optionalNull(Schema.String),
|
|
25
|
+
words: optionalNull(Schema.Array(Schema.Struct({
|
|
26
|
+
text: Schema.String,
|
|
27
|
+
start: Schema.Number,
|
|
28
|
+
end: Schema.Number,
|
|
29
|
+
confidence: optionalNull(Schema.Number),
|
|
30
|
+
speaker: optionalNull(Schema.String),
|
|
31
|
+
}))),
|
|
32
|
+
utterances: optionalNull(Schema.Array(Schema.Struct({
|
|
33
|
+
text: Schema.String,
|
|
34
|
+
start: Schema.Number,
|
|
35
|
+
end: Schema.Number,
|
|
36
|
+
speaker: optionalNull(Schema.String),
|
|
37
|
+
}))),
|
|
38
|
+
language_code: optionalNull(Schema.String),
|
|
39
|
+
audio_duration: optionalNull(Schema.Number),
|
|
40
|
+
speech_model_used: optionalNull(Schema.String),
|
|
41
|
+
error: optionalNull(Schema.String),
|
|
42
|
+
});
|
|
43
|
+
const STATUS = {
|
|
44
|
+
queued: "queued",
|
|
45
|
+
processing: "running",
|
|
46
|
+
completed: "completed",
|
|
47
|
+
error: "failed",
|
|
48
|
+
};
|
|
49
|
+
// ---------------------------------------------------------------------------
|
|
50
|
+
// 5. Request body construction
|
|
51
|
+
// ---------------------------------------------------------------------------
|
|
52
|
+
const decodeUpload = MediaProtocol.decodeJson(ADAPTER, NAME, Upload);
|
|
53
|
+
/** `/v2/transcript` only takes a URL, so inline audio is uploaded to `/v2/upload` first. */
|
|
54
|
+
const prepare = Effect.fn("AssemblyAITranscription.prepare")(function* (request, send) {
|
|
55
|
+
if (request.audio.source.type !== "bytes" && request.audio.source.type !== "base64")
|
|
56
|
+
return request;
|
|
57
|
+
const audio = yield* MediaInput.inlineBytes(ADAPTER, request.audio);
|
|
58
|
+
const uploaded = yield* send(UPLOAD_PATH, MediaProtocol.binary(audio, "application/octet-stream")).pipe(Effect.flatMap(decodeUpload));
|
|
59
|
+
return { ...request, audio: Media.url(uploaded.value.upload_url, { mediaType: request.audio.mediaType }) };
|
|
60
|
+
});
|
|
61
|
+
const fromRequest = Effect.fn("AssemblyAITranscription.fromRequest")(function* (request) {
|
|
62
|
+
const audio = yield* ProviderShared.mediaReference(request.audio, PROVIDER, NAME);
|
|
63
|
+
return MediaProtocol.json(mergeJsonRecords({
|
|
64
|
+
audio_url: audio.value,
|
|
65
|
+
speech_models: [request.model.id],
|
|
66
|
+
language_code: request.language,
|
|
67
|
+
language_detection: request.language === undefined ? true : undefined,
|
|
68
|
+
prompt: request.prompt,
|
|
69
|
+
// Turn-level `utterances`, the only segments AssemblyAI returns, require speaker labels.
|
|
70
|
+
speaker_labels: request.diarize === true || request.timestamps === "segment" ? true : undefined,
|
|
71
|
+
speakers_expected: request.speakers,
|
|
72
|
+
}, request.providerOptions, request.http?.body) ?? {});
|
|
73
|
+
});
|
|
74
|
+
// ---------------------------------------------------------------------------
|
|
75
|
+
// 6. Response decoding
|
|
76
|
+
// ---------------------------------------------------------------------------
|
|
77
|
+
const decodeTranscript = MediaProtocol.decodeJson(ADAPTER, NAME, Transcript);
|
|
78
|
+
const decodeStart = Effect.fn("AssemblyAITranscription.decodeStart")(function* (response) {
|
|
79
|
+
const output = yield* decodeTranscript(response);
|
|
80
|
+
const status = yield* MediaProtocol.status(STATUS, output.value.status, output);
|
|
81
|
+
return { token: { transcriptID: output.value.id }, snapshot: { id: output.value.id, status } };
|
|
82
|
+
});
|
|
83
|
+
const decodeStatus = Effect.fn("AssemblyAITranscription.decodeStatus")(function* (response, context) {
|
|
84
|
+
const output = yield* decodeTranscript(response);
|
|
85
|
+
const status = yield* MediaProtocol.status(STATUS, output.value.status, output);
|
|
86
|
+
return { id: context.token.transcriptID, status };
|
|
87
|
+
});
|
|
88
|
+
const seconds = (milliseconds) => milliseconds / 1000;
|
|
89
|
+
const decodeResult = Effect.fn("AssemblyAITranscription.decodeResult")(function* (response, context) {
|
|
90
|
+
const output = yield* decodeTranscript(response);
|
|
91
|
+
const transcript = output.value;
|
|
92
|
+
const status = yield* MediaProtocol.status(STATUS, transcript.status, output);
|
|
93
|
+
const error = transcript.error ?? undefined;
|
|
94
|
+
if (status === "failed")
|
|
95
|
+
return yield* output.ended("failed", `${NAME} transcription failed${error === undefined ? "" : `: ${error}`}`);
|
|
96
|
+
if (status !== "completed")
|
|
97
|
+
return yield* output.invalid(`${NAME} transcript ${context.token.transcriptID} has not finished`);
|
|
98
|
+
const duration = transcript.audio_duration ?? undefined;
|
|
99
|
+
return new TranscriptionResponse({
|
|
100
|
+
text: transcript.text ?? "",
|
|
101
|
+
segments: transcript.utterances?.map((utterance) => ({
|
|
102
|
+
text: utterance.text,
|
|
103
|
+
startSeconds: seconds(utterance.start),
|
|
104
|
+
endSeconds: seconds(utterance.end),
|
|
105
|
+
speaker: utterance.speaker ?? undefined,
|
|
106
|
+
})),
|
|
107
|
+
words: transcript.words?.map((word) => ({
|
|
108
|
+
text: word.text,
|
|
109
|
+
startSeconds: seconds(word.start),
|
|
110
|
+
endSeconds: seconds(word.end),
|
|
111
|
+
speaker: word.speaker ?? undefined,
|
|
112
|
+
confidence: word.confidence ?? undefined,
|
|
113
|
+
})),
|
|
114
|
+
language: transcript.language_code?.toLowerCase(),
|
|
115
|
+
durationSeconds: duration,
|
|
116
|
+
usage: duration === undefined ? undefined : { type: "seconds", seconds: duration },
|
|
117
|
+
providerMetadata: {
|
|
118
|
+
assemblyai: { transcriptId: transcript.id, speechModel: transcript.speech_model_used ?? undefined },
|
|
119
|
+
},
|
|
120
|
+
});
|
|
121
|
+
});
|
|
122
|
+
// ---------------------------------------------------------------------------
|
|
123
|
+
// 7. Protocol and route
|
|
124
|
+
// ---------------------------------------------------------------------------
|
|
125
|
+
const transcriptPath = (token) => `${PATH}/${token.transcriptID}`;
|
|
126
|
+
export const protocol = MediaProtocol.queued({
|
|
127
|
+
id: ADAPTER,
|
|
128
|
+
name: NAME,
|
|
129
|
+
token: Token,
|
|
130
|
+
start: { prepare, body: { from: fromRequest }, decode: decodeStart },
|
|
131
|
+
status: { path: transcriptPath, decode: decodeStatus },
|
|
132
|
+
result: { path: transcriptPath, decode: decodeResult },
|
|
133
|
+
});
|
|
134
|
+
export const model = (input) => TranscriptionModel.fromRoute({ id: ADAPTER, provider: PROVIDER, protocol, baseURL: DEFAULT_BASE_URL, path: PATH }, input);
|
|
135
|
+
export const AssemblyAITranscription = {
|
|
136
|
+
protocol,
|
|
137
|
+
model,
|
|
138
|
+
};
|
|
@@ -13,6 +13,7 @@ import { Lifecycle } from "./utils/lifecycle.js";
|
|
|
13
13
|
import { MistralToolID } from "./utils/mistral-tool-id.js";
|
|
14
14
|
import { ToolSchemaProjection } from "./utils/tool-schema.js";
|
|
15
15
|
import { ToolStream } from "./utils/tool-stream.js";
|
|
16
|
+
import { concatBytes } from "../utils/bytes.js";
|
|
16
17
|
const ADAPTER = "bedrock-converse";
|
|
17
18
|
// =============================================================================
|
|
18
19
|
// Request Body Schema
|
|
@@ -162,10 +163,10 @@ const lowerToolSpec = (tool, inputSchema) => ({
|
|
|
162
163
|
inputSchema: { json: inputSchema },
|
|
163
164
|
},
|
|
164
165
|
});
|
|
165
|
-
const lowerTools = (
|
|
166
|
+
const lowerTools = (model, breakpoints, tools) => {
|
|
166
167
|
const result = [];
|
|
167
168
|
for (const tool of tools) {
|
|
168
|
-
result.push(lowerToolSpec(tool, ToolSchemaProjection.modelCompatibility(tool.inputSchema,
|
|
169
|
+
result.push(lowerToolSpec(tool, ToolSchemaProjection.modelCompatibility(tool.inputSchema, model)));
|
|
169
170
|
const cachePoint = BedrockCache.block(breakpoints, tool.cache);
|
|
170
171
|
if (cachePoint)
|
|
171
172
|
result.push(cachePoint);
|
|
@@ -348,7 +349,7 @@ const fromRequest = Effect.fn("BedrockConverse.fromRequest")(function* (request)
|
|
|
348
349
|
if (flattened.tools.length === 0)
|
|
349
350
|
return undefined;
|
|
350
351
|
return {
|
|
351
|
-
tools: lowerTools(request.model
|
|
352
|
+
tools: lowerTools(request.model, breakpoints, flattened.tools),
|
|
352
353
|
// Converse has no native "none". Keep definitions stable for prompt
|
|
353
354
|
// caching and omit only the unsupported choice.
|
|
354
355
|
toolChoice,
|
|
@@ -413,14 +414,7 @@ const mapUsage = (usage, providerMetadataKey) => {
|
|
|
413
414
|
providerMetadata: { [providerMetadataKey]: usage },
|
|
414
415
|
});
|
|
415
416
|
};
|
|
416
|
-
const encodeRedactedContent = (chunks) =>
|
|
417
|
-
const bytes = new Uint8Array(chunks.reduce((total, chunk) => total + chunk.length, 0));
|
|
418
|
-
chunks.reduce((offset, chunk) => {
|
|
419
|
-
bytes.set(chunk, offset);
|
|
420
|
-
return offset + chunk.length;
|
|
421
|
-
}, 0);
|
|
422
|
-
return Encoding.encodeBase64(bytes);
|
|
423
|
-
};
|
|
417
|
+
const encodeRedactedContent = (chunks) => Encoding.encodeBase64(concatBytes(chunks));
|
|
424
418
|
const step = (state, event) => Effect.gen(function* () {
|
|
425
419
|
if (event.contentBlockStart?.start?.toolUse) {
|
|
426
420
|
const index = event.contentBlockStart.contentBlockIndex;
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import { Schema } from "effect";
|
|
2
|
+
import { ImageModel, ImageResponse, type ImageRequestFor } from "../image.js";
|
|
3
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
4
|
+
import { MediaRoute } from "../route/media.js";
|
|
5
|
+
export declare const DEFAULT_BASE_URL = "https://api.bfl.ai";
|
|
6
|
+
export type BlackForestLabsImageOptions = {
|
|
7
|
+
readonly safety_tolerance?: number;
|
|
8
|
+
readonly prompt_upsampling?: boolean;
|
|
9
|
+
readonly disable_pup?: boolean;
|
|
10
|
+
readonly raw?: boolean;
|
|
11
|
+
readonly guidance?: number;
|
|
12
|
+
readonly steps?: number;
|
|
13
|
+
} & Record<string, unknown>;
|
|
14
|
+
export type Request = ImageRequestFor<BlackForestLabsImageOptions>;
|
|
15
|
+
/** Regional clusters answer on different hosts, so the returned `polling_url` is followed verbatim. */
|
|
16
|
+
export declare const Token: Schema.Struct<{
|
|
17
|
+
readonly id: Schema.String;
|
|
18
|
+
readonly pollingURL: Schema.String;
|
|
19
|
+
}>;
|
|
20
|
+
export type Token = Schema.Schema.Type<typeof Token>;
|
|
21
|
+
export declare const protocol: MediaProtocol.Queued<Request, ImageResponse, {
|
|
22
|
+
readonly id: string;
|
|
23
|
+
readonly pollingURL: string;
|
|
24
|
+
}>;
|
|
25
|
+
export declare const model: (input: MediaRoute.ModelInput) => ImageModel<BlackForestLabsImageOptions>;
|
|
26
|
+
export declare const BlackForestLabsImages: {
|
|
27
|
+
readonly protocol: MediaProtocol.Queued<Request, ImageResponse, {
|
|
28
|
+
readonly id: string;
|
|
29
|
+
readonly pollingURL: string;
|
|
30
|
+
}>;
|
|
31
|
+
readonly model: (input: MediaRoute.ModelInput) => ImageModel<BlackForestLabsImageOptions>;
|
|
32
|
+
};
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
import { Effect, Schema } from "effect";
|
|
2
|
+
import { ImageModel, ImageResponse } from "../image.js";
|
|
3
|
+
import { Media } from "../media.js";
|
|
4
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
5
|
+
import { MediaRoute } from "../route/media.js";
|
|
6
|
+
import { ProviderID, mergeJsonRecords } from "../schema/index.js";
|
|
7
|
+
import { ProviderShared, optionalNull } from "./shared.js";
|
|
8
|
+
import { MediaInput } from "./utils/media-input.js";
|
|
9
|
+
const ADAPTER = "bfl-images";
|
|
10
|
+
const NAME = "Black Forest Labs";
|
|
11
|
+
const PROVIDER = ProviderID.make("black-forest-labs");
|
|
12
|
+
export const DEFAULT_BASE_URL = "https://api.bfl.ai";
|
|
13
|
+
// ---------------------------------------------------------------------------
|
|
14
|
+
// 2. Token and response schemas
|
|
15
|
+
// ---------------------------------------------------------------------------
|
|
16
|
+
/** Regional clusters answer on different hosts, so the returned `polling_url` is followed verbatim. */
|
|
17
|
+
export const Token = Schema.Struct({ id: Schema.String, pollingURL: Schema.String });
|
|
18
|
+
const StartResponse = Schema.Struct({
|
|
19
|
+
id: Schema.String,
|
|
20
|
+
polling_url: Schema.String,
|
|
21
|
+
});
|
|
22
|
+
const Result = Schema.Struct({
|
|
23
|
+
id: Schema.String,
|
|
24
|
+
status: Schema.String,
|
|
25
|
+
result: optionalNull(Schema.StructWithRest(Schema.Struct({ sample: Schema.String, seed: optionalNull(Schema.Number), prompt: optionalNull(Schema.String) }), [Schema.Record(Schema.String, Schema.Unknown)])),
|
|
26
|
+
cost: optionalNull(Schema.Number),
|
|
27
|
+
});
|
|
28
|
+
const STATUS = {
|
|
29
|
+
Pending: "running",
|
|
30
|
+
Reasoning: "running",
|
|
31
|
+
Generating: "running",
|
|
32
|
+
Ready: "completed",
|
|
33
|
+
Error: "failed",
|
|
34
|
+
// Moderation is terminal; `decodeResult` reports it as a content-policy failure.
|
|
35
|
+
"Content Moderated": "failed",
|
|
36
|
+
"Request Moderated": "failed",
|
|
37
|
+
"Task not found": "expired",
|
|
38
|
+
};
|
|
39
|
+
const isModerated = (status) => status === "Content Moderated" || status === "Request Moderated";
|
|
40
|
+
const capabilities = (model) => {
|
|
41
|
+
if (model.startsWith("flux-pro-1.0-fill"))
|
|
42
|
+
return { sizing: "none", imageField: "image", maxImages: 1, mask: true };
|
|
43
|
+
if (model.startsWith("flux-pro-1.0-expand"))
|
|
44
|
+
return { sizing: "none", imageField: "image", maxImages: 1, mask: false };
|
|
45
|
+
if (model.startsWith("flux-kontext"))
|
|
46
|
+
return { sizing: "aspectRatio", imageField: "input_image", maxImages: 4, mask: false };
|
|
47
|
+
if (model.startsWith("flux-pro-1.1-ultra"))
|
|
48
|
+
return { sizing: "aspectRatio", imageField: "image_prompt", maxImages: 1, mask: false };
|
|
49
|
+
if (model.startsWith("flux-pro-1.1") || model.startsWith("flux-dev"))
|
|
50
|
+
return { sizing: "dimensions", imageField: "image_prompt", maxImages: 1, mask: false };
|
|
51
|
+
if (model.startsWith("flux-2-klein"))
|
|
52
|
+
return { sizing: "dimensions", imageField: "input_image", maxImages: 4, mask: false };
|
|
53
|
+
return { sizing: "dimensions", imageField: "input_image", maxImages: 8, mask: false };
|
|
54
|
+
};
|
|
55
|
+
const unsupported = (model, field, message) => ProviderShared.unsupportedOperation({
|
|
56
|
+
operation: `media.${field}`,
|
|
57
|
+
provider: PROVIDER,
|
|
58
|
+
route: ADAPTER,
|
|
59
|
+
message: `${model} ${message}`,
|
|
60
|
+
});
|
|
61
|
+
const validate = (request, model) => {
|
|
62
|
+
const id = request.model.id;
|
|
63
|
+
const images = request.images?.length ?? 0;
|
|
64
|
+
if (request.n !== undefined && request.n > 1)
|
|
65
|
+
return Effect.fail(unsupported(id, "n", "generates one image per request; call it once per image"));
|
|
66
|
+
if (request.size !== undefined && model.sizing !== "dimensions")
|
|
67
|
+
return Effect.fail(unsupported(id, "size", "does not take size (width and height)"));
|
|
68
|
+
if (request.aspectRatio !== undefined && model.sizing !== "aspectRatio")
|
|
69
|
+
return Effect.fail(unsupported(id, "aspectRatio", "does not take aspectRatio"));
|
|
70
|
+
if (images > model.maxImages)
|
|
71
|
+
return Effect.fail(unsupported(id, "images", `takes at most ${model.maxImages} images`));
|
|
72
|
+
if (request.mask !== undefined && !model.mask)
|
|
73
|
+
return Effect.fail(unsupported(id, "mask", "does not inpaint; use flux-pro-1.0-fill"));
|
|
74
|
+
return Effect.void;
|
|
75
|
+
};
|
|
76
|
+
const imageInput = (asset) => {
|
|
77
|
+
const value = asset.inline()?.base64 ?? ProviderShared.mediaUrl(asset);
|
|
78
|
+
if (value === undefined)
|
|
79
|
+
return Effect.fail(ProviderShared.invalidRequest(`${NAME} accepts inline images or https URLs`));
|
|
80
|
+
return Effect.succeed(value);
|
|
81
|
+
};
|
|
82
|
+
const fromRequest = Effect.fn("BlackForestLabsImages.fromRequest")(function* (request) {
|
|
83
|
+
const model = capabilities(request.model.id);
|
|
84
|
+
yield* validate(request, model);
|
|
85
|
+
const images = yield* Effect.forEach(request.images ?? [], imageInput);
|
|
86
|
+
const fields = images.map((image, index) => [
|
|
87
|
+
index === 0 ? model.imageField : `${model.imageField}_${index + 1}`,
|
|
88
|
+
image,
|
|
89
|
+
]);
|
|
90
|
+
return MediaProtocol.json(mergeJsonRecords({
|
|
91
|
+
prompt: request.prompt,
|
|
92
|
+
...(request.size === undefined ? {} : MediaInput.dimensions(request.size)),
|
|
93
|
+
aspect_ratio: request.aspectRatio,
|
|
94
|
+
seed: request.seed,
|
|
95
|
+
output_format: request.format,
|
|
96
|
+
mask: request.mask === undefined ? undefined : yield* imageInput(request.mask),
|
|
97
|
+
...Object.fromEntries(fields),
|
|
98
|
+
}, request.providerOptions, request.http?.body) ?? {});
|
|
99
|
+
});
|
|
100
|
+
// ---------------------------------------------------------------------------
|
|
101
|
+
// 6. Response decoding
|
|
102
|
+
// ---------------------------------------------------------------------------
|
|
103
|
+
const decodeStart = MediaProtocol.decodeStarted(ADAPTER, NAME, StartResponse, (value) => ({
|
|
104
|
+
token: { id: value.id, pollingURL: value.polling_url },
|
|
105
|
+
snapshot: { id: value.id, status: "queued" },
|
|
106
|
+
}));
|
|
107
|
+
const decodeDocument = MediaProtocol.decodeJson(ADAPTER, NAME, Result);
|
|
108
|
+
const decodeStatus = Effect.fn("BlackForestLabsImages.decodeStatus")(function* (response, context) {
|
|
109
|
+
const output = yield* decodeDocument(response);
|
|
110
|
+
return { id: context.token.id, status: yield* MediaProtocol.status(STATUS, output.value.status, output) };
|
|
111
|
+
});
|
|
112
|
+
const decodeResult = Effect.fn("BlackForestLabsImages.decodeResult")(function* (response, context) {
|
|
113
|
+
const output = yield* decodeDocument(response);
|
|
114
|
+
const document = output.value;
|
|
115
|
+
const status = yield* MediaProtocol.status(STATUS, document.status, output);
|
|
116
|
+
if (isModerated(document.status))
|
|
117
|
+
return yield* output.contentPolicy(`${NAME} moderated the generation`);
|
|
118
|
+
if (status === "failed" || status === "expired")
|
|
119
|
+
return yield* output.ended(status, `${NAME} generation ${context.token.id} ended with ${document.status}`);
|
|
120
|
+
if (status !== "completed" || document.result === undefined || document.result === null)
|
|
121
|
+
return yield* output.invalid(`${NAME} generation ${context.token.id} has no result`);
|
|
122
|
+
const { sample, seed, prompt, ...rest } = document.result;
|
|
123
|
+
return new ImageResponse({
|
|
124
|
+
// `sample` is a signed URL that expires 10 minutes after the result is ready, so it is downloaded now.
|
|
125
|
+
images: [yield* context.materialize(Media.url(sample))],
|
|
126
|
+
usage: document.cost === undefined || document.cost === null ? undefined : { type: "credits", credits: document.cost },
|
|
127
|
+
providerMetadata: {
|
|
128
|
+
bfl: { id: context.token.id, seed: seed ?? undefined, prompt: prompt ?? undefined, ...rest },
|
|
129
|
+
},
|
|
130
|
+
});
|
|
131
|
+
});
|
|
132
|
+
// ---------------------------------------------------------------------------
|
|
133
|
+
// 7. Protocol and route
|
|
134
|
+
// ---------------------------------------------------------------------------
|
|
135
|
+
export const protocol = MediaProtocol.queued({
|
|
136
|
+
id: ADAPTER,
|
|
137
|
+
name: NAME,
|
|
138
|
+
token: Token,
|
|
139
|
+
start: { body: { from: fromRequest }, decode: decodeStart },
|
|
140
|
+
status: { path: (token) => token.pollingURL, decode: decodeStatus },
|
|
141
|
+
result: { path: (token) => token.pollingURL, decode: decodeResult },
|
|
142
|
+
});
|
|
143
|
+
export const model = (input) => ImageModel.fromRoute({
|
|
144
|
+
id: ADAPTER,
|
|
145
|
+
provider: PROVIDER,
|
|
146
|
+
protocol,
|
|
147
|
+
baseURL: DEFAULT_BASE_URL,
|
|
148
|
+
path: ({ request }) => `/v1/${request.model.id}`,
|
|
149
|
+
}, input);
|
|
150
|
+
export const BlackForestLabsImages = {
|
|
151
|
+
protocol,
|
|
152
|
+
model,
|
|
153
|
+
};
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
2
|
+
import { MediaRoute } from "../route/media.js";
|
|
3
|
+
import { SpeechModel, type SpeechRequestFor } from "../speech.js";
|
|
4
|
+
import { SpeechStream } from "./utils/speech-stream.js";
|
|
5
|
+
export declare const DEFAULT_BASE_URL = "https://api.cartesia.ai";
|
|
6
|
+
export declare const API_VERSION = "2026-08-14";
|
|
7
|
+
export declare const BYTES_PATH = "/tts/bytes";
|
|
8
|
+
export declare const SSE_PATH = "/tts/sse";
|
|
9
|
+
export type CartesiaSpeechString<Known extends string> = Known | (string & {});
|
|
10
|
+
export type CartesiaEncoding = SpeechStream.PcmEncoding;
|
|
11
|
+
export type CartesiaSpeechOptions = {
|
|
12
|
+
readonly sampleRate?: 8000 | 16000 | 22050 | 24000 | 44100 | 48000;
|
|
13
|
+
readonly bitRate?: 32000 | 64000 | 96000 | 128000 | 192000;
|
|
14
|
+
readonly encoding?: CartesiaEncoding;
|
|
15
|
+
readonly generation_config?: {
|
|
16
|
+
readonly volume?: number;
|
|
17
|
+
readonly emotion?: CartesiaSpeechString<"neutral" | "calm" | "angry" | "content" | "sad" | "scared">;
|
|
18
|
+
};
|
|
19
|
+
readonly pronunciation_dict_id?: string;
|
|
20
|
+
} & Record<string, unknown>;
|
|
21
|
+
export type Request = SpeechRequestFor<CartesiaSpeechOptions>;
|
|
22
|
+
interface State extends SpeechStream.Audio {
|
|
23
|
+
readonly done: boolean;
|
|
24
|
+
}
|
|
25
|
+
export declare const protocol: MediaProtocol.Streamed<Request, {
|
|
26
|
+
readonly type: "audio-delta";
|
|
27
|
+
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
28
|
+
} | {
|
|
29
|
+
readonly type: "timestamps";
|
|
30
|
+
readonly items: readonly {
|
|
31
|
+
readonly text: string;
|
|
32
|
+
readonly startSeconds: number;
|
|
33
|
+
readonly endSeconds: number;
|
|
34
|
+
}[];
|
|
35
|
+
} | {
|
|
36
|
+
readonly type: "finish";
|
|
37
|
+
readonly audio: import("../media.js").Asset;
|
|
38
|
+
readonly providerMetadata?: {
|
|
39
|
+
readonly [x: string]: {
|
|
40
|
+
readonly [x: string]: unknown;
|
|
41
|
+
};
|
|
42
|
+
} | undefined;
|
|
43
|
+
readonly usage?: {
|
|
44
|
+
readonly type: "tokens";
|
|
45
|
+
readonly input?: number | undefined;
|
|
46
|
+
readonly output?: number | undefined;
|
|
47
|
+
readonly total?: number | undefined;
|
|
48
|
+
readonly details?: {
|
|
49
|
+
readonly [x: string]: unknown;
|
|
50
|
+
} | undefined;
|
|
51
|
+
} | {
|
|
52
|
+
readonly type: "seconds";
|
|
53
|
+
readonly seconds: number;
|
|
54
|
+
} | {
|
|
55
|
+
readonly type: "characters";
|
|
56
|
+
readonly characters: number;
|
|
57
|
+
} | {
|
|
58
|
+
readonly type: "credits";
|
|
59
|
+
readonly credits: number;
|
|
60
|
+
} | {
|
|
61
|
+
readonly type: "compute";
|
|
62
|
+
readonly seconds: number;
|
|
63
|
+
} | undefined;
|
|
64
|
+
readonly notices?: readonly {
|
|
65
|
+
readonly type: "other" | "moderated" | "filtered";
|
|
66
|
+
readonly message: string;
|
|
67
|
+
readonly providerMetadata?: {
|
|
68
|
+
readonly [x: string]: {
|
|
69
|
+
readonly [x: string]: unknown;
|
|
70
|
+
};
|
|
71
|
+
} | undefined;
|
|
72
|
+
}[] | undefined;
|
|
73
|
+
}, string | Uint8Array<ArrayBufferLike>, State>;
|
|
74
|
+
export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<CartesiaSpeechOptions>;
|
|
75
|
+
export declare const CartesiaSpeech: {
|
|
76
|
+
readonly protocol: MediaProtocol.Streamed<Request, {
|
|
77
|
+
readonly type: "audio-delta";
|
|
78
|
+
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
79
|
+
} | {
|
|
80
|
+
readonly type: "timestamps";
|
|
81
|
+
readonly items: readonly {
|
|
82
|
+
readonly text: string;
|
|
83
|
+
readonly startSeconds: number;
|
|
84
|
+
readonly endSeconds: number;
|
|
85
|
+
}[];
|
|
86
|
+
} | {
|
|
87
|
+
readonly type: "finish";
|
|
88
|
+
readonly audio: import("../media.js").Asset;
|
|
89
|
+
readonly providerMetadata?: {
|
|
90
|
+
readonly [x: string]: {
|
|
91
|
+
readonly [x: string]: unknown;
|
|
92
|
+
};
|
|
93
|
+
} | undefined;
|
|
94
|
+
readonly usage?: {
|
|
95
|
+
readonly type: "tokens";
|
|
96
|
+
readonly input?: number | undefined;
|
|
97
|
+
readonly output?: number | undefined;
|
|
98
|
+
readonly total?: number | undefined;
|
|
99
|
+
readonly details?: {
|
|
100
|
+
readonly [x: string]: unknown;
|
|
101
|
+
} | undefined;
|
|
102
|
+
} | {
|
|
103
|
+
readonly type: "seconds";
|
|
104
|
+
readonly seconds: number;
|
|
105
|
+
} | {
|
|
106
|
+
readonly type: "characters";
|
|
107
|
+
readonly characters: number;
|
|
108
|
+
} | {
|
|
109
|
+
readonly type: "credits";
|
|
110
|
+
readonly credits: number;
|
|
111
|
+
} | {
|
|
112
|
+
readonly type: "compute";
|
|
113
|
+
readonly seconds: number;
|
|
114
|
+
} | undefined;
|
|
115
|
+
readonly notices?: readonly {
|
|
116
|
+
readonly type: "other" | "moderated" | "filtered";
|
|
117
|
+
readonly message: string;
|
|
118
|
+
readonly providerMetadata?: {
|
|
119
|
+
readonly [x: string]: {
|
|
120
|
+
readonly [x: string]: unknown;
|
|
121
|
+
};
|
|
122
|
+
} | undefined;
|
|
123
|
+
}[] | undefined;
|
|
124
|
+
}, string | Uint8Array<ArrayBufferLike>, State>;
|
|
125
|
+
readonly model: (input: MediaRoute.ModelInput) => SpeechModel<CartesiaSpeechOptions>;
|
|
126
|
+
};
|
|
127
|
+
export {};
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
import { Effect, Schema } from "effect";
|
|
2
|
+
import { classifyProviderFailure } from "../provider-error.js";
|
|
3
|
+
import { Framing } from "../route/framing.js";
|
|
4
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
5
|
+
import { MediaRoute } from "../route/media.js";
|
|
6
|
+
import { AIError, ProviderID, mergeJsonRecords } from "../schema/index.js";
|
|
7
|
+
import { SpeechModel } from "../speech.js";
|
|
8
|
+
import { ProviderShared, optionalNull } from "./shared.js";
|
|
9
|
+
import { SpeechStream } from "./utils/speech-stream.js";
|
|
10
|
+
const ADAPTER = "cartesia-speech";
|
|
11
|
+
const NAME = "Cartesia";
|
|
12
|
+
const PROVIDER = ProviderID.make("cartesia");
|
|
13
|
+
export const DEFAULT_BASE_URL = "https://api.cartesia.ai";
|
|
14
|
+
export const API_VERSION = "2026-08-14";
|
|
15
|
+
export const BYTES_PATH = "/tts/bytes";
|
|
16
|
+
export const SSE_PATH = "/tts/sse";
|
|
17
|
+
const DEFAULT_SAMPLE_RATE = 44100;
|
|
18
|
+
const DEFAULT_BIT_RATE = 128000;
|
|
19
|
+
// ---------------------------------------------------------------------------
|
|
20
|
+
// 3. Streaming event schema
|
|
21
|
+
// ---------------------------------------------------------------------------
|
|
22
|
+
/** `phoneme_timestamps` and future record types are ignored. */
|
|
23
|
+
const SseEvent = Schema.Struct({
|
|
24
|
+
type: Schema.String,
|
|
25
|
+
data: Schema.optional(Schema.Uint8ArrayFromBase64),
|
|
26
|
+
word_timestamps: Schema.optional(Schema.Struct({
|
|
27
|
+
words: Schema.Array(Schema.String),
|
|
28
|
+
start: Schema.Array(Schema.Number),
|
|
29
|
+
end: Schema.Array(Schema.Number),
|
|
30
|
+
})),
|
|
31
|
+
status_code: Schema.optional(Schema.Number),
|
|
32
|
+
title: Schema.optional(Schema.String),
|
|
33
|
+
message: Schema.optional(Schema.String),
|
|
34
|
+
error_code: optionalNull(Schema.String),
|
|
35
|
+
});
|
|
36
|
+
const decodeEvent = MediaProtocol.decodeFrame(ADAPTER, NAME, SseEvent);
|
|
37
|
+
// ---------------------------------------------------------------------------
|
|
38
|
+
// 5. Request body construction
|
|
39
|
+
// ---------------------------------------------------------------------------
|
|
40
|
+
/** Timestamps exist only on the SSE endpoint, so a `generate` that asks for them collects an SSE stream. */
|
|
41
|
+
const usesSse = (request) => request.mode === "stream" || request.timestamps === true;
|
|
42
|
+
const CONTAINERS = { pcm: "raw", wav: "wav", mp3: "mp3" };
|
|
43
|
+
const outputFormat = Effect.fn("CartesiaSpeech.outputFormat")(function* (request) {
|
|
44
|
+
const sse = usesSse(request);
|
|
45
|
+
const format = request.format ?? (sse ? "pcm" : "mp3");
|
|
46
|
+
const container = CONTAINERS[format];
|
|
47
|
+
if (container === undefined)
|
|
48
|
+
return yield* SpeechStream.unsupportedFormat(PROVIDER, ADAPTER, `${NAME} supports the pcm, wav, and mp3 formats, not "${format}"`);
|
|
49
|
+
if (sse && container !== "raw")
|
|
50
|
+
return yield* SpeechStream.unsupportedFormat(PROVIDER, ADAPTER, `${NAME} streams and timestamps only raw PCM; request format "pcm" instead of "${format}"`);
|
|
51
|
+
const sampleRate = request.providerOptions?.sampleRate ?? DEFAULT_SAMPLE_RATE;
|
|
52
|
+
if (container === "mp3")
|
|
53
|
+
return { container, sample_rate: sampleRate, bit_rate: request.providerOptions?.bitRate ?? DEFAULT_BIT_RATE };
|
|
54
|
+
return { container, encoding: request.providerOptions?.encoding ?? "pcm_s16le", sample_rate: sampleRate };
|
|
55
|
+
});
|
|
56
|
+
const fromRequest = Effect.fn("CartesiaSpeech.fromRequest")(function* (request) {
|
|
57
|
+
const voice = SpeechStream.voiceID(request.voice);
|
|
58
|
+
if (voice === undefined)
|
|
59
|
+
return yield* ProviderShared.invalidRequest(`${NAME} requires a voice id; pass it as \`voice\``);
|
|
60
|
+
const { sampleRate: _sampleRate, bitRate: _bitRate, encoding: _encoding, ...native } = request.providerOptions ?? {};
|
|
61
|
+
return MediaProtocol.json(mergeJsonRecords({
|
|
62
|
+
model_id: request.model.id,
|
|
63
|
+
transcript: request.text,
|
|
64
|
+
voice,
|
|
65
|
+
output_format: yield* outputFormat(request),
|
|
66
|
+
language: request.language,
|
|
67
|
+
generation_config: request.speed === undefined ? undefined : { speed: request.speed },
|
|
68
|
+
add_timestamps: request.timestamps === true ? true : undefined,
|
|
69
|
+
}, native, request.http?.body) ?? {});
|
|
70
|
+
});
|
|
71
|
+
// ---------------------------------------------------------------------------
|
|
72
|
+
// 6. Stream parsing
|
|
73
|
+
// ---------------------------------------------------------------------------
|
|
74
|
+
const onEvent = Effect.fn("CartesiaSpeech.onEvent")(function* (state, frame) {
|
|
75
|
+
const event = yield* decodeEvent(frame);
|
|
76
|
+
if (event.type === "chunk" && event.data !== undefined)
|
|
77
|
+
return SpeechStream.delta(state, event.data);
|
|
78
|
+
if (event.type === "timestamps" && event.word_timestamps !== undefined) {
|
|
79
|
+
const words = event.word_timestamps;
|
|
80
|
+
return [state, SpeechStream.timestamps(words.words, words.start, words.end)];
|
|
81
|
+
}
|
|
82
|
+
if (event.type === "done")
|
|
83
|
+
return [{ ...state, done: true }, []];
|
|
84
|
+
if (event.type === "error")
|
|
85
|
+
return yield* new AIError({
|
|
86
|
+
reason: classifyProviderFailure({
|
|
87
|
+
message: `${NAME} stream failed${event.title === undefined ? "" : ` (${event.title})`}: ${event.message ?? "unknown error"}`,
|
|
88
|
+
status: event.status_code,
|
|
89
|
+
rawBody: frame,
|
|
90
|
+
}),
|
|
91
|
+
});
|
|
92
|
+
return [state, []];
|
|
93
|
+
});
|
|
94
|
+
const finish = Effect.fn("CartesiaSpeech.finish")(function* (state, context) {
|
|
95
|
+
if (usesSse(context.request) && !state.done)
|
|
96
|
+
return yield* MediaProtocol.incomplete(ADAPTER);
|
|
97
|
+
const format = yield* outputFormat(context.request);
|
|
98
|
+
return yield* SpeechStream.finish(ADAPTER, state, format.container === "raw"
|
|
99
|
+
? SpeechStream.pcm(format.encoding, format.sample_rate)
|
|
100
|
+
: SpeechStream.container(format.container, format.sample_rate));
|
|
101
|
+
});
|
|
102
|
+
// ---------------------------------------------------------------------------
|
|
103
|
+
// 7. Protocol and route
|
|
104
|
+
// ---------------------------------------------------------------------------
|
|
105
|
+
export const protocol = MediaProtocol.stream({
|
|
106
|
+
id: ADAPTER,
|
|
107
|
+
name: NAME,
|
|
108
|
+
unsupported: ["instructions"],
|
|
109
|
+
body: { from: fromRequest },
|
|
110
|
+
frames: (bytes, context) => (usesSse(context.request) ? Framing.sse.frame(bytes) : bytes),
|
|
111
|
+
initial: () => ({ chunks: [], done: false }),
|
|
112
|
+
step: SpeechStream.step(onEvent),
|
|
113
|
+
finish,
|
|
114
|
+
});
|
|
115
|
+
export const model = (input) => SpeechModel.fromRoute({
|
|
116
|
+
id: ADAPTER,
|
|
117
|
+
provider: PROVIDER,
|
|
118
|
+
protocol,
|
|
119
|
+
baseURL: DEFAULT_BASE_URL,
|
|
120
|
+
headers: { "Cartesia-Version": API_VERSION },
|
|
121
|
+
path: ({ request }) => (usesSse(request) ? SSE_PATH : BYTES_PATH),
|
|
122
|
+
}, input);
|
|
123
|
+
export const CartesiaSpeech = {
|
|
124
|
+
protocol,
|
|
125
|
+
model,
|
|
126
|
+
};
|