@breeze.blue/sdk 0.16.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/README.md +36 -2
- package/dist/client.d.ts +24 -3
- package/dist/client.js +32 -2
- package/dist/index.d.ts +2 -0
- package/dist/index.js +1 -0
- package/dist/timestamps.d.ts +22 -0
- package/dist/timestamps.js +64 -0
- package/dist/types.d.ts +36 -5
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/dist/voice-metadata.d.ts +6 -2
- package/dist/voice-metadata.js +2 -1
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,15 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.18.0
|
|
4
|
+
|
|
5
|
+
- Create asynchronous speech jobs with word timestamps and read timing from completed job results.
|
|
6
|
+
|
|
7
|
+
## 0.17.0
|
|
8
|
+
|
|
9
|
+
- Generate synchronous speech or stream audio with word timestamps and safe stream cleanup.
|
|
10
|
+
- Create and poll voice localization previews before auditioning and saving them.
|
|
11
|
+
- Control TTS output volume alongside speaking speed.
|
|
12
|
+
|
|
3
13
|
## 0.16.0
|
|
4
14
|
|
|
5
15
|
- Clone previews use an automatically generated script in the reference audio language. Custom text, instructions, and language hints are not accepted.
|
package/README.md
CHANGED
|
@@ -436,11 +436,12 @@ console.log(metadataOptions.accentCodesByLanguage.en);
|
|
|
436
436
|
|
|
437
437
|
Voice listings keep the published default of stable creation-time order from
|
|
438
438
|
newest to oldest. Use `{ sort: "trend", voiceType: "default" }` for the shared
|
|
439
|
-
Daily Trend order.
|
|
439
|
+
Daily Trend order. Eligible voices not yet ranked follow the snapshot members;
|
|
440
|
+
returned page tokens pin the initial supplemental candidates. Search queries remain relevance-ranked; Trend is only a
|
|
440
441
|
weak prior within the same relevance tier.
|
|
441
442
|
|
|
442
443
|
Breeze voice creation is always two steps: produce a **preview**, let the
|
|
443
|
-
user accept it, then **save the preview** as a real voice.
|
|
444
|
+
user accept it, then **save the preview** as a real voice. Three ways to
|
|
444
445
|
produce a preview:
|
|
445
446
|
|
|
446
447
|
```ts
|
|
@@ -464,6 +465,29 @@ const design = await client.voices.createDesignPreview({
|
|
|
464
465
|
generatedVoiceId = design.previews[0].generatedVoiceId;
|
|
465
466
|
```
|
|
466
467
|
|
|
468
|
+
Option C keeps an existing voice and makes it speak another language.
|
|
469
|
+
Localization runs as a background job: start it, then poll until the status is
|
|
470
|
+
`ready`. `name` defaults to the source voice name.
|
|
471
|
+
|
|
472
|
+
```ts
|
|
473
|
+
const job = await client.voices.createLocalizePreview({
|
|
474
|
+
voiceId: firstVoiceId,
|
|
475
|
+
languageCode: "es",
|
|
476
|
+
name: "Documentary narrator (Spanish)",
|
|
477
|
+
});
|
|
478
|
+
|
|
479
|
+
let localized = await client.voices.getLocalizePreview(job.generationJobId);
|
|
480
|
+
while (localized.status !== "ready") {
|
|
481
|
+
if (localized.status === "failed" || localized.status === "cancelled") {
|
|
482
|
+
throw new Error(localized.error?.detail ?? localized.status);
|
|
483
|
+
}
|
|
484
|
+
await new Promise((resolve) => setTimeout(resolve, 2000));
|
|
485
|
+
localized = await client.voices.getLocalizePreview(job.generationJobId);
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
generatedVoiceId = localized.generatedVoiceId!;
|
|
489
|
+
```
|
|
490
|
+
|
|
467
491
|
For the fastest first audio, generate one design preview as a live 24 kHz mono
|
|
468
492
|
PCM response. The Node `stream()` helper consumes the response body directly;
|
|
469
493
|
a clean end means `generatedVoiceId` is ready to save:
|
|
@@ -578,3 +602,13 @@ Clone previews automatically generate a short script in the detected reference a
|
|
|
578
602
|
### HTTP TTS speed
|
|
579
603
|
|
|
580
604
|
Pass `voiceSettings: { speed: 1.25 }` to control speech speed without changing pitch, from 0.5 to 2.0. Omission always means 1.0; saved voice speed is not inherited. Sync, async and HTTP streaming support this parameter; Realtime does not.
|
|
605
|
+
|
|
606
|
+
Pass `voiceSettings: { volume: 1.5 }` to apply a request-level linear amplitude multiplier. Values range from 0.01 to 2.0; omission means 1.0. Values above 1.0 may clip peaks. Sync, async and HTTP streaming support this parameter; it does not change saved voice settings or apply to Realtime.
|
|
607
|
+
|
|
608
|
+
## Word timing
|
|
609
|
+
|
|
610
|
+
Stream audio and word/token timestamps with `client.textToSpeech.streamWithTimestamps(...)`. Use `for await` to consume chunks containing `audioBase64` and `wordTimestamps`; breaking iteration closes the stream. Decode the audio separately and merge repeated word indices. See [Speech timing](https://docs.breezeblue.ai/guides/speech-timing) for complete examples and timing semantics.
|
|
611
|
+
|
|
612
|
+
For a complete response, use `client.textToSpeech.convertWithTimestamps(...)`. It returns base64 audio, its content type, and the full word/token timestamp list. Decode the audio before saving or playback. See [Speech timing](https://docs.breezeblue.ai/guides/speech-timing) and [Convert with timestamps](https://docs.breezeblue.ai/api-reference/text-to-speech/convert-with-timestamps).
|
|
613
|
+
|
|
614
|
+
For background generation with timing, use `client.textToSpeech.createJobWithTimestamps(...)`. Poll with `client.generationJobs.get(jobId)`; a ready job includes `wordTimestamps` and the audio download URL.
|
package/dist/client.d.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
|
+
import { SpeechTimingStream, type SpeechWithTimestamps } from "./timestamps.js";
|
|
1
2
|
import { AudioResponse } from "./audio.js";
|
|
2
|
-
import type { AudioRequestOptions, BreezeBlueWebSocket, AsyncTextToSpeechJob, Balance, BreezeBlueClientOptions, CurrentApiKeyResponse, GenericStatus, GenerationJob, HistoryItem, HistoryList, HistoryListParams, Model, RequestOptions, RealtimeTextToSpeechConnectOptions, RealtimeTextToSpeechEvent, RealtimeTextToSpeechManagedConnectOptions, RealtimeTextToSpeechMessage, RealtimeTextToSpeechSession, RealtimeTextToSpeechSessionRequest, SavedVoice, SaveVoiceRequest, StreamTextToSpeechOptions, TextToSpeechRequest, TtsEnhanceRequest, TtsEnhanceResponse, Usage, UsageParams, Voice, VoiceClonePreview, VoiceClonePreviewRequest, VoiceDesignRequest, VoiceDesignResponse, VoiceDesignStreamRequest, VoiceEditRequest, VoiceList, VoiceMetadataOptions, VoiceRandomParams, VoiceSearchParams, VoiceSettings } from "./types.js";
|
|
3
|
+
import type { AudioRequestOptions, BreezeBlueWebSocket, AsyncTextToSpeechJob, Balance, BreezeBlueClientOptions, CurrentApiKeyResponse, GenericStatus, GenerationJob, HistoryItem, HistoryList, HistoryListParams, Model, RequestOptions, RealtimeTextToSpeechConnectOptions, RealtimeTextToSpeechEvent, RealtimeTextToSpeechManagedConnectOptions, RealtimeTextToSpeechMessage, RealtimeTextToSpeechSession, RealtimeTextToSpeechSessionRequest, SavedVoice, SaveVoiceRequest, StreamTextToSpeechOptions, TextToSpeechRequest, TtsEnhanceRequest, TtsEnhanceResponse, Usage, UsageParams, Voice, VoiceClonePreview, VoiceClonePreviewRequest, VoiceDesignRequest, VoiceDesignResponse, VoiceDesignStreamRequest, VoiceEditRequest, VoiceList, VoiceLocalizeJob, VoiceLocalizePreviewRequest, VoiceLocalizePreviewStatus, VoiceMetadataOptions, VoiceRandomParams, VoiceSearchParams, VoiceSettings } from "./types.js";
|
|
3
4
|
export declare class BreezeBlueClient {
|
|
4
5
|
readonly apiKey: string | undefined;
|
|
5
6
|
readonly baseUrl: string;
|
|
@@ -25,7 +26,12 @@ declare class TextToSpeechResource {
|
|
|
25
26
|
constructor(client: BreezeBlueClient);
|
|
26
27
|
convert(voiceId: string, request: TextToSpeechRequest, options?: AudioRequestOptions): Promise<AudioResponse>;
|
|
27
28
|
createJob(voiceId: string, request: TextToSpeechRequest, options?: AudioRequestOptions): Promise<AsyncTextToSpeechJob>;
|
|
29
|
+
createJobWithTimestamps(voiceId: string, request: TextToSpeechRequest, options?: AudioRequestOptions): Promise<AsyncTextToSpeechJob>;
|
|
28
30
|
stream(voiceId: string, request: TextToSpeechRequest, options?: StreamTextToSpeechOptions): Promise<AudioResponse>;
|
|
31
|
+
convertWithTimestamps(voiceId: string, request: TextToSpeechRequest, options?: AudioRequestOptions): Promise<SpeechWithTimestamps>;
|
|
32
|
+
streamWithTimestamps(voiceId: string, request: TextToSpeechRequest & {
|
|
33
|
+
timestampMode?: "chunk" | "lookahead";
|
|
34
|
+
}, options?: StreamTextToSpeechOptions): Promise<SpeechTimingStream>;
|
|
29
35
|
enhance(request: TtsEnhanceRequest, options?: RequestOptions): Promise<TtsEnhanceResponse>;
|
|
30
36
|
}
|
|
31
37
|
declare class RealtimeTextToSpeechResource {
|
|
@@ -197,8 +203,9 @@ declare class GenerationJobsResource {
|
|
|
197
203
|
* Voice CRUD plus the two-step preview → save flow for Breeze voices.
|
|
198
204
|
*
|
|
199
205
|
* Breeze's voice creation is always two steps: produce a preview
|
|
200
|
-
* (`createClonePreview` from an audio sample,
|
|
201
|
-
*
|
|
206
|
+
* (`createClonePreview` from an audio sample, `createDesignPreview` from a
|
|
207
|
+
* text description, or `createLocalizePreview` from an existing voice plus a
|
|
208
|
+
* target language), then persist it with `savePreview`. Preview generation
|
|
202
209
|
* never creates a saved voice until `savePreview` is called.
|
|
203
210
|
*/
|
|
204
211
|
declare class VoicesResource {
|
|
@@ -245,6 +252,20 @@ declare class VoicesResource {
|
|
|
245
252
|
* A clean end means `generatedVoiceId` is ready for {@link savePreview}.
|
|
246
253
|
*/
|
|
247
254
|
streamDesignPreview(request: VoiceDesignStreamRequest, options?: RequestOptions): Promise<AudioResponse>;
|
|
255
|
+
/**
|
|
256
|
+
* Start a preview that speaks an existing voice in another language.
|
|
257
|
+
* Localization runs as a background job: poll the returned
|
|
258
|
+
* `generationJobId` with {@link getLocalizePreview} until the status is
|
|
259
|
+
* `ready`, then audition it with {@link streamPreview} and persist it with
|
|
260
|
+
* {@link savePreview}. `name` defaults to the source voice name.
|
|
261
|
+
*/
|
|
262
|
+
createLocalizePreview(request: VoiceLocalizePreviewRequest, options?: RequestOptions): Promise<VoiceLocalizeJob>;
|
|
263
|
+
/**
|
|
264
|
+
* Read the current state of a localization preview job. `generatedVoiceId`,
|
|
265
|
+
* `text`, and `languageCode` stay null until the status is `ready`; `error`
|
|
266
|
+
* is set only when the job ends in `failed`.
|
|
267
|
+
*/
|
|
268
|
+
getLocalizePreview(generationJobId: string, options?: RequestOptions): Promise<VoiceLocalizePreviewStatus>;
|
|
248
269
|
/** Download the completed audio of a clone or design preview. */
|
|
249
270
|
streamPreview(generatedVoiceId: string, options?: AudioRequestOptions): Promise<AudioResponse>;
|
|
250
271
|
/** Persist a clone or design preview as a real voice asset. */
|
package/dist/client.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { SpeechTimingStream } from "./timestamps.js";
|
|
1
2
|
import { AudioResponse } from "./audio.js";
|
|
2
3
|
import { BreezeBlueAPIError, BreezeBlueConfigurationError, BreezeBlueRealtimeError, errorFromResponse, } from "./errors.js";
|
|
3
4
|
import { camelizeKeys, snakeizeKeys } from "./_transform.js";
|
|
@@ -112,6 +113,9 @@ class TextToSpeechResource {
|
|
|
112
113
|
createJob(voiceId, request, options = {}) {
|
|
113
114
|
return this.client.requestJson("POST", `/v1/text-to-speech/${encodeURIComponent(voiceId)}`, { outputFormat: options.outputFormat, delivery: "async" }, request, options);
|
|
114
115
|
}
|
|
116
|
+
createJobWithTimestamps(voiceId, request, options = {}) {
|
|
117
|
+
return this.client.requestJson("POST", `/v1/text-to-speech/${encodeURIComponent(voiceId)}/with-timestamps`, { outputFormat: options.outputFormat, delivery: "async" }, request, options);
|
|
118
|
+
}
|
|
115
119
|
stream(voiceId, request, options = {}) {
|
|
116
120
|
return this.client.requestAudio("POST", `/v1/text-to-speech/${encodeURIComponent(voiceId)}/stream`, {
|
|
117
121
|
outputFormat: options.outputFormat,
|
|
@@ -119,6 +123,13 @@ class TextToSpeechResource {
|
|
|
119
123
|
enableLogging: options.enableLogging,
|
|
120
124
|
}, request, options);
|
|
121
125
|
}
|
|
126
|
+
convertWithTimestamps(voiceId, request, options = {}) {
|
|
127
|
+
return this.client.requestJson("POST", `/v1/text-to-speech/${encodeURIComponent(voiceId)}/with-timestamps`, { outputFormat: options.outputFormat }, request, options);
|
|
128
|
+
}
|
|
129
|
+
async streamWithTimestamps(voiceId, request, options = {}) {
|
|
130
|
+
const response = await this.client.request("POST", `/v1/text-to-speech/${encodeURIComponent(voiceId)}/stream/with-timestamps`, { outputFormat: options.outputFormat, optimizeStreamingLatency: options.optimizeStreamingLatency, enableLogging: options.enableLogging }, request, options);
|
|
131
|
+
return new SpeechTimingStream(response);
|
|
132
|
+
}
|
|
122
133
|
enhance(request, options) {
|
|
123
134
|
return this.client.requestJson("POST", "/v1/text-to-speech/enhance", undefined, request, options);
|
|
124
135
|
}
|
|
@@ -1479,8 +1490,9 @@ class GenerationJobsResource {
|
|
|
1479
1490
|
* Voice CRUD plus the two-step preview → save flow for Breeze voices.
|
|
1480
1491
|
*
|
|
1481
1492
|
* Breeze's voice creation is always two steps: produce a preview
|
|
1482
|
-
* (`createClonePreview` from an audio sample,
|
|
1483
|
-
*
|
|
1493
|
+
* (`createClonePreview` from an audio sample, `createDesignPreview` from a
|
|
1494
|
+
* text description, or `createLocalizePreview` from an existing voice plus a
|
|
1495
|
+
* target language), then persist it with `savePreview`. Preview generation
|
|
1484
1496
|
* never creates a saved voice until `savePreview` is called.
|
|
1485
1497
|
*/
|
|
1486
1498
|
class VoicesResource {
|
|
@@ -1568,6 +1580,24 @@ class VoicesResource {
|
|
|
1568
1580
|
streamDesignPreview(request, options) {
|
|
1569
1581
|
return this.client.requestAudio("POST", "/v1/voice-previews/design/stream", undefined, request, options);
|
|
1570
1582
|
}
|
|
1583
|
+
/**
|
|
1584
|
+
* Start a preview that speaks an existing voice in another language.
|
|
1585
|
+
* Localization runs as a background job: poll the returned
|
|
1586
|
+
* `generationJobId` with {@link getLocalizePreview} until the status is
|
|
1587
|
+
* `ready`, then audition it with {@link streamPreview} and persist it with
|
|
1588
|
+
* {@link savePreview}. `name` defaults to the source voice name.
|
|
1589
|
+
*/
|
|
1590
|
+
createLocalizePreview(request, options) {
|
|
1591
|
+
return this.client.requestJson("POST", "/v1/voice-previews/localize", undefined, request, options);
|
|
1592
|
+
}
|
|
1593
|
+
/**
|
|
1594
|
+
* Read the current state of a localization preview job. `generatedVoiceId`,
|
|
1595
|
+
* `text`, and `languageCode` stay null until the status is `ready`; `error`
|
|
1596
|
+
* is set only when the job ends in `failed`.
|
|
1597
|
+
*/
|
|
1598
|
+
getLocalizePreview(generationJobId, options) {
|
|
1599
|
+
return this.client.requestJson("GET", `/v1/voice-previews/localize/${encodeURIComponent(generationJobId)}`, undefined, undefined, options);
|
|
1600
|
+
}
|
|
1571
1601
|
/** Download the completed audio of a clone or design preview. */
|
|
1572
1602
|
streamPreview(generatedVoiceId, options = {}) {
|
|
1573
1603
|
return this.client.requestAudio("GET", `/v1/voice-previews/${encodeURIComponent(generatedVoiceId)}/stream`, { outputFormat: options.outputFormat }, undefined, options);
|
package/dist/index.d.ts
CHANGED
|
@@ -4,3 +4,5 @@ export { BreezeBlueClient, ManagedRealtimeTextToSpeechConnection, RealtimeTextTo
|
|
|
4
4
|
export { SDK_NAME, SDK_VERSION } from "./version.js";
|
|
5
5
|
export type * from "./types.js";
|
|
6
6
|
export * from "./voice-metadata.js";
|
|
7
|
+
export { SpeechTimingStream } from "./timestamps.js";
|
|
8
|
+
export type { SpeechTimingChunk, SpeechWithTimestamps, WordTimestamp } from "./timestamps.js";
|
package/dist/index.js
CHANGED
|
@@ -3,3 +3,4 @@ export { BreezeBlueAPIError, BreezeBlueAuthenticationError, BreezeBlueBadRequest
|
|
|
3
3
|
export { BreezeBlueClient, ManagedRealtimeTextToSpeechConnection, RealtimeTextToSpeechConnection, } from "./client.js";
|
|
4
4
|
export { SDK_NAME, SDK_VERSION } from "./version.js";
|
|
5
5
|
export * from "./voice-metadata.js";
|
|
6
|
+
export { SpeechTimingStream } from "./timestamps.js";
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
export interface WordTimestamp {
|
|
2
|
+
index: number;
|
|
3
|
+
word: string;
|
|
4
|
+
start: number;
|
|
5
|
+
end: number;
|
|
6
|
+
}
|
|
7
|
+
export interface SpeechTimingChunk {
|
|
8
|
+
audioBase64: string;
|
|
9
|
+
wordTimestamps: WordTimestamp[];
|
|
10
|
+
}
|
|
11
|
+
export interface SpeechWithTimestamps extends SpeechTimingChunk {
|
|
12
|
+
contentType: string;
|
|
13
|
+
}
|
|
14
|
+
/** Single-use stream. Breaking iteration cancels the response body. */
|
|
15
|
+
export declare class SpeechTimingStream implements AsyncIterable<SpeechTimingChunk> {
|
|
16
|
+
private readonly response;
|
|
17
|
+
readonly historyItemId: string | null;
|
|
18
|
+
private reader?;
|
|
19
|
+
constructor(response: Response);
|
|
20
|
+
close(): Promise<void>;
|
|
21
|
+
[Symbol.asyncIterator](): AsyncGenerator<SpeechTimingChunk>;
|
|
22
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import { camelizeKeys } from "./_transform.js";
|
|
2
|
+
/** Single-use stream. Breaking iteration cancels the response body. */
|
|
3
|
+
export class SpeechTimingStream {
|
|
4
|
+
response;
|
|
5
|
+
historyItemId;
|
|
6
|
+
reader;
|
|
7
|
+
constructor(response) {
|
|
8
|
+
this.response = response;
|
|
9
|
+
this.historyItemId = response.headers.get("history-item-id");
|
|
10
|
+
}
|
|
11
|
+
async close() {
|
|
12
|
+
if (this.reader)
|
|
13
|
+
await this.reader.cancel();
|
|
14
|
+
else
|
|
15
|
+
await this.response.body?.cancel();
|
|
16
|
+
}
|
|
17
|
+
async *[Symbol.asyncIterator]() {
|
|
18
|
+
const reader = this.response.body?.getReader();
|
|
19
|
+
if (!reader)
|
|
20
|
+
throw new Error("Speech stream has no response body.");
|
|
21
|
+
this.reader = reader;
|
|
22
|
+
const decoder = new TextDecoder("utf-8", { fatal: true });
|
|
23
|
+
let pending = "";
|
|
24
|
+
const parse = (line) => {
|
|
25
|
+
if (line.length > 4 * 1024 * 1024)
|
|
26
|
+
throw new Error("Timestamp stream line is too large.");
|
|
27
|
+
const value = JSON.parse(line);
|
|
28
|
+
if (value.error)
|
|
29
|
+
throw new Error(`Speech stream failed: ${JSON.stringify(value.error)}`);
|
|
30
|
+
if (typeof value.audio_base64 !== "string" || !Array.isArray(value.word_timestamps))
|
|
31
|
+
throw new Error("Invalid timestamp stream event.");
|
|
32
|
+
return camelizeKeys(value);
|
|
33
|
+
};
|
|
34
|
+
try {
|
|
35
|
+
while (true) {
|
|
36
|
+
const { value, done } = await reader.read();
|
|
37
|
+
pending += decoder.decode(value, { stream: !done });
|
|
38
|
+
let boundary = pending.indexOf("\n");
|
|
39
|
+
while (boundary >= 0) {
|
|
40
|
+
const line = pending.slice(0, boundary);
|
|
41
|
+
pending = pending.slice(boundary + 1);
|
|
42
|
+
if (line.trim())
|
|
43
|
+
yield parse(line);
|
|
44
|
+
boundary = pending.indexOf("\n");
|
|
45
|
+
}
|
|
46
|
+
if (pending.length > 4 * 1024 * 1024)
|
|
47
|
+
throw new Error("Timestamp stream line is too large.");
|
|
48
|
+
if (done)
|
|
49
|
+
break;
|
|
50
|
+
}
|
|
51
|
+
if (pending.trim())
|
|
52
|
+
yield parse(pending);
|
|
53
|
+
}
|
|
54
|
+
finally {
|
|
55
|
+
try {
|
|
56
|
+
await reader.cancel();
|
|
57
|
+
}
|
|
58
|
+
finally {
|
|
59
|
+
reader.releaseLock();
|
|
60
|
+
this.reader = undefined;
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
}
|
package/dist/types.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type { WordTimestamp } from "./timestamps.js";
|
|
2
|
+
import type { VoiceAccent, VoiceWritableAccent, VoiceAge, VoiceGender, VoiceLanguageCode, VoiceTone } from "./voice-metadata.js";
|
|
2
3
|
export type JsonObject = Record<string, unknown>;
|
|
3
4
|
export type BreezeBlueFetch = typeof fetch;
|
|
4
5
|
export interface BreezeBlueClientOptions {
|
|
@@ -234,6 +235,11 @@ export interface VoiceSettings {
|
|
|
234
235
|
speed?: number;
|
|
235
236
|
guidanceScale?: number | null;
|
|
236
237
|
}
|
|
238
|
+
/** Request-only HTTP TTS settings. Volume is a linear amplitude multiplier. */
|
|
239
|
+
export interface TtsVoiceSettings extends VoiceSettings {
|
|
240
|
+
/** Linear amplitude multiplier (0.01–2.0); omitted means 1.0. */
|
|
241
|
+
volume?: number;
|
|
242
|
+
}
|
|
237
243
|
export interface PronunciationDictionaryLocator {
|
|
238
244
|
pronunciationDictionaryId: string;
|
|
239
245
|
versionId: string;
|
|
@@ -243,7 +249,7 @@ export interface TextToSpeechRequest {
|
|
|
243
249
|
modelId?: string;
|
|
244
250
|
languageCode?: string;
|
|
245
251
|
instructions?: string;
|
|
246
|
-
voiceSettings?:
|
|
252
|
+
voiceSettings?: TtsVoiceSettings;
|
|
247
253
|
seed?: number;
|
|
248
254
|
pronunciationDictionaryLocators?: PronunciationDictionaryLocator[];
|
|
249
255
|
previousText?: string;
|
|
@@ -271,6 +277,7 @@ export interface GenerationJobError {
|
|
|
271
277
|
detail: string;
|
|
272
278
|
}
|
|
273
279
|
export interface GenerationJob {
|
|
280
|
+
wordTimestamps?: WordTimestamp[];
|
|
274
281
|
generationJobId: string;
|
|
275
282
|
historyItemId: string;
|
|
276
283
|
status: string;
|
|
@@ -313,6 +320,9 @@ export interface Voice {
|
|
|
313
320
|
createdAtUnix?: number | null;
|
|
314
321
|
primaryCategoryCode?: string | null;
|
|
315
322
|
visibility?: string | null;
|
|
323
|
+
publicationStatus?: string | null;
|
|
324
|
+
publicationJobId?: string | null;
|
|
325
|
+
publicationErrorCode?: string | null;
|
|
316
326
|
languageCode: VoiceLanguageCode;
|
|
317
327
|
gender: VoiceGender | null;
|
|
318
328
|
age: VoiceAge | null;
|
|
@@ -351,7 +361,7 @@ export interface VoiceMetadataOptions {
|
|
|
351
361
|
ageCodes: VoiceAge[];
|
|
352
362
|
toneCodes: VoiceTone[];
|
|
353
363
|
toneMaxItems: number;
|
|
354
|
-
accentCodesByLanguage: Record<VoiceLanguageCode,
|
|
364
|
+
accentCodesByLanguage: Record<VoiceLanguageCode, VoiceWritableAccent[]>;
|
|
355
365
|
}
|
|
356
366
|
export interface VoiceRandomParams {
|
|
357
367
|
/** Only pick from voices whose saved audio language matches this code. */
|
|
@@ -367,6 +377,7 @@ export interface VoiceSearchParams {
|
|
|
367
377
|
age?: VoiceAge[];
|
|
368
378
|
tone?: VoiceTone[];
|
|
369
379
|
accent?: VoiceAccent;
|
|
380
|
+
accentMode?: "all" | "unmarked";
|
|
370
381
|
origin?: "designed" | "cloned" | string;
|
|
371
382
|
voiceType?: "all" | "default" | "personal" | string;
|
|
372
383
|
sort?: "created_at_unix" | "name" | "trend" | string;
|
|
@@ -379,6 +390,9 @@ export interface VoiceSearchParams {
|
|
|
379
390
|
}
|
|
380
391
|
export interface GenericStatus {
|
|
381
392
|
status: string;
|
|
393
|
+
visibility?: string | null;
|
|
394
|
+
publicationStatus?: string | null;
|
|
395
|
+
publicationJobId?: string | null;
|
|
382
396
|
}
|
|
383
397
|
export type UploadData = Blob | ArrayBuffer | Uint8Array | string;
|
|
384
398
|
export interface UploadFile {
|
|
@@ -407,7 +421,7 @@ export interface VoiceEditRequest {
|
|
|
407
421
|
gender?: VoiceGender | null;
|
|
408
422
|
age?: VoiceAge | null;
|
|
409
423
|
tone?: VoiceTone[];
|
|
410
|
-
accent?:
|
|
424
|
+
accent?: VoiceWritableAccent | null;
|
|
411
425
|
file?: UploadData | UploadFile;
|
|
412
426
|
files?: Array<UploadData | UploadFile>;
|
|
413
427
|
}
|
|
@@ -440,6 +454,23 @@ export interface VoiceDesignResponse {
|
|
|
440
454
|
previews: VoiceDesignPreview[];
|
|
441
455
|
text: string;
|
|
442
456
|
}
|
|
457
|
+
export interface VoiceLocalizePreviewRequest {
|
|
458
|
+
voiceId: string;
|
|
459
|
+
languageCode: VoiceLanguageCode;
|
|
460
|
+
name?: string;
|
|
461
|
+
}
|
|
462
|
+
export interface VoiceLocalizeJob {
|
|
463
|
+
generationJobId: string;
|
|
464
|
+
status: string;
|
|
465
|
+
}
|
|
466
|
+
export interface VoiceLocalizePreviewStatus {
|
|
467
|
+
generationJobId: string;
|
|
468
|
+
status: string;
|
|
469
|
+
languageCode?: VoiceLanguageCode | null;
|
|
470
|
+
generatedVoiceId?: string | null;
|
|
471
|
+
text?: string | null;
|
|
472
|
+
error?: GenerationJobError | null;
|
|
473
|
+
}
|
|
443
474
|
export interface SaveVoiceRequest {
|
|
444
475
|
generatedVoiceId: string;
|
|
445
476
|
voiceName: string;
|
|
@@ -452,7 +483,7 @@ export interface SaveVoiceRequest {
|
|
|
452
483
|
gender?: VoiceGender | null;
|
|
453
484
|
age?: VoiceAge | null;
|
|
454
485
|
tone?: VoiceTone[];
|
|
455
|
-
accent?:
|
|
486
|
+
accent?: VoiceWritableAccent | null;
|
|
456
487
|
}
|
|
457
488
|
export interface HistoryItem {
|
|
458
489
|
historyItemId: string;
|
package/dist/version.d.ts
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
export declare const SDK_VERSION = "0.
|
|
1
|
+
export declare const SDK_VERSION = "0.18.0";
|
|
2
2
|
export declare const SDK_NAME = "@breeze.blue/sdk";
|
package/dist/version.js
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
export const SDK_VERSION = "0.
|
|
1
|
+
export const SDK_VERSION = "0.18.0";
|
|
2
2
|
export const SDK_NAME = "@breeze.blue/sdk";
|
package/dist/voice-metadata.d.ts
CHANGED
|
@@ -5,10 +5,14 @@ export declare const VOICE_LANGUAGE_CODES: readonly ["af", "ar", "bg", "bn", "bs
|
|
|
5
5
|
export declare const VOICE_TONE_MAX_ITEMS: 3;
|
|
6
6
|
export declare const VOICE_ACCENT_CODES_BY_LANGUAGE: {
|
|
7
7
|
readonly en: readonly ["american", "british", "scottish", "irish", "australian", "canadian", "us_southern", "us_new_york", "indian", "south_african", "russian", "japanese", "korean", "chinese"];
|
|
8
|
-
readonly zh: readonly ["
|
|
8
|
+
readonly zh: readonly ["mandarin_northeastern", "mandarin_shaanxi", "mandarin_shanghai", "mandarin_sichuan", "mandarin_henan", "mandarin_beijing", "mandarin_taiwan", "cantonese"];
|
|
9
9
|
};
|
|
10
10
|
export type VoiceGender = (typeof VOICE_GENDER_CODES)[number];
|
|
11
11
|
export type VoiceAge = (typeof VOICE_AGE_CODES)[number];
|
|
12
12
|
export type VoiceTone = (typeof VOICE_TONE_CODES)[number];
|
|
13
13
|
export type VoiceLanguageCode = (typeof VOICE_LANGUAGE_CODES)[number];
|
|
14
|
-
export
|
|
14
|
+
export declare const VOICE_LEGACY_ACCENT_CODES_BY_LANGUAGE: {
|
|
15
|
+
readonly zh: readonly ["mandarin_guangdong", "mandarin_yunnan"];
|
|
16
|
+
};
|
|
17
|
+
export type VoiceWritableAccent = (typeof VOICE_ACCENT_CODES_BY_LANGUAGE)[keyof typeof VOICE_ACCENT_CODES_BY_LANGUAGE][number];
|
|
18
|
+
export type VoiceAccent = VoiceWritableAccent | "mandarin_guangdong" | "mandarin_yunnan";
|
package/dist/voice-metadata.js
CHANGED
|
@@ -6,5 +6,6 @@ export const VOICE_LANGUAGE_CODES = ["af", "ar", "bg", "bn", "bs", "ca", "cs", "
|
|
|
6
6
|
export const VOICE_TONE_MAX_ITEMS = 3;
|
|
7
7
|
export const VOICE_ACCENT_CODES_BY_LANGUAGE = {
|
|
8
8
|
en: ["american", "british", "scottish", "irish", "australian", "canadian", "us_southern", "us_new_york", "indian", "south_african", "russian", "japanese", "korean", "chinese"],
|
|
9
|
-
zh: ["
|
|
9
|
+
zh: ["mandarin_northeastern", "mandarin_shaanxi", "mandarin_shanghai", "mandarin_sichuan", "mandarin_henan", "mandarin_beijing", "mandarin_taiwan", "cantonese"],
|
|
10
10
|
};
|
|
11
|
+
export const VOICE_LEGACY_ACCENT_CODES_BY_LANGUAGE = { "zh": ["mandarin_guangdong", "mandarin_yunnan"] };
|