@breeze.blue/sdk 0.15.0 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,16 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.17.0
4
+
5
+ - Generate synchronous speech or stream audio with word timestamps and safe stream cleanup.
6
+ - Create and poll voice localization previews before auditioning and saving them.
7
+ - Control TTS output volume alongside speaking speed.
8
+
9
+ ## 0.16.0
10
+
11
+ - Clone previews use an automatically generated script in the reference audio language. Custom text, instructions, and language hints are not accepted.
12
+ - Breaking: remove `text`, `instructions`, `languageCode`, and `previewLanguageCode` from `voices.createClonePreview(...)` calls. Use Text to Speech with a saved voice for custom scripts and performance instructions.
13
+
3
14
  ## 0.15.0
4
15
 
5
16
  - Added `previewLanguageCode` to Voice Clone previews independently of the reference audio language hint.
package/README.md CHANGED
@@ -440,7 +440,7 @@ Daily Trend order. Search queries remain relevance-ranked; Trend is only a
440
440
  weak prior within the same relevance tier.
441
441
 
442
442
  Breeze voice creation is always two steps: produce a **preview**, let the
443
- user accept it, then **save the preview** as a real voice. Two ways to
443
+ user accept it, then **save the preview** as a real voice. Three ways to
444
444
  produce a preview:
445
445
 
446
446
  ```ts
@@ -454,7 +454,6 @@ const clonePreview = await client.voices.createClonePreview({
454
454
  filename: "sample.wav",
455
455
  contentType: "audio/wav",
456
456
  },
457
- text: "This is a short preview script.",
458
457
  });
459
458
  let generatedVoiceId = clonePreview.generatedVoiceId;
460
459
 
@@ -465,6 +464,29 @@ const design = await client.voices.createDesignPreview({
465
464
  generatedVoiceId = design.previews[0].generatedVoiceId;
466
465
  ```
467
466
 
467
+ Option C keeps an existing voice and makes it speak another language.
468
+ Localization runs as a background job: start it, then poll until the status is
469
+ `ready`. `name` defaults to the source voice name.
470
+
471
+ ```ts
472
+ const job = await client.voices.createLocalizePreview({
473
+ voiceId: firstVoiceId,
474
+ languageCode: "es",
475
+ name: "Documentary narrator (Spanish)",
476
+ });
477
+
478
+ let localized = await client.voices.getLocalizePreview(job.generationJobId);
479
+ while (localized.status !== "ready") {
480
+ if (localized.status === "failed" || localized.status === "cancelled") {
481
+ throw new Error(localized.error?.detail ?? localized.status);
482
+ }
483
+ await new Promise((resolve) => setTimeout(resolve, 2000));
484
+ localized = await client.voices.getLocalizePreview(job.generationJobId);
485
+ }
486
+
487
+ generatedVoiceId = localized.generatedVoiceId!;
488
+ ```
489
+
468
490
  For the fastest first audio, generate one design preview as a live 24 kHz mono
469
491
  PCM response. The Node `stream()` helper consumes the response body directly;
470
492
  a clean end means `generatedVoiceId` is ready to save:
@@ -574,8 +596,16 @@ try {
574
596
 
575
597
  Each API error exposes `status`, `code`, `detail`, `meta`, and `headers`.
576
598
 
577
- Clone previews identify the reference audio language and preview text language independently. Use `previewLanguageCode` for an explicit target hint when the text is ambiguous. Reference language hints do not override audio recognition; save cross-language previews with the original reference language.
599
+ Clone previews automatically generate a short script in the detected reference audio language. They accept a reference sample, name, and optional description. Save the preview as a voice, then use Text to Speech for custom scripts, instructions, or target languages.
578
600
 
579
601
  ### HTTP TTS speed
580
602
 
581
603
  Pass `voiceSettings: { speed: 1.25 }` to control speech speed without changing pitch, from 0.5 to 2.0. Omission always means 1.0; saved voice speed is not inherited. Sync, async and HTTP streaming support this parameter; Realtime does not.
604
+
605
+ Pass `voiceSettings: { volume: 1.5 }` to apply a request-level linear amplitude multiplier. Values range from 0.01 to 2.0; omission means 1.0. Values above 1.0 may clip peaks. Sync, async and HTTP streaming support this parameter; it does not change saved voice settings or apply to Realtime.
606
+
607
+ ## Word timing
608
+
609
+ Stream audio and word/token timestamps with `client.textToSpeech.streamWithTimestamps(...)`. Use `for await` to consume chunks containing `audioBase64` and `wordTimestamps`; breaking iteration closes the stream. Decode the audio separately and merge repeated word indices. See [Speech timing](https://docs.breezeblue.ai/guides/speech-timing) for complete examples and timing semantics.
610
+
611
+ For a complete response, use `client.textToSpeech.convertWithTimestamps(...)`. It returns base64 audio, its content type, and the full word/token timestamp list. Decode the audio before saving or playback. See [Speech timing](https://docs.breezeblue.ai/guides/speech-timing) and [Convert with timestamps](https://docs.breezeblue.ai/api-reference/text-to-speech/convert-with-timestamps).
package/dist/client.d.ts CHANGED
@@ -1,5 +1,6 @@
1
+ import { SpeechTimingStream, type SpeechWithTimestamps } from "./timestamps.js";
1
2
  import { AudioResponse } from "./audio.js";
2
- import type { AudioRequestOptions, BreezeBlueWebSocket, AsyncTextToSpeechJob, Balance, BreezeBlueClientOptions, CurrentApiKeyResponse, GenericStatus, GenerationJob, HistoryItem, HistoryList, HistoryListParams, Model, RequestOptions, RealtimeTextToSpeechConnectOptions, RealtimeTextToSpeechEvent, RealtimeTextToSpeechManagedConnectOptions, RealtimeTextToSpeechMessage, RealtimeTextToSpeechSession, RealtimeTextToSpeechSessionRequest, SavedVoice, SaveVoiceRequest, StreamTextToSpeechOptions, TextToSpeechRequest, TtsEnhanceRequest, TtsEnhanceResponse, Usage, UsageParams, Voice, VoiceClonePreview, VoiceClonePreviewRequest, VoiceDesignRequest, VoiceDesignResponse, VoiceDesignStreamRequest, VoiceEditRequest, VoiceList, VoiceMetadataOptions, VoiceRandomParams, VoiceSearchParams, VoiceSettings } from "./types.js";
3
+ import type { AudioRequestOptions, BreezeBlueWebSocket, AsyncTextToSpeechJob, Balance, BreezeBlueClientOptions, CurrentApiKeyResponse, GenericStatus, GenerationJob, HistoryItem, HistoryList, HistoryListParams, Model, RequestOptions, RealtimeTextToSpeechConnectOptions, RealtimeTextToSpeechEvent, RealtimeTextToSpeechManagedConnectOptions, RealtimeTextToSpeechMessage, RealtimeTextToSpeechSession, RealtimeTextToSpeechSessionRequest, SavedVoice, SaveVoiceRequest, StreamTextToSpeechOptions, TextToSpeechRequest, TtsEnhanceRequest, TtsEnhanceResponse, Usage, UsageParams, Voice, VoiceClonePreview, VoiceClonePreviewRequest, VoiceDesignRequest, VoiceDesignResponse, VoiceDesignStreamRequest, VoiceEditRequest, VoiceList, VoiceLocalizeJob, VoiceLocalizePreviewRequest, VoiceLocalizePreviewStatus, VoiceMetadataOptions, VoiceRandomParams, VoiceSearchParams, VoiceSettings } from "./types.js";
3
4
  export declare class BreezeBlueClient {
4
5
  readonly apiKey: string | undefined;
5
6
  readonly baseUrl: string;
@@ -26,6 +27,10 @@ declare class TextToSpeechResource {
26
27
  convert(voiceId: string, request: TextToSpeechRequest, options?: AudioRequestOptions): Promise<AudioResponse>;
27
28
  createJob(voiceId: string, request: TextToSpeechRequest, options?: AudioRequestOptions): Promise<AsyncTextToSpeechJob>;
28
29
  stream(voiceId: string, request: TextToSpeechRequest, options?: StreamTextToSpeechOptions): Promise<AudioResponse>;
30
+ convertWithTimestamps(voiceId: string, request: TextToSpeechRequest, options?: AudioRequestOptions): Promise<SpeechWithTimestamps>;
31
+ streamWithTimestamps(voiceId: string, request: TextToSpeechRequest & {
32
+ timestampMode?: "chunk" | "lookahead";
33
+ }, options?: StreamTextToSpeechOptions): Promise<SpeechTimingStream>;
29
34
  enhance(request: TtsEnhanceRequest, options?: RequestOptions): Promise<TtsEnhanceResponse>;
30
35
  }
31
36
  declare class RealtimeTextToSpeechResource {
@@ -197,8 +202,9 @@ declare class GenerationJobsResource {
197
202
  * Voice CRUD plus the two-step preview → save flow for Breeze voices.
198
203
  *
199
204
  * Breeze's voice creation is always two steps: produce a preview
200
- * (`createClonePreview` from an audio sample, or `createDesignPreview` from
201
- * a text description), then persist it with `savePreview`. Preview generation
205
+ * (`createClonePreview` from an audio sample, `createDesignPreview` from a
206
+ * text description, or `createLocalizePreview` from an existing voice plus a
207
+ * target language), then persist it with `savePreview`. Preview generation
202
208
  * never creates a saved voice until `savePreview` is called.
203
209
  */
204
210
  declare class VoicesResource {
@@ -245,6 +251,20 @@ declare class VoicesResource {
245
251
  * A clean end means `generatedVoiceId` is ready for {@link savePreview}.
246
252
  */
247
253
  streamDesignPreview(request: VoiceDesignStreamRequest, options?: RequestOptions): Promise<AudioResponse>;
254
+ /**
255
+ * Start a preview that speaks an existing voice in another language.
256
+ * Localization runs as a background job: poll the returned
257
+ * `generationJobId` with {@link getLocalizePreview} until the status is
258
+ * `ready`, then audition it with {@link streamPreview} and persist it with
259
+ * {@link savePreview}. `name` defaults to the source voice name.
260
+ */
261
+ createLocalizePreview(request: VoiceLocalizePreviewRequest, options?: RequestOptions): Promise<VoiceLocalizeJob>;
262
+ /**
263
+ * Read the current state of a localization preview job. `generatedVoiceId`,
264
+ * `text`, and `languageCode` stay null until the status is `ready`; `error`
265
+ * is set only when the job ends in `failed`.
266
+ */
267
+ getLocalizePreview(generationJobId: string, options?: RequestOptions): Promise<VoiceLocalizePreviewStatus>;
248
268
  /** Download the completed audio of a clone or design preview. */
249
269
  streamPreview(generatedVoiceId: string, options?: AudioRequestOptions): Promise<AudioResponse>;
250
270
  /** Persist a clone or design preview as a real voice asset. */
package/dist/client.js CHANGED
@@ -1,3 +1,4 @@
1
+ import { SpeechTimingStream } from "./timestamps.js";
1
2
  import { AudioResponse } from "./audio.js";
2
3
  import { BreezeBlueAPIError, BreezeBlueConfigurationError, BreezeBlueRealtimeError, errorFromResponse, } from "./errors.js";
3
4
  import { camelizeKeys, snakeizeKeys } from "./_transform.js";
@@ -119,6 +120,13 @@ class TextToSpeechResource {
119
120
  enableLogging: options.enableLogging,
120
121
  }, request, options);
121
122
  }
123
+ convertWithTimestamps(voiceId, request, options = {}) {
124
+ return this.client.requestJson("POST", `/v1/text-to-speech/${encodeURIComponent(voiceId)}/with-timestamps`, { outputFormat: options.outputFormat }, request, options);
125
+ }
126
+ async streamWithTimestamps(voiceId, request, options = {}) {
127
+ const response = await this.client.request("POST", `/v1/text-to-speech/${encodeURIComponent(voiceId)}/stream/with-timestamps`, { outputFormat: options.outputFormat, optimizeStreamingLatency: options.optimizeStreamingLatency, enableLogging: options.enableLogging }, request, options);
128
+ return new SpeechTimingStream(response);
129
+ }
122
130
  enhance(request, options) {
123
131
  return this.client.requestJson("POST", "/v1/text-to-speech/enhance", undefined, request, options);
124
132
  }
@@ -1479,8 +1487,9 @@ class GenerationJobsResource {
1479
1487
  * Voice CRUD plus the two-step preview → save flow for Breeze voices.
1480
1488
  *
1481
1489
  * Breeze's voice creation is always two steps: produce a preview
1482
- * (`createClonePreview` from an audio sample, or `createDesignPreview` from
1483
- * a text description), then persist it with `savePreview`. Preview generation
1490
+ * (`createClonePreview` from an audio sample, `createDesignPreview` from a
1491
+ * text description, or `createLocalizePreview` from an existing voice plus a
1492
+ * target language), then persist it with `savePreview`. Preview generation
1484
1493
  * never creates a saved voice until `savePreview` is called.
1485
1494
  */
1486
1495
  class VoicesResource {
@@ -1568,6 +1577,24 @@ class VoicesResource {
1568
1577
  streamDesignPreview(request, options) {
1569
1578
  return this.client.requestAudio("POST", "/v1/voice-previews/design/stream", undefined, request, options);
1570
1579
  }
1580
+ /**
1581
+ * Start a preview that speaks an existing voice in another language.
1582
+ * Localization runs as a background job: poll the returned
1583
+ * `generationJobId` with {@link getLocalizePreview} until the status is
1584
+ * `ready`, then audition it with {@link streamPreview} and persist it with
1585
+ * {@link savePreview}. `name` defaults to the source voice name.
1586
+ */
1587
+ createLocalizePreview(request, options) {
1588
+ return this.client.requestJson("POST", "/v1/voice-previews/localize", undefined, request, options);
1589
+ }
1590
+ /**
1591
+ * Read the current state of a localization preview job. `generatedVoiceId`,
1592
+ * `text`, and `languageCode` stay null until the status is `ready`; `error`
1593
+ * is set only when the job ends in `failed`.
1594
+ */
1595
+ getLocalizePreview(generationJobId, options) {
1596
+ return this.client.requestJson("GET", `/v1/voice-previews/localize/${encodeURIComponent(generationJobId)}`, undefined, undefined, options);
1597
+ }
1571
1598
  /** Download the completed audio of a clone or design preview. */
1572
1599
  streamPreview(generatedVoiceId, options = {}) {
1573
1600
  return this.client.requestAudio("GET", `/v1/voice-previews/${encodeURIComponent(generatedVoiceId)}/stream`, { outputFormat: options.outputFormat }, undefined, options);
@@ -1727,13 +1754,14 @@ function clonePreviewForm(request) {
1727
1754
  if (request.removeBackgroundNoise) {
1728
1755
  throw new BreezeBlueConfigurationError("removeBackgroundNoise is not supported by the Breeze Developer API.");
1729
1756
  }
1757
+ for (const field of ["text", "instructions", "languageCode", "previewLanguageCode"]) {
1758
+ if (field in request) {
1759
+ throw new BreezeBlueConfigurationError(`Clone previews do not accept ${field}; the script and language are generated automatically.`);
1760
+ }
1761
+ }
1730
1762
  const form = new FormData();
1731
1763
  form.set("name", request.name);
1732
1764
  appendOptional(form, "description", request.description);
1733
- appendOptional(form, "language_code", request.languageCode);
1734
- appendOptional(form, "preview_language_code", request.previewLanguageCode);
1735
- appendOptional(form, "text", request.text);
1736
- appendOptional(form, "instructions", request.instructions);
1737
1765
  for (const file of normalizeFiles(request.file, request.files, { required: true })) {
1738
1766
  appendFile(form, "files", file);
1739
1767
  }
package/dist/index.d.ts CHANGED
@@ -4,3 +4,5 @@ export { BreezeBlueClient, ManagedRealtimeTextToSpeechConnection, RealtimeTextTo
4
4
  export { SDK_NAME, SDK_VERSION } from "./version.js";
5
5
  export type * from "./types.js";
6
6
  export * from "./voice-metadata.js";
7
+ export { SpeechTimingStream } from "./timestamps.js";
8
+ export type { SpeechTimingChunk, SpeechWithTimestamps, WordTimestamp } from "./timestamps.js";
package/dist/index.js CHANGED
@@ -3,3 +3,4 @@ export { BreezeBlueAPIError, BreezeBlueAuthenticationError, BreezeBlueBadRequest
3
3
  export { BreezeBlueClient, ManagedRealtimeTextToSpeechConnection, RealtimeTextToSpeechConnection, } from "./client.js";
4
4
  export { SDK_NAME, SDK_VERSION } from "./version.js";
5
5
  export * from "./voice-metadata.js";
6
+ export { SpeechTimingStream } from "./timestamps.js";
@@ -0,0 +1,22 @@
1
+ export interface WordTimestamp {
2
+ index: number;
3
+ word: string;
4
+ start: number;
5
+ end: number;
6
+ }
7
+ export interface SpeechTimingChunk {
8
+ audioBase64: string;
9
+ wordTimestamps: WordTimestamp[];
10
+ }
11
+ export interface SpeechWithTimestamps extends SpeechTimingChunk {
12
+ contentType: string;
13
+ }
14
+ /** Single-use stream. Breaking iteration cancels the response body. */
15
+ export declare class SpeechTimingStream implements AsyncIterable<SpeechTimingChunk> {
16
+ private readonly response;
17
+ readonly historyItemId: string | null;
18
+ private reader?;
19
+ constructor(response: Response);
20
+ close(): Promise<void>;
21
+ [Symbol.asyncIterator](): AsyncGenerator<SpeechTimingChunk>;
22
+ }
@@ -0,0 +1,64 @@
1
+ import { camelizeKeys } from "./_transform.js";
2
+ /** Single-use stream. Breaking iteration cancels the response body. */
3
+ export class SpeechTimingStream {
4
+ response;
5
+ historyItemId;
6
+ reader;
7
+ constructor(response) {
8
+ this.response = response;
9
+ this.historyItemId = response.headers.get("history-item-id");
10
+ }
11
+ async close() {
12
+ if (this.reader)
13
+ await this.reader.cancel();
14
+ else
15
+ await this.response.body?.cancel();
16
+ }
17
+ async *[Symbol.asyncIterator]() {
18
+ const reader = this.response.body?.getReader();
19
+ if (!reader)
20
+ throw new Error("Speech stream has no response body.");
21
+ this.reader = reader;
22
+ const decoder = new TextDecoder("utf-8", { fatal: true });
23
+ let pending = "";
24
+ const parse = (line) => {
25
+ if (line.length > 4 * 1024 * 1024)
26
+ throw new Error("Timestamp stream line is too large.");
27
+ const value = JSON.parse(line);
28
+ if (value.error)
29
+ throw new Error(`Speech stream failed: ${JSON.stringify(value.error)}`);
30
+ if (typeof value.audio_base64 !== "string" || !Array.isArray(value.word_timestamps))
31
+ throw new Error("Invalid timestamp stream event.");
32
+ return camelizeKeys(value);
33
+ };
34
+ try {
35
+ while (true) {
36
+ const { value, done } = await reader.read();
37
+ pending += decoder.decode(value, { stream: !done });
38
+ let boundary = pending.indexOf("\n");
39
+ while (boundary >= 0) {
40
+ const line = pending.slice(0, boundary);
41
+ pending = pending.slice(boundary + 1);
42
+ if (line.trim())
43
+ yield parse(line);
44
+ boundary = pending.indexOf("\n");
45
+ }
46
+ if (pending.length > 4 * 1024 * 1024)
47
+ throw new Error("Timestamp stream line is too large.");
48
+ if (done)
49
+ break;
50
+ }
51
+ if (pending.trim())
52
+ yield parse(pending);
53
+ }
54
+ finally {
55
+ try {
56
+ await reader.cancel();
57
+ }
58
+ finally {
59
+ reader.releaseLock();
60
+ this.reader = undefined;
61
+ }
62
+ }
63
+ }
64
+ }
package/dist/types.d.ts CHANGED
@@ -234,6 +234,11 @@ export interface VoiceSettings {
234
234
  speed?: number;
235
235
  guidanceScale?: number | null;
236
236
  }
237
+ /** Request-only HTTP TTS settings. Volume is a linear amplitude multiplier. */
238
+ export interface TtsVoiceSettings extends VoiceSettings {
239
+ /** Linear amplitude multiplier (0.01–2.0); omitted means 1.0. */
240
+ volume?: number;
241
+ }
237
242
  export interface PronunciationDictionaryLocator {
238
243
  pronunciationDictionaryId: string;
239
244
  versionId: string;
@@ -243,7 +248,7 @@ export interface TextToSpeechRequest {
243
248
  modelId?: string;
244
249
  languageCode?: string;
245
250
  instructions?: string;
246
- voiceSettings?: VoiceSettings;
251
+ voiceSettings?: TtsVoiceSettings;
247
252
  seed?: number;
248
253
  pronunciationDictionaryLocators?: PronunciationDictionaryLocator[];
249
254
  previousText?: string;
@@ -313,6 +318,9 @@ export interface Voice {
313
318
  createdAtUnix?: number | null;
314
319
  primaryCategoryCode?: string | null;
315
320
  visibility?: string | null;
321
+ publicationStatus?: string | null;
322
+ publicationJobId?: string | null;
323
+ publicationErrorCode?: string | null;
316
324
  languageCode: VoiceLanguageCode;
317
325
  gender: VoiceGender | null;
318
326
  age: VoiceAge | null;
@@ -379,6 +387,9 @@ export interface VoiceSearchParams {
379
387
  }
380
388
  export interface GenericStatus {
381
389
  status: string;
390
+ visibility?: string | null;
391
+ publicationStatus?: string | null;
392
+ publicationJobId?: string | null;
382
393
  }
383
394
  export type UploadData = Blob | ArrayBuffer | Uint8Array | string;
384
395
  export interface UploadFile {
@@ -392,12 +403,6 @@ export interface VoiceClonePreviewRequest {
392
403
  files?: Array<UploadData | UploadFile>;
393
404
  description?: string;
394
405
  removeBackgroundNoise?: boolean;
395
- /** Reference language hint; audio recognition remains authoritative. */
396
- languageCode?: VoiceLanguageCode;
397
- /** Target hint used when preview text language is uncertain. */
398
- previewLanguageCode?: VoiceLanguageCode;
399
- text?: string;
400
- instructions?: string;
401
406
  }
402
407
  export interface VoiceClonePreview {
403
408
  generatedVoiceId: string;
@@ -446,6 +451,23 @@ export interface VoiceDesignResponse {
446
451
  previews: VoiceDesignPreview[];
447
452
  text: string;
448
453
  }
454
+ export interface VoiceLocalizePreviewRequest {
455
+ voiceId: string;
456
+ languageCode: VoiceLanguageCode;
457
+ name?: string;
458
+ }
459
+ export interface VoiceLocalizeJob {
460
+ generationJobId: string;
461
+ status: string;
462
+ }
463
+ export interface VoiceLocalizePreviewStatus {
464
+ generationJobId: string;
465
+ status: string;
466
+ languageCode?: VoiceLanguageCode | null;
467
+ generatedVoiceId?: string | null;
468
+ text?: string | null;
469
+ error?: GenerationJobError | null;
470
+ }
449
471
  export interface SaveVoiceRequest {
450
472
  generatedVoiceId: string;
451
473
  voiceName: string;
package/dist/version.d.ts CHANGED
@@ -1,2 +1,2 @@
1
- export declare const SDK_VERSION = "0.15.0";
1
+ export declare const SDK_VERSION = "0.17.0";
2
2
  export declare const SDK_NAME = "@breeze.blue/sdk";
package/dist/version.js CHANGED
@@ -1,2 +1,2 @@
1
- export const SDK_VERSION = "0.15.0";
1
+ export const SDK_VERSION = "0.17.0";
2
2
  export const SDK_NAME = "@breeze.blue/sdk";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@breeze.blue/sdk",
3
- "version": "0.15.0",
3
+ "version": "0.17.0",
4
4
  "description": "ESM-first TypeScript SDK for the Breeze Blue Developer API.",
5
5
  "license": "MIT",
6
6
  "author": "Breeze Blue <support@breezeblue.ai>",