eleven_rb 0.4.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +50 -0
- data/README.md +125 -4
- data/lib/eleven_rb/callbacks.rb +25 -1
- data/lib/eleven_rb/client.rb +8 -0
- data/lib/eleven_rb/configuration.rb +7 -3
- data/lib/eleven_rb/http/client.rb +32 -12
- data/lib/eleven_rb/model_capabilities.rb +103 -0
- data/lib/eleven_rb/objects/audio.rb +13 -4
- data/lib/eleven_rb/objects/cost_info.rb +8 -3
- data/lib/eleven_rb/objects/model.rb +10 -0
- data/lib/eleven_rb/objects/voice_settings.rb +33 -0
- data/lib/eleven_rb/resources/base.rb +18 -0
- data/lib/eleven_rb/resources/models.rb +37 -4
- data/lib/eleven_rb/resources/text_to_dialogue.rb +203 -0
- data/lib/eleven_rb/resources/text_to_speech.rb +150 -43
- data/lib/eleven_rb/version.rb +1 -1
- data/lib/eleven_rb.rb +3 -0
- metadata +7 -5
|
@@ -25,6 +25,39 @@ module ElevenRb
|
|
|
25
25
|
from_response(DEFAULTS.merge(overrides))
|
|
26
26
|
end
|
|
27
27
|
|
|
28
|
+
# Voice-setting keys the capability table knows about. Only these can be
|
|
29
|
+
# dropped for a model; any other key is passed through untouched so a
|
|
30
|
+
# future API field is never swallowed.
|
|
31
|
+
KNOWN_KEYS = ModelCapabilities::ALL_VOICE_SETTINGS
|
|
32
|
+
|
|
33
|
+
# Build the voice_settings hash a model actually honours
|
|
34
|
+
#
|
|
35
|
+
# Starts from DEFAULTS filtered to the model's supported keys and merges the
|
|
36
|
+
# overrides (keys symbolized). Known keys the model does not support are
|
|
37
|
+
# dropped; unknown keys pass through after the supported ones, in the
|
|
38
|
+
# caller's order; nil values are removed. Only non-nil override keys count
|
|
39
|
+
# as dropped: defaults the model does not take are filtered silently.
|
|
40
|
+
#
|
|
41
|
+
# @example
|
|
42
|
+
# VoiceSettings.for_model('eleven_multilingual_v2')
|
|
43
|
+
# # => [{ stability: 0.5, similarity_boost: 0.75, style: 0.0, use_speaker_boost: true }, []]
|
|
44
|
+
# VoiceSettings.for_model('eleven_v4', speed: 1.1)
|
|
45
|
+
# # => [{ stability: 0.5, similarity_boost: 0.75 }, [:speed]]
|
|
46
|
+
#
|
|
47
|
+
# @param model_id [String] the model ID
|
|
48
|
+
# @param overrides [Hash] caller settings (String or Symbol keys)
|
|
49
|
+
# @return [Array(Hash, Array<Symbol>)] the settings hash and the dropped override keys
|
|
50
|
+
def self.for_model(model_id, overrides = {})
|
|
51
|
+
supported = ModelCapabilities.supported_voice_settings(model_id)
|
|
52
|
+
requested = (overrides || {}).to_h.transform_keys(&:to_sym)
|
|
53
|
+
|
|
54
|
+
dropped = requested.compact.keys & (KNOWN_KEYS - supported)
|
|
55
|
+
unknown = requested.except(*KNOWN_KEYS)
|
|
56
|
+
settings = DEFAULTS.slice(*supported).merge(requested.slice(*supported)).merge(unknown).compact
|
|
57
|
+
|
|
58
|
+
[settings, dropped]
|
|
59
|
+
end
|
|
60
|
+
|
|
28
61
|
# Convert to hash suitable for API request
|
|
29
62
|
#
|
|
30
63
|
# @return [Hash]
|
|
@@ -51,6 +51,24 @@ module ElevenRb
|
|
|
51
51
|
http_client.post(path, body, response_type: :binary)
|
|
52
52
|
end
|
|
53
53
|
|
|
54
|
+
# Make a JSON POST request and also return the response headers
|
|
55
|
+
#
|
|
56
|
+
# @param path [String]
|
|
57
|
+
# @param body [Hash]
|
|
58
|
+
# @return [Hash] `{ body: Hash, headers: Hash<String, String> }`
|
|
59
|
+
def post_with_meta(path, body = {})
|
|
60
|
+
http_client.post(path, body, response_type: :json, with_meta: true)
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# Make a binary POST request and also return the response headers
|
|
64
|
+
#
|
|
65
|
+
# @param path [String]
|
|
66
|
+
# @param body [Hash]
|
|
67
|
+
# @return [Hash] `{ body: String, headers: Hash<String, String> }`
|
|
68
|
+
def post_binary_with_meta(path, body = {})
|
|
69
|
+
http_client.post(path, body, response_type: :binary, with_meta: true)
|
|
70
|
+
end
|
|
71
|
+
|
|
54
72
|
# Make a streaming POST request
|
|
55
73
|
#
|
|
56
74
|
# @param path [String]
|
|
@@ -9,23 +9,36 @@ module ElevenRb
|
|
|
9
9
|
#
|
|
10
10
|
# @example Find multilingual models
|
|
11
11
|
# client.models.multilingual
|
|
12
|
+
#
|
|
13
|
+
# @example Find one model
|
|
14
|
+
# client.models.find('eleven_v4')
|
|
12
15
|
class Models < Base
|
|
13
16
|
# List all available models
|
|
14
17
|
#
|
|
15
18
|
# @return [Array<Objects::Model>]
|
|
16
19
|
def list
|
|
17
|
-
|
|
20
|
+
# Call the HTTP client directly: #get below is the model lookup (kept for
|
|
21
|
+
# compatibility) and shadows Base#get, which made this method recurse.
|
|
22
|
+
response = http_client.get('/models')
|
|
18
23
|
response.map { |m| Objects::Model.from_response(m) }
|
|
19
24
|
end
|
|
20
25
|
|
|
21
|
-
#
|
|
26
|
+
# Find a specific model by ID
|
|
22
27
|
#
|
|
23
28
|
# @param model_id [String] the model ID
|
|
24
29
|
# @return [Objects::Model, nil]
|
|
25
|
-
def
|
|
30
|
+
def find(model_id)
|
|
26
31
|
list.find { |m| m.model_id == model_id }
|
|
27
32
|
end
|
|
28
33
|
|
|
34
|
+
# Alias of {#find}, kept for backwards compatibility
|
|
35
|
+
#
|
|
36
|
+
# @param model_id [String] the model ID
|
|
37
|
+
# @return [Objects::Model, nil]
|
|
38
|
+
def get(model_id)
|
|
39
|
+
find(model_id)
|
|
40
|
+
end
|
|
41
|
+
|
|
29
42
|
# Get all multilingual models
|
|
30
43
|
#
|
|
31
44
|
# @return [Array<Objects::Model>]
|
|
@@ -51,7 +64,20 @@ module ElevenRb
|
|
|
51
64
|
#
|
|
52
65
|
# @return [Objects::Model, nil]
|
|
53
66
|
def default
|
|
54
|
-
|
|
67
|
+
default_from(list)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# Get the latest/most capable model available to the account:
|
|
71
|
+
# eleven_v4, else eleven_v3, else {#default} (one /models request)
|
|
72
|
+
#
|
|
73
|
+
# @return [Objects::Model, nil]
|
|
74
|
+
def latest
|
|
75
|
+
models = list
|
|
76
|
+
%w[eleven_v4 eleven_v3].each do |model_id|
|
|
77
|
+
model = models.find { |m| m.model_id == model_id }
|
|
78
|
+
return model if model
|
|
79
|
+
end
|
|
80
|
+
default_from(models)
|
|
55
81
|
end
|
|
56
82
|
|
|
57
83
|
# Get model IDs as array
|
|
@@ -60,6 +86,13 @@ module ElevenRb
|
|
|
60
86
|
def ids
|
|
61
87
|
list.map(&:model_id)
|
|
62
88
|
end
|
|
89
|
+
|
|
90
|
+
private
|
|
91
|
+
|
|
92
|
+
# eleven_multilingual_v2, else the first TTS-capable model, from an already-fetched list
|
|
93
|
+
def default_from(models)
|
|
94
|
+
models.find { |m| m.model_id == 'eleven_multilingual_v2' } || models.find(&:can_do_text_to_speech)
|
|
95
|
+
end
|
|
63
96
|
end
|
|
64
97
|
end
|
|
65
98
|
end
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ElevenRb
|
|
4
|
+
module Resources
|
|
5
|
+
# Text-to-dialogue resource for multi-speaker audio generation
|
|
6
|
+
#
|
|
7
|
+
# @example Generate dialogue
|
|
8
|
+
# audio = client.text_to_dialogue.generate([
|
|
9
|
+
# { text: "[excited] Welcome!", voice_id: "voice_abc" },
|
|
10
|
+
# { text: "[laughs] Thanks!", voice_id: "voice_xyz" }
|
|
11
|
+
# ])
|
|
12
|
+
# audio.save_to_file("dialogue.mp3")
|
|
13
|
+
#
|
|
14
|
+
# @example Dialogue with timestamps and per-speaker segments
|
|
15
|
+
# result = client.dialogue.generate_with_timestamps(inputs, seed: 7)
|
|
16
|
+
# result[:voice_segments] # => [{ "voice_id" => ..., "start_time_seconds" => ... }, ...]
|
|
17
|
+
class TextToDialogue < Base
|
|
18
|
+
DEFAULT_MODEL = 'eleven_v4'
|
|
19
|
+
MAX_VOICES_PER_REQUEST = 10
|
|
20
|
+
|
|
21
|
+
# Kept for compatibility. The hard cap now comes from
|
|
22
|
+
# ModelCapabilities.max_text_length(model_id) (5,000 for eleven_v3).
|
|
23
|
+
MAX_TEXT_LENGTH = 5000
|
|
24
|
+
|
|
25
|
+
# Above this many characters the API recommends splitting the dialogue;
|
|
26
|
+
# the gem logs a warning but still sends the request.
|
|
27
|
+
RECOMMENDED_MAX_TEXT_LENGTH = 2_000
|
|
28
|
+
|
|
29
|
+
# Optional request-body keys, in the order they are written to the body.
|
|
30
|
+
# Each is omitted from the body when nil.
|
|
31
|
+
OPTIONAL_BODY_KEYS = %i[
|
|
32
|
+
language_code
|
|
33
|
+
settings
|
|
34
|
+
seed
|
|
35
|
+
use_pvc_as_ivc
|
|
36
|
+
previous_text
|
|
37
|
+
future_text
|
|
38
|
+
previous_request_ids
|
|
39
|
+
next_request_ids
|
|
40
|
+
pronunciation_dictionary_locators
|
|
41
|
+
].freeze
|
|
42
|
+
|
|
43
|
+
# Generate dialogue audio from multiple speaker inputs
|
|
44
|
+
#
|
|
45
|
+
# @param inputs [Array<Hash>] Array of { text:, voice_id: } hashes
|
|
46
|
+
# @param model_id [String] Model to use (default: eleven_v4)
|
|
47
|
+
# @param language_code [String, nil] ISO 639-1 language code
|
|
48
|
+
# @param settings [Hash, nil] Generation settings, sent unchanged (e.g. stability, similarity)
|
|
49
|
+
# @param seed [Integer, nil] Seed for reproducibility
|
|
50
|
+
# @param output_format [String] Audio output format
|
|
51
|
+
# @param apply_text_normalization [String] "auto", "on", or "off"
|
|
52
|
+
# @param use_pvc_as_ivc [Boolean, nil] use the IVC version of professional voices
|
|
53
|
+
# @param previous_text [String, nil] text that comes before this dialogue (continuity)
|
|
54
|
+
# @param future_text [String, nil] text that comes after this dialogue (continuity)
|
|
55
|
+
# @param previous_request_ids [Array<String>, nil] request IDs of preceding generations
|
|
56
|
+
# @param next_request_ids [Array<String>, nil] request IDs of following generations
|
|
57
|
+
# @param pronunciation_dictionary_locators [Array<Hash>, nil] pronunciation dictionaries to apply
|
|
58
|
+
# @return [Objects::Audio]
|
|
59
|
+
def generate(
|
|
60
|
+
inputs,
|
|
61
|
+
model_id: DEFAULT_MODEL,
|
|
62
|
+
language_code: nil,
|
|
63
|
+
settings: nil,
|
|
64
|
+
seed: nil,
|
|
65
|
+
output_format: 'mp3_44100_128',
|
|
66
|
+
apply_text_normalization: 'auto',
|
|
67
|
+
use_pvc_as_ivc: nil,
|
|
68
|
+
previous_text: nil,
|
|
69
|
+
future_text: nil,
|
|
70
|
+
previous_request_ids: nil,
|
|
71
|
+
next_request_ids: nil,
|
|
72
|
+
pronunciation_dictionary_locators: nil
|
|
73
|
+
)
|
|
74
|
+
validate_inputs!(inputs, model_id)
|
|
75
|
+
|
|
76
|
+
body = build_request_body(inputs, model_id, apply_text_normalization, optional_values(binding))
|
|
77
|
+
response = post_binary_with_meta("/text-to-dialogue?output_format=#{output_format}", body)
|
|
78
|
+
|
|
79
|
+
build_audio_response(response[:body], inputs, output_format, model_id,
|
|
80
|
+
request_id: response[:headers]['request-id'])
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# Generate dialogue audio with character timestamps and per-voice segments
|
|
84
|
+
#
|
|
85
|
+
# Takes the same keywords as {#generate}.
|
|
86
|
+
#
|
|
87
|
+
# @param inputs [Array<Hash>] Array of { text:, voice_id: } hashes
|
|
88
|
+
# @return [Hash] `{ audio:, alignment:, normalized_alignment:, voice_segments:, request_id: }`
|
|
89
|
+
def generate_with_timestamps(
|
|
90
|
+
inputs,
|
|
91
|
+
model_id: DEFAULT_MODEL,
|
|
92
|
+
language_code: nil,
|
|
93
|
+
settings: nil,
|
|
94
|
+
seed: nil,
|
|
95
|
+
output_format: 'mp3_44100_128',
|
|
96
|
+
apply_text_normalization: 'auto',
|
|
97
|
+
use_pvc_as_ivc: nil,
|
|
98
|
+
previous_text: nil,
|
|
99
|
+
future_text: nil,
|
|
100
|
+
previous_request_ids: nil,
|
|
101
|
+
next_request_ids: nil,
|
|
102
|
+
pronunciation_dictionary_locators: nil
|
|
103
|
+
)
|
|
104
|
+
validate_inputs!(inputs, model_id)
|
|
105
|
+
|
|
106
|
+
body = build_request_body(inputs, model_id, apply_text_normalization, optional_values(binding))
|
|
107
|
+
result = post_with_meta("/text-to-dialogue/with-timestamps?output_format=#{output_format}", body)
|
|
108
|
+
response = result[:body]
|
|
109
|
+
request_id = result[:headers]['request-id']
|
|
110
|
+
|
|
111
|
+
audio_data = Base64.decode64(response['audio_base64']) if response['audio_base64']
|
|
112
|
+
audio = (build_audio_response(audio_data, inputs, output_format, model_id, request_id: request_id) if audio_data)
|
|
113
|
+
|
|
114
|
+
{
|
|
115
|
+
audio: audio,
|
|
116
|
+
alignment: response['alignment'],
|
|
117
|
+
normalized_alignment: response['normalized_alignment'],
|
|
118
|
+
voice_segments: response['voice_segments'],
|
|
119
|
+
request_id: request_id
|
|
120
|
+
}
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
private
|
|
124
|
+
|
|
125
|
+
# The optional keyword values of the calling method, keyed by OPTIONAL_BODY_KEYS
|
|
126
|
+
def optional_values(caller_binding)
|
|
127
|
+
OPTIONAL_BODY_KEYS.to_h { |key| [key, caller_binding.local_variable_get(key)] }
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
def build_request_body(inputs, model_id, apply_text_normalization, options)
|
|
131
|
+
body = {
|
|
132
|
+
inputs: inputs.map { |i| { text: i[:text], voice_id: i[:voice_id] } },
|
|
133
|
+
model_id: model_id,
|
|
134
|
+
apply_text_normalization: apply_text_normalization
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
OPTIONAL_BODY_KEYS.each do |key|
|
|
138
|
+
body[key] = options[key] unless options[key].nil?
|
|
139
|
+
end
|
|
140
|
+
body
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
def build_audio_response(data, inputs, output_format, model_id, request_id: nil)
|
|
144
|
+
total_text = inputs.map { |i| i[:text] }.join("\n")
|
|
145
|
+
total_chars = inputs.sum { |i| i[:text].length }
|
|
146
|
+
primary_voice = inputs.first[:voice_id]
|
|
147
|
+
|
|
148
|
+
audio = Objects::Audio.new(
|
|
149
|
+
data: data, format: output_format,
|
|
150
|
+
voice_id: primary_voice, text: total_text, model_id: model_id,
|
|
151
|
+
request_id: request_id
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
cost_info = Objects::CostInfo.new(
|
|
155
|
+
character_count: total_chars, voice_id: primary_voice, model_id: model_id
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
http_client.config.trigger(
|
|
159
|
+
:on_audio_generated,
|
|
160
|
+
audio: audio, voice_id: primary_voice,
|
|
161
|
+
text: total_text, cost_info: cost_info.to_h,
|
|
162
|
+
request_id: request_id
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
audio
|
|
166
|
+
end
|
|
167
|
+
|
|
168
|
+
def validate_inputs!(inputs, model_id = DEFAULT_MODEL)
|
|
169
|
+
raise Errors::ValidationError, 'inputs must be a non-empty array' unless inputs.is_a?(Array) && !inputs.empty?
|
|
170
|
+
|
|
171
|
+
inputs.each_with_index do |input, i|
|
|
172
|
+
validate_presence!(input[:text], "inputs[#{i}].text")
|
|
173
|
+
validate_presence!(input[:voice_id], "inputs[#{i}].voice_id")
|
|
174
|
+
end
|
|
175
|
+
|
|
176
|
+
unique_voices = inputs.map { |i| i[:voice_id] }.uniq
|
|
177
|
+
if unique_voices.length > MAX_VOICES_PER_REQUEST
|
|
178
|
+
raise Errors::ValidationError,
|
|
179
|
+
"Maximum #{MAX_VOICES_PER_REQUEST} unique voices per request " \
|
|
180
|
+
"(got #{unique_voices.length})"
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
validate_text_length!(inputs.sum { |i| i[:text].length }, model_id)
|
|
184
|
+
end
|
|
185
|
+
|
|
186
|
+
def validate_text_length!(total_chars, model_id)
|
|
187
|
+
max_length = ModelCapabilities.max_text_length(model_id)
|
|
188
|
+
if total_chars > max_length
|
|
189
|
+
raise Errors::ValidationError,
|
|
190
|
+
"Total text length #{total_chars} exceeds maximum " \
|
|
191
|
+
"#{max_length} characters for #{model_id}"
|
|
192
|
+
end
|
|
193
|
+
|
|
194
|
+
return unless total_chars > RECOMMENDED_MAX_TEXT_LENGTH
|
|
195
|
+
|
|
196
|
+
http_client.config.logger&.warn(
|
|
197
|
+
"[ElevenRb] text-to-dialogue text is #{total_chars} characters; " \
|
|
198
|
+
"#{RECOMMENDED_MAX_TEXT_LENGTH} or fewer per request is recommended"
|
|
199
|
+
)
|
|
200
|
+
end
|
|
201
|
+
end
|
|
202
|
+
end
|
|
203
|
+
end
|
|
@@ -12,18 +12,67 @@ module ElevenRb
|
|
|
12
12
|
# client.tts.stream("Hello world", voice_id: "voice_id") do |chunk|
|
|
13
13
|
# io.write(chunk)
|
|
14
14
|
# end
|
|
15
|
+
#
|
|
16
|
+
# @example Eleven v4 with continuity and a fixed seed
|
|
17
|
+
# audio = client.tts.generate(
|
|
18
|
+
# "[sighs] Right. Let's try that again.",
|
|
19
|
+
# voice_id: "voice_id",
|
|
20
|
+
# model_id: "eleven_v4",
|
|
21
|
+
# seed: 42,
|
|
22
|
+
# previous_text: "That did not go to plan."
|
|
23
|
+
# )
|
|
24
|
+
# audio.request_id # => "abc123" (from the request-id response header)
|
|
15
25
|
class TextToSpeech < Base
|
|
16
26
|
DEFAULT_MODEL = 'eleven_multilingual_v2'
|
|
27
|
+
|
|
28
|
+
# Kept for compatibility. The per-request cap now comes from
|
|
29
|
+
# ModelCapabilities.max_text_length(model_id) (5,000 for eleven_v3).
|
|
17
30
|
MAX_TEXT_LENGTH = 5000
|
|
18
31
|
|
|
32
|
+
# Output formats the API accepts (documentation only; not validated)
|
|
19
33
|
OUTPUT_FORMATS = %w[
|
|
34
|
+
mp3_22050_32
|
|
35
|
+
mp3_24000_48
|
|
36
|
+
mp3_44100_32
|
|
37
|
+
mp3_44100_64
|
|
38
|
+
mp3_44100_96
|
|
20
39
|
mp3_44100_128
|
|
21
40
|
mp3_44100_192
|
|
41
|
+
opus_48000_32
|
|
42
|
+
opus_48000_64
|
|
43
|
+
opus_48000_96
|
|
44
|
+
opus_48000_128
|
|
45
|
+
opus_48000_192
|
|
46
|
+
pcm_8000
|
|
22
47
|
pcm_16000
|
|
23
48
|
pcm_22050
|
|
24
49
|
pcm_24000
|
|
50
|
+
pcm_32000
|
|
25
51
|
pcm_44100
|
|
52
|
+
pcm_48000
|
|
53
|
+
wav_8000
|
|
54
|
+
wav_16000
|
|
55
|
+
wav_22050
|
|
56
|
+
wav_24000
|
|
57
|
+
wav_32000
|
|
58
|
+
wav_44100
|
|
59
|
+
wav_48000
|
|
26
60
|
ulaw_8000
|
|
61
|
+
alaw_8000
|
|
62
|
+
].freeze
|
|
63
|
+
|
|
64
|
+
# Optional request-body keys, in the order they are written to the body.
|
|
65
|
+
# Each is omitted from the body when nil.
|
|
66
|
+
OPTIONAL_BODY_KEYS = %i[
|
|
67
|
+
language_code
|
|
68
|
+
apply_text_normalization
|
|
69
|
+
seed
|
|
70
|
+
previous_text
|
|
71
|
+
next_text
|
|
72
|
+
previous_request_ids
|
|
73
|
+
next_request_ids
|
|
74
|
+
pronunciation_dictionary_locators
|
|
75
|
+
use_pvc_as_ivc
|
|
27
76
|
].freeze
|
|
28
77
|
|
|
29
78
|
# Generate audio from text
|
|
@@ -31,30 +80,39 @@ module ElevenRb
|
|
|
31
80
|
# @param text [String] the text to convert
|
|
32
81
|
# @param voice_id [String] the voice ID to use
|
|
33
82
|
# @param model_id [String] the model to use (default: eleven_multilingual_v2)
|
|
34
|
-
# @param voice_settings [Hash] voice settings overrides
|
|
83
|
+
# @param voice_settings [Hash] voice settings overrides (keys the model ignores are dropped)
|
|
35
84
|
# @param output_format [String] audio output format
|
|
36
|
-
# @
|
|
37
|
-
|
|
38
|
-
|
|
85
|
+
# @param language_code [String, nil] ISO 639-1 language code to enforce
|
|
86
|
+
# @param apply_text_normalization [String, nil] "auto", "on" or "off"
|
|
87
|
+
# @param seed [Integer, nil] seed for reproducible generation
|
|
88
|
+
# @param previous_text [String, nil] text that comes before this request (continuity)
|
|
89
|
+
# @param next_text [String, nil] text that comes after this request (continuity)
|
|
90
|
+
# @param previous_request_ids [Array<String>, nil] request IDs of preceding generations
|
|
91
|
+
# @param next_request_ids [Array<String>, nil] request IDs of following generations
|
|
92
|
+
# @param pronunciation_dictionary_locators [Array<Hash>, nil] `{ pronunciation_dictionary_id:, version_id: }`
|
|
93
|
+
# @param use_pvc_as_ivc [Boolean, nil] use the IVC version of a professional voice
|
|
94
|
+
# @return [Objects::Audio] with request_id, character_cost and dropped_settings
|
|
95
|
+
def generate(text, voice_id:, model_id: DEFAULT_MODEL, voice_settings: {}, output_format: 'mp3_44100_128',
|
|
96
|
+
language_code: nil, apply_text_normalization: nil, seed: nil, previous_text: nil,
|
|
97
|
+
next_text: nil, previous_request_ids: nil, next_request_ids: nil,
|
|
98
|
+
pronunciation_dictionary_locators: nil, use_pvc_as_ivc: nil)
|
|
99
|
+
validate_text!(text, model_id)
|
|
39
100
|
validate_presence!(voice_id, 'voice_id')
|
|
40
|
-
|
|
41
|
-
settings = Objects::VoiceSettings::DEFAULTS.merge(voice_settings)
|
|
42
|
-
|
|
43
|
-
body = {
|
|
44
|
-
text: text,
|
|
45
|
-
model_id: model_id,
|
|
46
|
-
voice_settings: settings
|
|
47
|
-
}
|
|
101
|
+
body, dropped = build_body(text, model_id, voice_settings, optional_values(binding))
|
|
48
102
|
|
|
49
103
|
path = "/text-to-speech/#{voice_id}?output_format=#{output_format}"
|
|
50
|
-
response =
|
|
104
|
+
response = post_binary_with_meta(path, body)
|
|
105
|
+
headers = response[:headers]
|
|
51
106
|
|
|
52
107
|
audio = Objects::Audio.new(
|
|
53
|
-
data: response,
|
|
108
|
+
data: response[:body],
|
|
54
109
|
format: output_format,
|
|
55
110
|
voice_id: voice_id,
|
|
56
111
|
text: text,
|
|
57
|
-
model_id: model_id
|
|
112
|
+
model_id: model_id,
|
|
113
|
+
request_id: headers['request-id'],
|
|
114
|
+
character_cost: integer_header(headers, 'character-cost'),
|
|
115
|
+
dropped_settings: dropped
|
|
58
116
|
)
|
|
59
117
|
|
|
60
118
|
# Trigger cost tracking callback
|
|
@@ -64,7 +122,8 @@ module ElevenRb
|
|
|
64
122
|
audio: audio,
|
|
65
123
|
voice_id: voice_id,
|
|
66
124
|
text: text,
|
|
67
|
-
cost_info: cost_info.to_h
|
|
125
|
+
cost_info: cost_info.to_h,
|
|
126
|
+
request_id: audio.request_id
|
|
68
127
|
)
|
|
69
128
|
|
|
70
129
|
audio
|
|
@@ -72,6 +131,8 @@ module ElevenRb
|
|
|
72
131
|
|
|
73
132
|
# Stream audio from text
|
|
74
133
|
#
|
|
134
|
+
# Takes the same optional keywords as {#generate}.
|
|
135
|
+
#
|
|
75
136
|
# @param text [String] the text to convert
|
|
76
137
|
# @param voice_id [String] the voice ID to use
|
|
77
138
|
# @param model_id [String] the model to use
|
|
@@ -79,18 +140,15 @@ module ElevenRb
|
|
|
79
140
|
# @param output_format [String] audio output format
|
|
80
141
|
# @yield [String] each chunk of audio data
|
|
81
142
|
# @return [void]
|
|
82
|
-
def stream(text, voice_id:, model_id: DEFAULT_MODEL, voice_settings: {}, output_format: 'mp3_44100_128',
|
|
83
|
-
|
|
143
|
+
def stream(text, voice_id:, model_id: DEFAULT_MODEL, voice_settings: {}, output_format: 'mp3_44100_128',
|
|
144
|
+
language_code: nil, apply_text_normalization: nil, seed: nil, previous_text: nil,
|
|
145
|
+
next_text: nil, previous_request_ids: nil, next_request_ids: nil,
|
|
146
|
+
pronunciation_dictionary_locators: nil, use_pvc_as_ivc: nil, &block)
|
|
147
|
+
validate_text!(text, model_id)
|
|
84
148
|
validate_presence!(voice_id, 'voice_id')
|
|
85
149
|
raise ArgumentError, 'Block required for streaming' unless block_given?
|
|
86
150
|
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
body = {
|
|
90
|
-
text: text,
|
|
91
|
-
model_id: model_id,
|
|
92
|
-
voice_settings: settings
|
|
93
|
-
}
|
|
151
|
+
body, = build_body(text, model_id, voice_settings, optional_values(binding))
|
|
94
152
|
|
|
95
153
|
path = "/text-to-speech/#{voice_id}/stream?output_format=#{output_format}"
|
|
96
154
|
post_stream(path, body, &block)
|
|
@@ -102,33 +160,35 @@ module ElevenRb
|
|
|
102
160
|
audio: nil, # No audio object for streaming
|
|
103
161
|
voice_id: voice_id,
|
|
104
162
|
text: text,
|
|
105
|
-
cost_info: cost_info.to_h
|
|
163
|
+
cost_info: cost_info.to_h,
|
|
164
|
+
request_id: nil
|
|
106
165
|
)
|
|
107
166
|
end
|
|
108
167
|
|
|
109
168
|
# Generate audio with timestamps
|
|
110
169
|
#
|
|
170
|
+
# Takes the same optional keywords as {#generate}.
|
|
171
|
+
#
|
|
111
172
|
# @param text [String] the text to convert
|
|
112
173
|
# @param voice_id [String] the voice ID to use
|
|
113
174
|
# @param model_id [String] the model to use
|
|
114
175
|
# @param voice_settings [Hash] voice settings overrides
|
|
115
176
|
# @param output_format [String] audio output format
|
|
116
|
-
# @return [Hash]
|
|
177
|
+
# @return [Hash] `{ audio:, alignment:, normalized_alignment:, request_id:, character_cost: }`
|
|
117
178
|
def generate_with_timestamps(text, voice_id:, model_id: DEFAULT_MODEL, voice_settings: {},
|
|
118
|
-
output_format: 'mp3_44100_128'
|
|
119
|
-
|
|
179
|
+
output_format: 'mp3_44100_128', language_code: nil,
|
|
180
|
+
apply_text_normalization: nil, seed: nil, previous_text: nil,
|
|
181
|
+
next_text: nil, previous_request_ids: nil, next_request_ids: nil,
|
|
182
|
+
pronunciation_dictionary_locators: nil, use_pvc_as_ivc: nil)
|
|
183
|
+
validate_text!(text, model_id)
|
|
120
184
|
validate_presence!(voice_id, 'voice_id')
|
|
121
|
-
|
|
122
|
-
settings = Objects::VoiceSettings::DEFAULTS.merge(voice_settings)
|
|
123
|
-
|
|
124
|
-
body = {
|
|
125
|
-
text: text,
|
|
126
|
-
model_id: model_id,
|
|
127
|
-
voice_settings: settings
|
|
128
|
-
}
|
|
185
|
+
body, dropped = build_body(text, model_id, voice_settings, optional_values(binding))
|
|
129
186
|
|
|
130
187
|
path = "/text-to-speech/#{voice_id}/with-timestamps?output_format=#{output_format}"
|
|
131
|
-
|
|
188
|
+
result = post_with_meta(path, body)
|
|
189
|
+
response = result[:body]
|
|
190
|
+
request_id = result[:headers]['request-id']
|
|
191
|
+
character_cost = integer_header(result[:headers], 'character-cost')
|
|
132
192
|
|
|
133
193
|
# Decode base64 audio
|
|
134
194
|
audio_data = Base64.decode64(response['audio_base64']) if response['audio_base64']
|
|
@@ -139,25 +199,72 @@ module ElevenRb
|
|
|
139
199
|
format: output_format,
|
|
140
200
|
voice_id: voice_id,
|
|
141
201
|
text: text,
|
|
142
|
-
model_id: model_id
|
|
202
|
+
model_id: model_id,
|
|
203
|
+
request_id: request_id,
|
|
204
|
+
character_cost: character_cost,
|
|
205
|
+
dropped_settings: dropped
|
|
143
206
|
)
|
|
144
207
|
end
|
|
145
208
|
|
|
146
209
|
{
|
|
147
210
|
audio: audio,
|
|
148
|
-
alignment: response['alignment']
|
|
211
|
+
alignment: response['alignment'],
|
|
212
|
+
normalized_alignment: response['normalized_alignment'],
|
|
213
|
+
request_id: request_id,
|
|
214
|
+
character_cost: character_cost
|
|
149
215
|
}
|
|
150
216
|
end
|
|
151
217
|
|
|
152
218
|
private
|
|
153
219
|
|
|
154
|
-
|
|
220
|
+
# Shared request body for generate / stream / generate_with_timestamps
|
|
221
|
+
#
|
|
222
|
+
# @return [Array(Hash, Array<Symbol>)] the body and the dropped voice-setting keys
|
|
223
|
+
def build_body(text, model_id, voice_settings, options)
|
|
224
|
+
settings, dropped = resolve_voice_settings(model_id, voice_settings)
|
|
225
|
+
|
|
226
|
+
body = { text: text, model_id: model_id, voice_settings: settings }
|
|
227
|
+
OPTIONAL_BODY_KEYS.each do |key|
|
|
228
|
+
body[key] = options[key] unless options[key].nil?
|
|
229
|
+
end
|
|
230
|
+
|
|
231
|
+
[body, dropped]
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
# The optional keyword values of the calling method, keyed by OPTIONAL_BODY_KEYS
|
|
235
|
+
def optional_values(caller_binding)
|
|
236
|
+
OPTIONAL_BODY_KEYS.to_h { |key| [key, caller_binding.local_variable_get(key)] }
|
|
237
|
+
end
|
|
238
|
+
|
|
239
|
+
def resolve_voice_settings(model_id, voice_settings)
|
|
240
|
+
settings, dropped = Objects::VoiceSettings.for_model(model_id, voice_settings)
|
|
241
|
+
return [settings, dropped] if dropped.empty?
|
|
242
|
+
|
|
243
|
+
message = "voice settings #{dropped.join(', ')} are not supported by #{model_id} " \
|
|
244
|
+
"(supported: #{ModelCapabilities.supported_voice_settings(model_id).join(', ')})"
|
|
245
|
+
raise Errors::ValidationError, message if http_client.config.strict_voice_settings
|
|
246
|
+
|
|
247
|
+
http_client.config.logger&.warn("[ElevenRb] #{message}; dropped from the request")
|
|
248
|
+
[settings, dropped]
|
|
249
|
+
end
|
|
250
|
+
|
|
251
|
+
def integer_header(headers, name)
|
|
252
|
+
value = headers[name]
|
|
253
|
+
return nil if value.nil? || value.to_s.strip.empty?
|
|
254
|
+
|
|
255
|
+
Integer(value.to_s.strip, 10)
|
|
256
|
+
rescue ArgumentError
|
|
257
|
+
nil
|
|
258
|
+
end
|
|
259
|
+
|
|
260
|
+
def validate_text!(text, model_id = DEFAULT_MODEL)
|
|
155
261
|
validate_presence!(text, 'text')
|
|
156
262
|
|
|
157
|
-
|
|
263
|
+
max_length = ModelCapabilities.max_text_length(model_id)
|
|
264
|
+
return unless text.length > max_length
|
|
158
265
|
|
|
159
266
|
raise Errors::ValidationError,
|
|
160
|
-
"text exceeds maximum length of #{
|
|
267
|
+
"text exceeds maximum length of #{max_length} characters for #{model_id} (got #{text.length})"
|
|
161
268
|
end
|
|
162
269
|
end
|
|
163
270
|
end
|
data/lib/eleven_rb/version.rb
CHANGED
data/lib/eleven_rb.rb
CHANGED
|
@@ -60,6 +60,7 @@ module ElevenRb
|
|
|
60
60
|
retry_delay: config.retry_delay,
|
|
61
61
|
retry_statuses: config.retry_statuses,
|
|
62
62
|
logger: config.logger,
|
|
63
|
+
strict_voice_settings: config.strict_voice_settings,
|
|
63
64
|
on_request: config.on_request,
|
|
64
65
|
on_response: config.on_response,
|
|
65
66
|
on_error: config.on_error,
|
|
@@ -79,6 +80,7 @@ require_relative 'eleven_rb/errors'
|
|
|
79
80
|
require_relative 'eleven_rb/callbacks'
|
|
80
81
|
require_relative 'eleven_rb/instrumentation'
|
|
81
82
|
require_relative 'eleven_rb/configuration'
|
|
83
|
+
require_relative 'eleven_rb/model_capabilities'
|
|
82
84
|
|
|
83
85
|
# HTTP layer
|
|
84
86
|
require_relative 'eleven_rb/http/client'
|
|
@@ -109,6 +111,7 @@ require_relative 'eleven_rb/resources/user'
|
|
|
109
111
|
require_relative 'eleven_rb/resources/sound_effects'
|
|
110
112
|
require_relative 'eleven_rb/resources/music'
|
|
111
113
|
require_relative 'eleven_rb/resources/speech_to_speech'
|
|
114
|
+
require_relative 'eleven_rb/resources/text_to_dialogue'
|
|
112
115
|
|
|
113
116
|
# High-level components
|
|
114
117
|
require_relative 'eleven_rb/voice_slot_manager'
|