rime-api 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rime_api/SOURCE.json +9 -0
- rime_api/__init__.py +1 -0
- rime_api/py.typed +0 -0
- rime_api/schema/rime/text_to_speech.proto +303 -0
- rime_api/schema/text_to_speech.asyncapi.yaml +461 -0
- rime_api/text_to_speech_pb2.py +84 -0
- rime_api/text_to_speech_pb2.pyi +233 -0
- rime_api-0.0.1.dist-info/METADATA +87 -0
- rime_api-0.0.1.dist-info/RECORD +12 -0
- rime_api-0.0.1.dist-info/WHEEL +4 -0
- rime_api-0.0.1.dist-info/licenses/LICENSE +202 -0
- rime_api-0.0.1.dist-info/licenses/NOTICE +2 -0
rime_api/SOURCE.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": "0.0.1",
|
|
3
|
+
"schema": "rime/text_to_speech.proto",
|
|
4
|
+
"sha256": "b33b23caed621c19f848693d26301e43639a3a2fd63c8133a37f2f3673f7a412",
|
|
5
|
+
"asyncapi": {
|
|
6
|
+
"schema": "text_to_speech.asyncapi.yaml",
|
|
7
|
+
"sha256": "0ba7a67cf88127bcf21bfbffab4f7c752e823c3a80fc7e354d1cdc7a0fbcb4ee"
|
|
8
|
+
}
|
|
9
|
+
}
|
rime_api/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Generated Rime protocol definitions."""
|
rime_api/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,303 @@
|
|
|
1
|
+
syntax = "proto3";
|
|
2
|
+
|
|
3
|
+
package rime;
|
|
4
|
+
|
|
5
|
+
service TextToSpeech {
|
|
6
|
+
rpc Synthesize(SynthesisRequest) returns (stream SynthesisResponseStream) {}
|
|
7
|
+
// Streaming text input: the client feeds text incrementally into a single
|
|
8
|
+
// continuous synthesis. The FIRST client message MUST set `header` (all
|
|
9
|
+
// synthesis params; its `text` field may carry the first chunk or be empty);
|
|
10
|
+
// subsequent messages set `text_chunk`. End-of-input is signalled by the
|
|
11
|
+
// client half-closing the request stream. Audio frames stream back as they
|
|
12
|
+
// are produced.
|
|
13
|
+
rpc SynthesizeStreaming(stream StreamingSynthesisRequest) returns (stream SynthesisResponseStream) {}
|
|
14
|
+
rpc NormalizeText(NormalizeTextRequest) returns (NormalizeTextResponse) {}
|
|
15
|
+
rpc GetSupportedLanguages(GetSupportedLanguagesRequest) returns (GetSupportedLanguagesResponse) {}
|
|
16
|
+
rpc GetSupportedSpeakers(GetSupportedSpeakersRequest) returns (GetSupportedSpeakersResponse) {}
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
message StreamingSynthesisRequest {
|
|
20
|
+
reserved 3;
|
|
21
|
+
reserved "flush";
|
|
22
|
+
|
|
23
|
+
oneof payload {
|
|
24
|
+
// FIRST message only: all synthesis parameters. Its `text` field may carry
|
|
25
|
+
// the first complete sentence or be left empty.
|
|
26
|
+
SynthesisRequest header = 1;
|
|
27
|
+
|
|
28
|
+
// Subsequent messages: one complete sentence to append to the live
|
|
29
|
+
// synthesis.
|
|
30
|
+
string text_chunk = 2;
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
message SynthesisRequest {
|
|
35
|
+
reserved 1;
|
|
36
|
+
|
|
37
|
+
// BCP-47 language tag, e.g. "en".
|
|
38
|
+
// Used to route the frontend pipeline and set backend language.
|
|
39
|
+
optional string language = 2;
|
|
40
|
+
|
|
41
|
+
// Speaker name, e.g. "luna".
|
|
42
|
+
optional string speaker = 3;
|
|
43
|
+
|
|
44
|
+
// Text to synthesize.
|
|
45
|
+
string text = 4;
|
|
46
|
+
|
|
47
|
+
optional AudioParameters audio_parameters = 5;
|
|
48
|
+
|
|
49
|
+
// Controls how input text is split before synthesis.
|
|
50
|
+
// Defaults to SPLIT_STRATEGY_SENTENCE when unset.
|
|
51
|
+
optional SplitStrategy split_strategy = 6;
|
|
52
|
+
|
|
53
|
+
optional ArcanaParameters arcana_parameters = 7;
|
|
54
|
+
optional CodaParameters coda_parameters = 8;
|
|
55
|
+
optional MistParameters mist_parameters = 9;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
message SynthesisResponseStream {
|
|
59
|
+
bytes audio = 1;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
message AudioParameters {
|
|
63
|
+
// Equivalent to the HTTP `Accept` header.
|
|
64
|
+
// Examples: "audio/wav", "audio/mpeg", "audio/ogg;codecs=opus", "audio/l16".
|
|
65
|
+
optional string audio_format = 1;
|
|
66
|
+
|
|
67
|
+
optional int32 sampling_rate = 2;
|
|
68
|
+
optional float time_scale_factor = 3;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
message ArcanaParameters {
|
|
72
|
+
optional float repetition_penalty = 1;
|
|
73
|
+
optional float top_p = 2;
|
|
74
|
+
optional float temperature = 3;
|
|
75
|
+
optional int32 max_tokens = 4;
|
|
76
|
+
optional float presence_penalty = 5;
|
|
77
|
+
optional float frequency_penalty = 6;
|
|
78
|
+
optional int32 top_k = 7;
|
|
79
|
+
optional float min_p = 8;
|
|
80
|
+
optional int64 seed = 9;
|
|
81
|
+
optional int32 min_tokens = 13;
|
|
82
|
+
optional bool detokenize = 14;
|
|
83
|
+
optional bool skip_special_tokens = 15;
|
|
84
|
+
optional bool spaces_between_special_tokens = 16;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
message CodaParameters {
|
|
88
|
+
reserved 1, 2, 3, 4, 5, 6, 7;
|
|
89
|
+
reserved "temperature", "top_k", "top_p", "repetition_penalty", "max_tokens", "min_tokens", "seed";
|
|
90
|
+
|
|
91
|
+
// Streaming text input (`SynthesizeStreaming` / WebSocket `text`) only:
|
|
92
|
+
// number of leading text tokens prefilled as a lookahead block so the first
|
|
93
|
+
// audio frames have content context. Higher values improve content
|
|
94
|
+
// robustness at the start of the utterance but delay the first audio until
|
|
95
|
+
// that many text tokens have arrived. Clamped to the tokens available when
|
|
96
|
+
// generation starts. Defaults to 0, which reproduces upstream Qwen3-TTS's
|
|
97
|
+
// exact streaming layout. Ignored for one-shot synthesis (which always sees
|
|
98
|
+
// the complete text up front).
|
|
99
|
+
optional int32 text_lookahead_tokens = 8;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
message MistParameters {
|
|
103
|
+
optional bool pause_between_brackets = 1;
|
|
104
|
+
optional bool phonemize_between_brackets = 2;
|
|
105
|
+
repeated float inline_time_scale_factors = 3;
|
|
106
|
+
optional bool save_oovs = 4;
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
enum SplitStrategy {
|
|
110
|
+
SPLIT_STRATEGY_UNSPECIFIED = 0;
|
|
111
|
+
SPLIT_STRATEGY_SENTENCE = 1;
|
|
112
|
+
SPLIT_STRATEGY_NONE = 2;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
message NormalizeTextRequest {
|
|
116
|
+
// Text to normalize.
|
|
117
|
+
string text = 1;
|
|
118
|
+
|
|
119
|
+
// BCP-47 language tag, e.g. "en".
|
|
120
|
+
optional string language = 2;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
message NormalizeTextResponse {
|
|
124
|
+
// Deprecated: use normalized_sentences. Sentences are concatenated with no
|
|
125
|
+
// whitespace separator, so multi-sentence inputs come back glued together.
|
|
126
|
+
string normalized_text = 1;
|
|
127
|
+
|
|
128
|
+
// The normalized text, one entry per sentence emitted by the frontend
|
|
129
|
+
// pipeline. Each entry is already trimmed; join with a single space to
|
|
130
|
+
// reconstruct display text.
|
|
131
|
+
repeated string normalized_sentences = 2;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
message GetSupportedLanguagesRequest {
|
|
135
|
+
// Empty for now, but used instead of google.protobuf.Empty for extensibility.
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
message GetSupportedLanguagesResponse {
|
|
139
|
+
// List of supported BCP-47 language tags.
|
|
140
|
+
repeated string languages = 1;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
message GetSupportedSpeakersRequest {
|
|
144
|
+
optional string language = 1;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
message GetSupportedSpeakersResponse {
|
|
148
|
+
// List of supported speaker names for the given language.
|
|
149
|
+
repeated string speakers = 1;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
// ======================= WebSocket envelope =======================
|
|
153
|
+
//
|
|
154
|
+
// The engine serves a WebSocket synthesis interface at `GET /ws` on the HTTP
|
|
155
|
+
// port. Every WebSocket message carries exactly one envelope message:
|
|
156
|
+
// - text frames: canonical proto3 JSON encoding of the envelope
|
|
157
|
+
// - binary frames: protobuf binary encoding of the envelope
|
|
158
|
+
// Client frames are decoded according to their frame type. The response
|
|
159
|
+
// encoding is fixed at upgrade time by the WebSocket subprotocol: the first
|
|
160
|
+
// recognized entry in the client's offer list wins — `rime.v1.binary`
|
|
161
|
+
// selects binary frames, `rime.v1.json` selects JSON text frames — and the
|
|
162
|
+
// server echoes the selected subprotocol. Offering no subprotocol selects
|
|
163
|
+
// JSON. Offering only unrecognized values also selects JSON but echoes no
|
|
164
|
+
// subprotocol, which spec-compliant clients (RFC 6455 section 4.1; all
|
|
165
|
+
// browsers) treat as a failed handshake — offer one of the values above, or
|
|
166
|
+
// none.
|
|
167
|
+
//
|
|
168
|
+
// A *context* is one continuous synthesis. It is either one-shot (opened by a
|
|
169
|
+
// `start` whose `text` is the whole input) or streaming (opened by a `start`
|
|
170
|
+
// with empty `text`, then fed by `text` chunks and closed by `end`); either
|
|
171
|
+
// way the audio is a single continuous stream. Per context the server emits
|
|
172
|
+
// `started`, then zero or more `audio` payloads, then exactly one terminal
|
|
173
|
+
// event: `done`, `cancelled`, or `error`. Multiple contexts may be open
|
|
174
|
+
// concurrently on one connection; events are correlated by `context_id`.
|
|
175
|
+
//
|
|
176
|
+
// Authentication: preferred is an `Authorization` header on the upgrade
|
|
177
|
+
// request (rejected with HTTP 401 before the socket opens). Browsers cannot
|
|
178
|
+
// set WebSocket headers; they send `config.authorization` (or
|
|
179
|
+
// `config.license`) as the first message instead. Debug builds skip auth.
|
|
180
|
+
|
|
181
|
+
// Client -> server. Exactly one message per WebSocket frame.
|
|
182
|
+
message WebSocketRequest {
|
|
183
|
+
reserved 7;
|
|
184
|
+
reserved "flush";
|
|
185
|
+
|
|
186
|
+
// Client-chosen context identifier, at most 128 bytes. Empty selects the
|
|
187
|
+
// context named "default". An identifier may be reused after its context
|
|
188
|
+
// reaches a terminal event.
|
|
189
|
+
string context_id = 1;
|
|
190
|
+
|
|
191
|
+
oneof payload {
|
|
192
|
+
// Connection-level configuration. At most one, and only before the first
|
|
193
|
+
// `start`. Not required in debug mode or when the upgrade request
|
|
194
|
+
// carried an `Authorization` header.
|
|
195
|
+
WebSocketConfig config = 2;
|
|
196
|
+
|
|
197
|
+
// Opens the context with the full synthesis parameters. A non-empty
|
|
198
|
+
// `text` is the complete input for one-shot synthesis: audio streams
|
|
199
|
+
// and ends with `done`, no `end` required. An empty `text` opens a
|
|
200
|
+
// streaming context that stays open for `text` messages and finalizes
|
|
201
|
+
// on `end`.
|
|
202
|
+
SynthesisRequest start = 3;
|
|
203
|
+
|
|
204
|
+
// Appends a chunk of input text to an open streaming context (one opened
|
|
205
|
+
// with an empty `start.text`). The engine meters, segments, and
|
|
206
|
+
// synthesizes it into the same continuous audio as it arrives. Rejected
|
|
207
|
+
// with kind "invalid_input" if the context is not open for input (no
|
|
208
|
+
// `start`, or already `end`ed / one-shot).
|
|
209
|
+
string text = 4;
|
|
210
|
+
|
|
211
|
+
// Declares end of input for a streaming context. Remaining audio is
|
|
212
|
+
// delivered, followed by `done`. A no-op for a one-shot context. The
|
|
213
|
+
// connection stays open for further contexts.
|
|
214
|
+
WebSocketEnd end = 5;
|
|
215
|
+
|
|
216
|
+
// Aborts the context immediately, discarding in-flight synthesis.
|
|
217
|
+
// Acknowledged with `cancelled`; idempotent (cancelling an unknown or
|
|
218
|
+
// finished context is not an error).
|
|
219
|
+
WebSocketCancel cancel = 6;
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
message WebSocketConfig {
|
|
224
|
+
// Mirrors the HTTP `Authorization` header, e.g. "Api-Key <key>".
|
|
225
|
+
// Rejected if the upgrade request already carried an Authorization header.
|
|
226
|
+
optional string authorization = 1;
|
|
227
|
+
|
|
228
|
+
// License JSON for on-prem deployments; mirrors the `x-rime-license`
|
|
229
|
+
// gRPC metadata / the `license` field of the HTTP API.
|
|
230
|
+
optional string license = 2;
|
|
231
|
+
|
|
232
|
+
// Defaults merged (field-wise, shallow) under every subsequent `start`.
|
|
233
|
+
// Its `text` field is ignored.
|
|
234
|
+
optional SynthesisRequest defaults = 3;
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
message WebSocketEnd {}
|
|
238
|
+
|
|
239
|
+
message WebSocketCancel {}
|
|
240
|
+
|
|
241
|
+
// Server -> client. Exactly one message per WebSocket frame.
|
|
242
|
+
message WebSocketResponse {
|
|
243
|
+
// Context the payload applies to. Empty for connection-scoped payloads
|
|
244
|
+
// (`ready` and connection-level `error`s).
|
|
245
|
+
string context_id = 1;
|
|
246
|
+
|
|
247
|
+
oneof payload {
|
|
248
|
+
// Sent once per connection when the engine is ready to synthesize
|
|
249
|
+
// (pipelines loaded and serving). Gate `start` on this event.
|
|
250
|
+
WebSocketReady ready = 2;
|
|
251
|
+
|
|
252
|
+
// The context was accepted; precedes any of its audio.
|
|
253
|
+
WebSocketStarted started = 3;
|
|
254
|
+
|
|
255
|
+
// A chunk of encoded audio, in the format requested at `start`.
|
|
256
|
+
bytes audio = 4;
|
|
257
|
+
|
|
258
|
+
// Terminal: all audio for the context has been delivered.
|
|
259
|
+
WebSocketDone done = 5;
|
|
260
|
+
|
|
261
|
+
// Terminal: the context was aborted by `cancel`.
|
|
262
|
+
WebSocketCancelled cancelled = 6;
|
|
263
|
+
|
|
264
|
+
// Terminal for the context it names. Errors with an empty `context_id`
|
|
265
|
+
// are connection-scoped and leave open contexts running; the server
|
|
266
|
+
// closes the connection only when the failure is unrecoverable
|
|
267
|
+
// (undecodable frames, failed authentication).
|
|
268
|
+
WebSocketError error = 7;
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
message WebSocketReady {
|
|
273
|
+
// Protocol version. Currently 1.
|
|
274
|
+
uint32 protocol = 1;
|
|
275
|
+
|
|
276
|
+
// Supported BCP-47 language tags (mirrors GetSupportedLanguages).
|
|
277
|
+
repeated string languages = 2;
|
|
278
|
+
|
|
279
|
+
optional string default_language = 3;
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
message WebSocketStarted {
|
|
283
|
+
// Engine request id for this context; correlates with engine logs and
|
|
284
|
+
// matches the `x-request-id` gRPC/HTTP metadata.
|
|
285
|
+
string request_id = 1;
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
message WebSocketDone {}
|
|
289
|
+
|
|
290
|
+
message WebSocketCancelled {}
|
|
291
|
+
|
|
292
|
+
message WebSocketError {
|
|
293
|
+
// Stable machine-readable kind; same values as the HTTP error body
|
|
294
|
+
// (e.g. "invalid_input", "unauthenticated", "unavailable", "internal",
|
|
295
|
+
// "unimplemented").
|
|
296
|
+
string kind = 1;
|
|
297
|
+
|
|
298
|
+
// Human-readable description.
|
|
299
|
+
string message = 2;
|
|
300
|
+
|
|
301
|
+
// Engine request id, when the error occurred inside a synthesis request.
|
|
302
|
+
optional string request_id = 3;
|
|
303
|
+
}
|