rime-api 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
rime_api/SOURCE.json ADDED
@@ -0,0 +1,9 @@
1
+ {
2
+ "version": "0.0.1",
3
+ "schema": "rime/text_to_speech.proto",
4
+ "sha256": "b33b23caed621c19f848693d26301e43639a3a2fd63c8133a37f2f3673f7a412",
5
+ "asyncapi": {
6
+ "schema": "text_to_speech.asyncapi.yaml",
7
+ "sha256": "0ba7a67cf88127bcf21bfbffab4f7c752e823c3a80fc7e354d1cdc7a0fbcb4ee"
8
+ }
9
+ }
rime_api/__init__.py ADDED
@@ -0,0 +1 @@
1
+ """Generated Rime protocol definitions."""
rime_api/py.typed ADDED
File without changes
@@ -0,0 +1,303 @@
1
+ syntax = "proto3";
2
+
3
+ package rime;
4
+
5
+ service TextToSpeech {
6
+ rpc Synthesize(SynthesisRequest) returns (stream SynthesisResponseStream) {}
7
+ // Streaming text input: the client feeds text incrementally into a single
8
+ // continuous synthesis. The FIRST client message MUST set `header` (all
9
+ // synthesis params; its `text` field may carry the first chunk or be empty);
10
+ // subsequent messages set `text_chunk`. End-of-input is signalled by the
11
+ // client half-closing the request stream. Audio frames stream back as they
12
+ // are produced.
13
+ rpc SynthesizeStreaming(stream StreamingSynthesisRequest) returns (stream SynthesisResponseStream) {}
14
+ rpc NormalizeText(NormalizeTextRequest) returns (NormalizeTextResponse) {}
15
+ rpc GetSupportedLanguages(GetSupportedLanguagesRequest) returns (GetSupportedLanguagesResponse) {}
16
+ rpc GetSupportedSpeakers(GetSupportedSpeakersRequest) returns (GetSupportedSpeakersResponse) {}
17
+ }
18
+
19
+ message StreamingSynthesisRequest {
20
+ reserved 3;
21
+ reserved "flush";
22
+
23
+ oneof payload {
24
+ // FIRST message only: all synthesis parameters. Its `text` field may carry
25
+ // the first complete sentence or be left empty.
26
+ SynthesisRequest header = 1;
27
+
28
+ // Subsequent messages: one complete sentence to append to the live
29
+ // synthesis.
30
+ string text_chunk = 2;
31
+ }
32
+ }
33
+
34
+ message SynthesisRequest {
35
+ reserved 1;
36
+
37
+ // BCP-47 language tag, e.g. "en".
38
+ // Used to route the frontend pipeline and set backend language.
39
+ optional string language = 2;
40
+
41
+ // Speaker name, e.g. "luna".
42
+ optional string speaker = 3;
43
+
44
+ // Text to synthesize.
45
+ string text = 4;
46
+
47
+ optional AudioParameters audio_parameters = 5;
48
+
49
+ // Controls how input text is split before synthesis.
50
+ // Defaults to SPLIT_STRATEGY_SENTENCE when unset.
51
+ optional SplitStrategy split_strategy = 6;
52
+
53
+ optional ArcanaParameters arcana_parameters = 7;
54
+ optional CodaParameters coda_parameters = 8;
55
+ optional MistParameters mist_parameters = 9;
56
+ }
57
+
58
+ message SynthesisResponseStream {
59
+ bytes audio = 1;
60
+ }
61
+
62
+ message AudioParameters {
63
+ // Equivalent to the HTTP `Accept` header.
64
+ // Examples: "audio/wav", "audio/mpeg", "audio/ogg;codecs=opus", "audio/l16".
65
+ optional string audio_format = 1;
66
+
67
+ optional int32 sampling_rate = 2;
68
+ optional float time_scale_factor = 3;
69
+ }
70
+
71
+ message ArcanaParameters {
72
+ optional float repetition_penalty = 1;
73
+ optional float top_p = 2;
74
+ optional float temperature = 3;
75
+ optional int32 max_tokens = 4;
76
+ optional float presence_penalty = 5;
77
+ optional float frequency_penalty = 6;
78
+ optional int32 top_k = 7;
79
+ optional float min_p = 8;
80
+ optional int64 seed = 9;
81
+ optional int32 min_tokens = 13;
82
+ optional bool detokenize = 14;
83
+ optional bool skip_special_tokens = 15;
84
+ optional bool spaces_between_special_tokens = 16;
85
+ }
86
+
87
+ message CodaParameters {
88
+ reserved 1, 2, 3, 4, 5, 6, 7;
89
+ reserved "temperature", "top_k", "top_p", "repetition_penalty", "max_tokens", "min_tokens", "seed";
90
+
91
+ // Streaming text input (`SynthesizeStreaming` / WebSocket `text`) only:
92
+ // number of leading text tokens prefilled as a lookahead block so the first
93
+ // audio frames have content context. Higher values improve content
94
+ // robustness at the start of the utterance but delay the first audio until
95
+ // that many text tokens have arrived. Clamped to the tokens available when
96
+ // generation starts. Defaults to 0, which reproduces upstream Qwen3-TTS's
97
+ // exact streaming layout. Ignored for one-shot synthesis (which always sees
98
+ // the complete text up front).
99
+ optional int32 text_lookahead_tokens = 8;
100
+ }
101
+
102
+ message MistParameters {
103
+ optional bool pause_between_brackets = 1;
104
+ optional bool phonemize_between_brackets = 2;
105
+ repeated float inline_time_scale_factors = 3;
106
+ optional bool save_oovs = 4;
107
+ }
108
+
109
+ enum SplitStrategy {
110
+ SPLIT_STRATEGY_UNSPECIFIED = 0;
111
+ SPLIT_STRATEGY_SENTENCE = 1;
112
+ SPLIT_STRATEGY_NONE = 2;
113
+ }
114
+
115
+ message NormalizeTextRequest {
116
+ // Text to normalize.
117
+ string text = 1;
118
+
119
+ // BCP-47 language tag, e.g. "en".
120
+ optional string language = 2;
121
+ }
122
+
123
+ message NormalizeTextResponse {
124
+ // Deprecated: use normalized_sentences. Sentences are concatenated with no
125
+ // whitespace separator, so multi-sentence inputs come back glued together.
126
+ string normalized_text = 1;
127
+
128
+ // The normalized text, one entry per sentence emitted by the frontend
129
+ // pipeline. Each entry is already trimmed; join with a single space to
130
+ // reconstruct display text.
131
+ repeated string normalized_sentences = 2;
132
+ }
133
+
134
+ message GetSupportedLanguagesRequest {
135
+ // Empty for now, but used instead of google.protobuf.Empty for extensibility.
136
+ }
137
+
138
+ message GetSupportedLanguagesResponse {
139
+ // List of supported BCP-47 language tags.
140
+ repeated string languages = 1;
141
+ }
142
+
143
+ message GetSupportedSpeakersRequest {
144
+ optional string language = 1;
145
+ }
146
+
147
+ message GetSupportedSpeakersResponse {
148
+ // List of supported speaker names for the given language.
149
+ repeated string speakers = 1;
150
+ }
151
+
152
+ // ======================= WebSocket envelope =======================
153
+ //
154
+ // The engine serves a WebSocket synthesis interface at `GET /ws` on the HTTP
155
+ // port. Every WebSocket message carries exactly one envelope message:
156
+ // - text frames: canonical proto3 JSON encoding of the envelope
157
+ // - binary frames: protobuf binary encoding of the envelope
158
+ // Client frames are decoded according to their frame type. The response
159
+ // encoding is fixed at upgrade time by the WebSocket subprotocol: the first
160
+ // recognized entry in the client's offer list wins — `rime.v1.binary`
161
+ // selects binary frames, `rime.v1.json` selects JSON text frames — and the
162
+ // server echoes the selected subprotocol. Offering no subprotocol selects
163
+ // JSON. Offering only unrecognized values also selects JSON but echoes no
164
+ // subprotocol, which spec-compliant clients (RFC 6455 section 4.1; all
165
+ // browsers) treat as a failed handshake — offer one of the values above, or
166
+ // none.
167
+ //
168
+ // A *context* is one continuous synthesis. It is either one-shot (opened by a
169
+ // `start` whose `text` is the whole input) or streaming (opened by a `start`
170
+ // with empty `text`, then fed by `text` chunks and closed by `end`); either
171
+ // way the audio is a single continuous stream. Per context the server emits
172
+ // `started`, then zero or more `audio` payloads, then exactly one terminal
173
+ // event: `done`, `cancelled`, or `error`. Multiple contexts may be open
174
+ // concurrently on one connection; events are correlated by `context_id`.
175
+ //
176
+ // Authentication: preferred is an `Authorization` header on the upgrade
177
+ // request (rejected with HTTP 401 before the socket opens). Browsers cannot
178
+ // set WebSocket headers; they send `config.authorization` (or
179
+ // `config.license`) as the first message instead. Debug builds skip auth.
180
+
181
+ // Client -> server. Exactly one message per WebSocket frame.
182
+ message WebSocketRequest {
183
+ reserved 7;
184
+ reserved "flush";
185
+
186
+ // Client-chosen context identifier, at most 128 bytes. Empty selects the
187
+ // context named "default". An identifier may be reused after its context
188
+ // reaches a terminal event.
189
+ string context_id = 1;
190
+
191
+ oneof payload {
192
+ // Connection-level configuration. At most one, and only before the first
193
+ // `start`. Not required in debug mode or when the upgrade request
194
+ // carried an `Authorization` header.
195
+ WebSocketConfig config = 2;
196
+
197
+ // Opens the context with the full synthesis parameters. A non-empty
198
+ // `text` is the complete input for one-shot synthesis: audio streams
199
+ // and ends with `done`, no `end` required. An empty `text` opens a
200
+ // streaming context that stays open for `text` messages and finalizes
201
+ // on `end`.
202
+ SynthesisRequest start = 3;
203
+
204
+ // Appends a chunk of input text to an open streaming context (one opened
205
+ // with an empty `start.text`). The engine meters, segments, and
206
+ // synthesizes it into the same continuous audio as it arrives. Rejected
207
+ // with kind "invalid_input" if the context is not open for input (no
208
+ // `start`, or already `end`ed / one-shot).
209
+ string text = 4;
210
+
211
+ // Declares end of input for a streaming context. Remaining audio is
212
+ // delivered, followed by `done`. A no-op for a one-shot context. The
213
+ // connection stays open for further contexts.
214
+ WebSocketEnd end = 5;
215
+
216
+ // Aborts the context immediately, discarding in-flight synthesis.
217
+ // Acknowledged with `cancelled`; idempotent (cancelling an unknown or
218
+ // finished context is not an error).
219
+ WebSocketCancel cancel = 6;
220
+ }
221
+ }
222
+
223
+ message WebSocketConfig {
224
+ // Mirrors the HTTP `Authorization` header, e.g. "Api-Key <key>".
225
+ // Rejected if the upgrade request already carried an Authorization header.
226
+ optional string authorization = 1;
227
+
228
+ // License JSON for on-prem deployments; mirrors the `x-rime-license`
229
+ // gRPC metadata / the `license` field of the HTTP API.
230
+ optional string license = 2;
231
+
232
+ // Defaults merged (field-wise, shallow) under every subsequent `start`.
233
+ // Its `text` field is ignored.
234
+ optional SynthesisRequest defaults = 3;
235
+ }
236
+
237
+ message WebSocketEnd {}
238
+
239
+ message WebSocketCancel {}
240
+
241
+ // Server -> client. Exactly one message per WebSocket frame.
242
+ message WebSocketResponse {
243
+ // Context the payload applies to. Empty for connection-scoped payloads
244
+ // (`ready` and connection-level `error`s).
245
+ string context_id = 1;
246
+
247
+ oneof payload {
248
+ // Sent once per connection when the engine is ready to synthesize
249
+ // (pipelines loaded and serving). Gate `start` on this event.
250
+ WebSocketReady ready = 2;
251
+
252
+ // The context was accepted; precedes any of its audio.
253
+ WebSocketStarted started = 3;
254
+
255
+ // A chunk of encoded audio, in the format requested at `start`.
256
+ bytes audio = 4;
257
+
258
+ // Terminal: all audio for the context has been delivered.
259
+ WebSocketDone done = 5;
260
+
261
+ // Terminal: the context was aborted by `cancel`.
262
+ WebSocketCancelled cancelled = 6;
263
+
264
+ // Terminal for the context it names. Errors with an empty `context_id`
265
+ // are connection-scoped and leave open contexts running; the server
266
+ // closes the connection only when the failure is unrecoverable
267
+ // (undecodable frames, failed authentication).
268
+ WebSocketError error = 7;
269
+ }
270
+ }
271
+
272
+ message WebSocketReady {
273
+ // Protocol version. Currently 1.
274
+ uint32 protocol = 1;
275
+
276
+ // Supported BCP-47 language tags (mirrors GetSupportedLanguages).
277
+ repeated string languages = 2;
278
+
279
+ optional string default_language = 3;
280
+ }
281
+
282
+ message WebSocketStarted {
283
+ // Engine request id for this context; correlates with engine logs and
284
+ // matches the `x-request-id` gRPC/HTTP metadata.
285
+ string request_id = 1;
286
+ }
287
+
288
+ message WebSocketDone {}
289
+
290
+ message WebSocketCancelled {}
291
+
292
+ message WebSocketError {
293
+ // Stable machine-readable kind; same values as the HTTP error body
294
+ // (e.g. "invalid_input", "unauthenticated", "unavailable", "internal",
295
+ // "unimplemented").
296
+ string kind = 1;
297
+
298
+ // Human-readable description.
299
+ string message = 2;
300
+
301
+ // Engine request id, when the error occurred inside a synthesis request.
302
+ optional string request_id = 3;
303
+ }