@datalayer/agent-runtimes 1.3.62 → 1.3.64
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/chat/ChatFloating.d.ts +9 -1
- package/lib/chat/ChatFloating.js +69 -18
- package/lib/chat/assistant/AssistantStage.d.ts +14 -1
- package/lib/chat/assistant/AssistantStage.js +53 -1
- package/lib/chat/assistant/SpriteCharacter.js +1 -1
- package/lib/chat/assistant/characters.js +1 -5
- package/lib/chat/assistant/state.d.ts +7 -1
- package/lib/chat/assistant/state.js +3 -2
- package/lib/chat/base/ChatBase.js +32 -4
- package/lib/chat/messages/ChatMessageList.d.ts +0 -6
- package/lib/chat/messages/ChatMessageList.js +8 -2
- package/lib/config/AgentConfiguration.js +0 -6
- package/lib/examples/AgentA2ATeamExample.js +19 -2
- package/lib/examples/ChatAssistantExample.d.ts +3 -1
- package/lib/examples/ChatAssistantExample.js +75 -7
- package/lib/examples/ChatAssistantGalleryExample.d.ts +3 -2
- package/lib/examples/ChatAssistantGalleryExample.js +11 -6
- package/lib/examples/DecksAgent.js +4 -2
- package/lib/examples/LoopShellExample.js +4 -2
- package/lib/examples/VoiceChatExample.d.ts +20 -0
- package/lib/examples/VoiceChatExample.js +64 -0
- package/lib/examples/example-selector.js +1 -0
- package/lib/examples/main.js +12 -7
- package/lib/examples/utils/clippyJsCharacters.d.ts +18 -0
- package/lib/examples/utils/clippyJsCharacters.js +129 -0
- package/lib/loop/apps/appspec.d.ts +3 -1
- package/lib/loop/apps/appspec.js +31 -0
- package/lib/loop/apps/checks.js +1 -1
- package/lib/protocols/VercelAIAdapter.js +5 -0
- package/lib/specs/apps.js +112 -0
- package/lib/specs/appspecSchema.js +65 -0
- package/lib/specs/index.d.ts +1 -0
- package/lib/specs/index.js +1 -0
- package/lib/specs/voices.d.ts +72 -0
- package/lib/specs/voices.js +344 -0
- package/lib/types/agents.d.ts +1 -1
- package/lib/types/agentspecs.d.ts +16 -0
- package/lib/types/chat.d.ts +11 -0
- package/lib/voice/VoiceInput.d.ts +35 -0
- package/lib/voice/VoiceInput.js +233 -0
- package/lib/voice/capture.d.ts +30 -0
- package/lib/voice/capture.js +113 -0
- package/lib/voice/consent.d.ts +6 -0
- package/lib/voice/consent.js +34 -0
- package/lib/voice/hearing.d.ts +37 -0
- package/lib/voice/hearing.js +168 -0
- package/lib/voice/index.d.ts +47 -0
- package/lib/voice/index.js +13 -0
- package/lib/voice/pinned.d.ts +28 -0
- package/lib/voice/pinned.js +75 -0
- package/lib/voice/sentences.d.ts +26 -0
- package/lib/voice/sentences.js +71 -0
- package/lib/voice/speaker.d.ts +66 -0
- package/lib/voice/speaker.js +168 -0
- package/lib/voice/types.d.ts +92 -0
- package/lib/voice/types.js +5 -0
- package/lib/voice/useSpokenAnswers.d.ts +12 -0
- package/lib/voice/useSpokenAnswers.js +90 -0
- package/package.json +7 -3
- package/scripts/codegen/generate_voices.py +279 -0
- package/scripts/voice/measure.py +244 -0
- package/scripts/voice/pin_store.py +159 -0
- package/scripts/voice/serve_store.py +70 -0
- package/scripts/voice/transcribe.mjs +58 -0
|
@@ -0,0 +1,344 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright (c) 2025-2026 Datalayer, Inc.
|
|
3
|
+
* Distributed under the terms of the Modified BSD License.
|
|
4
|
+
*/
|
|
5
|
+
/** The licences the register allows anywhere, the browser included. */
|
|
6
|
+
export const SPEECH_ALLOWED_LICENCES = [
|
|
7
|
+
'MIT',
|
|
8
|
+
'Apache-2.0',
|
|
9
|
+
'BSD-2-Clause',
|
|
10
|
+
'BSD-3-Clause',
|
|
11
|
+
'0BSD',
|
|
12
|
+
'ISC',
|
|
13
|
+
'CC0-1.0',
|
|
14
|
+
'CC-BY-4.0',
|
|
15
|
+
];
|
|
16
|
+
export const VOICE_CATALOGUE = {
|
|
17
|
+
'kokoro-af-heart': {
|
|
18
|
+
id: 'kokoro-af-heart',
|
|
19
|
+
version: '0.0.1',
|
|
20
|
+
name: 'Heart',
|
|
21
|
+
description: "A warm American English voice, Kokoro's best graded (A).",
|
|
22
|
+
engine: 'kokoro',
|
|
23
|
+
model: 'kokoro-82m',
|
|
24
|
+
voice: 'af_heart',
|
|
25
|
+
languages: ['en-US'],
|
|
26
|
+
where: ['server'],
|
|
27
|
+
licence: {
|
|
28
|
+
weights: 'Apache-2.0',
|
|
29
|
+
},
|
|
30
|
+
attribution: '',
|
|
31
|
+
watermark: false,
|
|
32
|
+
sample: 'Hello. I read the answers aloud, sentence by sentence, as they are written.',
|
|
33
|
+
},
|
|
34
|
+
'kokoro-bf-emma': {
|
|
35
|
+
id: 'kokoro-bf-emma',
|
|
36
|
+
version: '0.0.1',
|
|
37
|
+
name: 'Emma',
|
|
38
|
+
description: 'A clear British English voice.',
|
|
39
|
+
engine: 'kokoro',
|
|
40
|
+
model: 'kokoro-82m',
|
|
41
|
+
voice: 'bf_emma',
|
|
42
|
+
languages: ['en-GB'],
|
|
43
|
+
where: ['server'],
|
|
44
|
+
licence: {
|
|
45
|
+
weights: 'Apache-2.0',
|
|
46
|
+
},
|
|
47
|
+
attribution: '',
|
|
48
|
+
watermark: false,
|
|
49
|
+
sample: 'Hello. I read the answers aloud, sentence by sentence, as they are written.',
|
|
50
|
+
},
|
|
51
|
+
'kokoro-ff-siwis': {
|
|
52
|
+
id: 'kokoro-ff-siwis',
|
|
53
|
+
version: '0.0.1',
|
|
54
|
+
name: 'Siwis',
|
|
55
|
+
description: "A French voice, Kokoro's only one, trained on the SIWIS French Speech Synthesis Database, whose licence asks for its attribution.",
|
|
56
|
+
engine: 'kokoro',
|
|
57
|
+
model: 'kokoro-82m',
|
|
58
|
+
voice: 'ff_siwis',
|
|
59
|
+
languages: ['fr-FR'],
|
|
60
|
+
where: ['server'],
|
|
61
|
+
licence: {
|
|
62
|
+
weights: 'Apache-2.0',
|
|
63
|
+
dataset: 'CC-BY-4.0',
|
|
64
|
+
},
|
|
65
|
+
attribution: 'Trained on the SIWIS French Speech Synthesis Database, by Pierre-Edouard Honnet, Alexandros Lazaridis, Philip N. Garner and Junichi Yamagishi (Idiap Research Institute), under CC BY 4.0.',
|
|
66
|
+
watermark: false,
|
|
67
|
+
sample: "Bonjour. Je lis les réponses à voix haute, phrase par phrase, à mesure qu'elles s'écrivent.",
|
|
68
|
+
},
|
|
69
|
+
};
|
|
70
|
+
export const SPEECH_MODEL_CATALOGUE = {
|
|
71
|
+
'kokoro-82m': {
|
|
72
|
+
id: 'kokoro-82m',
|
|
73
|
+
version: '0.0.1',
|
|
74
|
+
name: 'Kokoro 82M',
|
|
75
|
+
task: 'tts',
|
|
76
|
+
engine: 'kokoro-onnx',
|
|
77
|
+
dtype: 'fp32',
|
|
78
|
+
languages: ['en', 'fr', 'es', 'it', 'pt', 'hi', 'ja', 'zh'],
|
|
79
|
+
where: ['server'],
|
|
80
|
+
streaming: false,
|
|
81
|
+
licence: {
|
|
82
|
+
weights: 'Apache-2.0',
|
|
83
|
+
code: 'MIT',
|
|
84
|
+
},
|
|
85
|
+
attribution: '',
|
|
86
|
+
upstream: 'https://huggingface.co/hexgrad/Kokoro-82M',
|
|
87
|
+
revision: 'model-files-v1.0',
|
|
88
|
+
files: [
|
|
89
|
+
{
|
|
90
|
+
path: 'kokoro-v1.0.onnx',
|
|
91
|
+
sha256: '7d5df8ecf7d4b1878015a32686053fd0eebe2bc377234608764cc0ef3636a6c5',
|
|
92
|
+
size: 325532387,
|
|
93
|
+
},
|
|
94
|
+
{
|
|
95
|
+
path: 'voices-v1.0.bin',
|
|
96
|
+
sha256: 'bca610b8308e8d99f32e6fe4197e7ec01679264efed0cac9140fe9c29f1fbf7d',
|
|
97
|
+
size: 28214398,
|
|
98
|
+
},
|
|
99
|
+
],
|
|
100
|
+
},
|
|
101
|
+
'moonshine-base-en': {
|
|
102
|
+
id: 'moonshine-base-en',
|
|
103
|
+
version: '0.0.1',
|
|
104
|
+
name: 'Moonshine Base (English)',
|
|
105
|
+
task: 'stt',
|
|
106
|
+
engine: 'transformers.js',
|
|
107
|
+
dtype: 'q8',
|
|
108
|
+
languages: ['en'],
|
|
109
|
+
where: ['device'],
|
|
110
|
+
streaming: false,
|
|
111
|
+
licence: {
|
|
112
|
+
weights: 'MIT',
|
|
113
|
+
code: 'MIT',
|
|
114
|
+
},
|
|
115
|
+
attribution: '',
|
|
116
|
+
upstream: 'https://huggingface.co/UsefulSensors/moonshine-base',
|
|
117
|
+
revision: 'b1e9b6aae3c3c7298f10c3798393fdf38e8fbbad',
|
|
118
|
+
files: [
|
|
119
|
+
{
|
|
120
|
+
path: 'config.json',
|
|
121
|
+
sha256: 'fab7241d1e9fc6c2370c4c6dfb5da79bb54d67ed9ab6b507ac51d29d2abe01d1',
|
|
122
|
+
size: 922,
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
path: 'generation_config.json',
|
|
126
|
+
sha256: 'f9b3f711b57be7def2e50a8942f64f36ee0a55fad5b84ff93a687b6c5bcc1d44',
|
|
127
|
+
size: 147,
|
|
128
|
+
},
|
|
129
|
+
{
|
|
130
|
+
path: 'preprocessor_config.json',
|
|
131
|
+
sha256: 'fa43a7017ef85cd1d0fba0d9aae77c8adb16990ae6f11115631f41ec5d8aa679',
|
|
132
|
+
size: 128,
|
|
133
|
+
},
|
|
134
|
+
{
|
|
135
|
+
path: 'special_tokens_map.json',
|
|
136
|
+
sha256: 'ca3d163bab055381827226140568f3bef7eaac187cebd76878e0b63e9e442356',
|
|
137
|
+
size: 3,
|
|
138
|
+
},
|
|
139
|
+
{
|
|
140
|
+
path: 'tokenizer.json',
|
|
141
|
+
sha256: '7b913404bdd039af4756783218af4440bc07fb7d6d8258d677e34f95b3ec416f',
|
|
142
|
+
size: 3761754,
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
path: 'tokenizer_config.json',
|
|
146
|
+
sha256: 'edaee394565d428ea98a663ae7209cdcfeefc5585c42d7a570ff7c986df2cd15',
|
|
147
|
+
size: 135735,
|
|
148
|
+
},
|
|
149
|
+
{
|
|
150
|
+
path: 'onnx/encoder_model_quantized.onnx',
|
|
151
|
+
sha256: '1dd9ab0a7f987113d30affcba5a068d11c8f90fa0223caa3e491ade431ad9751',
|
|
152
|
+
size: 20513063,
|
|
153
|
+
},
|
|
154
|
+
{
|
|
155
|
+
path: 'onnx/decoder_model_merged_quantized.onnx',
|
|
156
|
+
sha256: 'cc9f3cd6698a369c6008b41aa60aa3fb3322e7f03c9bdf19d8e6b7200afca4f3',
|
|
157
|
+
size: 42498870,
|
|
158
|
+
},
|
|
159
|
+
],
|
|
160
|
+
},
|
|
161
|
+
'moonshine-tiny-en': {
|
|
162
|
+
id: 'moonshine-tiny-en',
|
|
163
|
+
version: '0.0.1',
|
|
164
|
+
name: 'Moonshine Tiny (English)',
|
|
165
|
+
task: 'stt',
|
|
166
|
+
engine: 'transformers.js',
|
|
167
|
+
dtype: 'q8',
|
|
168
|
+
languages: ['en'],
|
|
169
|
+
where: ['device'],
|
|
170
|
+
streaming: false,
|
|
171
|
+
licence: {
|
|
172
|
+
weights: 'MIT',
|
|
173
|
+
code: 'MIT',
|
|
174
|
+
},
|
|
175
|
+
attribution: '',
|
|
176
|
+
upstream: 'https://huggingface.co/UsefulSensors/moonshine-tiny',
|
|
177
|
+
revision: 'a6da1241cd305dcd64eab1edbd615f2bb9aabb95',
|
|
178
|
+
files: [
|
|
179
|
+
{
|
|
180
|
+
path: 'config.json',
|
|
181
|
+
sha256: '558e1e02069137c796ace1e50c48d8fe451f04a295929138e6bea885517f0edb',
|
|
182
|
+
size: 921,
|
|
183
|
+
},
|
|
184
|
+
{
|
|
185
|
+
path: 'generation_config.json',
|
|
186
|
+
sha256: 'f9b3f711b57be7def2e50a8942f64f36ee0a55fad5b84ff93a687b6c5bcc1d44',
|
|
187
|
+
size: 147,
|
|
188
|
+
},
|
|
189
|
+
{
|
|
190
|
+
path: 'preprocessor_config.json',
|
|
191
|
+
sha256: 'fa43a7017ef85cd1d0fba0d9aae77c8adb16990ae6f11115631f41ec5d8aa679',
|
|
192
|
+
size: 128,
|
|
193
|
+
},
|
|
194
|
+
{
|
|
195
|
+
path: 'special_tokens_map.json',
|
|
196
|
+
sha256: 'ca3d163bab055381827226140568f3bef7eaac187cebd76878e0b63e9e442356',
|
|
197
|
+
size: 3,
|
|
198
|
+
},
|
|
199
|
+
{
|
|
200
|
+
path: 'tokenizer.json',
|
|
201
|
+
sha256: '7b913404bdd039af4756783218af4440bc07fb7d6d8258d677e34f95b3ec416f',
|
|
202
|
+
size: 3761754,
|
|
203
|
+
},
|
|
204
|
+
{
|
|
205
|
+
path: 'tokenizer_config.json',
|
|
206
|
+
sha256: 'edaee394565d428ea98a663ae7209cdcfeefc5585c42d7a570ff7c986df2cd15',
|
|
207
|
+
size: 135735,
|
|
208
|
+
},
|
|
209
|
+
{
|
|
210
|
+
path: 'onnx/encoder_model_quantized.onnx',
|
|
211
|
+
sha256: 'c6fc4b7bc5af75c0591fd157a1f3829b533d18e9769a888fd95a62e470dd4f4a',
|
|
212
|
+
size: 7937661,
|
|
213
|
+
},
|
|
214
|
+
{
|
|
215
|
+
path: 'onnx/decoder_model_merged_quantized.onnx',
|
|
216
|
+
sha256: 'eed87831c3a6103534aae7d47a5d485025c659a1323901513961c39fe8a1a367',
|
|
217
|
+
size: 20243286,
|
|
218
|
+
},
|
|
219
|
+
],
|
|
220
|
+
},
|
|
221
|
+
'silero-vad': {
|
|
222
|
+
id: 'silero-vad',
|
|
223
|
+
version: '0.0.1',
|
|
224
|
+
name: 'Silero VAD',
|
|
225
|
+
task: 'vad',
|
|
226
|
+
engine: 'vad-web',
|
|
227
|
+
dtype: 'fp32',
|
|
228
|
+
languages: [],
|
|
229
|
+
where: ['device'],
|
|
230
|
+
streaming: true,
|
|
231
|
+
licence: {
|
|
232
|
+
weights: 'MIT',
|
|
233
|
+
code: 'ISC',
|
|
234
|
+
},
|
|
235
|
+
attribution: '',
|
|
236
|
+
upstream: 'https://github.com/snakers4/silero-vad',
|
|
237
|
+
revision: '0.0.31',
|
|
238
|
+
files: [
|
|
239
|
+
{
|
|
240
|
+
path: 'silero_vad_legacy.onnx',
|
|
241
|
+
sha256: 'a35ebf52fd3ce5f1469b2a36158dba761bc47b973ea3382b3186ca15b1f5af28',
|
|
242
|
+
size: 1807522,
|
|
243
|
+
},
|
|
244
|
+
],
|
|
245
|
+
},
|
|
246
|
+
'whisper-base': {
|
|
247
|
+
id: 'whisper-base',
|
|
248
|
+
version: '0.0.1',
|
|
249
|
+
name: 'Whisper Base',
|
|
250
|
+
task: 'stt',
|
|
251
|
+
engine: 'transformers.js',
|
|
252
|
+
dtype: 'q8',
|
|
253
|
+
languages: ['en', 'fr'],
|
|
254
|
+
where: ['device'],
|
|
255
|
+
streaming: false,
|
|
256
|
+
licence: {
|
|
257
|
+
weights: 'MIT',
|
|
258
|
+
code: 'MIT',
|
|
259
|
+
},
|
|
260
|
+
attribution: '',
|
|
261
|
+
upstream: 'https://huggingface.co/openai/whisper-base',
|
|
262
|
+
revision: '1846881b6b3a3024392c1eea3ad983695bc23925',
|
|
263
|
+
files: [
|
|
264
|
+
{
|
|
265
|
+
path: 'config.json',
|
|
266
|
+
sha256: 'f4d0608f7d918166da7edb3e188de5ef1bfe70d9802e785d271fd88111e9cf4b',
|
|
267
|
+
size: 2243,
|
|
268
|
+
},
|
|
269
|
+
{
|
|
270
|
+
path: 'generation_config.json',
|
|
271
|
+
sha256: '61070cf8de25b1e9256e8e102ded49d8d24a8369ed36ef84fdf21549e68125a0',
|
|
272
|
+
size: 3832,
|
|
273
|
+
},
|
|
274
|
+
{
|
|
275
|
+
path: 'preprocessor_config.json',
|
|
276
|
+
sha256: 'a6a76d28c93edb273669eb9e0b0636a2bddbb1272c3261e47b7ca6dfdbac1b8d',
|
|
277
|
+
size: 339,
|
|
278
|
+
},
|
|
279
|
+
{
|
|
280
|
+
path: 'special_tokens_map.json',
|
|
281
|
+
sha256: 'e67ae3a0aaa99abcd9f187138e12db1f65c16a14761c50ef10eef2c174a7a691',
|
|
282
|
+
size: 2194,
|
|
283
|
+
},
|
|
284
|
+
{
|
|
285
|
+
path: 'tokenizer.json',
|
|
286
|
+
sha256: '27fc476bfe7f17299480be2273fc0608e4d5a99aba2ab5dec5374b4482d1a566',
|
|
287
|
+
size: 2480466,
|
|
288
|
+
},
|
|
289
|
+
{
|
|
290
|
+
path: 'tokenizer_config.json',
|
|
291
|
+
sha256: '2e036e4dbacfdeb7242c7d4ec4149f4a16e86026048f94d1637e3a8ee9c6a573',
|
|
292
|
+
size: 282682,
|
|
293
|
+
},
|
|
294
|
+
{
|
|
295
|
+
path: 'added_tokens.json',
|
|
296
|
+
sha256: '9715fd2243b6f06a5858b5e32950d2853f73dd5bc201aafcf76f5082a2d8acd1',
|
|
297
|
+
size: 34604,
|
|
298
|
+
},
|
|
299
|
+
{
|
|
300
|
+
path: 'normalizer.json',
|
|
301
|
+
sha256: 'bf1c507dc8724ca9cf9903640dacfb69dae2f00edee4f21ceba106a7392f26dd',
|
|
302
|
+
size: 52666,
|
|
303
|
+
},
|
|
304
|
+
{
|
|
305
|
+
path: 'vocab.json',
|
|
306
|
+
sha256: '50d6a919f0a0601d56a04eb583c780d18553aa388254ba3158eb6a00f13e2c1a',
|
|
307
|
+
size: 1036584,
|
|
308
|
+
},
|
|
309
|
+
{
|
|
310
|
+
path: 'onnx/encoder_model_quantized.onnx',
|
|
311
|
+
sha256: '5862993336bf33acd23736071aae2b32261d3b1b2f37780194460d4ef974dd46',
|
|
312
|
+
size: 23201314,
|
|
313
|
+
},
|
|
314
|
+
{
|
|
315
|
+
path: 'onnx/decoder_model_merged_quantized.onnx',
|
|
316
|
+
sha256: 'fa3ef9902734ce5ae6f9ef2bdb2ba9a6c4b5785b09f4f420ce036573dc9d090b',
|
|
317
|
+
size: 53693315,
|
|
318
|
+
},
|
|
319
|
+
],
|
|
320
|
+
},
|
|
321
|
+
};
|
|
322
|
+
/** Whether a voice speaks a language: `fr` and `fr-FR` are spoken by a `fr-FR` voice. */
|
|
323
|
+
export function voiceSpeaks(voice, language) {
|
|
324
|
+
return voice.languages.some(own => own === language || own.split('-')[0] === language);
|
|
325
|
+
}
|
|
326
|
+
/** The voice a language is spoken with when none is said: the first that speaks it. */
|
|
327
|
+
export function voiceFor(language) {
|
|
328
|
+
const voices = Object.values(VOICE_CATALOGUE);
|
|
329
|
+
return (voices.find(voice => voice.languages.includes(language)) ??
|
|
330
|
+
voices.find(voice => voiceSpeaks(voice, language.split('-')[0])));
|
|
331
|
+
}
|
|
332
|
+
/**
|
|
333
|
+
* The speech-to-text model for a language, on the device: Moonshine where it
|
|
334
|
+
* hears it, Whisper otherwise (VOICE.md decision 3).
|
|
335
|
+
*/
|
|
336
|
+
export function transcriberFor(language) {
|
|
337
|
+
const base = language.split('-')[0];
|
|
338
|
+
const hearing = Object.values(SPEECH_MODEL_CATALOGUE).filter(model => model.task === 'stt' &&
|
|
339
|
+
model.where.includes('device') &&
|
|
340
|
+
model.languages.includes(base));
|
|
341
|
+
return (hearing.find(model => model.id === 'moonshine-tiny-en') ??
|
|
342
|
+
hearing.find(model => model.id.startsWith('moonshine-')) ??
|
|
343
|
+
hearing[0]);
|
|
344
|
+
}
|
package/lib/types/agents.d.ts
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
*/
|
|
4
4
|
import type { Agentspec } from './agentspecs';
|
|
5
5
|
import type { AgentConnection } from './connection';
|
|
6
|
-
export type AgentLibrary = 'pydantic-ai' | 'langchain'
|
|
6
|
+
export type AgentLibrary = 'pydantic-ai' | 'langchain';
|
|
7
7
|
/**
|
|
8
8
|
* Unified agent status covering runtime lifecycle and UI lifecycle.
|
|
9
9
|
*/
|
|
@@ -450,6 +450,20 @@ export interface AppSurfaceSpec {
|
|
|
450
450
|
composedBy: string;
|
|
451
451
|
composedAt: string;
|
|
452
452
|
}
|
|
453
|
+
/**
|
|
454
|
+
* An application's voice (VOICE.md VO-41), off unless said: whether a person
|
|
455
|
+
* may talk to it, whether its answers are heard, with which voice of the
|
|
456
|
+
* catalogue (`specs/voices`), in which language (BCP 47) and where its
|
|
457
|
+
* speech runs.
|
|
458
|
+
*/
|
|
459
|
+
export interface AppVoiceSpec {
|
|
460
|
+
enabled: boolean;
|
|
461
|
+
input: 'off' | 'push_to_talk' | 'hands_free';
|
|
462
|
+
output: 'off' | 'on_request' | 'always';
|
|
463
|
+
voice: string;
|
|
464
|
+
language: string;
|
|
465
|
+
where: 'auto' | 'device' | 'server';
|
|
466
|
+
}
|
|
453
467
|
/** What the user of an application sees. */
|
|
454
468
|
export interface AppInterfaceSpec {
|
|
455
469
|
layout: AppLayout;
|
|
@@ -465,6 +479,8 @@ export interface AppInterfaceSpec {
|
|
|
465
479
|
* Said, it wins over the one the person chose in their settings.
|
|
466
480
|
*/
|
|
467
481
|
assistant?: AppAssistantCharacter;
|
|
482
|
+
/** Its voice: off unless said (VO-41); absent from a spec made before voice. */
|
|
483
|
+
voice?: AppVoiceSpec;
|
|
468
484
|
}
|
|
469
485
|
export interface AppTestCaseSpec {
|
|
470
486
|
ask: string;
|
package/lib/types/chat.d.ts
CHANGED
|
@@ -29,6 +29,7 @@ import type { FrontendToolDefinition } from './tools';
|
|
|
29
29
|
import type { PoweredByTagProps } from '../chat/display/PoweredByTag';
|
|
30
30
|
import type { EphemeralRuntimeOverride } from '../chat/notebook/EphemeralNotebook';
|
|
31
31
|
import type { EphemeralDocumentCollaboration } from '../chat/document/EphemeralDocument';
|
|
32
|
+
import type { ChatVoice } from '../voice';
|
|
32
33
|
/**
|
|
33
34
|
* Context passed to tool-call pre-hooks.
|
|
34
35
|
* Fires when a tool call starts executing (backend or frontend).
|
|
@@ -976,6 +977,16 @@ export interface ChatBaseProps {
|
|
|
976
977
|
* with *Approve* and *Deny* (T-23).
|
|
977
978
|
*/
|
|
978
979
|
trailingContent?: ReactNode;
|
|
980
|
+
/**
|
|
981
|
+
* Its voice (VOICE.md V1): a microphone in the composer, push-to-talk,
|
|
982
|
+
* what is said heard on the device and put in the composer — or sent, with
|
|
983
|
+
* *Send what I say* — marked as spoken (VO-27).
|
|
984
|
+
*/
|
|
985
|
+
voice?: ChatVoice;
|
|
986
|
+
/** The agent is speaking: the composer offers *Stop speaking* (`Esc`). */
|
|
987
|
+
voiceSpeaking?: boolean;
|
|
988
|
+
/** Stops the agent's voice: the composer's *Stop speaking*, and the microphone opening (VO-13). */
|
|
989
|
+
onStopSpeaking?: () => void;
|
|
979
990
|
/**
|
|
980
991
|
* Show the information icon in the header.
|
|
981
992
|
* When clicked, fires onInformationClick.
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Push-to-talk in the composer (VOICE.md VO-10, VO-61, VO-62, VO-65 to VO-68).
|
|
3
|
+
*
|
|
4
|
+
* The microphone button: pressed and held, or clicked to start and clicked
|
|
5
|
+
* to stop; `Ctrl`+`Space` held does the same, and `Esc` cancels. While it
|
|
6
|
+
* listens the button says *Listening* and a meter shows the level (still,
|
|
7
|
+
* and said in words, for a reader who asks for reduced motion). Released,
|
|
8
|
+
* what was said is heard on the device and given to `onTranscript`: put in
|
|
9
|
+
* the composer, or sent at once when *Send what I say* is on. The first time,
|
|
10
|
+
* one sentence says where the audio goes before the browser asks.
|
|
11
|
+
*
|
|
12
|
+
* @module voice/VoiceInput
|
|
13
|
+
*/
|
|
14
|
+
import type { JSX } from 'react';
|
|
15
|
+
import type { DeviceHearing } from './hearing';
|
|
16
|
+
import type { Transcript } from './types';
|
|
17
|
+
/** What the microphone is doing, said in words (VO-67, VO-68). */
|
|
18
|
+
export type ListeningState = 'idle' | 'asking' | 'loading' | 'listening' | 'hearing';
|
|
19
|
+
export interface VoiceInputProps {
|
|
20
|
+
hearing: DeviceHearing;
|
|
21
|
+
/** The language listened to, BCP 47. */
|
|
22
|
+
language: string;
|
|
23
|
+
/** What was said, once heard. */
|
|
24
|
+
onTranscript: (transcript: Transcript) => void;
|
|
25
|
+
/** The microphone opens: the agent stops speaking (VO-13). */
|
|
26
|
+
onListen?: () => void;
|
|
27
|
+
/** Whose consent is remembered: the application's id. */
|
|
28
|
+
consentKey: string;
|
|
29
|
+
disabled?: boolean;
|
|
30
|
+
/** The agent is speaking: offer to stop it. */
|
|
31
|
+
speaking?: boolean;
|
|
32
|
+
onStopSpeaking?: () => void;
|
|
33
|
+
}
|
|
34
|
+
export declare function VoiceInput({ hearing, language, onTranscript, onListen, consentKey, disabled, speaking, onStopSpeaking, }: VoiceInputProps): JSX.Element;
|
|
35
|
+
export default VoiceInput;
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
import { jsx as _jsx, jsxs as _jsxs } from "react/jsx-runtime";
|
|
2
|
+
import { useCallback, useEffect, useRef, useState } from 'react';
|
|
3
|
+
import { Button, IconButton, Text, Tooltip } from '@primer/react';
|
|
4
|
+
import { Box } from '@datalayer/primer-addons';
|
|
5
|
+
import { MicrophoneIcon } from '@datalayer/icons-react';
|
|
6
|
+
import { MuteIcon } from '@primer/octicons-react';
|
|
7
|
+
import { MicrophoneRefused, openMicrophone } from './capture';
|
|
8
|
+
import { CONSENT_SENTENCE, consent, consented } from './consent';
|
|
9
|
+
const WORDS = {
|
|
10
|
+
idle: '',
|
|
11
|
+
asking: '',
|
|
12
|
+
loading: 'Getting ready to listen…',
|
|
13
|
+
listening: 'Listening… let go to stop, Esc to cancel',
|
|
14
|
+
hearing: 'Hearing what you said…',
|
|
15
|
+
};
|
|
16
|
+
/** A press shorter than this is a click: it starts, and the next click stops. */
|
|
17
|
+
const CLICK_MS = 300;
|
|
18
|
+
function prefersStill() {
|
|
19
|
+
return (typeof window !== 'undefined' &&
|
|
20
|
+
!!window.matchMedia?.('(prefers-reduced-motion: reduce)').matches);
|
|
21
|
+
}
|
|
22
|
+
export function VoiceInput({ hearing, language, onTranscript, onListen, consentKey, disabled = false, speaking = false, onStopSpeaking, }) {
|
|
23
|
+
const [state, setState] = useState('idle');
|
|
24
|
+
const [said, setSaid] = useState('');
|
|
25
|
+
const [problem, setProblem] = useState();
|
|
26
|
+
const [progress, setProgress] = useState();
|
|
27
|
+
const capture = useRef(undefined);
|
|
28
|
+
const pressedAt = useRef(0);
|
|
29
|
+
const stopOnRelease = useRef(false);
|
|
30
|
+
const meter = useRef(null);
|
|
31
|
+
const stateRef = useRef(state);
|
|
32
|
+
stateRef.current = state;
|
|
33
|
+
const start = useCallback(async () => {
|
|
34
|
+
if (disabled || stateRef.current !== 'idle') {
|
|
35
|
+
return;
|
|
36
|
+
}
|
|
37
|
+
if (!consented(consentKey)) {
|
|
38
|
+
setState('asking');
|
|
39
|
+
return;
|
|
40
|
+
}
|
|
41
|
+
setProblem(undefined);
|
|
42
|
+
onListen?.();
|
|
43
|
+
try {
|
|
44
|
+
setState('loading');
|
|
45
|
+
// The models load once, on first use, with their progress shown.
|
|
46
|
+
await hearing.ready(language, (loaded, total) => setProgress(total ? Math.round((loaded / total) * 100) : undefined));
|
|
47
|
+
setProgress(undefined);
|
|
48
|
+
if (stateRef.current !== 'loading') {
|
|
49
|
+
return;
|
|
50
|
+
}
|
|
51
|
+
capture.current = await openMicrophone();
|
|
52
|
+
setState('listening');
|
|
53
|
+
setSaid('Listening');
|
|
54
|
+
}
|
|
55
|
+
catch (error) {
|
|
56
|
+
setState('idle');
|
|
57
|
+
setProgress(undefined);
|
|
58
|
+
setProblem(error instanceof MicrophoneRefused || error instanceof Error
|
|
59
|
+
? error.message
|
|
60
|
+
: 'The microphone could not be opened.');
|
|
61
|
+
}
|
|
62
|
+
}, [consentKey, disabled, hearing, language, onListen]);
|
|
63
|
+
const stop = useCallback(async () => {
|
|
64
|
+
const open = capture.current;
|
|
65
|
+
capture.current = undefined;
|
|
66
|
+
if (!open) {
|
|
67
|
+
if (stateRef.current === 'loading') {
|
|
68
|
+
setState('idle');
|
|
69
|
+
}
|
|
70
|
+
return;
|
|
71
|
+
}
|
|
72
|
+
setState('hearing');
|
|
73
|
+
try {
|
|
74
|
+
const audio = await open.stop();
|
|
75
|
+
const transcript = await hearing.hear(audio, language);
|
|
76
|
+
setState('idle');
|
|
77
|
+
if (!transcript) {
|
|
78
|
+
setSaid('Nothing was heard.');
|
|
79
|
+
return;
|
|
80
|
+
}
|
|
81
|
+
setSaid(`Heard: ${transcript.text}`);
|
|
82
|
+
onTranscript(transcript);
|
|
83
|
+
}
|
|
84
|
+
catch (error) {
|
|
85
|
+
setState('idle');
|
|
86
|
+
setProblem(error instanceof Error
|
|
87
|
+
? error.message
|
|
88
|
+
: 'What you said could not be heard.');
|
|
89
|
+
}
|
|
90
|
+
}, [hearing, language, onTranscript]);
|
|
91
|
+
const cancel = useCallback(() => {
|
|
92
|
+
capture.current?.cancel();
|
|
93
|
+
capture.current = undefined;
|
|
94
|
+
if (stateRef.current !== 'idle') {
|
|
95
|
+
setState('idle');
|
|
96
|
+
setSaid('Cancelled.');
|
|
97
|
+
}
|
|
98
|
+
}, []);
|
|
99
|
+
// The key: Ctrl+Space held listens; Esc cancels, or stops the voice.
|
|
100
|
+
useEffect(() => {
|
|
101
|
+
const down = (event) => {
|
|
102
|
+
if (event.code === 'Space' && event.ctrlKey && !event.repeat) {
|
|
103
|
+
event.preventDefault();
|
|
104
|
+
pressedAt.current = Date.now();
|
|
105
|
+
void start();
|
|
106
|
+
}
|
|
107
|
+
else if (event.key === 'Escape') {
|
|
108
|
+
if (stateRef.current !== 'idle') {
|
|
109
|
+
cancel();
|
|
110
|
+
}
|
|
111
|
+
else if (speaking) {
|
|
112
|
+
onStopSpeaking?.();
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
};
|
|
116
|
+
const up = (event) => {
|
|
117
|
+
if (event.code === 'Space' || event.key === 'Control') {
|
|
118
|
+
if (stateRef.current === 'listening' ||
|
|
119
|
+
stateRef.current === 'loading') {
|
|
120
|
+
void stop();
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
};
|
|
124
|
+
// A hidden tab closes the microphone (VO-62).
|
|
125
|
+
const hidden = () => {
|
|
126
|
+
if (document.hidden) {
|
|
127
|
+
cancel();
|
|
128
|
+
}
|
|
129
|
+
};
|
|
130
|
+
window.addEventListener('keydown', down);
|
|
131
|
+
window.addEventListener('keyup', up);
|
|
132
|
+
document.addEventListener('visibilitychange', hidden);
|
|
133
|
+
return () => {
|
|
134
|
+
window.removeEventListener('keydown', down);
|
|
135
|
+
window.removeEventListener('keyup', up);
|
|
136
|
+
document.removeEventListener('visibilitychange', hidden);
|
|
137
|
+
};
|
|
138
|
+
}, [cancel, onStopSpeaking, speaking, start, stop]);
|
|
139
|
+
// The level meter, moving while it listens — still under reduced motion.
|
|
140
|
+
useEffect(() => {
|
|
141
|
+
if (state !== 'listening' || prefersStill()) {
|
|
142
|
+
return;
|
|
143
|
+
}
|
|
144
|
+
let frame = 0;
|
|
145
|
+
const draw = () => {
|
|
146
|
+
const level = capture.current?.level() ?? 0;
|
|
147
|
+
if (meter.current) {
|
|
148
|
+
meter.current.style.transform = `scaleX(${Math.max(0.04, level)})`;
|
|
149
|
+
}
|
|
150
|
+
frame = requestAnimationFrame(draw);
|
|
151
|
+
};
|
|
152
|
+
frame = requestAnimationFrame(draw);
|
|
153
|
+
return () => cancelAnimationFrame(frame);
|
|
154
|
+
}, [state]);
|
|
155
|
+
// Never left open behind the page.
|
|
156
|
+
useEffect(() => () => capture.current?.cancel(), []);
|
|
157
|
+
const listening = state === 'listening';
|
|
158
|
+
const label = listening
|
|
159
|
+
? 'Listening — let go to stop'
|
|
160
|
+
: 'Hold to talk (Ctrl+Space)';
|
|
161
|
+
return (_jsxs(Box, { "data-voice-input": state, sx: {
|
|
162
|
+
display: 'inline-flex',
|
|
163
|
+
alignItems: 'center',
|
|
164
|
+
gap: 1,
|
|
165
|
+
position: 'relative',
|
|
166
|
+
}, children: [speaking && onStopSpeaking && (_jsx(Tooltip, { text: "Stop speaking (Esc)", direction: "n", children: _jsx(IconButton, { icon: MuteIcon, "aria-label": "Stop speaking", size: "small", variant: "invisible", "data-voice-stop": "", onClick: onStopSpeaking }) })), listening && (_jsx(Box, { "aria-hidden": "true", sx: {
|
|
167
|
+
width: 40,
|
|
168
|
+
height: 4,
|
|
169
|
+
borderRadius: 2,
|
|
170
|
+
bg: 'neutral.muted',
|
|
171
|
+
overflow: 'hidden',
|
|
172
|
+
}, children: _jsx(Box, { ref: meter, sx: {
|
|
173
|
+
width: '100%',
|
|
174
|
+
height: '100%',
|
|
175
|
+
bg: 'danger.emphasis',
|
|
176
|
+
transformOrigin: 'left',
|
|
177
|
+
transform: prefersStill() ? 'scaleX(1)' : 'scaleX(0.04)',
|
|
178
|
+
} }) })), (state === 'loading' || state === 'hearing') && (_jsx(Text, { sx: { fontSize: 0, color: 'fg.muted' }, "data-voice-progress": "", children: state === 'loading' && progress !== undefined ? `${progress}%` : '…' })), _jsx(Tooltip, { text: label, direction: "n", children: _jsx(IconButton, { icon: () => _jsx(MicrophoneIcon, { size: 16 }), "aria-label": label, "aria-pressed": listening, size: "small", variant: listening ? 'danger' : 'invisible', disabled: disabled || state === 'hearing', "data-voice-mic": "", onPointerDown: event => {
|
|
179
|
+
if (event.button !== 0) {
|
|
180
|
+
return;
|
|
181
|
+
}
|
|
182
|
+
pressedAt.current = Date.now();
|
|
183
|
+
// A second press of click-to-talk stops at its release.
|
|
184
|
+
stopOnRelease.current = stateRef.current === 'listening';
|
|
185
|
+
if (stateRef.current === 'idle') {
|
|
186
|
+
void start();
|
|
187
|
+
}
|
|
188
|
+
}, onPointerUp: () => {
|
|
189
|
+
// Held: letting go stops. A click: it goes on until the next one.
|
|
190
|
+
if (stopOnRelease.current ||
|
|
191
|
+
Date.now() - pressedAt.current >= CLICK_MS) {
|
|
192
|
+
stopOnRelease.current = false;
|
|
193
|
+
void stop();
|
|
194
|
+
}
|
|
195
|
+
}, onKeyDown: event => {
|
|
196
|
+
// Enter or Space on the focused button toggles, for the keyboard (VO-66).
|
|
197
|
+
if (event.key === 'Enter' || event.key === ' ') {
|
|
198
|
+
event.preventDefault();
|
|
199
|
+
if (stateRef.current === 'listening') {
|
|
200
|
+
void stop();
|
|
201
|
+
}
|
|
202
|
+
else {
|
|
203
|
+
void start();
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
} }) }), state === 'asking' && (_jsxs(Box, { role: "dialog", "aria-label": "Talk instead of typing", "data-voice-consent": "", sx: {
|
|
207
|
+
position: 'absolute',
|
|
208
|
+
bottom: '100%',
|
|
209
|
+
right: 0,
|
|
210
|
+
mb: 2,
|
|
211
|
+
width: 280,
|
|
212
|
+
p: 3,
|
|
213
|
+
bg: 'canvas.overlay',
|
|
214
|
+
border: '1px solid',
|
|
215
|
+
borderColor: 'border.default',
|
|
216
|
+
borderRadius: 2,
|
|
217
|
+
boxShadow: 'shadow.large',
|
|
218
|
+
zIndex: 10,
|
|
219
|
+
}, children: [_jsx(Text, { as: "p", sx: { fontSize: 1, mb: 2 }, children: CONSENT_SENTENCE }), _jsxs(Box, { sx: { display: 'flex', gap: 2, justifyContent: 'flex-end' }, children: [_jsx(Button, { size: "small", onClick: () => setState('idle'), children: "Not now" }), _jsx(Button, { size: "small", variant: "primary", "data-voice-consent-yes": "", onClick: () => {
|
|
220
|
+
consent(consentKey);
|
|
221
|
+
// Asked, and said yes: listen now, without another press.
|
|
222
|
+
stateRef.current = 'idle';
|
|
223
|
+
setState('idle');
|
|
224
|
+
void start();
|
|
225
|
+
}, children: "Use the microphone" })] })] })), _jsx(Box, { as: "span", role: "status", "aria-live": "polite", sx: {
|
|
226
|
+
position: 'absolute',
|
|
227
|
+
width: 1,
|
|
228
|
+
height: 1,
|
|
229
|
+
overflow: 'hidden',
|
|
230
|
+
clip: 'rect(0 0 0 0)',
|
|
231
|
+
}, children: WORDS[state] || said }), problem && (_jsx(Text, { role: "alert", sx: { fontSize: 0, color: 'danger.fg', maxWidth: 220 }, "data-voice-problem": "", children: problem }))] }));
|
|
232
|
+
}
|
|
233
|
+
export default VoiceInput;
|