@datalayer/agent-runtimes 1.3.63 → 1.3.65
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/chat/ChatFloating.d.ts +9 -1
- package/lib/chat/ChatFloating.js +69 -18
- package/lib/chat/assistant/AssistantStage.d.ts +14 -1
- package/lib/chat/assistant/AssistantStage.js +48 -1
- package/lib/chat/assistant/state.d.ts +7 -1
- package/lib/chat/assistant/state.js +3 -2
- package/lib/chat/base/ChatBase.js +32 -4
- package/lib/chat/messages/ChatMessageList.d.ts +0 -6
- package/lib/chat/messages/ChatMessageList.js +8 -2
- package/lib/components/teams/A2ATeamGraph.d.ts +58 -0
- package/lib/components/teams/A2ATeamGraph.js +190 -0
- package/lib/components/teams/a2aTeamFlow.d.ts +33 -0
- package/lib/components/teams/a2aTeamFlow.js +32 -0
- package/lib/components/teams/index.d.ts +11 -0
- package/lib/components/teams/index.js +15 -0
- package/lib/components/teams/useA2ATeam.d.ts +57 -0
- package/lib/components/teams/useA2ATeam.js +233 -0
- package/lib/examples/AgentA2ATeamExample.d.ts +6 -3
- package/lib/examples/AgentA2ATeamExample.js +42 -208
- package/lib/examples/VoiceChatExample.d.ts +20 -0
- package/lib/examples/VoiceChatExample.js +64 -0
- package/lib/examples/example-selector.js +1 -0
- package/lib/examples/main.js +7 -1
- package/lib/loop/apps/appspec.d.ts +3 -1
- package/lib/loop/apps/appspec.js +31 -0
- package/lib/protocols/VercelAIAdapter.js +5 -0
- package/lib/specs/apps.js +112 -0
- package/lib/specs/appspecSchema.js +65 -0
- package/lib/specs/index.d.ts +1 -0
- package/lib/specs/index.js +1 -0
- package/lib/specs/voices.d.ts +72 -0
- package/lib/specs/voices.js +344 -0
- package/lib/types/agentspecs.d.ts +16 -0
- package/lib/types/chat.d.ts +11 -0
- package/lib/voice/VoiceInput.d.ts +35 -0
- package/lib/voice/VoiceInput.js +233 -0
- package/lib/voice/capture.d.ts +30 -0
- package/lib/voice/capture.js +113 -0
- package/lib/voice/consent.d.ts +6 -0
- package/lib/voice/consent.js +34 -0
- package/lib/voice/hearing.d.ts +37 -0
- package/lib/voice/hearing.js +168 -0
- package/lib/voice/index.d.ts +47 -0
- package/lib/voice/index.js +13 -0
- package/lib/voice/pinned.d.ts +28 -0
- package/lib/voice/pinned.js +75 -0
- package/lib/voice/sentences.d.ts +26 -0
- package/lib/voice/sentences.js +71 -0
- package/lib/voice/speaker.d.ts +66 -0
- package/lib/voice/speaker.js +168 -0
- package/lib/voice/types.d.ts +92 -0
- package/lib/voice/types.js +5 -0
- package/lib/voice/useSpokenAnswers.d.ts +12 -0
- package/lib/voice/useSpokenAnswers.js +90 -0
- package/package.json +7 -3
- package/scripts/codegen/generate_voices.py +279 -0
- package/scripts/voice/measure.py +244 -0
- package/scripts/voice/pin_store.py +159 -0
- package/scripts/voice/serve_store.py +70 -0
- package/scripts/voice/transcribe.mjs +58 -0
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# Copyright (c) 2025-2026 Datalayer, Inc.
|
|
3
|
+
# Distributed under the terms of the Modified BSD License.
|
|
4
|
+
|
|
5
|
+
"""
|
|
6
|
+
Generate the voice catalogue (VOICE.md VO-40) in Python and TypeScript.
|
|
7
|
+
|
|
8
|
+
Read through ``agentspecs.speech``, which refuses a voice or a model whose
|
|
9
|
+
licence the register does not allow: what is generated has passed it.
|
|
10
|
+
|
|
11
|
+
Usage:
|
|
12
|
+
python generate_voices.py \\
|
|
13
|
+
--python-output agent_runtimes/specs/voices.py \\
|
|
14
|
+
--typescript-output src/specs/voices.ts
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
import argparse
|
|
18
|
+
import json
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
from agentspecs.speech import list_speech_models, list_voices, register
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _voice(voice: Any) -> dict[str, Any]:
|
|
26
|
+
return {
|
|
27
|
+
"id": voice.id,
|
|
28
|
+
"version": voice.version,
|
|
29
|
+
"name": voice.name,
|
|
30
|
+
"description": " ".join(voice.description.split()),
|
|
31
|
+
"engine": voice.engine,
|
|
32
|
+
"model": voice.model,
|
|
33
|
+
"voice": voice.voice,
|
|
34
|
+
"languages": list(voice.languages),
|
|
35
|
+
"where": list(voice.where),
|
|
36
|
+
"licence": voice.licence.model_dump(exclude_defaults=True),
|
|
37
|
+
"attribution": " ".join(voice.attribution.split()),
|
|
38
|
+
"watermark": voice.watermark,
|
|
39
|
+
"sample": voice.sample,
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _model(model: Any) -> dict[str, Any]:
|
|
44
|
+
return {
|
|
45
|
+
"id": model.id,
|
|
46
|
+
"version": model.version,
|
|
47
|
+
"name": model.name,
|
|
48
|
+
"task": model.task,
|
|
49
|
+
"engine": model.engine,
|
|
50
|
+
"dtype": model.dtype,
|
|
51
|
+
"languages": list(model.languages),
|
|
52
|
+
"where": list(model.where),
|
|
53
|
+
"streaming": model.streaming,
|
|
54
|
+
"licence": model.licence.model_dump(exclude_defaults=True),
|
|
55
|
+
"attribution": " ".join(model.attribution.split()),
|
|
56
|
+
"upstream": model.source.upstream,
|
|
57
|
+
"revision": model.source.revision,
|
|
58
|
+
"files": [item.model_dump() for item in model.files],
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _py(value: Any) -> str:
|
|
63
|
+
"""A Python literal; `ruff format` lays it out."""
|
|
64
|
+
return repr(value)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def generate_python_code(
|
|
68
|
+
voices: list[dict[str, Any]], models: list[dict[str, Any]], allowed: list[str]
|
|
69
|
+
) -> str:
|
|
70
|
+
return "\n".join(
|
|
71
|
+
[
|
|
72
|
+
"# Copyright (c) 2025-2026 Datalayer, Inc.",
|
|
73
|
+
"# Distributed under the terms of the Modified BSD License.",
|
|
74
|
+
'"""',
|
|
75
|
+
"The voice catalogue (VOICE.md VO-40): voices and speech models.",
|
|
76
|
+
"",
|
|
77
|
+
"This file is AUTO-GENERATED from agentspecs (voices, speech-models).",
|
|
78
|
+
"DO NOT EDIT MANUALLY - run 'make specs' to regenerate.",
|
|
79
|
+
'"""',
|
|
80
|
+
"",
|
|
81
|
+
"from typing import Any, Dict, List, Optional",
|
|
82
|
+
"",
|
|
83
|
+
"#: The licences the register allows anywhere, the browser included.",
|
|
84
|
+
f"SPEECH_ALLOWED_LICENCES: List[str] = {_py(allowed)}",
|
|
85
|
+
"",
|
|
86
|
+
"#: The voices, by id.",
|
|
87
|
+
"VOICE_CATALOGUE: Dict[str, Dict[str, Any]] = {",
|
|
88
|
+
*[
|
|
89
|
+
f' "{voice["id"]}": {_py(voice)},'.replace("\n", "\n ")
|
|
90
|
+
for voice in voices
|
|
91
|
+
],
|
|
92
|
+
"}",
|
|
93
|
+
"",
|
|
94
|
+
"#: The speech models, by id, each file pinned by its SHA-256.",
|
|
95
|
+
"SPEECH_MODEL_CATALOGUE: Dict[str, Dict[str, Any]] = {",
|
|
96
|
+
*[
|
|
97
|
+
f' "{model["id"]}": {_py(model)},'.replace("\n", "\n ")
|
|
98
|
+
for model in models
|
|
99
|
+
],
|
|
100
|
+
"}",
|
|
101
|
+
"",
|
|
102
|
+
"",
|
|
103
|
+
"def get_voice_spec(voice_id: str) -> Optional[Dict[str, Any]]:",
|
|
104
|
+
' """A voice of the catalogue, or None."""',
|
|
105
|
+
" return VOICE_CATALOGUE.get(voice_id)",
|
|
106
|
+
"",
|
|
107
|
+
"",
|
|
108
|
+
"def get_speech_model_spec(model_id: str) -> Optional[Dict[str, Any]]:",
|
|
109
|
+
' """A speech model of the catalogue, or None."""',
|
|
110
|
+
" return SPEECH_MODEL_CATALOGUE.get(model_id)",
|
|
111
|
+
"",
|
|
112
|
+
"",
|
|
113
|
+
"def voice_speaks(voice: Dict[str, Any], language: str) -> bool:",
|
|
114
|
+
' """Whether a voice speaks a language: `fr` and `fr-FR` are spoken by a `fr-FR` voice."""',
|
|
115
|
+
" return any(own == language or own.split('-')[0] == language for own in voice['languages'])",
|
|
116
|
+
"",
|
|
117
|
+
"",
|
|
118
|
+
"def voice_for(language: str) -> Optional[Dict[str, Any]]:",
|
|
119
|
+
' """The first voice of the catalogue that speaks a language, or None."""',
|
|
120
|
+
" base = (language or '').strip()",
|
|
121
|
+
" exact = [v for v in VOICE_CATALOGUE.values() if base in v['languages']]",
|
|
122
|
+
" near = [v for v in VOICE_CATALOGUE.values() if voice_speaks(v, base.split('-')[0])]",
|
|
123
|
+
" return (exact or near or [None])[0]",
|
|
124
|
+
"",
|
|
125
|
+
]
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def generate_typescript_code(
|
|
130
|
+
voices: list[dict[str, Any]], models: list[dict[str, Any]], allowed: list[str]
|
|
131
|
+
) -> str:
|
|
132
|
+
def ts(value: Any) -> str:
|
|
133
|
+
return json.dumps(value, indent=2, ensure_ascii=False)
|
|
134
|
+
|
|
135
|
+
return "\n".join(
|
|
136
|
+
[
|
|
137
|
+
"/*",
|
|
138
|
+
" * Copyright (c) 2025-2026 Datalayer, Inc.",
|
|
139
|
+
" * Distributed under the terms of the Modified BSD License.",
|
|
140
|
+
" */",
|
|
141
|
+
"",
|
|
142
|
+
"/**",
|
|
143
|
+
" * The voice catalogue (VOICE.md VO-40): voices and speech models.",
|
|
144
|
+
" *",
|
|
145
|
+
" * This file is AUTO-GENERATED from agentspecs (voices, speech-models).",
|
|
146
|
+
" * DO NOT EDIT MANUALLY - run 'make specs' to regenerate.",
|
|
147
|
+
" *",
|
|
148
|
+
" * @module specs/voices",
|
|
149
|
+
" */",
|
|
150
|
+
"",
|
|
151
|
+
"/** Where a step of speech runs: the person's browser, or Datalayer's servers. */",
|
|
152
|
+
"export type SpeechWhere = 'device' | 'server';",
|
|
153
|
+
"",
|
|
154
|
+
"/** The licences a voice or a model is admitted by. */",
|
|
155
|
+
"export interface SpeechLicence {",
|
|
156
|
+
" weights: string;",
|
|
157
|
+
" code?: string;",
|
|
158
|
+
" dataset?: string;",
|
|
159
|
+
"}",
|
|
160
|
+
"",
|
|
161
|
+
"/** A voice an application may speak with. */",
|
|
162
|
+
"export interface VoiceSpec {",
|
|
163
|
+
" id: string;",
|
|
164
|
+
" version: string;",
|
|
165
|
+
" name: string;",
|
|
166
|
+
" description: string;",
|
|
167
|
+
" engine: string;",
|
|
168
|
+
" /** The speech model it is a voice of. */",
|
|
169
|
+
" model: string;",
|
|
170
|
+
" /** The engine's own name for it. */",
|
|
171
|
+
" voice: string;",
|
|
172
|
+
" /** BCP 47. */",
|
|
173
|
+
" languages: string[];",
|
|
174
|
+
" where: SpeechWhere[];",
|
|
175
|
+
" licence: SpeechLicence;",
|
|
176
|
+
" /** What a page listing the voices shows, when the licence asks. */",
|
|
177
|
+
" attribution: string;",
|
|
178
|
+
" watermark: boolean;",
|
|
179
|
+
" sample: string;",
|
|
180
|
+
"}",
|
|
181
|
+
"",
|
|
182
|
+
"/** One file of a model, pinned by its hash and its size. */",
|
|
183
|
+
"export interface PinnedFile {",
|
|
184
|
+
" path: string;",
|
|
185
|
+
" sha256: string;",
|
|
186
|
+
" size: number;",
|
|
187
|
+
"}",
|
|
188
|
+
"",
|
|
189
|
+
"/** A model of speech: to text, to speech, or voice activity. */",
|
|
190
|
+
"export interface SpeechModelSpec {",
|
|
191
|
+
" id: string;",
|
|
192
|
+
" version: string;",
|
|
193
|
+
" name: string;",
|
|
194
|
+
" task: 'stt' | 'tts' | 'vad';",
|
|
195
|
+
" engine: string;",
|
|
196
|
+
" dtype: string;",
|
|
197
|
+
" languages: string[];",
|
|
198
|
+
" where: SpeechWhere[];",
|
|
199
|
+
" streaming: boolean;",
|
|
200
|
+
" licence: SpeechLicence;",
|
|
201
|
+
" attribution: string;",
|
|
202
|
+
" upstream: string;",
|
|
203
|
+
" revision: string;",
|
|
204
|
+
" files: PinnedFile[];",
|
|
205
|
+
"}",
|
|
206
|
+
"",
|
|
207
|
+
"/** The licences the register allows anywhere, the browser included. */",
|
|
208
|
+
f"export const SPEECH_ALLOWED_LICENCES: readonly string[] = {ts(allowed)};",
|
|
209
|
+
"",
|
|
210
|
+
"export const VOICE_CATALOGUE: Record<string, VoiceSpec> = {",
|
|
211
|
+
*[
|
|
212
|
+
f" {json.dumps(voice['id'])}: {ts(voice)},".replace("\n", "\n ")
|
|
213
|
+
for voice in voices
|
|
214
|
+
],
|
|
215
|
+
"};",
|
|
216
|
+
"",
|
|
217
|
+
"export const SPEECH_MODEL_CATALOGUE: Record<string, SpeechModelSpec> = {",
|
|
218
|
+
*[
|
|
219
|
+
f" {json.dumps(model['id'])}: {ts(model)},".replace("\n", "\n ")
|
|
220
|
+
for model in models
|
|
221
|
+
],
|
|
222
|
+
"};",
|
|
223
|
+
"",
|
|
224
|
+
"/** Whether a voice speaks a language: `fr` and `fr-FR` are spoken by a `fr-FR` voice. */",
|
|
225
|
+
"export function voiceSpeaks(voice: VoiceSpec, language: string): boolean {",
|
|
226
|
+
" return voice.languages.some(",
|
|
227
|
+
" own => own === language || own.split('-')[0] === language,",
|
|
228
|
+
" );",
|
|
229
|
+
"}",
|
|
230
|
+
"",
|
|
231
|
+
"/** The voice a language is spoken with when none is said: the first that speaks it. */",
|
|
232
|
+
"export function voiceFor(language: string): VoiceSpec | undefined {",
|
|
233
|
+
" const voices = Object.values(VOICE_CATALOGUE);",
|
|
234
|
+
" return (",
|
|
235
|
+
" voices.find(voice => voice.languages.includes(language)) ??",
|
|
236
|
+
" voices.find(voice => voiceSpeaks(voice, language.split('-')[0]))",
|
|
237
|
+
" );",
|
|
238
|
+
"}",
|
|
239
|
+
"",
|
|
240
|
+
"/**",
|
|
241
|
+
" * The speech-to-text model for a language, on the device: Moonshine where it",
|
|
242
|
+
" * hears it, Whisper otherwise (VOICE.md decision 3).",
|
|
243
|
+
" */",
|
|
244
|
+
"export function transcriberFor(language: string): SpeechModelSpec | undefined {",
|
|
245
|
+
" const base = language.split('-')[0];",
|
|
246
|
+
" const hearing = Object.values(SPEECH_MODEL_CATALOGUE).filter(",
|
|
247
|
+
" model =>",
|
|
248
|
+
" model.task === 'stt' &&",
|
|
249
|
+
" model.where.includes('device') &&",
|
|
250
|
+
" model.languages.includes(base),",
|
|
251
|
+
" );",
|
|
252
|
+
" return (",
|
|
253
|
+
" hearing.find(model => model.id === 'moonshine-tiny-en') ??",
|
|
254
|
+
" hearing.find(model => model.id.startsWith('moonshine-')) ??",
|
|
255
|
+
" hearing[0]",
|
|
256
|
+
" );",
|
|
257
|
+
"}",
|
|
258
|
+
"",
|
|
259
|
+
]
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def main() -> None:
|
|
264
|
+
parser = argparse.ArgumentParser(
|
|
265
|
+
description="Generate the voice catalogue from agentspecs"
|
|
266
|
+
)
|
|
267
|
+
parser.add_argument("--python-output", type=Path, required=True)
|
|
268
|
+
parser.add_argument("--typescript-output", type=Path, required=True)
|
|
269
|
+
args = parser.parse_args()
|
|
270
|
+
voices = [_voice(voice) for voice in list_voices()]
|
|
271
|
+
models = [_model(model) for model in list_speech_models()]
|
|
272
|
+
allowed = list(register().allowed)
|
|
273
|
+
args.python_output.write_text(generate_python_code(voices, models, allowed))
|
|
274
|
+
args.typescript_output.write_text(generate_typescript_code(voices, models, allowed))
|
|
275
|
+
print(f"✓ Generated {len(voices)} voices and {len(models)} speech models")
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
if __name__ == "__main__":
|
|
279
|
+
main()
|
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# Copyright (c) 2025-2026 Datalayer, Inc.
|
|
3
|
+
# Distributed under the terms of the Modified BSD License.
|
|
4
|
+
|
|
5
|
+
"""Measure the speech models on the recorded fixtures (VOICE.md VO-05, VO-51, VO-52).
|
|
6
|
+
|
|
7
|
+
python scripts/voice/measure.py --store ~/.cache/speech-store
|
|
8
|
+
|
|
9
|
+
Writes ``tests/voice/measured/*.json``, which ``test_voice_wer.py`` checks
|
|
10
|
+
against ``tests/voice/baselines.json``:
|
|
11
|
+
|
|
12
|
+
- **speech to text**, each device model of the catalogue on the fixtures of
|
|
13
|
+
each language it hears, run as the browser runs it (transformers.js, the
|
|
14
|
+
same pinned files; in Node, so the timings are the CPU's, not the
|
|
15
|
+
browser's): what it heard, and how long it took;
|
|
16
|
+
- **text to speech**, each voice saying its sample and a few of the
|
|
17
|
+
product's sentences (Kokoro on the CPU, as ai-agents-speech runs it): how
|
|
18
|
+
long the synthesis took against the length of the speech (the real-time
|
|
19
|
+
factor), and what the device model of its language hears of it — the
|
|
20
|
+
intelligibility of the voice, a round trip.
|
|
21
|
+
|
|
22
|
+
Needs ``kokoro-onnx`` and ``soundfile`` (the speech service's own), and
|
|
23
|
+
``@huggingface/transformers`` where ``node`` resolves it (``--node-modules``
|
|
24
|
+
when it is not this checkout's). One heavy process at a time: the models run
|
|
25
|
+
one after the other.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import argparse
|
|
31
|
+
import importlib.util
|
|
32
|
+
import json
|
|
33
|
+
import os
|
|
34
|
+
import subprocess
|
|
35
|
+
import sys
|
|
36
|
+
import tempfile
|
|
37
|
+
import time
|
|
38
|
+
from pathlib import Path
|
|
39
|
+
from typing import Any, Dict, List
|
|
40
|
+
|
|
41
|
+
import numpy as np
|
|
42
|
+
import soundfile as sf
|
|
43
|
+
|
|
44
|
+
ROOT = Path(__file__).resolve().parents[2]
|
|
45
|
+
VOICE_TESTS = ROOT / "tests" / "voice"
|
|
46
|
+
|
|
47
|
+
#: Sentences the product says, spoken by each voice of their language.
|
|
48
|
+
PHRASES = {
|
|
49
|
+
"en": [
|
|
50
|
+
"Open the notebook and run the first cell.",
|
|
51
|
+
"Datalayer saved the report at three forty five.",
|
|
52
|
+
"Ask the agent to compare the two suppliers.",
|
|
53
|
+
],
|
|
54
|
+
"fr": [
|
|
55
|
+
"Ouvre le notebook et lance la première cellule.",
|
|
56
|
+
"Le rapport est prêt, je l'envoie à l'équipe.",
|
|
57
|
+
"Demande à l'agent de comparer les deux fournisseurs.",
|
|
58
|
+
],
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def catalogue() -> Any:
|
|
63
|
+
"""The generated catalogue, read without importing agent_runtimes."""
|
|
64
|
+
spec = importlib.util.spec_from_file_location(
|
|
65
|
+
"voices", ROOT / "agent_runtimes" / "specs" / "voices.py"
|
|
66
|
+
)
|
|
67
|
+
module = importlib.util.module_from_spec(spec) # type: ignore[arg-type]
|
|
68
|
+
spec.loader.exec_module(module) # type: ignore[union-attr]
|
|
69
|
+
return module
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def resample(samples: np.ndarray, rate: int, to: int = 16000) -> np.ndarray:
|
|
73
|
+
if rate == to:
|
|
74
|
+
return samples.astype(np.float32)
|
|
75
|
+
count = int(round(len(samples) * to / rate))
|
|
76
|
+
return np.interp(
|
|
77
|
+
np.linspace(0, len(samples) - 1, count), np.arange(len(samples)), samples
|
|
78
|
+
).astype(np.float32)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def transcribe(
|
|
82
|
+
store: Path, model: str, language: str, wavs: List[Path], env: Dict[str, str]
|
|
83
|
+
) -> Dict[str, Any]:
|
|
84
|
+
script = ROOT / "scripts" / "voice" / "transcribe.mjs"
|
|
85
|
+
done = subprocess.run(
|
|
86
|
+
["node", str(script), str(store), model, language, *map(str, wavs)],
|
|
87
|
+
capture_output=True,
|
|
88
|
+
text=True,
|
|
89
|
+
env=env,
|
|
90
|
+
check=True,
|
|
91
|
+
)
|
|
92
|
+
return json.loads(done.stdout.strip().splitlines()[-1])
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def main() -> int:
|
|
96
|
+
parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
|
|
97
|
+
parser.add_argument("--store", type=Path, required=True)
|
|
98
|
+
parser.add_argument(
|
|
99
|
+
"--node-modules",
|
|
100
|
+
type=Path,
|
|
101
|
+
default=None,
|
|
102
|
+
help="Where @huggingface/transformers is installed, when not in this checkout",
|
|
103
|
+
)
|
|
104
|
+
args = parser.parse_args()
|
|
105
|
+
voices = catalogue()
|
|
106
|
+
env = dict(os.environ)
|
|
107
|
+
script_dir = ROOT / "scripts" / "voice"
|
|
108
|
+
linked = None
|
|
109
|
+
if args.node_modules is not None:
|
|
110
|
+
# Node resolves a package from the script's folders up: lend it one.
|
|
111
|
+
linked = script_dir / "node_modules"
|
|
112
|
+
if not linked.exists():
|
|
113
|
+
linked.symlink_to(args.node_modules)
|
|
114
|
+
fixtures = json.loads((VOICE_TESTS / "fixtures.json").read_text())["clips"]
|
|
115
|
+
measured = VOICE_TESTS / "measured"
|
|
116
|
+
measured.mkdir(exist_ok=True)
|
|
117
|
+
try:
|
|
118
|
+
with tempfile.TemporaryDirectory() as scratch:
|
|
119
|
+
wavs: Dict[str, Path] = {}
|
|
120
|
+
for clip in fixtures:
|
|
121
|
+
samples, rate = sf.read(
|
|
122
|
+
VOICE_TESTS / "fixtures" / clip["file"], dtype="float32"
|
|
123
|
+
)
|
|
124
|
+
wav = Path(scratch) / f"{clip['id']}.wav"
|
|
125
|
+
sf.write(wav, resample(samples, rate), 16000, subtype="PCM_16")
|
|
126
|
+
wavs[clip["id"]] = wav
|
|
127
|
+
# --- speech to text, on the device's models ---------------------------------
|
|
128
|
+
for model in voices.SPEECH_MODEL_CATALOGUE.values():
|
|
129
|
+
if model["task"] != "stt" or "device" not in model["where"]:
|
|
130
|
+
continue
|
|
131
|
+
by_language = {}
|
|
132
|
+
for language in model["languages"]:
|
|
133
|
+
clips = [clip for clip in fixtures if clip["language"] == language]
|
|
134
|
+
if not clips:
|
|
135
|
+
continue
|
|
136
|
+
print(f"{model['id']} on {len(clips)} {language} clips", flush=True)
|
|
137
|
+
heard = transcribe(
|
|
138
|
+
args.store,
|
|
139
|
+
model["id"],
|
|
140
|
+
language,
|
|
141
|
+
[wavs[c["id"]] for c in clips],
|
|
142
|
+
env,
|
|
143
|
+
)
|
|
144
|
+
by_language[language] = {
|
|
145
|
+
"load_ms": heard["load_ms"],
|
|
146
|
+
"clips": [
|
|
147
|
+
{
|
|
148
|
+
"id": clip["id"],
|
|
149
|
+
"heard": result["text"],
|
|
150
|
+
"ms": result["ms"],
|
|
151
|
+
"audio_s": result["audio_s"],
|
|
152
|
+
}
|
|
153
|
+
for clip, result in zip(clips, heard["results"])
|
|
154
|
+
],
|
|
155
|
+
}
|
|
156
|
+
(measured / f"stt-{model['id']}.json").write_text(
|
|
157
|
+
json.dumps(
|
|
158
|
+
{
|
|
159
|
+
"model": model["id"],
|
|
160
|
+
"runtime": "transformers.js 4.3.0 on onnxruntime-node, CPU",
|
|
161
|
+
"languages": by_language,
|
|
162
|
+
},
|
|
163
|
+
indent=2,
|
|
164
|
+
ensure_ascii=False,
|
|
165
|
+
)
|
|
166
|
+
+ "\n"
|
|
167
|
+
)
|
|
168
|
+
# --- text to speech, and what the device hears of it -----------------------
|
|
169
|
+
from kokoro_onnx import Kokoro
|
|
170
|
+
|
|
171
|
+
kokoro_dir = args.store / "kokoro-82m"
|
|
172
|
+
files = {
|
|
173
|
+
item["path"]
|
|
174
|
+
for item in voices.SPEECH_MODEL_CATALOGUE["kokoro-82m"]["files"]
|
|
175
|
+
}
|
|
176
|
+
onnx = next(name for name in files if name.endswith(".onnx"))
|
|
177
|
+
started = time.monotonic()
|
|
178
|
+
kokoro = Kokoro(str(kokoro_dir / onnx), str(kokoro_dir / "voices-v1.0.bin"))
|
|
179
|
+
load_ms = round((time.monotonic() - started) * 1000)
|
|
180
|
+
spoken: List[Dict[str, Any]] = []
|
|
181
|
+
for voice in voices.VOICE_CATALOGUE.values():
|
|
182
|
+
language = voice["languages"][0]
|
|
183
|
+
base = language.split("-")[0]
|
|
184
|
+
for index, text in enumerate([voice["sample"], *PHRASES.get(base, [])]):
|
|
185
|
+
started = time.monotonic()
|
|
186
|
+
samples, rate = kokoro.create(
|
|
187
|
+
text, voice=voice["voice"], lang=language.lower()
|
|
188
|
+
)
|
|
189
|
+
took = time.monotonic() - started
|
|
190
|
+
wav = Path(scratch) / f"{voice['id']}-{index}.wav"
|
|
191
|
+
sf.write(wav, resample(samples, rate), 16000, subtype="PCM_16")
|
|
192
|
+
spoken.append(
|
|
193
|
+
{
|
|
194
|
+
"voice": voice["id"],
|
|
195
|
+
"language": base,
|
|
196
|
+
"text": text,
|
|
197
|
+
"wav": wav,
|
|
198
|
+
"synthesis_ms": round(took * 1000),
|
|
199
|
+
"audio_s": round(len(samples) / rate, 2),
|
|
200
|
+
}
|
|
201
|
+
)
|
|
202
|
+
round_trips = []
|
|
203
|
+
for base in sorted({item["language"] for item in spoken}):
|
|
204
|
+
model = voices.SPEECH_MODEL_CATALOGUE[
|
|
205
|
+
"moonshine-tiny-en" if base == "en" else "whisper-base"
|
|
206
|
+
]
|
|
207
|
+
items = [item for item in spoken if item["language"] == base]
|
|
208
|
+
heard = transcribe(
|
|
209
|
+
args.store, model["id"], base, [item["wav"] for item in items], env
|
|
210
|
+
)
|
|
211
|
+
for item, result in zip(items, heard["results"]):
|
|
212
|
+
round_trips.append(
|
|
213
|
+
{
|
|
214
|
+
**{
|
|
215
|
+
key: value
|
|
216
|
+
for key, value in item.items()
|
|
217
|
+
if key != "wav"
|
|
218
|
+
},
|
|
219
|
+
"heard_by": model["id"],
|
|
220
|
+
"heard": result["text"],
|
|
221
|
+
}
|
|
222
|
+
)
|
|
223
|
+
(measured / "tts-kokoro-82m.json").write_text(
|
|
224
|
+
json.dumps(
|
|
225
|
+
{
|
|
226
|
+
"model": "kokoro-82m",
|
|
227
|
+
"runtime": "kokoro-onnx 0.6.1, onnxruntime CPU",
|
|
228
|
+
"load_ms": load_ms,
|
|
229
|
+
"spoken": round_trips,
|
|
230
|
+
},
|
|
231
|
+
indent=2,
|
|
232
|
+
ensure_ascii=False,
|
|
233
|
+
)
|
|
234
|
+
+ "\n"
|
|
235
|
+
)
|
|
236
|
+
finally:
|
|
237
|
+
if linked is not None and linked.is_symlink():
|
|
238
|
+
linked.unlink()
|
|
239
|
+
print(f"wrote {measured}")
|
|
240
|
+
return 0
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
if __name__ == "__main__":
|
|
244
|
+
sys.exit(main())
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# Copyright (c) 2025-2026 Datalayer, Inc.
|
|
3
|
+
# Distributed under the terms of the Modified BSD License.
|
|
4
|
+
|
|
5
|
+
"""The pinned-file store of the speech models (VOICE.md VO-03, VO-48, VO-49).
|
|
6
|
+
|
|
7
|
+
Every file of every speech model of the catalogue, laid out as
|
|
8
|
+
``<store>/<model id>/<path>`` and checked against the SHA-256 and the size
|
|
9
|
+
the catalogue pins. The store is filled **once**, from where each model was
|
|
10
|
+
published, and then copied to Datalayer's storage; the browser and the speech
|
|
11
|
+
service read it from there and never from a third party's hub.
|
|
12
|
+
|
|
13
|
+
python scripts/voice/pin_store.py --store ~/.cache/speech-store [--from DIR]
|
|
14
|
+
python scripts/voice/pin_store.py --store ~/.cache/speech-store --verify
|
|
15
|
+
|
|
16
|
+
``--from`` takes files already downloaded (any layout) whose hash matches,
|
|
17
|
+
instead of downloading them again. ``--verify`` downloads nothing and fails
|
|
18
|
+
on a missing or changed file.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import argparse
|
|
24
|
+
import hashlib
|
|
25
|
+
import shutil
|
|
26
|
+
import sys
|
|
27
|
+
import tarfile
|
|
28
|
+
import tempfile
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
from typing import Any, Dict, Iterator, Optional
|
|
31
|
+
|
|
32
|
+
import httpx
|
|
33
|
+
|
|
34
|
+
from agent_runtimes.specs.voices import SPEECH_MODEL_CATALOGUE
|
|
35
|
+
|
|
36
|
+
#: Where the npm package that ships Silero VAD is published.
|
|
37
|
+
NPM_TARBALL = "https://registry.npmjs.org/@ricky0123/vad-web/-/vad-web-{revision}.tgz"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def sha256_of(path: Path) -> str:
|
|
41
|
+
digest = hashlib.sha256()
|
|
42
|
+
with path.open("rb") as stream:
|
|
43
|
+
for block in iter(lambda: stream.read(1 << 20), b""):
|
|
44
|
+
digest.update(block)
|
|
45
|
+
return digest.hexdigest()
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def upstream_url(model: Dict[str, Any], path: str) -> Optional[str]:
|
|
49
|
+
"""Where a file was published: a Hugging Face revision or a GitHub release; None for npm."""
|
|
50
|
+
repository = _repository(model)
|
|
51
|
+
if repository.startswith("https://huggingface.co/"):
|
|
52
|
+
return f"{repository}/resolve/{model['revision']}/{path}"
|
|
53
|
+
if "/releases/tag/" in repository:
|
|
54
|
+
return f"{repository.replace('/releases/tag/', '/releases/download/')}/{path}"
|
|
55
|
+
return None
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _repository(model: Dict[str, Any]) -> str:
|
|
59
|
+
from agentspecs.speech import get_speech_model
|
|
60
|
+
|
|
61
|
+
spec = get_speech_model(model["id"])
|
|
62
|
+
if spec is None:
|
|
63
|
+
raise SystemExit(
|
|
64
|
+
f"{model['id']} is in the generated catalogue and not in agentspecs: run make specs"
|
|
65
|
+
)
|
|
66
|
+
return spec.source.repository
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _found(local: Optional[Path], sha256: str, size: int) -> Optional[Path]:
|
|
70
|
+
if local is None:
|
|
71
|
+
return None
|
|
72
|
+
for candidate in local.rglob("*"):
|
|
73
|
+
if (
|
|
74
|
+
candidate.is_file()
|
|
75
|
+
and candidate.stat().st_size == size
|
|
76
|
+
and sha256_of(candidate) == sha256
|
|
77
|
+
):
|
|
78
|
+
return candidate
|
|
79
|
+
return None
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _npm_file(model: Dict[str, Any], path: str, into: Path) -> None:
|
|
83
|
+
with tempfile.TemporaryDirectory() as scratch:
|
|
84
|
+
tarball = Path(scratch) / "package.tgz"
|
|
85
|
+
_download(NPM_TARBALL.format(revision=model["revision"]), tarball)
|
|
86
|
+
with tarfile.open(tarball) as archive:
|
|
87
|
+
member = archive.getmember(f"package/dist/{path}")
|
|
88
|
+
extracted = archive.extractfile(member)
|
|
89
|
+
if extracted is None:
|
|
90
|
+
raise SystemExit(f"{path} is not a file of the vad-web package")
|
|
91
|
+
into.write_bytes(extracted.read())
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _download(url: str, target: Path) -> None:
|
|
95
|
+
"""Copy a published file, once, from where it was published (HTTPS only)."""
|
|
96
|
+
if not url.startswith("https://"):
|
|
97
|
+
raise SystemExit(f"{url} is not an HTTPS address: refused.")
|
|
98
|
+
with httpx.stream("GET", url, follow_redirects=True, timeout=120.0) as answered:
|
|
99
|
+
answered.raise_for_status()
|
|
100
|
+
with target.open("wb") as stream:
|
|
101
|
+
for block in answered.iter_bytes(1 << 20):
|
|
102
|
+
stream.write(block)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def files() -> Iterator[tuple[Dict[str, Any], Dict[str, Any]]]:
|
|
106
|
+
for model in SPEECH_MODEL_CATALOGUE.values():
|
|
107
|
+
for item in model["files"]:
|
|
108
|
+
yield model, item
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def main() -> int:
|
|
112
|
+
parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
|
|
113
|
+
parser.add_argument("--store", type=Path, required=True)
|
|
114
|
+
parser.add_argument("--from", dest="local", type=Path, default=None)
|
|
115
|
+
parser.add_argument("--verify", action="store_true")
|
|
116
|
+
args = parser.parse_args()
|
|
117
|
+
problems = []
|
|
118
|
+
total = 0
|
|
119
|
+
for model, item in files():
|
|
120
|
+
target = args.store / model["id"] / item["path"]
|
|
121
|
+
total += item["size"]
|
|
122
|
+
if (
|
|
123
|
+
target.is_file()
|
|
124
|
+
and target.stat().st_size == item["size"]
|
|
125
|
+
and sha256_of(target) == item["sha256"]
|
|
126
|
+
):
|
|
127
|
+
continue
|
|
128
|
+
if args.verify:
|
|
129
|
+
problems.append(
|
|
130
|
+
f"{model['id']}/{item['path']} is missing or does not match its pin"
|
|
131
|
+
)
|
|
132
|
+
continue
|
|
133
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
134
|
+
local = _found(args.local, item["sha256"], item["size"])
|
|
135
|
+
if local is not None:
|
|
136
|
+
shutil.copyfile(local, target)
|
|
137
|
+
else:
|
|
138
|
+
url = upstream_url(model, item["path"])
|
|
139
|
+
print(f"downloading {model['id']}/{item['path']}", flush=True)
|
|
140
|
+
if url is None:
|
|
141
|
+
_npm_file(model, item["path"], target)
|
|
142
|
+
else:
|
|
143
|
+
_download(url, target)
|
|
144
|
+
if sha256_of(target) != item["sha256"]:
|
|
145
|
+
target.unlink()
|
|
146
|
+
problems.append(
|
|
147
|
+
f"{model['id']}/{item['path']} was published with another hash: refused"
|
|
148
|
+
)
|
|
149
|
+
for problem in problems:
|
|
150
|
+
print(problem, file=sys.stderr)
|
|
151
|
+
if not problems:
|
|
152
|
+
print(
|
|
153
|
+
f"{args.store}: {sum(1 for _ in files())} files, {total / 1e6:.0f} MB, every one as pinned"
|
|
154
|
+
)
|
|
155
|
+
return 1 if problems else 0
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
if __name__ == "__main__":
|
|
159
|
+
raise SystemExit(main())
|