@datalayer/agent-runtimes 1.3.62 → 1.3.64

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/lib/chat/ChatFloating.d.ts +9 -1
  2. package/lib/chat/ChatFloating.js +69 -18
  3. package/lib/chat/assistant/AssistantStage.d.ts +14 -1
  4. package/lib/chat/assistant/AssistantStage.js +53 -1
  5. package/lib/chat/assistant/SpriteCharacter.js +1 -1
  6. package/lib/chat/assistant/characters.js +1 -5
  7. package/lib/chat/assistant/state.d.ts +7 -1
  8. package/lib/chat/assistant/state.js +3 -2
  9. package/lib/chat/base/ChatBase.js +32 -4
  10. package/lib/chat/messages/ChatMessageList.d.ts +0 -6
  11. package/lib/chat/messages/ChatMessageList.js +8 -2
  12. package/lib/config/AgentConfiguration.js +0 -6
  13. package/lib/examples/AgentA2ATeamExample.js +19 -2
  14. package/lib/examples/ChatAssistantExample.d.ts +3 -1
  15. package/lib/examples/ChatAssistantExample.js +75 -7
  16. package/lib/examples/ChatAssistantGalleryExample.d.ts +3 -2
  17. package/lib/examples/ChatAssistantGalleryExample.js +11 -6
  18. package/lib/examples/DecksAgent.js +4 -2
  19. package/lib/examples/LoopShellExample.js +4 -2
  20. package/lib/examples/VoiceChatExample.d.ts +20 -0
  21. package/lib/examples/VoiceChatExample.js +64 -0
  22. package/lib/examples/example-selector.js +1 -0
  23. package/lib/examples/main.js +12 -7
  24. package/lib/examples/utils/clippyJsCharacters.d.ts +18 -0
  25. package/lib/examples/utils/clippyJsCharacters.js +129 -0
  26. package/lib/loop/apps/appspec.d.ts +3 -1
  27. package/lib/loop/apps/appspec.js +31 -0
  28. package/lib/loop/apps/checks.js +1 -1
  29. package/lib/protocols/VercelAIAdapter.js +5 -0
  30. package/lib/specs/apps.js +112 -0
  31. package/lib/specs/appspecSchema.js +65 -0
  32. package/lib/specs/index.d.ts +1 -0
  33. package/lib/specs/index.js +1 -0
  34. package/lib/specs/voices.d.ts +72 -0
  35. package/lib/specs/voices.js +344 -0
  36. package/lib/types/agents.d.ts +1 -1
  37. package/lib/types/agentspecs.d.ts +16 -0
  38. package/lib/types/chat.d.ts +11 -0
  39. package/lib/voice/VoiceInput.d.ts +35 -0
  40. package/lib/voice/VoiceInput.js +233 -0
  41. package/lib/voice/capture.d.ts +30 -0
  42. package/lib/voice/capture.js +113 -0
  43. package/lib/voice/consent.d.ts +6 -0
  44. package/lib/voice/consent.js +34 -0
  45. package/lib/voice/hearing.d.ts +37 -0
  46. package/lib/voice/hearing.js +168 -0
  47. package/lib/voice/index.d.ts +47 -0
  48. package/lib/voice/index.js +13 -0
  49. package/lib/voice/pinned.d.ts +28 -0
  50. package/lib/voice/pinned.js +75 -0
  51. package/lib/voice/sentences.d.ts +26 -0
  52. package/lib/voice/sentences.js +71 -0
  53. package/lib/voice/speaker.d.ts +66 -0
  54. package/lib/voice/speaker.js +168 -0
  55. package/lib/voice/types.d.ts +92 -0
  56. package/lib/voice/types.js +5 -0
  57. package/lib/voice/useSpokenAnswers.d.ts +12 -0
  58. package/lib/voice/useSpokenAnswers.js +90 -0
  59. package/package.json +7 -3
  60. package/scripts/codegen/generate_voices.py +279 -0
  61. package/scripts/voice/measure.py +244 -0
  62. package/scripts/voice/pin_store.py +159 -0
  63. package/scripts/voice/serve_store.py +70 -0
  64. package/scripts/voice/transcribe.mjs +58 -0
@@ -0,0 +1,279 @@
1
+ #!/usr/bin/env python3
2
+ # Copyright (c) 2025-2026 Datalayer, Inc.
3
+ # Distributed under the terms of the Modified BSD License.
4
+
5
+ """
6
+ Generate the voice catalogue (VOICE.md VO-40) in Python and TypeScript.
7
+
8
+ Read through ``agentspecs.speech``, which refuses a voice or a model whose
9
+ licence the register does not allow: what is generated has passed it.
10
+
11
+ Usage:
12
+ python generate_voices.py \\
13
+ --python-output agent_runtimes/specs/voices.py \\
14
+ --typescript-output src/specs/voices.ts
15
+ """
16
+
17
+ import argparse
18
+ import json
19
+ from pathlib import Path
20
+ from typing import Any
21
+
22
+ from agentspecs.speech import list_speech_models, list_voices, register
23
+
24
+
25
+ def _voice(voice: Any) -> dict[str, Any]:
26
+ return {
27
+ "id": voice.id,
28
+ "version": voice.version,
29
+ "name": voice.name,
30
+ "description": " ".join(voice.description.split()),
31
+ "engine": voice.engine,
32
+ "model": voice.model,
33
+ "voice": voice.voice,
34
+ "languages": list(voice.languages),
35
+ "where": list(voice.where),
36
+ "licence": voice.licence.model_dump(exclude_defaults=True),
37
+ "attribution": " ".join(voice.attribution.split()),
38
+ "watermark": voice.watermark,
39
+ "sample": voice.sample,
40
+ }
41
+
42
+
43
+ def _model(model: Any) -> dict[str, Any]:
44
+ return {
45
+ "id": model.id,
46
+ "version": model.version,
47
+ "name": model.name,
48
+ "task": model.task,
49
+ "engine": model.engine,
50
+ "dtype": model.dtype,
51
+ "languages": list(model.languages),
52
+ "where": list(model.where),
53
+ "streaming": model.streaming,
54
+ "licence": model.licence.model_dump(exclude_defaults=True),
55
+ "attribution": " ".join(model.attribution.split()),
56
+ "upstream": model.source.upstream,
57
+ "revision": model.source.revision,
58
+ "files": [item.model_dump() for item in model.files],
59
+ }
60
+
61
+
62
+ def _py(value: Any) -> str:
63
+ """A Python literal; `ruff format` lays it out."""
64
+ return repr(value)
65
+
66
+
67
+ def generate_python_code(
68
+ voices: list[dict[str, Any]], models: list[dict[str, Any]], allowed: list[str]
69
+ ) -> str:
70
+ return "\n".join(
71
+ [
72
+ "# Copyright (c) 2025-2026 Datalayer, Inc.",
73
+ "# Distributed under the terms of the Modified BSD License.",
74
+ '"""',
75
+ "The voice catalogue (VOICE.md VO-40): voices and speech models.",
76
+ "",
77
+ "This file is AUTO-GENERATED from agentspecs (voices, speech-models).",
78
+ "DO NOT EDIT MANUALLY - run 'make specs' to regenerate.",
79
+ '"""',
80
+ "",
81
+ "from typing import Any, Dict, List, Optional",
82
+ "",
83
+ "#: The licences the register allows anywhere, the browser included.",
84
+ f"SPEECH_ALLOWED_LICENCES: List[str] = {_py(allowed)}",
85
+ "",
86
+ "#: The voices, by id.",
87
+ "VOICE_CATALOGUE: Dict[str, Dict[str, Any]] = {",
88
+ *[
89
+ f' "{voice["id"]}": {_py(voice)},'.replace("\n", "\n ")
90
+ for voice in voices
91
+ ],
92
+ "}",
93
+ "",
94
+ "#: The speech models, by id, each file pinned by its SHA-256.",
95
+ "SPEECH_MODEL_CATALOGUE: Dict[str, Dict[str, Any]] = {",
96
+ *[
97
+ f' "{model["id"]}": {_py(model)},'.replace("\n", "\n ")
98
+ for model in models
99
+ ],
100
+ "}",
101
+ "",
102
+ "",
103
+ "def get_voice_spec(voice_id: str) -> Optional[Dict[str, Any]]:",
104
+ ' """A voice of the catalogue, or None."""',
105
+ " return VOICE_CATALOGUE.get(voice_id)",
106
+ "",
107
+ "",
108
+ "def get_speech_model_spec(model_id: str) -> Optional[Dict[str, Any]]:",
109
+ ' """A speech model of the catalogue, or None."""',
110
+ " return SPEECH_MODEL_CATALOGUE.get(model_id)",
111
+ "",
112
+ "",
113
+ "def voice_speaks(voice: Dict[str, Any], language: str) -> bool:",
114
+ ' """Whether a voice speaks a language: `fr` and `fr-FR` are spoken by a `fr-FR` voice."""',
115
+ " return any(own == language or own.split('-')[0] == language for own in voice['languages'])",
116
+ "",
117
+ "",
118
+ "def voice_for(language: str) -> Optional[Dict[str, Any]]:",
119
+ ' """The first voice of the catalogue that speaks a language, or None."""',
120
+ " base = (language or '').strip()",
121
+ " exact = [v for v in VOICE_CATALOGUE.values() if base in v['languages']]",
122
+ " near = [v for v in VOICE_CATALOGUE.values() if voice_speaks(v, base.split('-')[0])]",
123
+ " return (exact or near or [None])[0]",
124
+ "",
125
+ ]
126
+ )
127
+
128
+
129
+ def generate_typescript_code(
130
+ voices: list[dict[str, Any]], models: list[dict[str, Any]], allowed: list[str]
131
+ ) -> str:
132
+ def ts(value: Any) -> str:
133
+ return json.dumps(value, indent=2, ensure_ascii=False)
134
+
135
+ return "\n".join(
136
+ [
137
+ "/*",
138
+ " * Copyright (c) 2025-2026 Datalayer, Inc.",
139
+ " * Distributed under the terms of the Modified BSD License.",
140
+ " */",
141
+ "",
142
+ "/**",
143
+ " * The voice catalogue (VOICE.md VO-40): voices and speech models.",
144
+ " *",
145
+ " * This file is AUTO-GENERATED from agentspecs (voices, speech-models).",
146
+ " * DO NOT EDIT MANUALLY - run 'make specs' to regenerate.",
147
+ " *",
148
+ " * @module specs/voices",
149
+ " */",
150
+ "",
151
+ "/** Where a step of speech runs: the person's browser, or Datalayer's servers. */",
152
+ "export type SpeechWhere = 'device' | 'server';",
153
+ "",
154
+ "/** The licences a voice or a model is admitted by. */",
155
+ "export interface SpeechLicence {",
156
+ " weights: string;",
157
+ " code?: string;",
158
+ " dataset?: string;",
159
+ "}",
160
+ "",
161
+ "/** A voice an application may speak with. */",
162
+ "export interface VoiceSpec {",
163
+ " id: string;",
164
+ " version: string;",
165
+ " name: string;",
166
+ " description: string;",
167
+ " engine: string;",
168
+ " /** The speech model it is a voice of. */",
169
+ " model: string;",
170
+ " /** The engine's own name for it. */",
171
+ " voice: string;",
172
+ " /** BCP 47. */",
173
+ " languages: string[];",
174
+ " where: SpeechWhere[];",
175
+ " licence: SpeechLicence;",
176
+ " /** What a page listing the voices shows, when the licence asks. */",
177
+ " attribution: string;",
178
+ " watermark: boolean;",
179
+ " sample: string;",
180
+ "}",
181
+ "",
182
+ "/** One file of a model, pinned by its hash and its size. */",
183
+ "export interface PinnedFile {",
184
+ " path: string;",
185
+ " sha256: string;",
186
+ " size: number;",
187
+ "}",
188
+ "",
189
+ "/** A model of speech: to text, to speech, or voice activity. */",
190
+ "export interface SpeechModelSpec {",
191
+ " id: string;",
192
+ " version: string;",
193
+ " name: string;",
194
+ " task: 'stt' | 'tts' | 'vad';",
195
+ " engine: string;",
196
+ " dtype: string;",
197
+ " languages: string[];",
198
+ " where: SpeechWhere[];",
199
+ " streaming: boolean;",
200
+ " licence: SpeechLicence;",
201
+ " attribution: string;",
202
+ " upstream: string;",
203
+ " revision: string;",
204
+ " files: PinnedFile[];",
205
+ "}",
206
+ "",
207
+ "/** The licences the register allows anywhere, the browser included. */",
208
+ f"export const SPEECH_ALLOWED_LICENCES: readonly string[] = {ts(allowed)};",
209
+ "",
210
+ "export const VOICE_CATALOGUE: Record<string, VoiceSpec> = {",
211
+ *[
212
+ f" {json.dumps(voice['id'])}: {ts(voice)},".replace("\n", "\n ")
213
+ for voice in voices
214
+ ],
215
+ "};",
216
+ "",
217
+ "export const SPEECH_MODEL_CATALOGUE: Record<string, SpeechModelSpec> = {",
218
+ *[
219
+ f" {json.dumps(model['id'])}: {ts(model)},".replace("\n", "\n ")
220
+ for model in models
221
+ ],
222
+ "};",
223
+ "",
224
+ "/** Whether a voice speaks a language: `fr` and `fr-FR` are spoken by a `fr-FR` voice. */",
225
+ "export function voiceSpeaks(voice: VoiceSpec, language: string): boolean {",
226
+ " return voice.languages.some(",
227
+ " own => own === language || own.split('-')[0] === language,",
228
+ " );",
229
+ "}",
230
+ "",
231
+ "/** The voice a language is spoken with when none is said: the first that speaks it. */",
232
+ "export function voiceFor(language: string): VoiceSpec | undefined {",
233
+ " const voices = Object.values(VOICE_CATALOGUE);",
234
+ " return (",
235
+ " voices.find(voice => voice.languages.includes(language)) ??",
236
+ " voices.find(voice => voiceSpeaks(voice, language.split('-')[0]))",
237
+ " );",
238
+ "}",
239
+ "",
240
+ "/**",
241
+ " * The speech-to-text model for a language, on the device: Moonshine where it",
242
+ " * hears it, Whisper otherwise (VOICE.md decision 3).",
243
+ " */",
244
+ "export function transcriberFor(language: string): SpeechModelSpec | undefined {",
245
+ " const base = language.split('-')[0];",
246
+ " const hearing = Object.values(SPEECH_MODEL_CATALOGUE).filter(",
247
+ " model =>",
248
+ " model.task === 'stt' &&",
249
+ " model.where.includes('device') &&",
250
+ " model.languages.includes(base),",
251
+ " );",
252
+ " return (",
253
+ " hearing.find(model => model.id === 'moonshine-tiny-en') ??",
254
+ " hearing.find(model => model.id.startsWith('moonshine-')) ??",
255
+ " hearing[0]",
256
+ " );",
257
+ "}",
258
+ "",
259
+ ]
260
+ )
261
+
262
+
263
+ def main() -> None:
264
+ parser = argparse.ArgumentParser(
265
+ description="Generate the voice catalogue from agentspecs"
266
+ )
267
+ parser.add_argument("--python-output", type=Path, required=True)
268
+ parser.add_argument("--typescript-output", type=Path, required=True)
269
+ args = parser.parse_args()
270
+ voices = [_voice(voice) for voice in list_voices()]
271
+ models = [_model(model) for model in list_speech_models()]
272
+ allowed = list(register().allowed)
273
+ args.python_output.write_text(generate_python_code(voices, models, allowed))
274
+ args.typescript_output.write_text(generate_typescript_code(voices, models, allowed))
275
+ print(f"✓ Generated {len(voices)} voices and {len(models)} speech models")
276
+
277
+
278
+ if __name__ == "__main__":
279
+ main()
@@ -0,0 +1,244 @@
1
+ #!/usr/bin/env python3
2
+ # Copyright (c) 2025-2026 Datalayer, Inc.
3
+ # Distributed under the terms of the Modified BSD License.
4
+
5
+ """Measure the speech models on the recorded fixtures (VOICE.md VO-05, VO-51, VO-52).
6
+
7
+ python scripts/voice/measure.py --store ~/.cache/speech-store
8
+
9
+ Writes ``tests/voice/measured/*.json``, which ``test_voice_wer.py`` checks
10
+ against ``tests/voice/baselines.json``:
11
+
12
+ - **speech to text**, each device model of the catalogue on the fixtures of
13
+ each language it hears, run as the browser runs it (transformers.js, the
14
+ same pinned files; in Node, so the timings are the CPU's, not the
15
+ browser's): what it heard, and how long it took;
16
+ - **text to speech**, each voice saying its sample and a few of the
17
+ product's sentences (Kokoro on the CPU, as ai-agents-speech runs it): how
18
+ long the synthesis took against the length of the speech (the real-time
19
+ factor), and what the device model of its language hears of it — the
20
+ intelligibility of the voice, a round trip.
21
+
22
+ Needs ``kokoro-onnx`` and ``soundfile`` (the speech service's own), and
23
+ ``@huggingface/transformers`` where ``node`` resolves it (``--node-modules``
24
+ when it is not this checkout's). One heavy process at a time: the models run
25
+ one after the other.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import argparse
31
+ import importlib.util
32
+ import json
33
+ import os
34
+ import subprocess
35
+ import sys
36
+ import tempfile
37
+ import time
38
+ from pathlib import Path
39
+ from typing import Any, Dict, List
40
+
41
+ import numpy as np
42
+ import soundfile as sf
43
+
44
+ ROOT = Path(__file__).resolve().parents[2]
45
+ VOICE_TESTS = ROOT / "tests" / "voice"
46
+
47
+ #: Sentences the product says, spoken by each voice of their language.
48
+ PHRASES = {
49
+ "en": [
50
+ "Open the notebook and run the first cell.",
51
+ "Datalayer saved the report at three forty five.",
52
+ "Ask the agent to compare the two suppliers.",
53
+ ],
54
+ "fr": [
55
+ "Ouvre le notebook et lance la première cellule.",
56
+ "Le rapport est prêt, je l'envoie à l'équipe.",
57
+ "Demande à l'agent de comparer les deux fournisseurs.",
58
+ ],
59
+ }
60
+
61
+
62
+ def catalogue() -> Any:
63
+ """The generated catalogue, read without importing agent_runtimes."""
64
+ spec = importlib.util.spec_from_file_location(
65
+ "voices", ROOT / "agent_runtimes" / "specs" / "voices.py"
66
+ )
67
+ module = importlib.util.module_from_spec(spec) # type: ignore[arg-type]
68
+ spec.loader.exec_module(module) # type: ignore[union-attr]
69
+ return module
70
+
71
+
72
+ def resample(samples: np.ndarray, rate: int, to: int = 16000) -> np.ndarray:
73
+ if rate == to:
74
+ return samples.astype(np.float32)
75
+ count = int(round(len(samples) * to / rate))
76
+ return np.interp(
77
+ np.linspace(0, len(samples) - 1, count), np.arange(len(samples)), samples
78
+ ).astype(np.float32)
79
+
80
+
81
+ def transcribe(
82
+ store: Path, model: str, language: str, wavs: List[Path], env: Dict[str, str]
83
+ ) -> Dict[str, Any]:
84
+ script = ROOT / "scripts" / "voice" / "transcribe.mjs"
85
+ done = subprocess.run(
86
+ ["node", str(script), str(store), model, language, *map(str, wavs)],
87
+ capture_output=True,
88
+ text=True,
89
+ env=env,
90
+ check=True,
91
+ )
92
+ return json.loads(done.stdout.strip().splitlines()[-1])
93
+
94
+
95
+ def main() -> int:
96
+ parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
97
+ parser.add_argument("--store", type=Path, required=True)
98
+ parser.add_argument(
99
+ "--node-modules",
100
+ type=Path,
101
+ default=None,
102
+ help="Where @huggingface/transformers is installed, when not in this checkout",
103
+ )
104
+ args = parser.parse_args()
105
+ voices = catalogue()
106
+ env = dict(os.environ)
107
+ script_dir = ROOT / "scripts" / "voice"
108
+ linked = None
109
+ if args.node_modules is not None:
110
+ # Node resolves a package from the script's folders up: lend it one.
111
+ linked = script_dir / "node_modules"
112
+ if not linked.exists():
113
+ linked.symlink_to(args.node_modules)
114
+ fixtures = json.loads((VOICE_TESTS / "fixtures.json").read_text())["clips"]
115
+ measured = VOICE_TESTS / "measured"
116
+ measured.mkdir(exist_ok=True)
117
+ try:
118
+ with tempfile.TemporaryDirectory() as scratch:
119
+ wavs: Dict[str, Path] = {}
120
+ for clip in fixtures:
121
+ samples, rate = sf.read(
122
+ VOICE_TESTS / "fixtures" / clip["file"], dtype="float32"
123
+ )
124
+ wav = Path(scratch) / f"{clip['id']}.wav"
125
+ sf.write(wav, resample(samples, rate), 16000, subtype="PCM_16")
126
+ wavs[clip["id"]] = wav
127
+ # --- speech to text, on the device's models ---------------------------------
128
+ for model in voices.SPEECH_MODEL_CATALOGUE.values():
129
+ if model["task"] != "stt" or "device" not in model["where"]:
130
+ continue
131
+ by_language = {}
132
+ for language in model["languages"]:
133
+ clips = [clip for clip in fixtures if clip["language"] == language]
134
+ if not clips:
135
+ continue
136
+ print(f"{model['id']} on {len(clips)} {language} clips", flush=True)
137
+ heard = transcribe(
138
+ args.store,
139
+ model["id"],
140
+ language,
141
+ [wavs[c["id"]] for c in clips],
142
+ env,
143
+ )
144
+ by_language[language] = {
145
+ "load_ms": heard["load_ms"],
146
+ "clips": [
147
+ {
148
+ "id": clip["id"],
149
+ "heard": result["text"],
150
+ "ms": result["ms"],
151
+ "audio_s": result["audio_s"],
152
+ }
153
+ for clip, result in zip(clips, heard["results"])
154
+ ],
155
+ }
156
+ (measured / f"stt-{model['id']}.json").write_text(
157
+ json.dumps(
158
+ {
159
+ "model": model["id"],
160
+ "runtime": "transformers.js 4.3.0 on onnxruntime-node, CPU",
161
+ "languages": by_language,
162
+ },
163
+ indent=2,
164
+ ensure_ascii=False,
165
+ )
166
+ + "\n"
167
+ )
168
+ # --- text to speech, and what the device hears of it -----------------------
169
+ from kokoro_onnx import Kokoro
170
+
171
+ kokoro_dir = args.store / "kokoro-82m"
172
+ files = {
173
+ item["path"]
174
+ for item in voices.SPEECH_MODEL_CATALOGUE["kokoro-82m"]["files"]
175
+ }
176
+ onnx = next(name for name in files if name.endswith(".onnx"))
177
+ started = time.monotonic()
178
+ kokoro = Kokoro(str(kokoro_dir / onnx), str(kokoro_dir / "voices-v1.0.bin"))
179
+ load_ms = round((time.monotonic() - started) * 1000)
180
+ spoken: List[Dict[str, Any]] = []
181
+ for voice in voices.VOICE_CATALOGUE.values():
182
+ language = voice["languages"][0]
183
+ base = language.split("-")[0]
184
+ for index, text in enumerate([voice["sample"], *PHRASES.get(base, [])]):
185
+ started = time.monotonic()
186
+ samples, rate = kokoro.create(
187
+ text, voice=voice["voice"], lang=language.lower()
188
+ )
189
+ took = time.monotonic() - started
190
+ wav = Path(scratch) / f"{voice['id']}-{index}.wav"
191
+ sf.write(wav, resample(samples, rate), 16000, subtype="PCM_16")
192
+ spoken.append(
193
+ {
194
+ "voice": voice["id"],
195
+ "language": base,
196
+ "text": text,
197
+ "wav": wav,
198
+ "synthesis_ms": round(took * 1000),
199
+ "audio_s": round(len(samples) / rate, 2),
200
+ }
201
+ )
202
+ round_trips = []
203
+ for base in sorted({item["language"] for item in spoken}):
204
+ model = voices.SPEECH_MODEL_CATALOGUE[
205
+ "moonshine-tiny-en" if base == "en" else "whisper-base"
206
+ ]
207
+ items = [item for item in spoken if item["language"] == base]
208
+ heard = transcribe(
209
+ args.store, model["id"], base, [item["wav"] for item in items], env
210
+ )
211
+ for item, result in zip(items, heard["results"]):
212
+ round_trips.append(
213
+ {
214
+ **{
215
+ key: value
216
+ for key, value in item.items()
217
+ if key != "wav"
218
+ },
219
+ "heard_by": model["id"],
220
+ "heard": result["text"],
221
+ }
222
+ )
223
+ (measured / "tts-kokoro-82m.json").write_text(
224
+ json.dumps(
225
+ {
226
+ "model": "kokoro-82m",
227
+ "runtime": "kokoro-onnx 0.6.1, onnxruntime CPU",
228
+ "load_ms": load_ms,
229
+ "spoken": round_trips,
230
+ },
231
+ indent=2,
232
+ ensure_ascii=False,
233
+ )
234
+ + "\n"
235
+ )
236
+ finally:
237
+ if linked is not None and linked.is_symlink():
238
+ linked.unlink()
239
+ print(f"wrote {measured}")
240
+ return 0
241
+
242
+
243
+ if __name__ == "__main__":
244
+ sys.exit(main())
@@ -0,0 +1,159 @@
1
+ #!/usr/bin/env python3
2
+ # Copyright (c) 2025-2026 Datalayer, Inc.
3
+ # Distributed under the terms of the Modified BSD License.
4
+
5
+ """The pinned-file store of the speech models (VOICE.md VO-03, VO-48, VO-49).
6
+
7
+ Every file of every speech model of the catalogue, laid out as
8
+ ``<store>/<model id>/<path>`` and checked against the SHA-256 and the size
9
+ the catalogue pins. The store is filled **once**, from where each model was
10
+ published, and then copied to Datalayer's storage; the browser and the speech
11
+ service read it from there and never from a third party's hub.
12
+
13
+ python scripts/voice/pin_store.py --store ~/.cache/speech-store [--from DIR]
14
+ python scripts/voice/pin_store.py --store ~/.cache/speech-store --verify
15
+
16
+ ``--from`` takes files already downloaded (any layout) whose hash matches,
17
+ instead of downloading them again. ``--verify`` downloads nothing and fails
18
+ on a missing or changed file.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import argparse
24
+ import hashlib
25
+ import shutil
26
+ import sys
27
+ import tarfile
28
+ import tempfile
29
+ from pathlib import Path
30
+ from typing import Any, Dict, Iterator, Optional
31
+
32
+ import httpx
33
+
34
+ from agent_runtimes.specs.voices import SPEECH_MODEL_CATALOGUE
35
+
36
+ #: Where the npm package that ships Silero VAD is published.
37
+ NPM_TARBALL = "https://registry.npmjs.org/@ricky0123/vad-web/-/vad-web-{revision}.tgz"
38
+
39
+
40
+ def sha256_of(path: Path) -> str:
41
+ digest = hashlib.sha256()
42
+ with path.open("rb") as stream:
43
+ for block in iter(lambda: stream.read(1 << 20), b""):
44
+ digest.update(block)
45
+ return digest.hexdigest()
46
+
47
+
48
+ def upstream_url(model: Dict[str, Any], path: str) -> Optional[str]:
49
+ """Where a file was published: a Hugging Face revision or a GitHub release; None for npm."""
50
+ repository = _repository(model)
51
+ if repository.startswith("https://huggingface.co/"):
52
+ return f"{repository}/resolve/{model['revision']}/{path}"
53
+ if "/releases/tag/" in repository:
54
+ return f"{repository.replace('/releases/tag/', '/releases/download/')}/{path}"
55
+ return None
56
+
57
+
58
+ def _repository(model: Dict[str, Any]) -> str:
59
+ from agentspecs.speech import get_speech_model
60
+
61
+ spec = get_speech_model(model["id"])
62
+ if spec is None:
63
+ raise SystemExit(
64
+ f"{model['id']} is in the generated catalogue and not in agentspecs: run make specs"
65
+ )
66
+ return spec.source.repository
67
+
68
+
69
+ def _found(local: Optional[Path], sha256: str, size: int) -> Optional[Path]:
70
+ if local is None:
71
+ return None
72
+ for candidate in local.rglob("*"):
73
+ if (
74
+ candidate.is_file()
75
+ and candidate.stat().st_size == size
76
+ and sha256_of(candidate) == sha256
77
+ ):
78
+ return candidate
79
+ return None
80
+
81
+
82
+ def _npm_file(model: Dict[str, Any], path: str, into: Path) -> None:
83
+ with tempfile.TemporaryDirectory() as scratch:
84
+ tarball = Path(scratch) / "package.tgz"
85
+ _download(NPM_TARBALL.format(revision=model["revision"]), tarball)
86
+ with tarfile.open(tarball) as archive:
87
+ member = archive.getmember(f"package/dist/{path}")
88
+ extracted = archive.extractfile(member)
89
+ if extracted is None:
90
+ raise SystemExit(f"{path} is not a file of the vad-web package")
91
+ into.write_bytes(extracted.read())
92
+
93
+
94
+ def _download(url: str, target: Path) -> None:
95
+ """Copy a published file, once, from where it was published (HTTPS only)."""
96
+ if not url.startswith("https://"):
97
+ raise SystemExit(f"{url} is not an HTTPS address: refused.")
98
+ with httpx.stream("GET", url, follow_redirects=True, timeout=120.0) as answered:
99
+ answered.raise_for_status()
100
+ with target.open("wb") as stream:
101
+ for block in answered.iter_bytes(1 << 20):
102
+ stream.write(block)
103
+
104
+
105
+ def files() -> Iterator[tuple[Dict[str, Any], Dict[str, Any]]]:
106
+ for model in SPEECH_MODEL_CATALOGUE.values():
107
+ for item in model["files"]:
108
+ yield model, item
109
+
110
+
111
+ def main() -> int:
112
+ parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
113
+ parser.add_argument("--store", type=Path, required=True)
114
+ parser.add_argument("--from", dest="local", type=Path, default=None)
115
+ parser.add_argument("--verify", action="store_true")
116
+ args = parser.parse_args()
117
+ problems = []
118
+ total = 0
119
+ for model, item in files():
120
+ target = args.store / model["id"] / item["path"]
121
+ total += item["size"]
122
+ if (
123
+ target.is_file()
124
+ and target.stat().st_size == item["size"]
125
+ and sha256_of(target) == item["sha256"]
126
+ ):
127
+ continue
128
+ if args.verify:
129
+ problems.append(
130
+ f"{model['id']}/{item['path']} is missing or does not match its pin"
131
+ )
132
+ continue
133
+ target.parent.mkdir(parents=True, exist_ok=True)
134
+ local = _found(args.local, item["sha256"], item["size"])
135
+ if local is not None:
136
+ shutil.copyfile(local, target)
137
+ else:
138
+ url = upstream_url(model, item["path"])
139
+ print(f"downloading {model['id']}/{item['path']}", flush=True)
140
+ if url is None:
141
+ _npm_file(model, item["path"], target)
142
+ else:
143
+ _download(url, target)
144
+ if sha256_of(target) != item["sha256"]:
145
+ target.unlink()
146
+ problems.append(
147
+ f"{model['id']}/{item['path']} was published with another hash: refused"
148
+ )
149
+ for problem in problems:
150
+ print(problem, file=sys.stderr)
151
+ if not problems:
152
+ print(
153
+ f"{args.store}: {sum(1 for _ in files())} files, {total / 1e6:.0f} MB, every one as pinned"
154
+ )
155
+ return 1 if problems else 0
156
+
157
+
158
+ if __name__ == "__main__":
159
+ raise SystemExit(main())