@datalayer/agent-runtimes 1.3.62 → 1.3.64

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/lib/chat/ChatFloating.d.ts +9 -1
  2. package/lib/chat/ChatFloating.js +69 -18
  3. package/lib/chat/assistant/AssistantStage.d.ts +14 -1
  4. package/lib/chat/assistant/AssistantStage.js +53 -1
  5. package/lib/chat/assistant/SpriteCharacter.js +1 -1
  6. package/lib/chat/assistant/characters.js +1 -5
  7. package/lib/chat/assistant/state.d.ts +7 -1
  8. package/lib/chat/assistant/state.js +3 -2
  9. package/lib/chat/base/ChatBase.js +32 -4
  10. package/lib/chat/messages/ChatMessageList.d.ts +0 -6
  11. package/lib/chat/messages/ChatMessageList.js +8 -2
  12. package/lib/config/AgentConfiguration.js +0 -6
  13. package/lib/examples/AgentA2ATeamExample.js +19 -2
  14. package/lib/examples/ChatAssistantExample.d.ts +3 -1
  15. package/lib/examples/ChatAssistantExample.js +75 -7
  16. package/lib/examples/ChatAssistantGalleryExample.d.ts +3 -2
  17. package/lib/examples/ChatAssistantGalleryExample.js +11 -6
  18. package/lib/examples/DecksAgent.js +4 -2
  19. package/lib/examples/LoopShellExample.js +4 -2
  20. package/lib/examples/VoiceChatExample.d.ts +20 -0
  21. package/lib/examples/VoiceChatExample.js +64 -0
  22. package/lib/examples/example-selector.js +1 -0
  23. package/lib/examples/main.js +12 -7
  24. package/lib/examples/utils/clippyJsCharacters.d.ts +18 -0
  25. package/lib/examples/utils/clippyJsCharacters.js +129 -0
  26. package/lib/loop/apps/appspec.d.ts +3 -1
  27. package/lib/loop/apps/appspec.js +31 -0
  28. package/lib/loop/apps/checks.js +1 -1
  29. package/lib/protocols/VercelAIAdapter.js +5 -0
  30. package/lib/specs/apps.js +112 -0
  31. package/lib/specs/appspecSchema.js +65 -0
  32. package/lib/specs/index.d.ts +1 -0
  33. package/lib/specs/index.js +1 -0
  34. package/lib/specs/voices.d.ts +72 -0
  35. package/lib/specs/voices.js +344 -0
  36. package/lib/types/agents.d.ts +1 -1
  37. package/lib/types/agentspecs.d.ts +16 -0
  38. package/lib/types/chat.d.ts +11 -0
  39. package/lib/voice/VoiceInput.d.ts +35 -0
  40. package/lib/voice/VoiceInput.js +233 -0
  41. package/lib/voice/capture.d.ts +30 -0
  42. package/lib/voice/capture.js +113 -0
  43. package/lib/voice/consent.d.ts +6 -0
  44. package/lib/voice/consent.js +34 -0
  45. package/lib/voice/hearing.d.ts +37 -0
  46. package/lib/voice/hearing.js +168 -0
  47. package/lib/voice/index.d.ts +47 -0
  48. package/lib/voice/index.js +13 -0
  49. package/lib/voice/pinned.d.ts +28 -0
  50. package/lib/voice/pinned.js +75 -0
  51. package/lib/voice/sentences.d.ts +26 -0
  52. package/lib/voice/sentences.js +71 -0
  53. package/lib/voice/speaker.d.ts +66 -0
  54. package/lib/voice/speaker.js +168 -0
  55. package/lib/voice/types.d.ts +92 -0
  56. package/lib/voice/types.js +5 -0
  57. package/lib/voice/useSpokenAnswers.d.ts +12 -0
  58. package/lib/voice/useSpokenAnswers.js +90 -0
  59. package/package.json +7 -3
  60. package/scripts/codegen/generate_voices.py +279 -0
  61. package/scripts/voice/measure.py +244 -0
  62. package/scripts/voice/pin_store.py +159 -0
  63. package/scripts/voice/serve_store.py +70 -0
  64. package/scripts/voice/transcribe.mjs +58 -0
@@ -0,0 +1,30 @@
1
+ /**
2
+ * The microphone, for one press (VOICE.md VO-10, VO-62): opened by a
3
+ * gesture, shown while open, closed when the press ends or is cancelled.
4
+ * What it heard is kept in memory only, until it is transcribed.
5
+ *
6
+ * @module voice/capture
7
+ */
8
+ /** The rate the speech models hear at. */
9
+ export declare const HEARING_RATE = 16000;
10
+ /** One opening of the microphone. */
11
+ export interface Capture {
12
+ /** How loud it is now, between 0 and 1: what the level meter shows. */
13
+ level: () => number;
14
+ /** Close it and give what it heard, mono at 16 kHz. */
15
+ stop: () => Promise<Float32Array>;
16
+ /** Close it and forget what it heard. */
17
+ cancel: () => void;
18
+ }
19
+ /** The microphone could not be opened: said in a sentence, never a trace. */
20
+ export declare class MicrophoneRefused extends Error {
21
+ }
22
+ /** The loudness of what an analyser holds now, between 0 and 1. */
23
+ export declare function levelOf(analyser: AnalyserNode, buffer: Float32Array): number;
24
+ /** Mono samples at one rate, brought to another. */
25
+ export declare function resample(samples: Float32Array, from: number, to?: number): Float32Array;
26
+ /**
27
+ * Open the microphone. The browser's echo cancellation is asked for, so
28
+ * the agent's own voice is not heard back (VO-13).
29
+ */
30
+ export declare function openMicrophone(): Promise<Capture>;
@@ -0,0 +1,113 @@
1
+ /*
2
+ * Copyright (c) 2025-2026 Datalayer, Inc.
3
+ * Distributed under the terms of the Modified BSD License.
4
+ */
5
+ /**
6
+ * The microphone, for one press (VOICE.md VO-10, VO-62): opened by a
7
+ * gesture, shown while open, closed when the press ends or is cancelled.
8
+ * What it heard is kept in memory only, until it is transcribed.
9
+ *
10
+ * @module voice/capture
11
+ */
12
+ /** The rate the speech models hear at. */
13
+ export const HEARING_RATE = 16000;
14
+ /** The microphone could not be opened: said in a sentence, never a trace. */
15
+ export class MicrophoneRefused extends Error {
16
+ }
17
+ /** The loudness of what an analyser holds now, between 0 and 1. */
18
+ export function levelOf(analyser, buffer) {
19
+ analyser.getFloatTimeDomainData(buffer);
20
+ let sum = 0;
21
+ for (let index = 0; index < buffer.length; index += 1) {
22
+ sum += buffer[index] * buffer[index];
23
+ }
24
+ // Speech sits around 0.05 to 0.2 RMS: brought to a scale the eye reads.
25
+ return Math.min(1, Math.sqrt(sum / buffer.length) * 5);
26
+ }
27
+ /** Mono samples at one rate, brought to another. */
28
+ export function resample(samples, from, to = HEARING_RATE) {
29
+ if (from === to) {
30
+ return samples;
31
+ }
32
+ const length = Math.round((samples.length * to) / from);
33
+ const out = new Float32Array(length);
34
+ const step = from / to;
35
+ for (let index = 0; index < length; index += 1) {
36
+ const at = index * step;
37
+ const left = Math.floor(at);
38
+ const right = Math.min(left + 1, samples.length - 1);
39
+ out[index] = samples[left] + (samples[right] - samples[left]) * (at - left);
40
+ }
41
+ return out;
42
+ }
43
+ /**
44
+ * Open the microphone. The browser's echo cancellation is asked for, so
45
+ * the agent's own voice is not heard back (VO-13).
46
+ */
47
+ export async function openMicrophone() {
48
+ if (!navigator.mediaDevices?.getUserMedia) {
49
+ throw new MicrophoneRefused('This browser gives no page a microphone.');
50
+ }
51
+ let stream;
52
+ try {
53
+ stream = await navigator.mediaDevices.getUserMedia({
54
+ audio: {
55
+ channelCount: 1,
56
+ echoCancellation: true,
57
+ noiseSuppression: true,
58
+ autoGainControl: true,
59
+ },
60
+ });
61
+ }
62
+ catch (error) {
63
+ const name = error?.name;
64
+ throw new MicrophoneRefused(name === 'NotAllowedError'
65
+ ? 'The microphone was not allowed: allow it for this page to talk, or type.'
66
+ : name === 'NotFoundError'
67
+ ? 'No microphone was found on this device.'
68
+ : 'The microphone could not be opened.');
69
+ }
70
+ const context = new AudioContext();
71
+ const source = context.createMediaStreamSource(stream);
72
+ const analyser = context.createAnalyser();
73
+ analyser.fftSize = 1024;
74
+ source.connect(analyser);
75
+ // The samples as they come, kept in memory until the press ends.
76
+ const recorder = context.createScriptProcessor(4096, 1, 1);
77
+ const chunks = [];
78
+ recorder.onaudioprocess = event => {
79
+ chunks.push(new Float32Array(event.inputBuffer.getChannelData(0)));
80
+ };
81
+ source.connect(recorder);
82
+ // A processor runs only when connected; a silent gain keeps it off the speakers.
83
+ const mute = context.createGain();
84
+ mute.gain.value = 0;
85
+ recorder.connect(mute);
86
+ mute.connect(context.destination);
87
+ const buffer = new Float32Array(analyser.fftSize);
88
+ const close = () => {
89
+ recorder.onaudioprocess = null;
90
+ stream.getTracks().forEach(track => track.stop());
91
+ void context.close();
92
+ };
93
+ return {
94
+ level: () => levelOf(analyser, buffer),
95
+ stop: async () => {
96
+ const rate = context.sampleRate;
97
+ close();
98
+ const length = chunks.reduce((total, chunk) => total + chunk.length, 0);
99
+ const joined = new Float32Array(length);
100
+ let at = 0;
101
+ for (const chunk of chunks) {
102
+ joined.set(chunk, at);
103
+ at += chunk.length;
104
+ }
105
+ chunks.length = 0;
106
+ return resample(joined, rate);
107
+ },
108
+ cancel: () => {
109
+ close();
110
+ chunks.length = 0;
111
+ },
112
+ };
113
+ }
@@ -0,0 +1,6 @@
1
+ /** Whether the person said yes for this application. */
2
+ export declare function consented(app: string): boolean;
3
+ /** Remember the person's yes for this application. */
4
+ export declare function consent(app: string): void;
5
+ /** The sentence said before the first use: where the audio goes (VO-61). */
6
+ export declare const CONSENT_SENTENCE = "What you say is transcribed on this device: the sound never leaves it, and is not kept. The words go in the message box, for you to send.";
@@ -0,0 +1,34 @@
1
+ /*
2
+ * Copyright (c) 2025-2026 Datalayer, Inc.
3
+ * Distributed under the terms of the Modified BSD License.
4
+ */
5
+ /**
6
+ * The person's yes to the microphone, asked once per application before the
7
+ * browser's own prompt (VOICE.md VO-61), and remembered where the page keeps
8
+ * its preferences. A page without storage asks again next time.
9
+ *
10
+ * @module voice/consent
11
+ */
12
+ const KEY = 'datalayer-voice-consent';
13
+ /** Whether the person said yes for this application. */
14
+ export function consented(app) {
15
+ try {
16
+ const kept = JSON.parse(window.localStorage.getItem(KEY) || '{}');
17
+ return kept?.[app] === true;
18
+ }
19
+ catch {
20
+ return false;
21
+ }
22
+ }
23
+ /** Remember the person's yes for this application. */
24
+ export function consent(app) {
25
+ try {
26
+ const kept = JSON.parse(window.localStorage.getItem(KEY) || '{}');
27
+ window.localStorage.setItem(KEY, JSON.stringify({ ...kept, [app]: true }));
28
+ }
29
+ catch {
30
+ // Without storage, the yes holds for the page.
31
+ }
32
+ }
33
+ /** The sentence said before the first use: where the audio goes (VO-61). */
34
+ export const CONSENT_SENTENCE = 'What you say is transcribed on this device: the sound never leaves it, and is not kept. The words go in the message box, for you to send.';
@@ -0,0 +1,37 @@
1
+ import type { Transcript, VoiceEngines } from './types';
2
+ /** How far the first use has come: bytes loaded of the bytes to load. */
3
+ export type HearingProgress = (loaded: number, total: number) => void;
4
+ /** A language no device model hears. */
5
+ export declare class NotHeardHere extends Error {
6
+ }
7
+ /** Where onnxruntime-web's own files are served, beside the models. */
8
+ export declare function runtimeFiles(modelsUrl: string, version: string): {
9
+ mjs: string;
10
+ wasm: string;
11
+ };
12
+ /** Joins the speech a VAD found, in order. */
13
+ export declare function joinSegments(segments: Float32Array[]): Float32Array;
14
+ /**
15
+ * The page's hearing: loaded once, on first use, from Datalayer's origin
16
+ * (`modelsUrl`), every file checked against its pin.
17
+ */
18
+ export declare class DeviceHearing {
19
+ private readonly modelsUrl;
20
+ private readonly engines;
21
+ private recognizers;
22
+ private vad?;
23
+ constructor(modelsUrl: string, engines: VoiceEngines);
24
+ /** The model that hears a language here, or a refusal in a sentence. */
25
+ modelFor(language: string): import("..").SpeechModelSpec;
26
+ /** What the first use downloads for a language, in bytes. */
27
+ sizeFor(language: string): number;
28
+ /** Load what hears a language, saying how far it has come. */
29
+ ready(language: string, progress?: HearingProgress): Promise<void>;
30
+ private recognizer;
31
+ private activity;
32
+ /**
33
+ * What was said in a press, or `undefined` when nothing was: its speech
34
+ * found by the VAD, then transcribed in the language.
35
+ */
36
+ hear(audio: Float32Array, language: string): Promise<Transcript | undefined>;
37
+ }
@@ -0,0 +1,168 @@
1
+ /*
2
+ * Copyright (c) 2025-2026 Datalayer, Inc.
3
+ * Distributed under the terms of the Modified BSD License.
4
+ */
5
+ /**
6
+ * Hearing on the device (VOICE.md VO-10, VO-14, decision 3): what was said
7
+ * in one press, kept to its speech by Silero VAD and transcribed by the
8
+ * catalogue's model for the language — Moonshine for English, Whisper for
9
+ * French — in the page, with transformers.js. The audio never leaves the
10
+ * browser, and is forgotten once heard.
11
+ *
12
+ * @module voice/hearing
13
+ */
14
+ import { SPEECH_MODEL_CATALOGUE, transcriberFor } from '../specs/voices';
15
+ import { pinnedBytes, pinnedFetch, downloadSize } from './pinned';
16
+ import { HEARING_RATE } from './capture';
17
+ /** A language no device model hears. */
18
+ export class NotHeardHere extends Error {
19
+ }
20
+ /** Where onnxruntime-web's own files are served, beside the models. */
21
+ export function runtimeFiles(modelsUrl, version) {
22
+ const root = `${modelsUrl.replace(/\/+$/, '')}/onnxruntime-web/${version}/`;
23
+ return {
24
+ mjs: `${root}ort-wasm-simd-threaded.asyncify.mjs`,
25
+ wasm: `${root}ort-wasm-simd-threaded.asyncify.wasm`,
26
+ };
27
+ }
28
+ /** Joins the speech a VAD found, in order. */
29
+ export function joinSegments(segments) {
30
+ const out = new Float32Array(segments.reduce((n, s) => n + s.length, 0));
31
+ let at = 0;
32
+ for (const segment of segments) {
33
+ out.set(segment, at);
34
+ at += segment.length;
35
+ }
36
+ return out;
37
+ }
38
+ /**
39
+ * The page's hearing: loaded once, on first use, from Datalayer's origin
40
+ * (`modelsUrl`), every file checked against its pin.
41
+ */
42
+ export class DeviceHearing {
43
+ modelsUrl;
44
+ engines;
45
+ recognizers = new Map();
46
+ vad;
47
+ constructor(modelsUrl, engines) {
48
+ this.modelsUrl = modelsUrl;
49
+ this.engines = engines;
50
+ }
51
+ /** The model that hears a language here, or a refusal in a sentence. */
52
+ modelFor(language) {
53
+ const model = transcriberFor(language);
54
+ if (!model) {
55
+ throw new NotHeardHere(`Nothing in this browser hears ${language} yet: type, or choose English or French.`);
56
+ }
57
+ return model;
58
+ }
59
+ /** What the first use downloads for a language, in bytes. */
60
+ sizeFor(language) {
61
+ return (downloadSize(this.modelFor(language).id) + downloadSize('silero-vad'));
62
+ }
63
+ /** Load what hears a language, saying how far it has come. */
64
+ async ready(language, progress) {
65
+ await Promise.all([this.recognizer(language, progress), this.activity()]);
66
+ }
67
+ recognizer(language, progress) {
68
+ const model = this.modelFor(language);
69
+ let loading = this.recognizers.get(model.id);
70
+ if (!loading) {
71
+ loading = (async () => {
72
+ const transformers = await this.engines.transformers();
73
+ const { env } = transformers;
74
+ env.allowLocalModels = false;
75
+ env.allowRemoteModels = true;
76
+ env.remoteHost = `${this.modelsUrl.replace(/\/+$/, '')}/`;
77
+ env.remotePathTemplate = '{model}/';
78
+ env.useBrowserCache = true;
79
+ env.fetch = pinnedFetch(this.modelsUrl);
80
+ const wasm = env.backends.onnx.wasm;
81
+ const version = env.backends.onnx.versions?.web;
82
+ if (wasm && version) {
83
+ wasm.wasmPaths = runtimeFiles(this.modelsUrl, version);
84
+ }
85
+ const total = downloadSize(model.id);
86
+ const loaded = new Map();
87
+ const recognizer = await transformers.pipeline('automatic-speech-recognition', model.id, {
88
+ dtype: model.dtype,
89
+ device: 'wasm',
90
+ progress_callback: (info) => {
91
+ if (info?.status === 'progress' &&
92
+ typeof info.loaded === 'number') {
93
+ loaded.set(String(info.file), info.loaded);
94
+ progress?.(Array.from(loaded.values()).reduce((a, b) => a + b, 0), total);
95
+ }
96
+ },
97
+ });
98
+ progress?.(total, total);
99
+ return recognizer;
100
+ })();
101
+ loading.catch(() => this.recognizers.delete(model.id));
102
+ this.recognizers.set(model.id, loading);
103
+ }
104
+ return loading;
105
+ }
106
+ activity() {
107
+ if (!this.vad) {
108
+ const file = SPEECH_MODEL_CATALOGUE['silero-vad'].files[0].path;
109
+ const url = `${this.modelsUrl.replace(/\/+$/, '')}/silero-vad/${file}`;
110
+ this.vad = this.engines.vad().then(module => module.NonRealTimeVAD.new({
111
+ modelURL: url,
112
+ modelFetcher: path => pinnedBytes(this.modelsUrl, path),
113
+ ortConfig: ort => {
114
+ const version = ort?.env?.versions?.web;
115
+ if (ort?.env?.wasm && version) {
116
+ ort.env.wasm.wasmPaths = runtimeFiles(this.modelsUrl, version);
117
+ }
118
+ },
119
+ // A press is short: keep a word said alone.
120
+ minSpeechMs: 150,
121
+ preSpeechPadMs: 200,
122
+ redemptionMs: 600,
123
+ }));
124
+ this.vad.catch(() => {
125
+ this.vad = undefined;
126
+ });
127
+ }
128
+ return this.vad;
129
+ }
130
+ /**
131
+ * What was said in a press, or `undefined` when nothing was: its speech
132
+ * found by the VAD, then transcribed in the language.
133
+ */
134
+ async hear(audio, language) {
135
+ const started = performance.now();
136
+ const model = this.modelFor(language);
137
+ const [recognize, vad] = await Promise.all([
138
+ this.recognizer(language),
139
+ this.activity(),
140
+ ]);
141
+ const segments = [];
142
+ for await (const segment of vad.run(audio, HEARING_RATE)) {
143
+ segments.push(segment.audio);
144
+ }
145
+ const speech = joinSegments(segments);
146
+ if (speech.length < HEARING_RATE * 0.2) {
147
+ return undefined;
148
+ }
149
+ // Whisper is told the language; Moonshine hears English only.
150
+ const options = model.id.startsWith('whisper')
151
+ ? {
152
+ language: language.startsWith('fr') ? 'french' : 'english',
153
+ task: 'transcribe',
154
+ }
155
+ : {};
156
+ const heard = await recognize(speech, options);
157
+ const text = (Array.isArray(heard) ? heard.map(h => h.text).join(' ') : heard.text).trim();
158
+ if (!text) {
159
+ return undefined;
160
+ }
161
+ return {
162
+ text,
163
+ metadata: { input: 'voice', language, engine: model.id, where: 'device' },
164
+ seconds: speech.length / HEARING_RATE,
165
+ ms: Math.round(performance.now() - started),
166
+ };
167
+ }
168
+ }
@@ -0,0 +1,47 @@
1
+ /**
2
+ * Voice: agents that listen and speak (VOICE.md).
3
+ *
4
+ * Push-to-talk heard on the device (Moonshine, Whisper, Silero VAD, from
5
+ * Datalayer's origin and as pinned), answers spoken by Datalayer's speech
6
+ * service (Kokoro on ai-agents), the assistant's mouth moving with them.
7
+ *
8
+ * @module voice
9
+ */
10
+ import type { VoiceEngines } from './types';
11
+ export * from './types';
12
+ export * from './sentences';
13
+ export * from './pinned';
14
+ export * from './capture';
15
+ export * from './hearing';
16
+ export * from './speaker';
17
+ export * from './consent';
18
+ export * from './useSpokenAnswers';
19
+ export { VoiceInput } from './VoiceInput';
20
+ export type { VoiceInputProps, ListeningState } from './VoiceInput';
21
+ /**
22
+ * A chat's voice: what a host gives `ChatBase` and `ChatFloating` to let a
23
+ * person talk to the agent and hear it — the Appspec's `interface.voice`,
24
+ * with where the models and the speech service are.
25
+ */
26
+ export interface ChatVoice {
27
+ /** The language listened to and spoken, BCP 47. */
28
+ language: string;
29
+ /** `push_to_talk`, or `off`. Hands-free is Phase V3. */
30
+ input: 'off' | 'push_to_talk';
31
+ /** When answers are heard: `always`, or `off`. */
32
+ output: 'off' | 'always';
33
+ /** The voice of the catalogue answers are spoken with. */
34
+ voice: string;
35
+ /** Datalayer's origin for the browser's models: the pinned store. */
36
+ modelsUrl: string;
37
+ /** The packages that hear, loaded when voice is first used. */
38
+ engines: VoiceEngines;
39
+ /** ai-agents, whose speech service says the answers. */
40
+ speechUrl?: string;
41
+ /** The token the speech service is asked with. */
42
+ token?: string;
43
+ /** Send what is said at once, rather than put it in the composer. */
44
+ sendWhatISay?: boolean;
45
+ /** Whose consent to the microphone is remembered: the application's id. */
46
+ consentKey?: string;
47
+ }
@@ -0,0 +1,13 @@
1
+ /*
2
+ * Copyright (c) 2025-2026 Datalayer, Inc.
3
+ * Distributed under the terms of the Modified BSD License.
4
+ */
5
+ export * from './types';
6
+ export * from './sentences';
7
+ export * from './pinned';
8
+ export * from './capture';
9
+ export * from './hearing';
10
+ export * from './speaker';
11
+ export * from './consent';
12
+ export * from './useSpokenAnswers';
13
+ export { VoiceInput } from './VoiceInput';
@@ -0,0 +1,28 @@
1
+ /**
2
+ * The browser's models, from Datalayer's origin and as pinned (VOICE.md
3
+ * VO-49, VO-03): every file a model is made of is fetched from the models'
4
+ * origin, laid out as the pinned store is (`<model id>/<path>`), and its
5
+ * SHA-256 checked against the catalogue before anything reads it. A file
6
+ * that is not the one pinned is refused; nothing is asked of another origin.
7
+ *
8
+ * @module voice/pinned
9
+ */
10
+ import { type PinnedFile } from '../specs/voices';
11
+ /** A file of a model that is not the one the catalogue pins, or not reachable. */
12
+ export declare class PinRefused extends Error {
13
+ }
14
+ /** The pinned file at a URL under the models' origin, if it is one. */
15
+ export declare function pinOf(base: string, url: string): PinnedFile | undefined;
16
+ /** The SHA-256 of some bytes, in hex. */
17
+ export declare function sha256Hex(bytes: ArrayBuffer): Promise<string>;
18
+ /**
19
+ * A `fetch` that reaches the models' origin only, and answers a model's
20
+ * file only once its hash is the one pinned. The runtime's own files
21
+ * (onnxruntime-web's WASM, under `onnxruntime-web/`) come from the same
22
+ * origin, at the version the page bundles.
23
+ */
24
+ export declare function pinnedFetch(base: string, fetcher?: typeof fetch): (input: string | URL, init?: RequestInit) => Promise<Response>;
25
+ /** A model's file as bytes, checked against its pin. */
26
+ export declare function pinnedBytes(base: string, url: string, fetcher?: typeof fetch): Promise<ArrayBuffer>;
27
+ /** What a first use downloads for a model, in bytes. */
28
+ export declare function downloadSize(modelId: string): number;
@@ -0,0 +1,75 @@
1
+ /*
2
+ * Copyright (c) 2025-2026 Datalayer, Inc.
3
+ * Distributed under the terms of the Modified BSD License.
4
+ */
5
+ /**
6
+ * The browser's models, from Datalayer's origin and as pinned (VOICE.md
7
+ * VO-49, VO-03): every file a model is made of is fetched from the models'
8
+ * origin, laid out as the pinned store is (`<model id>/<path>`), and its
9
+ * SHA-256 checked against the catalogue before anything reads it. A file
10
+ * that is not the one pinned is refused; nothing is asked of another origin.
11
+ *
12
+ * @module voice/pinned
13
+ */
14
+ import { SPEECH_MODEL_CATALOGUE } from '../specs/voices';
15
+ /** A file of a model that is not the one the catalogue pins, or not reachable. */
16
+ export class PinRefused extends Error {
17
+ }
18
+ /** The pinned file at a URL under the models' origin, if it is one. */
19
+ export function pinOf(base, url) {
20
+ const root = base.replace(/\/+$/, '') + '/';
21
+ if (!url.startsWith(root)) {
22
+ return undefined;
23
+ }
24
+ const [model, ...rest] = url.slice(root.length).split('?')[0].split('/');
25
+ const path = rest.join('/');
26
+ return SPEECH_MODEL_CATALOGUE[model]?.files.find(file => file.path === path);
27
+ }
28
+ /** The SHA-256 of some bytes, in hex. */
29
+ export async function sha256Hex(bytes) {
30
+ const digest = await crypto.subtle.digest('SHA-256', bytes);
31
+ return Array.from(new Uint8Array(digest))
32
+ .map(byte => byte.toString(16).padStart(2, '0'))
33
+ .join('');
34
+ }
35
+ /**
36
+ * A `fetch` that reaches the models' origin only, and answers a model's
37
+ * file only once its hash is the one pinned. The runtime's own files
38
+ * (onnxruntime-web's WASM, under `onnxruntime-web/`) come from the same
39
+ * origin, at the version the page bundles.
40
+ */
41
+ export function pinnedFetch(base, fetcher = (input, init) => fetch(input, init)) {
42
+ const root = base.replace(/\/+$/, '') + '/';
43
+ return async (input, init) => {
44
+ const url = typeof input === 'string' ? input : input.toString();
45
+ if (!url.startsWith(root)) {
46
+ throw new PinRefused(`Voice reads its models from ${root} only, not ${url}.`);
47
+ }
48
+ const answered = await fetcher(url, init);
49
+ const pin = pinOf(base, url);
50
+ if (!pin || !answered.ok) {
51
+ return answered;
52
+ }
53
+ const bytes = await answered.arrayBuffer();
54
+ if (bytes.byteLength !== pin.size ||
55
+ (await sha256Hex(bytes)) !== pin.sha256) {
56
+ throw new PinRefused(`${url.slice(root.length)} is not the file the voice catalogue pins: refused.`);
57
+ }
58
+ return new Response(bytes, {
59
+ status: answered.status,
60
+ headers: answered.headers,
61
+ });
62
+ };
63
+ }
64
+ /** A model's file as bytes, checked against its pin. */
65
+ export async function pinnedBytes(base, url, fetcher) {
66
+ const answered = await pinnedFetch(base, fetcher)(url);
67
+ if (!answered.ok) {
68
+ throw new PinRefused(`${url} could not be read (${answered.status}).`);
69
+ }
70
+ return answered.arrayBuffer();
71
+ }
72
+ /** What a first use downloads for a model, in bytes. */
73
+ export function downloadSize(modelId) {
74
+ return (SPEECH_MODEL_CATALOGUE[modelId]?.files ?? []).reduce((total, file) => total + file.size, 0);
75
+ }
@@ -0,0 +1,26 @@
1
+ /**
2
+ * An answer as it is heard (VOICE.md VO-20): read as plain words, the way
3
+ * the balloon reads it (T-23), and cut into sentences as it is written, so
4
+ * that the first one is said before the answer is finished.
5
+ *
6
+ * @module voice/sentences
7
+ */
8
+ /**
9
+ * The words of an answer: no Markdown signs; a code block said as *a code
10
+ * block*, a table as *a table*, a link by its text, an image by its words.
11
+ */
12
+ export declare function plainWords(markdown: string): string;
13
+ /**
14
+ * Cuts an answer into the sentences to say, as it grows.
15
+ *
16
+ * `feed` takes the answer as written so far (its whole text, each time) and
17
+ * gives back the sentences now complete and not given before; `end` gives
18
+ * what is left once the answer is finished. A sentence is complete at its
19
+ * stop followed by a space — never at the very end of what has arrived,
20
+ * which may be cut in the middle of a number (`3.5`).
21
+ */
22
+ export declare class SentenceCutter {
23
+ private said;
24
+ feed(written: string): string[];
25
+ end(written: string): string[];
26
+ }
@@ -0,0 +1,71 @@
1
+ /*
2
+ * Copyright (c) 2025-2026 Datalayer, Inc.
3
+ * Distributed under the terms of the Modified BSD License.
4
+ */
5
+ /**
6
+ * An answer as it is heard (VOICE.md VO-20): read as plain words, the way
7
+ * the balloon reads it (T-23), and cut into sentences as it is written, so
8
+ * that the first one is said before the answer is finished.
9
+ *
10
+ * @module voice/sentences
11
+ */
12
+ /**
13
+ * The words of an answer: no Markdown signs; a code block said as *a code
14
+ * block*, a table as *a table*, a link by its text, an image by its words.
15
+ */
16
+ export function plainWords(markdown) {
17
+ return (markdown
18
+ .replace(/```[\s\S]*?(```|$)/g, ' A code block. ')
19
+ // A table: its rows, said once as what they are.
20
+ .replace(/(^|\n)(\|[^\n]*\|[ \t]*(\n|$))+/g, '$1 A table. \n')
21
+ .replace(/!\[([^\]]*)\]\([^)]*\)/g, '$1')
22
+ .replace(/\[([^\]]*)\]\([^)]*\)/g, '$1')
23
+ .replace(/`([^`]*)`/g, '$1')
24
+ .replace(/^\s{0,3}#{1,6}\s+(.*)$/gm, '$1.')
25
+ .replace(/^\s*(?:[-*+]|\d+[.)])\s+/gm, '')
26
+ .replace(/^\s*>\s?/gm, '')
27
+ .replace(/[*_~]+/g, '')
28
+ .replace(/\.\s*\./g, '.')
29
+ .replace(/\s+/g, ' ')
30
+ .trim());
31
+ }
32
+ /** Where a sentence ends: its stop, then a space or the end of what is written. */
33
+ const END = /[.!?…:;](?=["'»)\]]*(\s|$))/g;
34
+ /** Below this, a sentence waits for the next one to be said with it. */
35
+ const SHORTEST = 12;
36
+ /**
37
+ * Cuts an answer into the sentences to say, as it grows.
38
+ *
39
+ * `feed` takes the answer as written so far (its whole text, each time) and
40
+ * gives back the sentences now complete and not given before; `end` gives
41
+ * what is left once the answer is finished. A sentence is complete at its
42
+ * stop followed by a space — never at the very end of what has arrived,
43
+ * which may be cut in the middle of a number (`3.5`).
44
+ */
45
+ export class SentenceCutter {
46
+ said = 0;
47
+ feed(written) {
48
+ const words = plainWords(written);
49
+ const out = [];
50
+ END.lastIndex = this.said;
51
+ let match;
52
+ while ((match = END.exec(words)) !== null) {
53
+ const stop = match.index + match[0].length;
54
+ // At the very end, more may still come: wait.
55
+ if (stop >= words.length) {
56
+ break;
57
+ }
58
+ const sentence = words.slice(this.said, stop).trim();
59
+ if (sentence.length >= SHORTEST) {
60
+ out.push(sentence);
61
+ this.said = stop;
62
+ }
63
+ }
64
+ return out;
65
+ }
66
+ end(written) {
67
+ const rest = plainWords(written).slice(this.said).trim();
68
+ this.said = plainWords(written).length;
69
+ return rest ? [rest] : [];
70
+ }
71
+ }