@datalayer/agent-runtimes 1.3.63 → 1.3.64

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/lib/chat/ChatFloating.d.ts +9 -1
  2. package/lib/chat/ChatFloating.js +69 -18
  3. package/lib/chat/assistant/AssistantStage.d.ts +14 -1
  4. package/lib/chat/assistant/AssistantStage.js +48 -1
  5. package/lib/chat/assistant/state.d.ts +7 -1
  6. package/lib/chat/assistant/state.js +3 -2
  7. package/lib/chat/base/ChatBase.js +32 -4
  8. package/lib/chat/messages/ChatMessageList.d.ts +0 -6
  9. package/lib/chat/messages/ChatMessageList.js +8 -2
  10. package/lib/examples/VoiceChatExample.d.ts +20 -0
  11. package/lib/examples/VoiceChatExample.js +64 -0
  12. package/lib/examples/example-selector.js +1 -0
  13. package/lib/examples/main.js +7 -1
  14. package/lib/loop/apps/appspec.d.ts +3 -1
  15. package/lib/loop/apps/appspec.js +31 -0
  16. package/lib/protocols/VercelAIAdapter.js +5 -0
  17. package/lib/specs/apps.js +112 -0
  18. package/lib/specs/appspecSchema.js +65 -0
  19. package/lib/specs/index.d.ts +1 -0
  20. package/lib/specs/index.js +1 -0
  21. package/lib/specs/voices.d.ts +72 -0
  22. package/lib/specs/voices.js +344 -0
  23. package/lib/types/agentspecs.d.ts +16 -0
  24. package/lib/types/chat.d.ts +11 -0
  25. package/lib/voice/VoiceInput.d.ts +35 -0
  26. package/lib/voice/VoiceInput.js +233 -0
  27. package/lib/voice/capture.d.ts +30 -0
  28. package/lib/voice/capture.js +113 -0
  29. package/lib/voice/consent.d.ts +6 -0
  30. package/lib/voice/consent.js +34 -0
  31. package/lib/voice/hearing.d.ts +37 -0
  32. package/lib/voice/hearing.js +168 -0
  33. package/lib/voice/index.d.ts +47 -0
  34. package/lib/voice/index.js +13 -0
  35. package/lib/voice/pinned.d.ts +28 -0
  36. package/lib/voice/pinned.js +75 -0
  37. package/lib/voice/sentences.d.ts +26 -0
  38. package/lib/voice/sentences.js +71 -0
  39. package/lib/voice/speaker.d.ts +66 -0
  40. package/lib/voice/speaker.js +168 -0
  41. package/lib/voice/types.d.ts +92 -0
  42. package/lib/voice/types.js +5 -0
  43. package/lib/voice/useSpokenAnswers.d.ts +12 -0
  44. package/lib/voice/useSpokenAnswers.js +90 -0
  45. package/package.json +7 -3
  46. package/scripts/codegen/generate_voices.py +279 -0
  47. package/scripts/voice/measure.py +244 -0
  48. package/scripts/voice/pin_store.py +159 -0
  49. package/scripts/voice/serve_store.py +70 -0
  50. package/scripts/voice/transcribe.mjs +58 -0
@@ -187,6 +187,11 @@ export class VercelAIAdapter extends BaseProtocolAdapter {
187
187
  id: msg.id,
188
188
  role: msg.role,
189
189
  parts,
190
+ // How a message was heard, when it was said (VOICE.md VO-27):
191
+ // the runtime keeps the turn marked spoken.
192
+ ...(msg.role === 'user' && msg.metadata?.input === 'voice'
193
+ ? { metadata: msg.metadata }
194
+ : {}),
190
195
  };
191
196
  });
192
197
  // For tool-result messages, add them as separate tool parts
package/lib/specs/apps.js CHANGED
@@ -67,6 +67,14 @@ export const ACCOUNTING_APP_0_0_1 = {
67
67
  settings: [],
68
68
  components: [],
69
69
  assistant: 'wizard',
70
+ voice: {
71
+ enabled: false,
72
+ input: 'push_to_talk',
73
+ output: 'on_request',
74
+ voice: '',
75
+ language: '',
76
+ where: 'auto',
77
+ },
70
78
  },
71
79
  tests: {
72
80
  readyAt: 0.8,
@@ -196,6 +204,14 @@ export const CUSTOMER_INTERVIEW_APP_0_0_1 = {
196
204
  ],
197
205
  components: [],
198
206
  assistant: 'cat',
207
+ voice: {
208
+ enabled: false,
209
+ input: 'push_to_talk',
210
+ output: 'on_request',
211
+ voice: '',
212
+ language: '',
213
+ where: 'auto',
214
+ },
199
215
  },
200
216
  tests: {
201
217
  readyAt: 0.8,
@@ -303,6 +319,14 @@ export const DATA_QUALITY_APP_0_0_1 = {
303
319
  'TextField',
304
320
  'Button',
305
321
  ],
322
+ voice: {
323
+ enabled: false,
324
+ input: 'push_to_talk',
325
+ output: 'on_request',
326
+ voice: '',
327
+ language: '',
328
+ where: 'auto',
329
+ },
306
330
  },
307
331
  tests: {
308
332
  readyAt: 0.8,
@@ -450,6 +474,14 @@ export const DECIDE_APP_0_0_1 = {
450
474
  settings: [],
451
475
  components: [],
452
476
  assistant: 'wizard',
477
+ voice: {
478
+ enabled: false,
479
+ input: 'push_to_talk',
480
+ output: 'on_request',
481
+ voice: '',
482
+ language: '',
483
+ where: 'auto',
484
+ },
453
485
  },
454
486
  tests: {
455
487
  readyAt: 0.8,
@@ -593,6 +625,14 @@ export const INBOX_TRIAGE_APP_0_0_1 = {
593
625
  ],
594
626
  settings: [],
595
627
  components: [],
628
+ voice: {
629
+ enabled: false,
630
+ input: 'push_to_talk',
631
+ output: 'on_request',
632
+ voice: '',
633
+ language: '',
634
+ where: 'auto',
635
+ },
596
636
  },
597
637
  tests: {
598
638
  readyAt: 0.9,
@@ -722,6 +762,14 @@ export const MODEL_CHOICE_APP_0_0_1 = {
722
762
  'TextField',
723
763
  'Button',
724
764
  ],
765
+ voice: {
766
+ enabled: false,
767
+ input: 'push_to_talk',
768
+ output: 'on_request',
769
+ voice: '',
770
+ language: '',
771
+ where: 'auto',
772
+ },
725
773
  },
726
774
  tests: {
727
775
  readyAt: 0.8,
@@ -1046,6 +1094,14 @@ export const PIPELINE_REPORT_APP_0_0_1 = {
1046
1094
  composedAt: '',
1047
1095
  },
1048
1096
  assistant: 'eyes',
1097
+ voice: {
1098
+ enabled: false,
1099
+ input: 'push_to_talk',
1100
+ output: 'on_request',
1101
+ voice: '',
1102
+ language: '',
1103
+ where: 'auto',
1104
+ },
1049
1105
  },
1050
1106
  tests: {
1051
1107
  readyAt: 0.9,
@@ -1332,6 +1388,14 @@ export const QUOTE_CALCULATOR_APP_0_0_1 = {
1332
1388
  composedBy: 'template',
1333
1389
  composedAt: '',
1334
1390
  },
1391
+ voice: {
1392
+ enabled: false,
1393
+ input: 'push_to_talk',
1394
+ output: 'on_request',
1395
+ voice: '',
1396
+ language: '',
1397
+ where: 'auto',
1398
+ },
1335
1399
  },
1336
1400
  tests: {
1337
1401
  readyAt: 1.0,
@@ -1559,6 +1623,14 @@ export const REPORT_FROM_A_FILE_APP_0_0_1 = {
1559
1623
  composedAt: '',
1560
1624
  },
1561
1625
  assistant: 'wizard',
1626
+ voice: {
1627
+ enabled: false,
1628
+ input: 'push_to_talk',
1629
+ output: 'on_request',
1630
+ voice: '',
1631
+ language: '',
1632
+ where: 'auto',
1633
+ },
1562
1634
  },
1563
1635
  tests: {
1564
1636
  readyAt: 0.8,
@@ -1672,6 +1744,14 @@ export const SALES_APP_0_0_1 = {
1672
1744
  settings: [],
1673
1745
  components: [],
1674
1746
  assistant: 'paperclip',
1747
+ voice: {
1748
+ enabled: false,
1749
+ input: 'push_to_talk',
1750
+ output: 'on_request',
1751
+ voice: '',
1752
+ language: '',
1753
+ where: 'auto',
1754
+ },
1675
1755
  },
1676
1756
  tests: {
1677
1757
  readyAt: 0.8,
@@ -1778,6 +1858,14 @@ export const SHIP_OR_FIX_APP_0_0_1 = {
1778
1858
  'TextField',
1779
1859
  'Button',
1780
1860
  ],
1861
+ voice: {
1862
+ enabled: false,
1863
+ input: 'push_to_talk',
1864
+ output: 'on_request',
1865
+ voice: '',
1866
+ language: '',
1867
+ where: 'auto',
1868
+ },
1781
1869
  },
1782
1870
  tests: {
1783
1871
  readyAt: 0.8,
@@ -1945,6 +2033,14 @@ export const SUPPLIER_COMPARISON_APP_0_0_1 = {
1945
2033
  'TextField',
1946
2034
  'Button',
1947
2035
  ],
2036
+ voice: {
2037
+ enabled: false,
2038
+ input: 'push_to_talk',
2039
+ output: 'on_request',
2040
+ voice: '',
2041
+ language: '',
2042
+ where: 'auto',
2043
+ },
1948
2044
  },
1949
2045
  tests: {
1950
2046
  readyAt: 0.8,
@@ -2230,6 +2326,14 @@ export const SUPPORT_DESK_APP_0_0_1 = {
2230
2326
  composedAt: '',
2231
2327
  },
2232
2328
  assistant: 'paperclip',
2329
+ voice: {
2330
+ enabled: false,
2331
+ input: 'push_to_talk',
2332
+ output: 'on_request',
2333
+ voice: '',
2334
+ language: '',
2335
+ where: 'auto',
2336
+ },
2233
2337
  },
2234
2338
  tests: {
2235
2339
  readyAt: 0.8,
@@ -2356,6 +2460,14 @@ export const WEB_RESEARCH_APP_0_0_1 = {
2356
2460
  },
2357
2461
  ],
2358
2462
  components: [],
2463
+ voice: {
2464
+ enabled: false,
2465
+ input: 'push_to_talk',
2466
+ output: 'on_request',
2467
+ voice: '',
2468
+ language: '',
2469
+ where: 'auto',
2470
+ },
2359
2471
  },
2360
2472
  tests: {
2361
2473
  readyAt: 0.8,
@@ -326,6 +326,10 @@ export const APPSPEC_SCHEMA = {
326
326
  description: "The character its floating assistant shows, by the id a plugin contributes it under (lowercase letters and digits, words joined by a hyphen): Datalayer's are `paperclip`, `wizard`, `cat` and `eyes`. The paper clip when unsaid; an id no enabled plugin contributes is refused where the plugins are known, the runtime and the page. Said here, it wins over a person's own choice in their settings",
327
327
  title: 'Assistant',
328
328
  },
329
+ voice: {
330
+ $ref: '#/$defs/AppVoice',
331
+ description: 'Its voice: whether it listens and speaks, with which voice, in which language (off unless said)',
332
+ },
329
333
  },
330
334
  title: 'AppInterface',
331
335
  type: 'object',
@@ -716,6 +720,48 @@ export const APPSPEC_SCHEMA = {
716
720
  title: 'AppVerified',
717
721
  type: 'object',
718
722
  },
723
+ AppVoice: {
724
+ additionalProperties: false,
725
+ description: 'Its voice (VOICE.md VO-41): whether it listens, whether it speaks, with which voice, in which language.\n\nOff unless said. What is said becomes a message, and what is heard is the\nanswer the conversation shows: the text stays the truth.',
726
+ properties: {
727
+ enabled: {
728
+ default: false,
729
+ description: 'Whether it has a voice at all; off unless said',
730
+ title: 'Enabled',
731
+ type: 'boolean',
732
+ },
733
+ input: {
734
+ $ref: '#/$defs/VoiceInput',
735
+ default: 'push_to_talk',
736
+ description: '`off`, `push_to_talk` (hold a key or the microphone, speak, let go) or `hands_free`',
737
+ },
738
+ output: {
739
+ $ref: '#/$defs/VoiceOutput',
740
+ default: 'on_request',
741
+ description: 'When its answers are heard: `off`, `on_request` (a Read aloud on each answer) or `always`',
742
+ },
743
+ voice: {
744
+ default: '',
745
+ description: "The voice it speaks with, an id of the voice catalogue (`kokoro-af-heart`); the language's first when unsaid",
746
+ title: 'Voice',
747
+ type: 'string',
748
+ },
749
+ language: {
750
+ default: '',
751
+ description: "The language it listens and speaks in, BCP 47 (`en-US`, `fr-FR`); the person's when unsaid",
752
+ pattern: '^(?:[a-z]{2,3}(?:-[A-Z]{2})?)?$',
753
+ title: 'Language',
754
+ type: 'string',
755
+ },
756
+ where: {
757
+ $ref: '#/$defs/VoiceWhere',
758
+ default: 'auto',
759
+ description: "Where its speech runs: `auto`, `device` (the person's browser) or `server` (Datalayer's)",
760
+ },
761
+ },
762
+ title: 'AppVoice',
763
+ type: 'object',
764
+ },
719
765
  Behaviour: {
720
766
  description: 'What an application does when it meets an action: the four a person chooses from.',
721
767
  enum: ['do_it', 'if_asked', 'ask_first', 'leave_to_me'],
@@ -791,6 +837,7 @@ export const APPSPEC_SCHEMA = {
791
837
  'sources',
792
838
  'outputs',
793
839
  'feedback',
840
+ 'audio',
794
841
  ],
795
842
  title: 'RecordItem',
796
843
  type: 'string',
@@ -813,6 +860,24 @@ export const APPSPEC_SCHEMA = {
813
860
  title: 'Visibility',
814
861
  type: 'string',
815
862
  },
863
+ VoiceInput: {
864
+ description: 'How a person talks to the application (VOICE.md VO-10, VO-12).',
865
+ enum: ['off', 'push_to_talk', 'hands_free'],
866
+ title: 'VoiceInput',
867
+ type: 'string',
868
+ },
869
+ VoiceOutput: {
870
+ description: 'When its answers are heard (VO-21).',
871
+ enum: ['off', 'on_request', 'always'],
872
+ title: 'VoiceOutput',
873
+ type: 'string',
874
+ },
875
+ VoiceWhere: {
876
+ description: "Where its speech runs: in the person's browser, on Datalayer's servers, or the better of the two (§5).",
877
+ enum: ['auto', 'device', 'server'],
878
+ title: 'VoiceWhere',
879
+ type: 'string',
880
+ },
816
881
  },
817
882
  additionalProperties: false,
818
883
  description: 'Specification for an application.',
@@ -14,6 +14,7 @@ export * from './memory';
14
14
  export * from './models';
15
15
  export * from './modelProviders';
16
16
  export * from './notifications';
17
+ export * from './voices';
17
18
  export * from './outputs';
18
19
  export * from './skills';
19
20
  export * from './backendTools';
@@ -18,6 +18,7 @@ export * from './memory';
18
18
  export * from './models';
19
19
  export * from './modelProviders';
20
20
  export * from './notifications';
21
+ export * from './voices';
21
22
  export * from './outputs';
22
23
  export * from './skills';
23
24
  export * from './backendTools';
@@ -0,0 +1,72 @@
1
+ /**
2
+ * The voice catalogue (VOICE.md VO-40): voices and speech models.
3
+ *
4
+ * This file is AUTO-GENERATED from agentspecs (voices, speech-models).
5
+ * DO NOT EDIT MANUALLY - run 'make specs' to regenerate.
6
+ *
7
+ * @module specs/voices
8
+ */
9
+ /** Where a step of speech runs: the person's browser, or Datalayer's servers. */
10
+ export type SpeechWhere = 'device' | 'server';
11
+ /** The licences a voice or a model is admitted by. */
12
+ export interface SpeechLicence {
13
+ weights: string;
14
+ code?: string;
15
+ dataset?: string;
16
+ }
17
+ /** A voice an application may speak with. */
18
+ export interface VoiceSpec {
19
+ id: string;
20
+ version: string;
21
+ name: string;
22
+ description: string;
23
+ engine: string;
24
+ /** The speech model it is a voice of. */
25
+ model: string;
26
+ /** The engine's own name for it. */
27
+ voice: string;
28
+ /** BCP 47. */
29
+ languages: string[];
30
+ where: SpeechWhere[];
31
+ licence: SpeechLicence;
32
+ /** What a page listing the voices shows, when the licence asks. */
33
+ attribution: string;
34
+ watermark: boolean;
35
+ sample: string;
36
+ }
37
+ /** One file of a model, pinned by its hash and its size. */
38
+ export interface PinnedFile {
39
+ path: string;
40
+ sha256: string;
41
+ size: number;
42
+ }
43
+ /** A model of speech: to text, to speech, or voice activity. */
44
+ export interface SpeechModelSpec {
45
+ id: string;
46
+ version: string;
47
+ name: string;
48
+ task: 'stt' | 'tts' | 'vad';
49
+ engine: string;
50
+ dtype: string;
51
+ languages: string[];
52
+ where: SpeechWhere[];
53
+ streaming: boolean;
54
+ licence: SpeechLicence;
55
+ attribution: string;
56
+ upstream: string;
57
+ revision: string;
58
+ files: PinnedFile[];
59
+ }
60
+ /** The licences the register allows anywhere, the browser included. */
61
+ export declare const SPEECH_ALLOWED_LICENCES: readonly string[];
62
+ export declare const VOICE_CATALOGUE: Record<string, VoiceSpec>;
63
+ export declare const SPEECH_MODEL_CATALOGUE: Record<string, SpeechModelSpec>;
64
+ /** Whether a voice speaks a language: `fr` and `fr-FR` are spoken by a `fr-FR` voice. */
65
+ export declare function voiceSpeaks(voice: VoiceSpec, language: string): boolean;
66
+ /** The voice a language is spoken with when none is said: the first that speaks it. */
67
+ export declare function voiceFor(language: string): VoiceSpec | undefined;
68
+ /**
69
+ * The speech-to-text model for a language, on the device: Moonshine where it
70
+ * hears it, Whisper otherwise (VOICE.md decision 3).
71
+ */
72
+ export declare function transcriberFor(language: string): SpeechModelSpec | undefined;