@voicethere/agent 0.8.3 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@voicethere/agent",
3
- "version": "0.8.3",
3
+ "version": "0.9.1",
4
4
  "description": "VoiceThere customer agent SDK — IPC types and runtime helpers for sandboxed child bundles",
5
5
  "type": "module",
6
6
  "exports": {
@@ -17,10 +17,10 @@ import {
17
17
 
18
18
  ## Product vs e2e
19
19
 
20
- | Kind | Dashboard create | Prebuilt seed bundle | Typical consumer |
21
- | ----------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------ | ----------------------- |
22
- | **product** | Yes (`echo`, `echo-dc`, `voice-starter`, `world-sync`, `world-sync-binary`, `game-sync`, `voice-showcase`, `recording-consent`, `positional-tts`, `spatial-showcase`, `webhooks`, `webhooks-redis`) | Yes — `dist/templates/<id>/agent.js` | Platform project create |
23
- | **e2e** | No | No — build from sources at test time | `voicethere/e2e` smokes |
20
+ | Kind | Dashboard create | Prebuilt seed bundle | Typical consumer |
21
+ | ----------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------ | ----------------------- |
22
+ | **product** | Yes (`echo`, `echo-dc`, `voice-starter`, `language-switch`, `world-sync`, `world-sync-binary`, `game-sync`, `voice-showcase`, `recording-consent`, `positional-tts`, `spatial-showcase`, `webhooks`, `webhooks-redis`) | Yes — `dist/templates/<id>/agent.js` | Platform project create |
23
+ | **e2e** | No | No — build from sources at test time | `voicethere/e2e` smokes |
24
24
 
25
25
  Product templates always set `seedOnCreate: true`. CI fails if a product template is missing its prebuilt bundle after `npm run build`.
26
26
 
@@ -47,11 +47,11 @@ npm run build:templates
47
47
 
48
48
  Three product templates cover positional / world sync, from JSON to Redis:
49
49
 
50
- | Id | Channel | State |
51
- | -------------------- | ------------------------------- | ----------------------------- |
52
- | `world-sync` | `onDataChannelMessage` (JSON) | One agent, in-memory, no Redis |
53
- | `world-sync-binary` | `onDataChannelBinary` + `broadCastBinaryToClients` (`ArrayBuffer`) | One agent, in-memory, no Redis |
54
- | `game-sync` | JSON control + binary world snapshots | Redis when `AGENT_REDIS_URL` is set |
50
+ | Id | Channel | State |
51
+ | ------------------- | ------------------------------------------------------------------ | ----------------------------------- |
52
+ | `world-sync` | `onDataChannelMessage` (JSON) | One agent, in-memory, no Redis |
53
+ | `world-sync-binary` | `onDataChannelBinary` + `broadCastBinaryToClients` (`ArrayBuffer`) | One agent, in-memory, no Redis |
54
+ | `game-sync` | JSON control + binary world snapshots | Redis when `AGENT_REDIS_URL` is set |
55
55
 
56
56
  See each folder README for the wire format.
57
57
 
@@ -59,32 +59,33 @@ See each folder README for the wire format.
59
59
 
60
60
  Each folder has its own README. Summary:
61
61
 
62
- | Id | Folder | Summary |
63
- | -- | ------ | ------- |
64
- | `echo` | `echo/` | Voice + chat echo |
65
- | `echo-dc` | `echo-dc/` | Data-channel echo, no TTS |
66
- | `voice-starter` | `voice-starter/` | Every speech event |
67
- | `world-sync` | `world-sync/` | JSON pose broadcast |
68
- | `world-sync-binary` | `world-sync-binary/` | Binary pose `ArrayBuffer` |
69
- | `game-sync` | `game-sync/` | Authoritative sim, Redis + binary snapshots |
70
- | `voice-showcase` | `voice-showcase/` | Conversational landing demo |
71
- | `recording-consent` | `recording-consent/` | Recording consent flow |
72
- | `positional-tts` | `positional-tts/` | Orbiting TTS |
73
- | `spatial-showcase` | `spatial-showcase/` | Orbit / soundboard / proximity |
74
- | `webhooks` | `webhooks/` | Inbound HMAC webhooks |
75
- | `webhooks-redis` | `webhooks-redis/` | Webhooks plus Redis counter |
62
+ | Id | Folder | Summary |
63
+ | ------------------- | -------------------- | ----------------------------------------------------------------- |
64
+ | `echo` | `echo/` | Voice + chat echo |
65
+ | `echo-dc` | `echo-dc/` | Data-channel echo, no TTS |
66
+ | `voice-starter` | `voice-starter/` | Every speech event |
67
+ | `language-switch` | `language-switch/` | Manual `setVoiceLanguage` or runner auto-switch + prompt handlers |
68
+ | `world-sync` | `world-sync/` | JSON pose broadcast |
69
+ | `world-sync-binary` | `world-sync-binary/` | Binary pose `ArrayBuffer` |
70
+ | `game-sync` | `game-sync/` | Authoritative sim, Redis + binary snapshots |
71
+ | `voice-showcase` | `voice-showcase/` | Conversational landing demo |
72
+ | `recording-consent` | `recording-consent/` | Recording consent flow |
73
+ | `positional-tts` | `positional-tts/` | Orbiting TTS |
74
+ | `spatial-showcase` | `spatial-showcase/` | Orbit / soundboard / proximity |
75
+ | `webhooks` | `webhooks/` | Inbound HMAC webhooks |
76
+ | `webhooks-redis` | `webhooks-redis/` | Webhooks plus Redis counter |
76
77
 
77
78
  ## E2e templates
78
79
 
79
80
  These mirror former `e2e/fixtures/*` sources. E2E resolves entries from the package, builds into ephemeral workdirs, and uploads `dist/agent.js`.
80
81
 
81
- | Id | Source | Purpose |
82
- | ----------------- | ---------------------------------------------- | --------------------------------------------- |
83
- | `echo-smoke` | `echo-smoke/agent.ts` | voice-smoke, agent-smoke, cli-smoke |
84
- | `crash` | `crash/agent.ts` | session-errors-smoke, crash-policy smokes |
85
- | `game-sync-smoke` | `game-sync-smoke/agent.ts` | deploy-smoke, shared-child, idle smokes |
86
- | `redis-sync` | `redis-sync/agent.ts` + `world-layout.ts` | redis-sync-smoke (binary positions + Redis world blob) |
87
- | `mix-smoke` | `mix-smoke/agent.ts` | voice-data-mix-smoke |
82
+ | Id | Source | Purpose |
83
+ | ----------------- | ----------------------------------------- | ------------------------------------------------------ |
84
+ | `echo-smoke` | `echo-smoke/agent.ts` | voice-smoke, agent-smoke, cli-smoke |
85
+ | `crash` | `crash/agent.ts` | session-errors-smoke, crash-policy smokes |
86
+ | `game-sync-smoke` | `game-sync-smoke/agent.ts` | deploy-smoke, shared-child, idle smokes |
87
+ | `redis-sync` | `redis-sync/agent.ts` + `world-layout.ts` | redis-sync-smoke (binary positions + Redis world blob) |
88
+ | `mix-smoke` | `mix-smoke/agent.ts` | voice-data-mix-smoke |
88
89
 
89
90
  **Note:** Product `echo` is not the same as e2e `echo-smoke` — keep both ids.
90
91
 
@@ -0,0 +1,31 @@
1
+ # language-switch
2
+
3
+ Spoken-language detection tells the agent which ISO 639-1 code it heard. It does **not** change the TTS voice or the STT model by itself. This template shows how to react in two ways:
4
+
5
+ ## Mode A — Manual switch (default)
6
+
7
+ Leave the project **spoken-language auto-switch** setting off. When `onUserLanguage` fires, this agent calls `setVoiceLanguage` separately for each side:
8
+
9
+ - `scope: "tts"` switches only the speaking voice
10
+ - `scope: "stt"` switches only the listening model
11
+
12
+ A Sherpa language with no STT id (`it`, `pt`, `nl`, `pl`, `hi`) fails the STT call and still switches TTS.
13
+
14
+ ## Mode B — Runner auto-switch
15
+
16
+ Enable auto-switch in the project voice settings (runner applies STT/TTS when LID detects a new language). When `session_start.env` includes a truthy `SHERPA_LID_AUTO_SWITCH`, this template **does not** call `setVoiceLanguage` from `onUserLanguage` — it logs that the runner owns the switch and updates prompts only. Use `onVoiceLanguageChanged` when you need the committed language after the runner applies the change.
17
+
18
+ Detection and chat commands are unchanged: `/tts` and `/stt` still call `setVoiceLanguage` for one vendor at a time.
19
+
20
+ Chat commands change one vendor while the session stays connected. API keys are project secrets on the running deploy, not arguments:
21
+
22
+ - `/tts sherpa de` — Sherpa TTS only
23
+ - `/stt sherpa de` — Sherpa STT only
24
+ - `/tts elevenlabs` — ElevenLabs TTS, current STT stays
25
+ - `/stt deepgram de` — Deepgram STT, current voice stays
26
+
27
+ English finals are echoed as `you said: …` after a short wait, so a language event for the same utterance can cancel the echo. The revealing utterance is not spoken back as an English transcript.
28
+
29
+ `voice` and `stt` are Sherpa catalog ids (`de`, `en-lessac`, `en-small`), listed in the spoken-language docs. Vendor ids are `local-sherpa`, `openai`, `deepgram`, `assemblyai`, `google`, `elevenlabs`, and `cartesia`.
30
+
31
+ Entry: `templates/language-switch/agent.ts`.
@@ -0,0 +1,233 @@
1
+ /**
2
+ * Change STT and TTS separately, including the vendor, while the call stays up.
3
+ *
4
+ * Two deployment modes (see README):
5
+ *
6
+ * (A) Project auto-switch off — this agent calls `setVoiceLanguage` from
7
+ * `onUserLanguage` when LID detects a new language.
8
+ * (B) Project enables runner auto-switch — STT/TTS are runner-owned; use
9
+ * `onUserLanguage` / `onVoiceLanguageChanged` for prompts only.
10
+ *
11
+ * Chat commands change a single vendor mid-conversation:
12
+ *
13
+ * - `/tts sherpa de` — Sherpa TTS only
14
+ * - `/stt sherpa de` — Sherpa STT only
15
+ * - `/tts elevenlabs` — ElevenLabs TTS, current STT stays
16
+ * - `/stt deepgram de` — Deepgram STT, current voice stays
17
+ *
18
+ * Vendor API keys are project secrets already on the running deploy. Do not
19
+ * put keys in this file. See /docs/spoken-language.
20
+ */
21
+ import {
22
+ agentLog,
23
+ defineAgent,
24
+ isRunnerLidAutoSwitchEnabled,
25
+ parseChatText,
26
+ setVoiceLanguage,
27
+ speak,
28
+ type VoiceLanguageResult,
29
+ } from "@voicethere/agent";
30
+
31
+ const ECHO_PREFIX = "you said:";
32
+ const ECHO_WAIT_MS = 4000;
33
+
34
+ const REPLIES: Record<string, string> = {
35
+ de: "Guten Tag. Ich antworte jetzt auf Deutsch.",
36
+ es: "Hola. Ahora respondo en español.",
37
+ fr: "Bonjour. Je réponds maintenant en français.",
38
+ it: "Ciao. Adesso rispondo in italiano.",
39
+ pt: "Olá. Agora respondo em português.",
40
+ nl: "Hallo. Ik antwoord nu in het Nederlands.",
41
+ pl: "Dzień dobry. Teraz odpowiadam po polsku.",
42
+ ru: "Здравствуйте. Теперь я отвечаю по-русски.",
43
+ };
44
+
45
+ type SessionLanguage = {
46
+ language: string;
47
+ runnerAutoSwitch: boolean;
48
+ echoTimer: ReturnType<typeof setTimeout> | undefined;
49
+ suppressNextFinal: boolean;
50
+ };
51
+
52
+ const sessions = new Map<string, SessionLanguage>();
53
+
54
+ function stateFor(
55
+ sessionId: string,
56
+ env?: Record<string, string>,
57
+ ): SessionLanguage {
58
+ const existing = sessions.get(sessionId);
59
+ if (existing) return existing;
60
+ const created: SessionLanguage = {
61
+ language: "en",
62
+ runnerAutoSwitch: env ? isRunnerLidAutoSwitchEnabled(env) : false,
63
+ echoTimer: undefined,
64
+ suppressNextFinal: false,
65
+ };
66
+ sessions.set(sessionId, created);
67
+ return created;
68
+ }
69
+
70
+ function replyFor(language: string): string {
71
+ return REPLIES[language] ?? `Continuing in ${language}.`;
72
+ }
73
+
74
+ function logSwitch(
75
+ sessionId: string,
76
+ side: string,
77
+ result: VoiceLanguageResult,
78
+ ): void {
79
+ if (!result.ok) {
80
+ agentLog(
81
+ "warn",
82
+ `setVoiceLanguage ${side} failed: ${result.reason ?? "unknown"}`,
83
+ );
84
+ return;
85
+ }
86
+ const provider = side === "tts" ? result.ttsProvider : result.sttProvider;
87
+ const model = side === "tts" ? result.voice : result.stt;
88
+ agentLog(
89
+ "info",
90
+ `voice ${side} ${sessionId} ${result.language ?? ""} provider=${provider ?? "unchanged"} model=${model ?? "unchanged"}`,
91
+ );
92
+ }
93
+
94
+ function prepareLanguageTransition(state: SessionLanguage): void {
95
+ if (state.echoTimer) {
96
+ clearTimeout(state.echoTimer);
97
+ state.echoTimer = undefined;
98
+ } else {
99
+ state.suppressNextFinal = true;
100
+ }
101
+ }
102
+
103
+ defineAgent({
104
+ onSessionStart({ sessionId, env }) {
105
+ const state = stateFor(sessionId, env);
106
+ if (state.runnerAutoSwitch) {
107
+ agentLog(
108
+ "info",
109
+ `language-switch session_start ${sessionId} runner LID auto-switch owns STT/TTS`,
110
+ );
111
+ } else {
112
+ agentLog(
113
+ "info",
114
+ `language-switch session_start ${sessionId} manual setVoiceLanguage`,
115
+ );
116
+ }
117
+ },
118
+
119
+ async onUserLanguage({ sessionId, language }) {
120
+ const state = stateFor(sessionId);
121
+ if (!language || language === state.language) return;
122
+
123
+ prepareLanguageTransition(state);
124
+
125
+ if (state.runnerAutoSwitch) {
126
+ agentLog(
127
+ "info",
128
+ `LID detected ${language}; runner auto-switch applies STT/TTS — agent updates prompts only`,
129
+ );
130
+ state.language = language;
131
+ speak(sessionId, replyFor(language));
132
+ return;
133
+ }
134
+
135
+ const tts = await setVoiceLanguage(sessionId, {
136
+ scope: "tts",
137
+ language,
138
+ voice: language,
139
+ });
140
+ logSwitch(sessionId, "tts", tts);
141
+
142
+ const stt = await setVoiceLanguage(sessionId, {
143
+ scope: "stt",
144
+ language,
145
+ stt: language,
146
+ });
147
+ logSwitch(sessionId, "stt", stt);
148
+
149
+ if (!tts.ok) {
150
+ state.suppressNextFinal = false;
151
+ return;
152
+ }
153
+
154
+ state.language = language;
155
+ speak(sessionId, replyFor(language));
156
+ },
157
+
158
+ async onVoiceLanguageChanged({ sessionId, language }) {
159
+ const state = stateFor(sessionId);
160
+ if (!state.runnerAutoSwitch || !language || language === state.language) {
161
+ return;
162
+ }
163
+ agentLog(
164
+ "info",
165
+ `runner committed voice language ${language} for ${sessionId}`,
166
+ );
167
+ state.language = language;
168
+ },
169
+
170
+ async onDataChannelMessage(ctx) {
171
+ const text = parseChatText(ctx.message);
172
+ if (!text) return;
173
+ const [command, vendor, language] = text.trim().split(/\s+/);
174
+ if (command === "/tts" && vendor === "sherpa" && language) {
175
+ const result = await setVoiceLanguage(ctx.sessionId, {
176
+ scope: "tts",
177
+ language,
178
+ ttsVendor: { provider: "local-sherpa", voice: language },
179
+ });
180
+ logSwitch(ctx.sessionId, "tts", result);
181
+ if (result.ok) speak(ctx.sessionId, replyFor(language));
182
+ return;
183
+ }
184
+ if (command === "/stt" && vendor === "sherpa" && language) {
185
+ const result = await setVoiceLanguage(ctx.sessionId, {
186
+ scope: "stt",
187
+ language,
188
+ sttVendor: { provider: "local-sherpa", model: language },
189
+ });
190
+ logSwitch(ctx.sessionId, "stt", result);
191
+ return;
192
+ }
193
+ if (command === "/tts" && vendor === "elevenlabs") {
194
+ const result = await setVoiceLanguage(ctx.sessionId, {
195
+ scope: "tts",
196
+ ttsVendor: { provider: "elevenlabs", model: "eleven_multilingual_v2" },
197
+ });
198
+ logSwitch(ctx.sessionId, "tts", result);
199
+ if (result.ok) speak(ctx.sessionId, "Speaking with ElevenLabs.");
200
+ return;
201
+ }
202
+ if (command === "/stt" && vendor === "deepgram") {
203
+ const result = await setVoiceLanguage(ctx.sessionId, {
204
+ scope: "stt",
205
+ sttVendor: {
206
+ provider: "deepgram",
207
+ model: "nova-3",
208
+ language: language || "en",
209
+ },
210
+ });
211
+ logSwitch(ctx.sessionId, "stt", result);
212
+ }
213
+ },
214
+
215
+ onUserSpeechFinal({ sessionId, text }) {
216
+ const state = stateFor(sessionId);
217
+ if (state.suppressNextFinal) {
218
+ state.suppressNextFinal = false;
219
+ return;
220
+ }
221
+ if (state.echoTimer) clearTimeout(state.echoTimer);
222
+ state.echoTimer = setTimeout(() => {
223
+ state.echoTimer = undefined;
224
+ speak(sessionId, `${ECHO_PREFIX} ${text}`.trim());
225
+ }, ECHO_WAIT_MS);
226
+ },
227
+
228
+ onSessionEnd({ sessionId }) {
229
+ const state = sessions.get(sessionId);
230
+ if (state?.echoTimer) clearTimeout(state.echoTimer);
231
+ sessions.delete(sessionId);
232
+ },
233
+ });
@@ -179,7 +179,9 @@ defineAgent({
179
179
  },
180
180
 
181
181
  onUserLanguage({ sessionId, language }) {
182
- // Route prompts / TTS voice from the detected ISO 639-1 code
182
+ // Detection does not change TTS or STT. Call setVoiceLanguage with
183
+ // scope "tts" or "stt" — or pass sttVendor / ttsVendor to change vendor.
184
+ // See templates/language-switch/agent.ts.
183
185
  agentLog("info", `user_language ${sessionId} ${language}`);
184
186
  },
185
187