@voicethere/agent 0.9.0 → 0.9.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@voicethere/agent",
3
- "version": "0.9.0",
3
+ "version": "0.9.2",
4
4
  "description": "VoiceThere customer agent SDK — IPC types and runtime helpers for sandboxed child bundles",
5
5
  "type": "module",
6
6
  "exports": {
@@ -17,10 +17,10 @@ import {
17
17
 
18
18
  ## Product vs e2e
19
19
 
20
- | Kind | Dashboard create | Prebuilt seed bundle | Typical consumer |
21
- | ----------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------ | ----------------------- |
20
+ | Kind | Dashboard create | Prebuilt seed bundle | Typical consumer |
21
+ | ----------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------ | ----------------------- |
22
22
  | **product** | Yes (`echo`, `echo-dc`, `voice-starter`, `language-switch`, `world-sync`, `world-sync-binary`, `game-sync`, `voice-showcase`, `recording-consent`, `positional-tts`, `spatial-showcase`, `webhooks`, `webhooks-redis`) | Yes — `dist/templates/<id>/agent.js` | Platform project create |
23
- | **e2e** | No | No — build from sources at test time | `voicethere/e2e` smokes |
23
+ | **e2e** | No | No — build from sources at test time | `voicethere/e2e` smokes |
24
24
 
25
25
  Product templates always set `seedOnCreate: true`. CI fails if a product template is missing its prebuilt bundle after `npm run build`.
26
26
 
@@ -47,11 +47,11 @@ npm run build:templates
47
47
 
48
48
  Three product templates cover positional / world sync, from JSON to Redis:
49
49
 
50
- | Id | Channel | State |
51
- | -------------------- | ------------------------------- | ----------------------------- |
52
- | `world-sync` | `onDataChannelMessage` (JSON) | One agent, in-memory, no Redis |
53
- | `world-sync-binary` | `onDataChannelBinary` + `broadCastBinaryToClients` (`ArrayBuffer`) | One agent, in-memory, no Redis |
54
- | `game-sync` | JSON control + binary world snapshots | Redis when `AGENT_REDIS_URL` is set |
50
+ | Id | Channel | State |
51
+ | ------------------- | ------------------------------------------------------------------ | ----------------------------------- |
52
+ | `world-sync` | `onDataChannelMessage` (JSON) | One agent, in-memory, no Redis |
53
+ | `world-sync-binary` | `onDataChannelBinary` + `broadCastBinaryToClients` (`ArrayBuffer`) | One agent, in-memory, no Redis |
54
+ | `game-sync` | JSON control + binary world snapshots | Redis when `AGENT_REDIS_URL` is set |
55
55
 
56
56
  See each folder README for the wire format.
57
57
 
@@ -59,33 +59,33 @@ See each folder README for the wire format.
59
59
 
60
60
  Each folder has its own README. Summary:
61
61
 
62
- | Id | Folder | Summary |
63
- | -- | ------ | ------- |
64
- | `echo` | `echo/` | Voice + chat echo |
65
- | `echo-dc` | `echo-dc/` | Data-channel echo, no TTS |
66
- | `voice-starter` | `voice-starter/` | Every speech event |
67
- | `language-switch` | `language-switch/` | Separate `setVoiceLanguage` calls for STT, TTS, and vendor |
68
- | `world-sync` | `world-sync/` | JSON pose broadcast |
69
- | `world-sync-binary` | `world-sync-binary/` | Binary pose `ArrayBuffer` |
70
- | `game-sync` | `game-sync/` | Authoritative sim, Redis + binary snapshots |
71
- | `voice-showcase` | `voice-showcase/` | Conversational landing demo |
72
- | `recording-consent` | `recording-consent/` | Recording consent flow |
73
- | `positional-tts` | `positional-tts/` | Orbiting TTS |
74
- | `spatial-showcase` | `spatial-showcase/` | Orbit / soundboard / proximity |
75
- | `webhooks` | `webhooks/` | Inbound HMAC webhooks |
76
- | `webhooks-redis` | `webhooks-redis/` | Webhooks plus Redis counter |
62
+ | Id | Folder | Summary |
63
+ | ------------------- | -------------------- | ----------------------------------------------------------------- |
64
+ | `echo` | `echo/` | Voice + chat echo |
65
+ | `echo-dc` | `echo-dc/` | Data-channel echo, no TTS |
66
+ | `voice-starter` | `voice-starter/` | Every speech event |
67
+ | `language-switch` | `language-switch/` | Manual `setVoiceLanguage`, or runner auto-switch with a localized echo of the recognized text |
68
+ | `world-sync` | `world-sync/` | JSON pose broadcast |
69
+ | `world-sync-binary` | `world-sync-binary/` | Binary pose `ArrayBuffer` |
70
+ | `game-sync` | `game-sync/` | Authoritative sim, Redis + binary snapshots |
71
+ | `voice-showcase` | `voice-showcase/` | Conversational landing demo |
72
+ | `recording-consent` | `recording-consent/` | Recording consent flow |
73
+ | `positional-tts` | `positional-tts/` | Orbiting TTS |
74
+ | `spatial-showcase` | `spatial-showcase/` | Orbit / soundboard / proximity |
75
+ | `webhooks` | `webhooks/` | Inbound HMAC webhooks |
76
+ | `webhooks-redis` | `webhooks-redis/` | Webhooks plus Redis counter |
77
77
 
78
78
  ## E2e templates
79
79
 
80
80
  These mirror former `e2e/fixtures/*` sources. E2E resolves entries from the package, builds into ephemeral workdirs, and uploads `dist/agent.js`.
81
81
 
82
- | Id | Source | Purpose |
83
- | ----------------- | ---------------------------------------------- | --------------------------------------------- |
84
- | `echo-smoke` | `echo-smoke/agent.ts` | voice-smoke, agent-smoke, cli-smoke |
85
- | `crash` | `crash/agent.ts` | session-errors-smoke, crash-policy smokes |
86
- | `game-sync-smoke` | `game-sync-smoke/agent.ts` | deploy-smoke, shared-child, idle smokes |
87
- | `redis-sync` | `redis-sync/agent.ts` + `world-layout.ts` | redis-sync-smoke (binary positions + Redis world blob) |
88
- | `mix-smoke` | `mix-smoke/agent.ts` | voice-data-mix-smoke |
82
+ | Id | Source | Purpose |
83
+ | ----------------- | ----------------------------------------- | ------------------------------------------------------ |
84
+ | `echo-smoke` | `echo-smoke/agent.ts` | voice-smoke, agent-smoke, cli-smoke |
85
+ | `crash` | `crash/agent.ts` | session-errors-smoke, crash-policy smokes |
86
+ | `game-sync-smoke` | `game-sync-smoke/agent.ts` | deploy-smoke, shared-child, idle smokes |
87
+ | `redis-sync` | `redis-sync/agent.ts` + `world-layout.ts` | redis-sync-smoke (binary positions + Redis world blob) |
88
+ | `mix-smoke` | `mix-smoke/agent.ts` | voice-data-mix-smoke |
89
89
 
90
90
  **Note:** Product `echo` is not the same as e2e `echo-smoke` — keep both ids.
91
91
 
@@ -1,12 +1,22 @@
1
1
  # language-switch
2
2
 
3
- Spoken-language detection tells the agent which ISO 639-1 code it heard. It does **not** change the TTS voice or the STT model. This template changes the two sides separately with `setVoiceLanguage`:
3
+ Spoken-language detection tells the agent which ISO 639-1 code it heard. It does **not** change the TTS voice or the STT model by itself. This template shows how to react in two ways:
4
+
5
+ ## Mode A — Manual switch (default)
6
+
7
+ Leave the project **spoken-language auto-switch** setting off. When `onUserLanguage` fires, this agent calls `setVoiceLanguage` separately for each side:
4
8
 
5
9
  - `scope: "tts"` switches only the speaking voice
6
10
  - `scope: "stt"` switches only the listening model
7
11
 
8
12
  A Sherpa language with no STT id (`it`, `pt`, `nl`, `pl`, `hi`) fails the STT call and still switches TTS.
9
13
 
14
+ ## Mode B — Runner auto-switch
15
+
16
+ Enable auto-switch in the project voice settings (runner applies STT/TTS when LID detects a new language). When `session_start.env` includes a truthy `SHERPA_LID_AUTO_SWITCH`, this template **does not** call `setVoiceLanguage`. The runner replays the utterance into the new language's STT, so the next final is the recognized text in the new language. The agent remembers the language from `onUserLanguage` / `onVoiceLanguageChanged` and answers each final right away with a short localized prefix, for example `you said: …` in English and `Du hast gesagt: …` in German. Use `onVoiceLanguageChanged` when you need the committed language after the runner applies the change.
17
+
18
+ Detection and chat commands are unchanged: `/tts` and `/stt` still call `setVoiceLanguage` for one vendor at a time.
19
+
10
20
  Chat commands change one vendor while the session stays connected. API keys are project secrets on the running deploy, not arguments:
11
21
 
12
22
  - `/tts sherpa de` — Sherpa TTS only
@@ -14,7 +24,7 @@ Chat commands change one vendor while the session stays connected. API keys are
14
24
  - `/tts elevenlabs` — ElevenLabs TTS, current STT stays
15
25
  - `/stt deepgram de` — Deepgram STT, current voice stays
16
26
 
17
- English finals are echoed as `you said: …` after a short wait, so a language event for the same utterance can cancel the echo. The revealing utterance is not spoken back as an English transcript.
27
+ In manual mode, English finals are echoed as `you said: …` after a short wait, so a language event for the same utterance can cancel the echo. The revealing utterance is not spoken back as an English transcript.
18
28
 
19
29
  `voice` and `stt` are Sherpa catalog ids (`de`, `en-lessac`, `en-small`), listed in the spoken-language docs. Vendor ids are `local-sherpa`, `openai`, `deepgram`, `assemblyai`, `google`, `elevenlabs`, and `cartesia`.
20
30
 
@@ -1,10 +1,17 @@
1
1
  /**
2
2
  * Change STT and TTS separately, including the vendor, while the call stays up.
3
3
  *
4
- * The runner reports `user_language` and does not change either side. This
5
- * agent switches the Sherpa voice and the Sherpa STT model as two calls, so
6
- * one side can fail without blocking the other. Chat commands change a single
7
- * vendor mid-conversation:
4
+ * Two deployment modes (see README):
5
+ *
6
+ * (A) Project auto-switch off — this agent calls `setVoiceLanguage` from
7
+ * `onUserLanguage` when LID detects a new language.
8
+ * (B) Project enables runner auto-switch — STT/TTS are runner-owned. The
9
+ * runner replays the utterance into the new language's STT, so the next
10
+ * final is the correctly recognized text. The agent never calls
11
+ * `setVoiceLanguage`; it remembers the language and answers each final
12
+ * right away with a short localized prefix ("Du hast gesagt: …").
13
+ *
14
+ * Chat commands change a single vendor mid-conversation:
8
15
  *
9
16
  * - `/tts sherpa de` — Sherpa TTS only
10
17
  * - `/stt sherpa de` — Sherpa STT only
@@ -17,6 +24,7 @@
17
24
  import {
18
25
  agentLog,
19
26
  defineAgent,
27
+ isRunnerLidAutoSwitchEnabled,
20
28
  parseChatText,
21
29
  setVoiceLanguage,
22
30
  speak,
@@ -26,6 +34,19 @@ import {
26
34
  const ECHO_PREFIX = "you said:";
27
35
  const ECHO_WAIT_MS = 4000;
28
36
 
37
+ /** Echo prefix per language for runner auto-switch mode. */
38
+ const ECHO_PREFIXES: Record<string, string> = {
39
+ en: "you said:",
40
+ de: "Du hast gesagt:",
41
+ es: "Dijiste:",
42
+ fr: "Tu as dit :",
43
+ it: "Hai detto:",
44
+ pt: "Você disse:",
45
+ nl: "Je zei:",
46
+ pl: "Powiedziałeś:",
47
+ ru: "Вы сказали:",
48
+ };
49
+
29
50
  const REPLIES: Record<string, string> = {
30
51
  de: "Guten Tag. Ich antworte jetzt auf Deutsch.",
31
52
  es: "Hola. Ahora respondo en español.",
@@ -39,17 +60,22 @@ const REPLIES: Record<string, string> = {
39
60
 
40
61
  type SessionLanguage = {
41
62
  language: string;
63
+ runnerAutoSwitch: boolean;
42
64
  echoTimer: ReturnType<typeof setTimeout> | undefined;
43
65
  suppressNextFinal: boolean;
44
66
  };
45
67
 
46
68
  const sessions = new Map<string, SessionLanguage>();
47
69
 
48
- function stateFor(sessionId: string): SessionLanguage {
70
+ function stateFor(
71
+ sessionId: string,
72
+ env?: Record<string, string>,
73
+ ): SessionLanguage {
49
74
  const existing = sessions.get(sessionId);
50
75
  if (existing) return existing;
51
76
  const created: SessionLanguage = {
52
77
  language: "en",
78
+ runnerAutoSwitch: env ? isRunnerLidAutoSwitchEnabled(env) : false,
53
79
  echoTimer: undefined,
54
80
  suppressNextFinal: false,
55
81
  };
@@ -61,9 +87,20 @@ function replyFor(language: string): string {
61
87
  return REPLIES[language] ?? `Continuing in ${language}.`;
62
88
  }
63
89
 
64
- function logSwitch(sessionId: string, side: string, result: VoiceLanguageResult): void {
90
+ function echoPrefixFor(language: string): string {
91
+ return ECHO_PREFIXES[language] ?? ECHO_PREFIXES.en!;
92
+ }
93
+
94
+ function logSwitch(
95
+ sessionId: string,
96
+ side: string,
97
+ result: VoiceLanguageResult,
98
+ ): void {
65
99
  if (!result.ok) {
66
- agentLog("warn", `setVoiceLanguage ${side} failed: ${result.reason ?? "unknown"}`);
100
+ agentLog(
101
+ "warn",
102
+ `setVoiceLanguage ${side} failed: ${result.reason ?? "unknown"}`,
103
+ );
67
104
  return;
68
105
  }
69
106
  const provider = side === "tts" ? result.ttsProvider : result.sttProvider;
@@ -74,23 +111,49 @@ function logSwitch(sessionId: string, side: string, result: VoiceLanguageResult)
74
111
  );
75
112
  }
76
113
 
114
+ function prepareLanguageTransition(state: SessionLanguage): void {
115
+ if (state.echoTimer) {
116
+ clearTimeout(state.echoTimer);
117
+ state.echoTimer = undefined;
118
+ } else {
119
+ state.suppressNextFinal = true;
120
+ }
121
+ }
122
+
77
123
  defineAgent({
78
- onSessionStart({ sessionId }) {
79
- stateFor(sessionId);
80
- agentLog("info", `language-switch session_start ${sessionId}`);
124
+ onSessionStart({ sessionId, env }) {
125
+ const state = stateFor(sessionId, env);
126
+ if (state.runnerAutoSwitch) {
127
+ agentLog(
128
+ "info",
129
+ `language-switch session_start ${sessionId} runner LID auto-switch owns STT/TTS`,
130
+ );
131
+ } else {
132
+ agentLog(
133
+ "info",
134
+ `language-switch session_start ${sessionId} manual setVoiceLanguage`,
135
+ );
136
+ }
81
137
  },
82
138
 
83
139
  async onUserLanguage({ sessionId, language }) {
84
140
  const state = stateFor(sessionId);
85
141
  if (!language || language === state.language) return;
86
142
 
87
- if (state.echoTimer) {
88
- clearTimeout(state.echoTimer);
89
- state.echoTimer = undefined;
90
- } else {
91
- state.suppressNextFinal = true;
143
+ if (state.runnerAutoSwitch) {
144
+ // The runner switches STT/TTS and replays the utterance into the new
145
+ // STT; the next final is the real utterance, so nothing is suppressed
146
+ // and no fixed sentence is spoken.
147
+ agentLog(
148
+ "info",
149
+ `LID detected ${language}; runner auto-switch applies STT/TTS — agent only remembers the language`,
150
+ );
151
+ state.language = language;
152
+ return;
92
153
  }
93
154
 
155
+ prepareLanguageTransition(state);
156
+
94
157
  const tts = await setVoiceLanguage(sessionId, {
95
158
  scope: "tts",
96
159
  language,
@@ -114,6 +177,18 @@ defineAgent({
114
177
  speak(sessionId, replyFor(language));
115
178
  },
116
179
 
180
+ async onVoiceLanguageChanged({ sessionId, language }) {
181
+ const state = stateFor(sessionId);
182
+ if (!state.runnerAutoSwitch || !language || language === state.language) {
183
+ return;
184
+ }
185
+ agentLog(
186
+ "info",
187
+ `runner committed voice language ${language} for ${sessionId}`,
188
+ );
189
+ state.language = language;
190
+ },
191
+
117
192
  async onDataChannelMessage(ctx) {
118
193
  const text = parseChatText(ctx.message);
119
194
  if (!text) return;
@@ -161,6 +236,12 @@ defineAgent({
161
236
 
162
237
  onUserSpeechFinal({ sessionId, text }) {
163
238
  const state = stateFor(sessionId);
239
+ if (state.runnerAutoSwitch) {
240
+ // Reply immediately: the runner already holds the final until LID has
241
+ // decided, so no extra wait is needed to avoid a wrong-language echo.
242
+ speak(sessionId, `${echoPrefixFor(state.language)} ${text}`.trim());
243
+ return;
244
+ }
164
245
  if (state.suppressNextFinal) {
165
246
  state.suppressNextFinal = false;
166
247
  return;