felona-voice 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent.d.ts +105 -0
- package/dist/agent.d.ts.map +1 -0
- package/dist/agent.js +287 -0
- package/dist/agent.js.map +1 -0
- package/dist/analytics/index.d.ts +3 -0
- package/dist/analytics/index.d.ts.map +1 -0
- package/dist/analytics/index.js +2 -0
- package/dist/analytics/index.js.map +1 -0
- package/dist/analytics/logger.d.ts +92 -0
- package/dist/analytics/logger.d.ts.map +1 -0
- package/dist/analytics/logger.js +75 -0
- package/dist/analytics/logger.js.map +1 -0
- package/dist/index.d.ts +45 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +54 -0
- package/dist/index.js.map +1 -0
- package/dist/jev/action-space.d.ts +38 -0
- package/dist/jev/action-space.d.ts.map +1 -0
- package/dist/jev/action-space.js +92 -0
- package/dist/jev/action-space.js.map +1 -0
- package/dist/jev/embeddings.d.ts +23 -0
- package/dist/jev/embeddings.d.ts.map +1 -0
- package/dist/jev/embeddings.js +63 -0
- package/dist/jev/embeddings.js.map +1 -0
- package/dist/jev/engine.d.ts +101 -0
- package/dist/jev/engine.d.ts.map +1 -0
- package/dist/jev/engine.js +173 -0
- package/dist/jev/engine.js.map +1 -0
- package/dist/jev/index.d.ts +4 -0
- package/dist/jev/index.d.ts.map +1 -0
- package/dist/jev/index.js +4 -0
- package/dist/jev/index.js.map +1 -0
- package/dist/llm/index.d.ts +2 -0
- package/dist/llm/index.d.ts.map +1 -0
- package/dist/llm/index.js +2 -0
- package/dist/llm/index.js.map +1 -0
- package/dist/llm/openai.d.ts +22 -0
- package/dist/llm/openai.d.ts.map +1 -0
- package/dist/llm/openai.js +140 -0
- package/dist/llm/openai.js.map +1 -0
- package/dist/memory/context.d.ts +53 -0
- package/dist/memory/context.d.ts.map +1 -0
- package/dist/memory/context.js +95 -0
- package/dist/memory/context.js.map +1 -0
- package/dist/memory/index.d.ts +2 -0
- package/dist/memory/index.d.ts.map +1 -0
- package/dist/memory/index.js +2 -0
- package/dist/memory/index.js.map +1 -0
- package/dist/pipeline.d.ts +93 -0
- package/dist/pipeline.d.ts.map +1 -0
- package/dist/pipeline.js +255 -0
- package/dist/pipeline.js.map +1 -0
- package/dist/stt/deepgram.d.ts +27 -0
- package/dist/stt/deepgram.d.ts.map +1 -0
- package/dist/stt/deepgram.js +137 -0
- package/dist/stt/deepgram.js.map +1 -0
- package/dist/stt/index.d.ts +2 -0
- package/dist/stt/index.d.ts.map +1 -0
- package/dist/stt/index.js +2 -0
- package/dist/stt/index.js.map +1 -0
- package/dist/tools/index.d.ts +2 -0
- package/dist/tools/index.d.ts.map +1 -0
- package/dist/tools/index.js +2 -0
- package/dist/tools/index.js.map +1 -0
- package/dist/tools/registry.d.ts +38 -0
- package/dist/tools/registry.d.ts.map +1 -0
- package/dist/tools/registry.js +77 -0
- package/dist/tools/registry.js.map +1 -0
- package/dist/transport/index.d.ts +2 -0
- package/dist/transport/index.d.ts.map +1 -0
- package/dist/transport/index.js +2 -0
- package/dist/transport/index.js.map +1 -0
- package/dist/transport/websocket.d.ts +38 -0
- package/dist/transport/websocket.d.ts.map +1 -0
- package/dist/transport/websocket.js +180 -0
- package/dist/transport/websocket.js.map +1 -0
- package/dist/tts/eleven-labs.d.ts +27 -0
- package/dist/tts/eleven-labs.d.ts.map +1 -0
- package/dist/tts/eleven-labs.js +73 -0
- package/dist/tts/eleven-labs.js.map +1 -0
- package/dist/tts/index.d.ts +2 -0
- package/dist/tts/index.d.ts.map +1 -0
- package/dist/tts/index.js +2 -0
- package/dist/tts/index.js.map +1 -0
- package/dist/types.d.ts +370 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +13 -0
- package/dist/types.js.map +1 -0
- package/dist/vad/energy.d.ts +49 -0
- package/dist/vad/energy.d.ts.map +1 -0
- package/dist/vad/energy.js +108 -0
- package/dist/vad/energy.js.map +1 -0
- package/dist/vad/index.d.ts +2 -0
- package/dist/vad/index.d.ts.map +1 -0
- package/dist/vad/index.js +2 -0
- package/dist/vad/index.js.map +1 -0
- package/package.json +51 -0
- package/src/agent.ts +373 -0
- package/src/analytics/index.ts +7 -0
- package/src/analytics/logger.ts +152 -0
- package/src/index.ts +121 -0
- package/src/jev/action-space.ts +119 -0
- package/src/jev/embeddings.ts +81 -0
- package/src/jev/engine.ts +225 -0
- package/src/jev/index.ts +3 -0
- package/src/llm/index.ts +1 -0
- package/src/llm/openai.ts +178 -0
- package/src/memory/context.ts +124 -0
- package/src/memory/index.ts +1 -0
- package/src/pipeline.ts +331 -0
- package/src/stt/deepgram.ts +182 -0
- package/src/stt/index.ts +1 -0
- package/src/tools/index.ts +1 -0
- package/src/tools/registry.ts +99 -0
- package/src/transport/index.ts +1 -0
- package/src/transport/websocket.ts +227 -0
- package/src/tts/eleven-labs.ts +100 -0
- package/src/tts/index.ts +1 -0
- package/src/types.ts +429 -0
- package/src/vad/energy.ts +135 -0
- package/src/vad/index.ts +1 -0
- package/tests/jev.test.ts +198 -0
- package/tests/memory.test.ts +169 -0
- package/tests/vad.test.ts +94 -0
- package/tsconfig.json +9 -0
- package/tsconfig.tsbuildinfo +1 -0
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
import type { LLMProvider, LLMInterface, LLMOptions, LLMContext } from "../types.js";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* OpenAILLM — LLM provider using OpenAI's chat completions API.
|
|
5
|
+
*
|
|
6
|
+
* The LLM in Felona Voice is used for **content generation**, not flow control.
|
|
7
|
+
* JEV decides what to do; the LLM decides how to say it.
|
|
8
|
+
*/
|
|
9
|
+
export class OpenAILLM implements LLMProvider {
|
|
10
|
+
readonly name = "openai";
|
|
11
|
+
private readonly apiKey: string;
|
|
12
|
+
private readonly baseUrl: string;
|
|
13
|
+
|
|
14
|
+
constructor(options: { apiKey: string; baseUrl?: string }) {
|
|
15
|
+
this.apiKey = options.apiKey;
|
|
16
|
+
this.baseUrl = options.baseUrl ?? "https://api.openai.com/v1";
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
create(options: LLMOptions): LLMInterface {
|
|
20
|
+
return new OpenAILLMInterface({
|
|
21
|
+
apiKey: this.apiKey,
|
|
22
|
+
baseUrl: this.baseUrl,
|
|
23
|
+
model: options.model,
|
|
24
|
+
temperature: options.temperature ?? 0.7,
|
|
25
|
+
maxTokens: options.maxTokens ?? 256,
|
|
26
|
+
});
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
class OpenAILLMInterface implements LLMInterface {
|
|
31
|
+
private readonly apiKey: string;
|
|
32
|
+
private readonly baseUrl: string;
|
|
33
|
+
private readonly model: string;
|
|
34
|
+
private readonly temperature: number;
|
|
35
|
+
private readonly maxTokens: number;
|
|
36
|
+
|
|
37
|
+
constructor(options: {
|
|
38
|
+
apiKey: string;
|
|
39
|
+
baseUrl: string;
|
|
40
|
+
model: string;
|
|
41
|
+
temperature: number;
|
|
42
|
+
maxTokens: number;
|
|
43
|
+
}) {
|
|
44
|
+
this.apiKey = options.apiKey;
|
|
45
|
+
this.baseUrl = options.baseUrl;
|
|
46
|
+
this.model = options.model;
|
|
47
|
+
this.temperature = options.temperature;
|
|
48
|
+
this.maxTokens = options.maxTokens;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
async generate(prompt: string, context?: LLMContext): Promise<string> {
|
|
52
|
+
const messages = this.buildMessages(prompt, context);
|
|
53
|
+
|
|
54
|
+
const response = await fetch(`${this.baseUrl}/chat/completions`, {
|
|
55
|
+
method: "POST",
|
|
56
|
+
headers: {
|
|
57
|
+
"Content-Type": "application/json",
|
|
58
|
+
Authorization: `Bearer ${this.apiKey}`,
|
|
59
|
+
},
|
|
60
|
+
body: JSON.stringify({
|
|
61
|
+
model: this.model,
|
|
62
|
+
messages,
|
|
63
|
+
temperature: this.temperature,
|
|
64
|
+
max_tokens: this.maxTokens,
|
|
65
|
+
}),
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
if (!response.ok) {
|
|
69
|
+
const error = await response.text();
|
|
70
|
+
throw new Error(`OpenAI LLM failed: ${response.status} ${error}`);
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const data = (await response.json()) as {
|
|
74
|
+
choices: Array<{ message: { content: string } }>;
|
|
75
|
+
};
|
|
76
|
+
return data.choices[0]?.message?.content ?? "";
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
async *generateStream(
|
|
80
|
+
prompt: string,
|
|
81
|
+
context?: LLMContext,
|
|
82
|
+
): AsyncIterable<string> {
|
|
83
|
+
const messages = this.buildMessages(prompt, context);
|
|
84
|
+
|
|
85
|
+
const response = await fetch(`${this.baseUrl}/chat/completions`, {
|
|
86
|
+
method: "POST",
|
|
87
|
+
headers: {
|
|
88
|
+
"Content-Type": "application/json",
|
|
89
|
+
Authorization: `Bearer ${this.apiKey}`,
|
|
90
|
+
},
|
|
91
|
+
body: JSON.stringify({
|
|
92
|
+
model: this.model,
|
|
93
|
+
messages,
|
|
94
|
+
temperature: this.temperature,
|
|
95
|
+
max_tokens: this.maxTokens,
|
|
96
|
+
stream: true,
|
|
97
|
+
}),
|
|
98
|
+
});
|
|
99
|
+
|
|
100
|
+
if (!response.ok) {
|
|
101
|
+
const error = await response.text();
|
|
102
|
+
throw new Error(`OpenAI LLM stream failed: ${response.status} ${error}`);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
if (!response.body) {
|
|
106
|
+
throw new Error("OpenAI LLM returned no body");
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
const reader = response.body.getReader();
|
|
110
|
+
const decoder = new TextDecoder();
|
|
111
|
+
let buffer = "";
|
|
112
|
+
|
|
113
|
+
try {
|
|
114
|
+
while (true) {
|
|
115
|
+
const { done, value } = await reader.read();
|
|
116
|
+
if (done) break;
|
|
117
|
+
|
|
118
|
+
buffer += decoder.decode(value, { stream: true });
|
|
119
|
+
const lines = buffer.split("\n");
|
|
120
|
+
buffer = lines.pop() ?? "";
|
|
121
|
+
|
|
122
|
+
for (const line of lines) {
|
|
123
|
+
const trimmed = line.trim();
|
|
124
|
+
if (!trimmed || !trimmed.startsWith("data: ")) continue;
|
|
125
|
+
|
|
126
|
+
const jsonStr = trimmed.slice(6);
|
|
127
|
+
if (jsonStr === "[DONE]") return;
|
|
128
|
+
|
|
129
|
+
try {
|
|
130
|
+
const parsed = JSON.parse(jsonStr);
|
|
131
|
+
const content = parsed.choices?.[0]?.delta?.content;
|
|
132
|
+
if (content) {
|
|
133
|
+
yield content;
|
|
134
|
+
}
|
|
135
|
+
} catch {
|
|
136
|
+
// Ignore malformed JSON chunks
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
} finally {
|
|
141
|
+
reader.releaseLock();
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
private buildMessages(
|
|
146
|
+
prompt: string,
|
|
147
|
+
context?: LLMContext,
|
|
148
|
+
): Array<{ role: string; content: string }> {
|
|
149
|
+
const messages: Array<{ role: string; content: string }> = [];
|
|
150
|
+
|
|
151
|
+
// System prompt
|
|
152
|
+
if (context?.systemPrompt) {
|
|
153
|
+
messages.push({ role: "system", content: context.systemPrompt });
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
// Additional instructions
|
|
157
|
+
if (context?.instructions) {
|
|
158
|
+
messages.push({ role: "system", content: context.instructions });
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
// Conversation history
|
|
162
|
+
if (context?.history) {
|
|
163
|
+
messages.push(...context.history);
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
// The actual prompt (what JEV decided the agent should do)
|
|
167
|
+
messages.push({ role: "user", content: prompt });
|
|
168
|
+
|
|
169
|
+
return messages;
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
export function createOpenAILLM(options: {
|
|
174
|
+
apiKey: string;
|
|
175
|
+
baseUrl?: string;
|
|
176
|
+
}): OpenAILLM {
|
|
177
|
+
return new OpenAILLM(options);
|
|
178
|
+
}
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
import { EventEmitter } from "node:events";
|
|
2
|
+
import { randomUUID } from "node:crypto";
|
|
3
|
+
import type {
|
|
4
|
+
ConversationTurn,
|
|
5
|
+
MemoryManager,
|
|
6
|
+
ConversationContext,
|
|
7
|
+
Session,
|
|
8
|
+
} from "../types.js";
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* ConversationMemory — Manages the sliding window of conversation turns,
|
|
12
|
+
* extracted slots, and builds the full context for JEV.
|
|
13
|
+
*
|
|
14
|
+
* This is the agent's short-term memory for a single session.
|
|
15
|
+
*/
|
|
16
|
+
export class ConversationMemory extends EventEmitter implements MemoryManager {
|
|
17
|
+
private turns: ConversationTurn[] = [];
|
|
18
|
+
private slots: Map<string, unknown> = new Map();
|
|
19
|
+
private readonly maxTurns: number;
|
|
20
|
+
|
|
21
|
+
constructor(options?: { maxTurns?: number }) {
|
|
22
|
+
super();
|
|
23
|
+
this.maxTurns = options?.maxTurns ?? 50;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
addTurn(turn: ConversationTurn): void {
|
|
27
|
+
this.turns.push(turn);
|
|
28
|
+
|
|
29
|
+
// Sliding window — drop oldest turns if we exceed the max
|
|
30
|
+
if (this.turns.length > this.maxTurns) {
|
|
31
|
+
const dropped = this.turns.shift();
|
|
32
|
+
this.emit("turnDropped", dropped);
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
this.emit("turnAdded", turn);
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
getTurns(): ConversationTurn[] {
|
|
39
|
+
return [...this.turns];
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
getRecentTurns(n: number): ConversationTurn[] {
|
|
43
|
+
return this.turns.slice(-n);
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
setSlot(key: string, value: unknown): void {
|
|
47
|
+
const previous = this.slots.get(key);
|
|
48
|
+
this.slots.set(key, value);
|
|
49
|
+
this.emit("slotUpdated", { key, value, previous });
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
getSlot(key: string): unknown {
|
|
53
|
+
return this.slots.get(key);
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
getSlots(): Record<string, unknown> {
|
|
57
|
+
return Object.fromEntries(this.slots);
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
clear(): void {
|
|
61
|
+
this.turns = [];
|
|
62
|
+
this.slots.clear();
|
|
63
|
+
this.emit("cleared");
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Build the full conversation context for JEV.
|
|
68
|
+
* This is the primary input to the JEV engine's encode() method.
|
|
69
|
+
*/
|
|
70
|
+
buildContext(
|
|
71
|
+
session: Session,
|
|
72
|
+
systemPrompt: string,
|
|
73
|
+
currentUtterance?: string,
|
|
74
|
+
): ConversationContext {
|
|
75
|
+
return {
|
|
76
|
+
session,
|
|
77
|
+
turns: this.getTurns(),
|
|
78
|
+
currentUtterance: currentUtterance ?? "",
|
|
79
|
+
slots: this.getSlots(),
|
|
80
|
+
systemPrompt,
|
|
81
|
+
};
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Serialize the conversation to a format suitable for LLM context.
|
|
86
|
+
* Returns an array of {role, content} messages.
|
|
87
|
+
*/
|
|
88
|
+
toLLMHistory(): Array<{ role: "user" | "assistant"; content: string }> {
|
|
89
|
+
return this.turns.map((turn) => ({
|
|
90
|
+
role: turn.role === "user" ? ("user" as const) : ("assistant" as const),
|
|
91
|
+
content: turn.content,
|
|
92
|
+
}));
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Get a text summary of the conversation so far.
|
|
97
|
+
* Useful for embedding into the JEV context vector.
|
|
98
|
+
*/
|
|
99
|
+
toContextString(): string {
|
|
100
|
+
return this.turns
|
|
101
|
+
.map((t) => `${t.role === "user" ? "User" : "Agent"}: ${t.content}`)
|
|
102
|
+
.join("\n");
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** Number of turns in memory */
|
|
106
|
+
get length(): number {
|
|
107
|
+
return this.turns.length;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/** Get the last turn (or undefined if empty) */
|
|
111
|
+
get lastTurn(): ConversationTurn | undefined {
|
|
112
|
+
return this.turns[this.turns.length - 1];
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* Create a new ConversationMemory instance.
|
|
118
|
+
* Factory function for convenience.
|
|
119
|
+
*/
|
|
120
|
+
export function createMemory(
|
|
121
|
+
options?: { maxTurns?: number },
|
|
122
|
+
): ConversationMemory {
|
|
123
|
+
return new ConversationMemory(options);
|
|
124
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export { ConversationMemory, createMemory } from "./context.js";
|
package/src/pipeline.ts
ADDED
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
import { EventEmitter } from "node:events";
|
|
2
|
+
import type {
|
|
3
|
+
AudioChunk,
|
|
4
|
+
Session,
|
|
5
|
+
STTProvider,
|
|
6
|
+
STTStream,
|
|
7
|
+
TTSProvider,
|
|
8
|
+
VADProvider,
|
|
9
|
+
LLMInterface,
|
|
10
|
+
ConversationContext,
|
|
11
|
+
ActionMatch,
|
|
12
|
+
AgentHooks,
|
|
13
|
+
} from "./types.js";
|
|
14
|
+
import type { JEVEngine } from "./jev/engine.js";
|
|
15
|
+
import type { ConversationMemory } from "./memory/context.js";
|
|
16
|
+
import type { ToolRegistry } from "./tools/registry.js";
|
|
17
|
+
import type { CallLogger, JEVDecisionLog } from "./analytics/logger.js";
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* VoicePipeline — Orchestrates the full voice conversation loop.
|
|
21
|
+
*
|
|
22
|
+
* Audio In → VAD → STT → JEV Decision → LLM Generate → TTS → Audio Out
|
|
23
|
+
*
|
|
24
|
+
* One pipeline instance is created per active session (call).
|
|
25
|
+
* It manages:
|
|
26
|
+
* - Audio buffering and VAD-based turn detection
|
|
27
|
+
* - Streaming STT transcription
|
|
28
|
+
* - JEV-powered action selection
|
|
29
|
+
* - LLM response generation
|
|
30
|
+
* - Streaming TTS playback
|
|
31
|
+
* - Barge-in (user interruption) handling
|
|
32
|
+
*/
|
|
33
|
+
export class VoicePipeline extends EventEmitter {
|
|
34
|
+
private readonly sessionId: string;
|
|
35
|
+
private readonly session: Session;
|
|
36
|
+
private readonly stt: STTProvider;
|
|
37
|
+
private readonly tts: TTSProvider;
|
|
38
|
+
private readonly vad: VADProvider;
|
|
39
|
+
private readonly llm: LLMInterface;
|
|
40
|
+
private readonly jev: JEVEngine;
|
|
41
|
+
private readonly memory: ConversationMemory;
|
|
42
|
+
private readonly tools: ToolRegistry;
|
|
43
|
+
private readonly logger: CallLogger;
|
|
44
|
+
private readonly hooks: AgentHooks;
|
|
45
|
+
private readonly systemPrompt: string;
|
|
46
|
+
|
|
47
|
+
private sttStream: STTStream | null = null;
|
|
48
|
+
private isSpeaking = false; // Is the agent currently speaking?
|
|
49
|
+
private isProcessing = false; // Is the pipeline processing a user turn?
|
|
50
|
+
private currentTranscript = "";
|
|
51
|
+
private decisions: JEVDecisionLog[] = [];
|
|
52
|
+
|
|
53
|
+
// Callback to send audio back to the client
|
|
54
|
+
private sendAudioFn: ((chunk: AudioChunk) => Promise<void>) | null = null;
|
|
55
|
+
|
|
56
|
+
constructor(options: {
|
|
57
|
+
sessionId: string;
|
|
58
|
+
session: Session;
|
|
59
|
+
stt: STTProvider;
|
|
60
|
+
tts: TTSProvider;
|
|
61
|
+
vad: VADProvider;
|
|
62
|
+
llm: LLMInterface;
|
|
63
|
+
jev: JEVEngine;
|
|
64
|
+
memory: ConversationMemory;
|
|
65
|
+
tools: ToolRegistry;
|
|
66
|
+
logger: CallLogger;
|
|
67
|
+
hooks: AgentHooks;
|
|
68
|
+
systemPrompt: string;
|
|
69
|
+
sendAudio: (chunk: AudioChunk) => Promise<void>;
|
|
70
|
+
}) {
|
|
71
|
+
super();
|
|
72
|
+
this.sessionId = options.sessionId;
|
|
73
|
+
this.session = options.session;
|
|
74
|
+
this.stt = options.stt;
|
|
75
|
+
this.tts = options.tts;
|
|
76
|
+
this.vad = options.vad;
|
|
77
|
+
this.llm = options.llm;
|
|
78
|
+
this.jev = options.jev;
|
|
79
|
+
this.memory = options.memory;
|
|
80
|
+
this.tools = options.tools;
|
|
81
|
+
this.logger = options.logger;
|
|
82
|
+
this.hooks = options.hooks;
|
|
83
|
+
this.systemPrompt = options.systemPrompt;
|
|
84
|
+
this.sendAudioFn = options.sendAudio;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Start the pipeline — initialize STT stream and begin processing.
|
|
89
|
+
*/
|
|
90
|
+
async start(): Promise<void> {
|
|
91
|
+
this.logger.log("info", `Pipeline started for session ${this.sessionId}`);
|
|
92
|
+
|
|
93
|
+
// Create STT stream
|
|
94
|
+
this.sttStream = this.stt.createStream({
|
|
95
|
+
interimResults: true,
|
|
96
|
+
language: "en-US",
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
// Handle STT results
|
|
100
|
+
this.sttStream.onResult((result) => {
|
|
101
|
+
this.handleSTTResult(result);
|
|
102
|
+
});
|
|
103
|
+
|
|
104
|
+
// Notify hooks
|
|
105
|
+
await this.hooks.onCallStart?.(this.session);
|
|
106
|
+
|
|
107
|
+
this.emit("started");
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Process incoming audio from the transport.
|
|
112
|
+
* This is called for every audio chunk received from the client.
|
|
113
|
+
*/
|
|
114
|
+
processAudio(chunk: AudioChunk): void {
|
|
115
|
+
// Step 1: VAD — detect if the user is speaking
|
|
116
|
+
const vadResult = this.vad.process(chunk);
|
|
117
|
+
|
|
118
|
+
// Step 2: Handle barge-in — user starts speaking while agent is talking
|
|
119
|
+
if (vadResult.event?.type === "speech_start" && this.isSpeaking) {
|
|
120
|
+
this.handleBargeIn();
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
// Step 3: Feed audio to STT (only when speech is detected)
|
|
124
|
+
if (vadResult.isSpeech && this.sttStream) {
|
|
125
|
+
this.sttStream.write(chunk);
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// Step 4: Handle speech end — user stopped speaking, process the turn
|
|
129
|
+
if (vadResult.event?.type === "speech_end" && !this.isProcessing) {
|
|
130
|
+
this.handleUserTurnComplete();
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Stop the pipeline — clean up resources.
|
|
136
|
+
*/
|
|
137
|
+
async stop(): Promise<void> {
|
|
138
|
+
// Log the call
|
|
139
|
+
await this.logger.logCall({
|
|
140
|
+
session: this.session,
|
|
141
|
+
turns: this.memory.getTurns(),
|
|
142
|
+
decisions: this.decisions,
|
|
143
|
+
metrics: {
|
|
144
|
+
totalTurns: this.memory.length,
|
|
145
|
+
avgJEVLatencyMs: this.decisions.length > 0
|
|
146
|
+
? this.decisions.reduce((sum, d) => sum + d.latencyMs, 0) / this.decisions.length
|
|
147
|
+
: 0,
|
|
148
|
+
bargeInCount: 0, // TODO: track this
|
|
149
|
+
avgConfidence: this.decisions.length > 0
|
|
150
|
+
? this.decisions.reduce((sum, d) => sum + d.confidence, 0) / this.decisions.length
|
|
151
|
+
: 0,
|
|
152
|
+
},
|
|
153
|
+
});
|
|
154
|
+
|
|
155
|
+
// Close STT stream
|
|
156
|
+
await this.sttStream?.close();
|
|
157
|
+
|
|
158
|
+
// Notify hooks
|
|
159
|
+
await this.hooks.onCallEnd?.(this.session, this.memory.getTurns());
|
|
160
|
+
|
|
161
|
+
this.logger.log("info", `Pipeline stopped for session ${this.sessionId}`);
|
|
162
|
+
this.emit("stopped");
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* Handle a STT transcription result.
|
|
167
|
+
*/
|
|
168
|
+
private handleSTTResult(result: {
|
|
169
|
+
text: string;
|
|
170
|
+
isFinal: boolean;
|
|
171
|
+
confidence: number;
|
|
172
|
+
}): void {
|
|
173
|
+
if (result.isFinal) {
|
|
174
|
+
// Accumulate final transcripts
|
|
175
|
+
this.currentTranscript += (this.currentTranscript ? " " : "") + result.text;
|
|
176
|
+
this.emit("transcript", {
|
|
177
|
+
text: this.currentTranscript,
|
|
178
|
+
isFinal: true,
|
|
179
|
+
});
|
|
180
|
+
} else {
|
|
181
|
+
// Emit interim for real-time feedback
|
|
182
|
+
this.emit("transcript", {
|
|
183
|
+
text: result.text,
|
|
184
|
+
isFinal: false,
|
|
185
|
+
});
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* Handle user turn completion — the user has stopped speaking.
|
|
191
|
+
* This triggers the JEV → LLM → TTS pipeline.
|
|
192
|
+
*/
|
|
193
|
+
private async handleUserTurnComplete(): Promise<void> {
|
|
194
|
+
const userText = this.currentTranscript.trim();
|
|
195
|
+
if (!userText) return; // Ignore empty turns
|
|
196
|
+
|
|
197
|
+
this.isProcessing = true;
|
|
198
|
+
this.currentTranscript = "";
|
|
199
|
+
|
|
200
|
+
try {
|
|
201
|
+
// Add user turn to memory
|
|
202
|
+
this.memory.addTurn({
|
|
203
|
+
role: "user",
|
|
204
|
+
content: userText,
|
|
205
|
+
timestampMs: Date.now() - this.session.startedAt.getTime(),
|
|
206
|
+
});
|
|
207
|
+
|
|
208
|
+
// Notify hooks
|
|
209
|
+
await this.hooks.onUserSpoke?.(userText, this.session);
|
|
210
|
+
|
|
211
|
+
// Build conversation context for JEV
|
|
212
|
+
const context = this.memory.buildContext(
|
|
213
|
+
this.session,
|
|
214
|
+
this.systemPrompt,
|
|
215
|
+
userText,
|
|
216
|
+
);
|
|
217
|
+
|
|
218
|
+
// JEV Decision — what should the agent do?
|
|
219
|
+
const startTime = performance.now();
|
|
220
|
+
const match = await this.jev.decide(context);
|
|
221
|
+
const jevLatencyMs = performance.now() - startTime;
|
|
222
|
+
|
|
223
|
+
// Log the decision
|
|
224
|
+
const decisionLog: JEVDecisionLog = {
|
|
225
|
+
timestampMs: Date.now() - this.session.startedAt.getTime(),
|
|
226
|
+
contextSummary: userText.slice(0, 200),
|
|
227
|
+
selectedAction: match.action.id,
|
|
228
|
+
confidence: match.confidence,
|
|
229
|
+
candidates: match.candidates.slice(0, 5),
|
|
230
|
+
latencyMs: jevLatencyMs,
|
|
231
|
+
};
|
|
232
|
+
this.decisions.push(decisionLog);
|
|
233
|
+
this.logger.logDecision(decisionLog);
|
|
234
|
+
|
|
235
|
+
// Notify hooks
|
|
236
|
+
await this.hooks.onActionSelected?.(match.action, match.confidence, context);
|
|
237
|
+
|
|
238
|
+
this.logger.log(
|
|
239
|
+
"info",
|
|
240
|
+
`JEV selected: "${match.action.id}" (confidence: ${match.confidence.toFixed(3)}, latency: ${jevLatencyMs.toFixed(1)}ms)`,
|
|
241
|
+
);
|
|
242
|
+
|
|
243
|
+
// Execute the action handler — this calls the LLM to generate response text
|
|
244
|
+
const actionContext = {
|
|
245
|
+
conversation: context,
|
|
246
|
+
llm: this.llm,
|
|
247
|
+
tools: this.tools,
|
|
248
|
+
memory: this.memory,
|
|
249
|
+
session: this.session,
|
|
250
|
+
};
|
|
251
|
+
|
|
252
|
+
const responseText = await match.action.handler(actionContext);
|
|
253
|
+
|
|
254
|
+
// Add agent turn to memory
|
|
255
|
+
this.memory.addTurn({
|
|
256
|
+
role: "agent",
|
|
257
|
+
content: responseText,
|
|
258
|
+
timestampMs: Date.now() - this.session.startedAt.getTime(),
|
|
259
|
+
actionId: match.action.id,
|
|
260
|
+
confidence: match.confidence,
|
|
261
|
+
});
|
|
262
|
+
|
|
263
|
+
// Notify hooks
|
|
264
|
+
await this.hooks.onAgentSpoke?.(responseText, this.session);
|
|
265
|
+
|
|
266
|
+
// TTS — speak the response
|
|
267
|
+
await this.speak(responseText);
|
|
268
|
+
} catch (error) {
|
|
269
|
+
const err = error instanceof Error ? error : new Error(String(error));
|
|
270
|
+
this.logger.log("error", `Pipeline error: ${err.message}`);
|
|
271
|
+
await this.hooks.onError?.(err, this.session);
|
|
272
|
+
this.emit("error", err);
|
|
273
|
+
} finally {
|
|
274
|
+
this.isProcessing = false;
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
/**
|
|
279
|
+
* Speak text through TTS and send audio back to the client.
|
|
280
|
+
*/
|
|
281
|
+
private async speak(text: string): Promise<void> {
|
|
282
|
+
if (!this.sendAudioFn) return;
|
|
283
|
+
|
|
284
|
+
this.isSpeaking = true;
|
|
285
|
+
this.emit("agentSpeaking", true);
|
|
286
|
+
|
|
287
|
+
try {
|
|
288
|
+
for await (const chunk of this.tts.synthesize(text)) {
|
|
289
|
+
// Check if we've been interrupted (barge-in)
|
|
290
|
+
if (!this.isSpeaking) break;
|
|
291
|
+
|
|
292
|
+
await this.sendAudioFn(chunk);
|
|
293
|
+
}
|
|
294
|
+
} catch (error) {
|
|
295
|
+
this.logger.log(
|
|
296
|
+
"error",
|
|
297
|
+
`TTS error: ${error instanceof Error ? error.message : String(error)}`,
|
|
298
|
+
);
|
|
299
|
+
} finally {
|
|
300
|
+
this.isSpeaking = false;
|
|
301
|
+
this.emit("agentSpeaking", false);
|
|
302
|
+
}
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
/**
|
|
306
|
+
* Handle barge-in — user interrupts while agent is speaking.
|
|
307
|
+
* Stop TTS playback immediately and re-enter listening mode.
|
|
308
|
+
*/
|
|
309
|
+
private handleBargeIn(): void {
|
|
310
|
+
if (!this.isSpeaking) return;
|
|
311
|
+
|
|
312
|
+
this.logger.log("info", "Barge-in detected — stopping agent speech");
|
|
313
|
+
this.isSpeaking = false;
|
|
314
|
+
this.hooks.onBargeIn?.(this.session);
|
|
315
|
+
this.emit("bargeIn");
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
/** Get current pipeline state */
|
|
319
|
+
get state(): { isSpeaking: boolean; isProcessing: boolean } {
|
|
320
|
+
return {
|
|
321
|
+
isSpeaking: this.isSpeaking,
|
|
322
|
+
isProcessing: this.isProcessing,
|
|
323
|
+
};
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
export function createPipeline(
|
|
328
|
+
options: ConstructorParameters<typeof VoicePipeline>[0],
|
|
329
|
+
): VoicePipeline {
|
|
330
|
+
return new VoicePipeline(options);
|
|
331
|
+
}
|