osborn 0.9.180 → 0.9.182
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config.d.ts +6 -46
- package/dist/config.js +20 -56
- package/dist/index.js +94 -849
- package/dist/pipeline-fastbrain.d.ts +10 -17
- package/dist/pipeline-fastbrain.js +269 -246
- package/dist/turn-detector-shim.js +6 -2
- package/dist/voice-io.d.ts +3 -34
- package/dist/voice-io.js +30 -76
- package/package.json +3 -5
- package/tests/autocompact-pct.test.ts +0 -77
- package/tests/voice-io-stt-config.test.ts +0 -177
|
@@ -21,9 +21,13 @@ export class CloudTurnDetector {
|
|
|
21
21
|
model = 'lk_end_of_utterance_multilingual';
|
|
22
22
|
provider = 'livekit';
|
|
23
23
|
constructor() {
|
|
24
|
-
|
|
24
|
+
const raw = process.env.LIVEKIT_REMOTE_EOT_URL;
|
|
25
|
+
// EOT endpoint is HTTP — convert wss:// → https:// if the env var uses WebSocket scheme
|
|
26
|
+
this.#remoteUrl = raw
|
|
27
|
+
? raw.replace(/^wss:\/\//, 'https://').replace(/^ws:\/\//, 'http://')
|
|
28
|
+
: undefined;
|
|
25
29
|
if (this.#remoteUrl) {
|
|
26
|
-
console.log(`🧠 Turn detector: LiveKit Cloud remote inference`);
|
|
30
|
+
console.log(`🧠 Turn detector: LiveKit Cloud remote inference (${this.#remoteUrl})`);
|
|
27
31
|
}
|
|
28
32
|
else {
|
|
29
33
|
console.log('🧠 Turn detector: No LIVEKIT_REMOTE_EOT_URL — STT endpointing fallback');
|
package/dist/voice-io.d.ts
CHANGED
|
@@ -1,16 +1,10 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Voice I/O Module
|
|
3
|
-
* Handles STT (Speech-to-Text)
|
|
4
|
-
*
|
|
5
|
-
* Supports two modes:
|
|
6
|
-
* - Direct mode: STT (Deepgram) → Claude Agent SDK → TTS (Deepgram)
|
|
7
|
-
* - Realtime mode: OpenAI/Gemini native speech-to-speech models
|
|
3
|
+
* Handles STT (Speech-to-Text) and TTS (Text-to-Speech) for pipeline mode.
|
|
8
4
|
*/
|
|
9
5
|
import * as deepgram from '@livekit/agents-plugin-deepgram';
|
|
10
|
-
import * as google from '@livekit/agents-plugin-google';
|
|
11
6
|
import * as openai from '@livekit/agents-plugin-openai';
|
|
12
7
|
import * as silero from '@livekit/agents-plugin-silero';
|
|
13
|
-
import type { RealtimeConfig } from './config.js';
|
|
14
8
|
export interface STTConfig {
|
|
15
9
|
provider: 'deepgram' | 'deepgram-flux' | 'groq-whisper' | 'openai-whisper';
|
|
16
10
|
model?: string;
|
|
@@ -21,7 +15,7 @@ export interface STTConfig {
|
|
|
21
15
|
eotTimeoutMs?: number;
|
|
22
16
|
}
|
|
23
17
|
export interface TTSConfig {
|
|
24
|
-
provider: '
|
|
18
|
+
provider: 'openai' | 'deepgram' | 'groq-orpheus' | 'fishaudio' | 'rime';
|
|
25
19
|
voice?: string;
|
|
26
20
|
model?: string;
|
|
27
21
|
}
|
|
@@ -36,7 +30,6 @@ export interface VoiceIOConfig {
|
|
|
36
30
|
export declare function createSTT(config: STTConfig): deepgram.STT | deepgram.STTv2 | openai.STT;
|
|
37
31
|
/**
|
|
38
32
|
* Create TTS (Text-to-Speech) instance based on config
|
|
39
|
-
* Using Gemini TTS as default (cheaper, good quality)
|
|
40
33
|
*/
|
|
41
34
|
export declare function createTTS(config: TTSConfig): any;
|
|
42
35
|
/**
|
|
@@ -49,32 +42,8 @@ export declare function createTTS(config: TTSConfig): any;
|
|
|
49
42
|
*/
|
|
50
43
|
export declare function createVAD(): Promise<silero.VAD>;
|
|
51
44
|
/**
|
|
52
|
-
*
|
|
53
|
-
* Uses Deepgram STT (fast, accurate) + Deepgram TTS (fast, good)
|
|
54
|
-
*/
|
|
55
|
-
export declare const DEFAULT_VOICE_IO_CONFIG: VoiceIOConfig;
|
|
56
|
-
/**
|
|
57
|
-
* Direct mode voice config — centralized here for easy provider swapping.
|
|
45
|
+
* Pipeline mode voice config — centralized here for easy provider swapping.
|
|
58
46
|
* To switch providers: comment out the active line, uncomment the alternative.
|
|
59
47
|
*/
|
|
60
48
|
export declare const DIRECT_MODE_STT: STTConfig;
|
|
61
49
|
export declare const DIRECT_MODE_TTS: TTSConfig;
|
|
62
|
-
export interface RealtimeModelConfig {
|
|
63
|
-
provider: 'openai' | 'gemini';
|
|
64
|
-
openaiVoice?: 'alloy' | 'echo' | 'fable' | 'onyx' | 'nova' | 'shimmer';
|
|
65
|
-
openaiModel?: string;
|
|
66
|
-
geminiVoice?: 'Charon' | 'Puck' | 'Kore' | 'Fenrir' | 'Aoede';
|
|
67
|
-
geminiModel?: string;
|
|
68
|
-
instructions?: string;
|
|
69
|
-
}
|
|
70
|
-
/**
|
|
71
|
-
* Create Realtime Model for native speech-to-speech
|
|
72
|
-
* Supports OpenAI Realtime API and Gemini Live API
|
|
73
|
-
*
|
|
74
|
-
* Note: Instructions are passed to voice.Agent, not to the RealtimeModel
|
|
75
|
-
*/
|
|
76
|
-
export declare function createRealtimeModel(config: RealtimeModelConfig): google.beta.realtime.RealtimeModel | openai.realtime.RealtimeModel;
|
|
77
|
-
/**
|
|
78
|
-
* Create realtime model from config
|
|
79
|
-
*/
|
|
80
|
-
export declare function createRealtimeModelFromConfig(realtimeConfig: RealtimeConfig, instructions?: string): google.beta.realtime.RealtimeModel | openai.realtime.RealtimeModel;
|
package/dist/voice-io.js
CHANGED
|
@@ -1,14 +1,11 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Voice I/O Module
|
|
3
|
-
* Handles STT (Speech-to-Text)
|
|
4
|
-
*
|
|
5
|
-
* Supports two modes:
|
|
6
|
-
* - Direct mode: STT (Deepgram) → Claude Agent SDK → TTS (Deepgram)
|
|
7
|
-
* - Realtime mode: OpenAI/Gemini native speech-to-speech models
|
|
3
|
+
* Handles STT (Speech-to-Text) and TTS (Text-to-Speech) for pipeline mode.
|
|
8
4
|
*/
|
|
9
5
|
import * as deepgram from '@livekit/agents-plugin-deepgram';
|
|
10
|
-
import * as
|
|
6
|
+
import * as fishaudio from '@livekit/agents-plugin-fishaudio';
|
|
11
7
|
import * as openai from '@livekit/agents-plugin-openai';
|
|
8
|
+
import * as rime from '@livekit/agents-plugin-rime';
|
|
12
9
|
import * as silero from '@livekit/agents-plugin-silero';
|
|
13
10
|
/**
|
|
14
11
|
* Create STT (Speech-to-Text) instance based on config
|
|
@@ -48,18 +45,10 @@ export function createSTT(config) {
|
|
|
48
45
|
}
|
|
49
46
|
/**
|
|
50
47
|
* Create TTS (Text-to-Speech) instance based on config
|
|
51
|
-
* Using Gemini TTS as default (cheaper, good quality)
|
|
52
48
|
*/
|
|
53
49
|
export function createTTS(config) {
|
|
54
50
|
let tts;
|
|
55
51
|
switch (config.provider) {
|
|
56
|
-
case 'gemini':
|
|
57
|
-
// Gemini TTS via google plugin
|
|
58
|
-
tts = new google.beta.TTS({
|
|
59
|
-
model: config.model || 'gemini-2.5-flash-preview-tts',
|
|
60
|
-
voice: config.voice || 'apollo',
|
|
61
|
-
});
|
|
62
|
-
break;
|
|
63
52
|
case 'openai':
|
|
64
53
|
tts = new openai.TTS({
|
|
65
54
|
voice: config.voice || 'alloy',
|
|
@@ -81,6 +70,27 @@ export function createTTS(config) {
|
|
|
81
70
|
baseURL: 'https://api.groq.com/openai/v1',
|
|
82
71
|
});
|
|
83
72
|
break;
|
|
73
|
+
case 'fishaudio':
|
|
74
|
+
// Fish Audio s2-pro ($15/M chars) — blind test winner, half price of OpenAI tts-1-hd
|
|
75
|
+
// voiceId: pick from Fish Audio voice library (leave undefined for system default)
|
|
76
|
+
// Requires FISH_AUDIO_API_KEY env var
|
|
77
|
+
tts = new fishaudio.TTS({
|
|
78
|
+
model: (config.model || 's2-pro'),
|
|
79
|
+
voiceId: config.voice,
|
|
80
|
+
latencyMode: 'low',
|
|
81
|
+
});
|
|
82
|
+
break;
|
|
83
|
+
case 'rime':
|
|
84
|
+
// Rime Mist v3 ($30/M chars, 37ms TTFB) — fastest commercial TTS, conversation-trained
|
|
85
|
+
// useWebsocket: true → WebSocket streaming = clean abort on interruption (like Deepgram)
|
|
86
|
+
// speaker: pick from Rime voice library, e.g. 'aurora', 'ember', 'cove'
|
|
87
|
+
// Requires RIME_API_KEY env var
|
|
88
|
+
tts = new rime.TTS({
|
|
89
|
+
modelId: (config.model || 'mistv3'),
|
|
90
|
+
speaker: config.voice || 'cove',
|
|
91
|
+
useWebsocket: true,
|
|
92
|
+
});
|
|
93
|
+
break;
|
|
84
94
|
default:
|
|
85
95
|
throw new Error(`Unknown TTS provider: ${config.provider}`);
|
|
86
96
|
}
|
|
@@ -110,76 +120,20 @@ export async function createVAD() {
|
|
|
110
120
|
});
|
|
111
121
|
}
|
|
112
122
|
/**
|
|
113
|
-
*
|
|
114
|
-
* Uses Deepgram STT (fast, accurate) + Deepgram TTS (fast, good)
|
|
115
|
-
*/
|
|
116
|
-
export const DEFAULT_VOICE_IO_CONFIG = {
|
|
117
|
-
stt: {
|
|
118
|
-
provider: 'deepgram',
|
|
119
|
-
model: 'nova-3',
|
|
120
|
-
language: 'en',
|
|
121
|
-
},
|
|
122
|
-
tts: {
|
|
123
|
-
provider: 'deepgram',
|
|
124
|
-
voice: 'aura-2-asteria-en',
|
|
125
|
-
},
|
|
126
|
-
};
|
|
127
|
-
/**
|
|
128
|
-
* Direct mode voice config — centralized here for easy provider swapping.
|
|
123
|
+
* Pipeline mode voice config — centralized here for easy provider swapping.
|
|
129
124
|
* To switch providers: comment out the active line, uncomment the alternative.
|
|
130
125
|
*/
|
|
131
126
|
export const DIRECT_MODE_STT = {
|
|
132
127
|
// provider: 'groq-whisper', model: 'whisper-large-v3-turbo', // Batch — needs VAD
|
|
133
128
|
// provider: 'openai-whisper', model: 'whisper-1', // Batch — needs VAD
|
|
134
|
-
// provider: 'deepgram', model: '
|
|
135
|
-
provider: 'deepgram
|
|
129
|
+
// provider: 'deepgram-flux', model: 'flux-general-en', language: 'en', // Streaming, ML-based turn detection — Flux V2 has silent keepalive timeout bug (~30s silence kills connection)
|
|
130
|
+
provider: 'deepgram', model: 'nova-3', language: 'en', // Streaming, silence-based endpointing (550ms)
|
|
136
131
|
};
|
|
137
132
|
export const DIRECT_MODE_TTS = {
|
|
138
|
-
// provider: 'deepgram', model: 'aura-2-asteria-en', // WebSocket-based: handles TTS abort cleanly
|
|
139
|
-
// provider: 'gemini', model: 'gemini-2.5-flash-preview-tts', voice: 'apollo',
|
|
133
|
+
// provider: 'deepgram', model: 'aura-2-asteria-en', // WebSocket-based: handles TTS abort cleanly — but quality rejected (run-on sentences)
|
|
140
134
|
// provider: 'openai', model: 'tts-1', voice: 'fable', // HTTP streaming: throws APIUserAbortError on interrupt → unrecoverable session crash
|
|
141
135
|
provider: 'openai', model: 'tts-1-hd', voice: 'fable', // 0.9.70: test tts-1-hd — tts-1 had chronic per-sentence HTTP hangs (40s SDK watchdog → APIUserAbortError mid-message)
|
|
142
136
|
// provider: 'groq-orpheus', model: 'canopylabs/orpheus-v1-english', voice: 'autumn', // $22/M chars — voices: autumn, diana, hannah, austin, daniel, troy
|
|
137
|
+
// provider: 'fishaudio', model: 's2-pro', voice: '<voice-id>', // $15/M chars — blind test winner, HTTP streaming (test abort behavior)
|
|
138
|
+
// provider: 'rime', model: 'mistv3', voice: 'cove', // $30/M chars, 37ms TTFB — WebSocket (safe on interruption); voices: aurora, ember, cove
|
|
143
139
|
};
|
|
144
|
-
/**
|
|
145
|
-
* Create Realtime Model for native speech-to-speech
|
|
146
|
-
* Supports OpenAI Realtime API and Gemini Live API
|
|
147
|
-
*
|
|
148
|
-
* Note: Instructions are passed to voice.Agent, not to the RealtimeModel
|
|
149
|
-
*/
|
|
150
|
-
export function createRealtimeModel(config) {
|
|
151
|
-
if (config.provider === 'gemini') {
|
|
152
|
-
console.log('📱 Using Gemini Live API (realtime)');
|
|
153
|
-
// Using 'latest' alias — 12-2025 had a known 1008 crash bug during interruptions + tool calls
|
|
154
|
-
return new google.beta.realtime.RealtimeModel({
|
|
155
|
-
model: config.geminiModel || 'gemini-2.5-flash-native-audio-latest',
|
|
156
|
-
voice: config.geminiVoice || 'Charon',
|
|
157
|
-
// Gemini supports instructions at model level
|
|
158
|
-
instructions: config.instructions,
|
|
159
|
-
// Enable transcription so we get text of what the agent says
|
|
160
|
-
inputAudioTranscription: {},
|
|
161
|
-
outputAudioTranscription: {},
|
|
162
|
-
});
|
|
163
|
-
}
|
|
164
|
-
else {
|
|
165
|
-
console.log('📱 Using OpenAI Realtime API');
|
|
166
|
-
// OpenAI RealtimeModel - instructions go to voice.Agent instead
|
|
167
|
-
return new openai.realtime.RealtimeModel({
|
|
168
|
-
model: config.openaiModel || 'gpt-4o-realtime-preview',
|
|
169
|
-
voice: config.openaiVoice || 'alloy',
|
|
170
|
-
});
|
|
171
|
-
}
|
|
172
|
-
}
|
|
173
|
-
/**
|
|
174
|
-
* Create realtime model from config
|
|
175
|
-
*/
|
|
176
|
-
export function createRealtimeModelFromConfig(realtimeConfig, instructions) {
|
|
177
|
-
return createRealtimeModel({
|
|
178
|
-
provider: realtimeConfig.provider || 'openai',
|
|
179
|
-
openaiVoice: realtimeConfig.openaiVoice,
|
|
180
|
-
openaiModel: realtimeConfig.openaiModel,
|
|
181
|
-
geminiVoice: realtimeConfig.geminiVoice,
|
|
182
|
-
geminiModel: realtimeConfig.geminiModel,
|
|
183
|
-
instructions,
|
|
184
|
-
});
|
|
185
|
-
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "osborn",
|
|
3
|
-
"version": "0.9.
|
|
3
|
+
"version": "0.9.182",
|
|
4
4
|
"description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -32,17 +32,15 @@
|
|
|
32
32
|
"dependencies": {
|
|
33
33
|
"@anthropic-ai/claude-agent-sdk": "^0.2.91",
|
|
34
34
|
"@anthropic-ai/sdk": "^0.80.0",
|
|
35
|
-
"@google/genai": "^1.0.0",
|
|
36
35
|
"@livekit/agents": "1.4.6",
|
|
37
36
|
"@livekit/agents-plugin-deepgram": "1.4.6",
|
|
38
|
-
"@livekit/agents-plugin-
|
|
39
|
-
"@livekit/agents-plugin-google": "1.4.6",
|
|
37
|
+
"@livekit/agents-plugin-fishaudio": "1.4.6",
|
|
40
38
|
"@livekit/agents-plugin-livekit": "1.4.6",
|
|
41
39
|
"@livekit/agents-plugin-openai": "1.4.6",
|
|
40
|
+
"@livekit/agents-plugin-rime": "1.4.6",
|
|
42
41
|
"@livekit/agents-plugin-silero": "1.4.6",
|
|
43
42
|
"@livekit/rtc-node": "0.13.29",
|
|
44
43
|
"@modelcontextprotocol/sdk": "^1.29.0",
|
|
45
|
-
"@openai/codex-sdk": "^0.77.0",
|
|
46
44
|
"@smithery/api": "^0.48.0",
|
|
47
45
|
"@types/diff": "^8.0.0",
|
|
48
46
|
"@vscode/ripgrep": "^1.17.1",
|
|
@@ -1,77 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Regression tests for CLAUDE_AUTOCOMPACT_PCT_OVERRIDE in claude-llm.ts
|
|
3
|
-
*
|
|
4
|
-
* Scope: verifies that the compaction threshold is set to a value that
|
|
5
|
-
* leaves meaningful context headroom before compaction fires.
|
|
6
|
-
*
|
|
7
|
-
* Derived from: requirements documented in CLAUDE.md (ENABLE_1M_CONTEXT,
|
|
8
|
-
* compaction rationale comments in claude-llm.ts) and the 0.9.177 → 0.9.178
|
|
9
|
-
* change that lowered the threshold from 92 → 60.
|
|
10
|
-
* NOT derived from the implementation diff.
|
|
11
|
-
*/
|
|
12
|
-
|
|
13
|
-
import { strict as assert } from 'node:assert'
|
|
14
|
-
import { readFileSync } from 'node:fs'
|
|
15
|
-
import { dirname, join } from 'node:path'
|
|
16
|
-
import { fileURLToPath } from 'node:url'
|
|
17
|
-
|
|
18
|
-
const here = dirname(fileURLToPath(import.meta.url))
|
|
19
|
-
const src = readFileSync(join(here, '../src/claude-llm.ts'), 'utf8')
|
|
20
|
-
|
|
21
|
-
let passed = 0
|
|
22
|
-
let failed = 0
|
|
23
|
-
|
|
24
|
-
function test(name: string, fn: () => void) {
|
|
25
|
-
try {
|
|
26
|
-
fn()
|
|
27
|
-
console.log(` ✅ ${name}`)
|
|
28
|
-
passed++
|
|
29
|
-
} catch (err: any) {
|
|
30
|
-
console.error(` ❌ ${name}`)
|
|
31
|
-
console.error(` ${err.message}`)
|
|
32
|
-
failed++
|
|
33
|
-
}
|
|
34
|
-
}
|
|
35
|
-
|
|
36
|
-
// Extract the numeric value of CLAUDE_AUTOCOMPACT_PCT_OVERRIDE from source
|
|
37
|
-
const match = src.match(/process\.env\.CLAUDE_AUTOCOMPACT_PCT_OVERRIDE\s*=\s*'(\d+)'/)
|
|
38
|
-
const pctValue = match ? parseInt(match[1], 10) : NaN
|
|
39
|
-
|
|
40
|
-
console.log('\n=== autocompact PCT override regression tests ===\n')
|
|
41
|
-
console.log(` (detected value: CLAUDE_AUTOCOMPACT_PCT_OVERRIDE = '${pctValue}')`)
|
|
42
|
-
console.log()
|
|
43
|
-
|
|
44
|
-
test('CLAUDE_AUTOCOMPACT_PCT_OVERRIDE is set in claude-llm.ts', () => {
|
|
45
|
-
assert.ok(
|
|
46
|
-
!isNaN(pctValue),
|
|
47
|
-
'CLAUDE_AUTOCOMPACT_PCT_OVERRIDE must be explicitly set in claude-llm.ts'
|
|
48
|
-
)
|
|
49
|
-
})
|
|
50
|
-
|
|
51
|
-
test('CLAUDE_AUTOCOMPACT_PCT_OVERRIDE is a valid percentage (1–99)', () => {
|
|
52
|
-
assert.ok(
|
|
53
|
-
pctValue >= 1 && pctValue <= 99,
|
|
54
|
-
`CLAUDE_AUTOCOMPACT_PCT_OVERRIDE must be between 1 and 99, got ${pctValue}`
|
|
55
|
-
)
|
|
56
|
-
})
|
|
57
|
-
|
|
58
|
-
test('CLAUDE_AUTOCOMPACT_PCT_OVERRIDE is ≤ 92 (prevents compaction at extreme tail)', () => {
|
|
59
|
-
// Values above 92 leave almost no headroom before hitting the token limit.
|
|
60
|
-
// The long-standing default was 92; anything higher than that risks the
|
|
61
|
-
// "compacting every message" bug described in memory/osborn-1m-context-and-worker-restart.md
|
|
62
|
-
assert.ok(
|
|
63
|
-
pctValue <= 92,
|
|
64
|
-
`CLAUDE_AUTOCOMPACT_PCT_OVERRIDE is ${pctValue} which is above 92 — ` +
|
|
65
|
-
'values this high risk compaction firing on every message at the context ceiling'
|
|
66
|
-
)
|
|
67
|
-
})
|
|
68
|
-
|
|
69
|
-
test('ENABLE_1M_CONTEXT is also set in claude-llm.ts', () => {
|
|
70
|
-
assert.ok(
|
|
71
|
-
src.includes("process.env.ENABLE_1M_CONTEXT = '1'"),
|
|
72
|
-
"ENABLE_1M_CONTEXT = '1' must be set alongside CLAUDE_AUTOCOMPACT_PCT_OVERRIDE"
|
|
73
|
-
)
|
|
74
|
-
})
|
|
75
|
-
|
|
76
|
-
console.log(`\n--- ${passed + failed} tests: ${passed} passed, ${failed} failed ---\n`)
|
|
77
|
-
if (failed > 0) process.exit(1)
|
|
@@ -1,177 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Regression tests for voice-io.ts STT configuration
|
|
3
|
-
*
|
|
4
|
-
* Scope: covers the 0.9.175 → 0.9.177 change that switches DIRECT_MODE_STT
|
|
5
|
-
* from deepgram-flux (ML turn detection) back to deepgram (silence-based
|
|
6
|
-
* endpointing), due to Flux V2 keepalive timeout bug (~30s of silence kills
|
|
7
|
-
* the connection).
|
|
8
|
-
*
|
|
9
|
-
* Derived from: requirements in CLAUDE.md (Three Voice Modes), voice-io.ts
|
|
10
|
-
* interface definitions, and the documented bug. NOT derived from the diff.
|
|
11
|
-
*/
|
|
12
|
-
|
|
13
|
-
import { strict as assert } from 'node:assert'
|
|
14
|
-
|
|
15
|
-
// ──────────────────────────────────────────────────────────────────────────────
|
|
16
|
-
// Inline the minimal types — avoid importing from the module which requires
|
|
17
|
-
// live LK plugins (can't instantiate in a unit test without credentials).
|
|
18
|
-
// ──────────────────────────────────────────────────────────────────────────────
|
|
19
|
-
interface STTConfig {
|
|
20
|
-
provider: 'deepgram' | 'deepgram-flux' | 'groq-whisper' | 'openai-whisper'
|
|
21
|
-
model?: string
|
|
22
|
-
language?: string
|
|
23
|
-
eotThreshold?: number
|
|
24
|
-
eotTimeoutMs?: number
|
|
25
|
-
}
|
|
26
|
-
|
|
27
|
-
// ──────────────────────────────────────────────────────────────────────────────
|
|
28
|
-
// Read the compiled output so we get the real runtime values, not just TS types
|
|
29
|
-
// ──────────────────────────────────────────────────────────────────────────────
|
|
30
|
-
// We import from the transpiled dist to avoid plugin side-effects at import time.
|
|
31
|
-
// If dist is missing, fall back to a direct static-analysis of the source file.
|
|
32
|
-
|
|
33
|
-
let DIRECT_MODE_STT_PROVIDER: string | undefined
|
|
34
|
-
let DIRECT_MODE_STT_MODEL: string | undefined
|
|
35
|
-
let DIRECT_MODE_STT_LANG: string | undefined
|
|
36
|
-
|
|
37
|
-
try {
|
|
38
|
-
// dist/voice-io.js is the compiled output of the build step
|
|
39
|
-
const mod = await import('../dist/voice-io.js')
|
|
40
|
-
const cfg: STTConfig = mod.DIRECT_MODE_STT
|
|
41
|
-
DIRECT_MODE_STT_PROVIDER = cfg.provider
|
|
42
|
-
DIRECT_MODE_STT_MODEL = cfg.model
|
|
43
|
-
DIRECT_MODE_STT_LANG = cfg.language
|
|
44
|
-
} catch (e) {
|
|
45
|
-
// Fallback: parse source file statically to extract the active provider line
|
|
46
|
-
const { readFileSync } = await import('node:fs')
|
|
47
|
-
const { fileURLToPath } = await import('node:url')
|
|
48
|
-
const { dirname, join } = await import('node:path')
|
|
49
|
-
const here = dirname(fileURLToPath(import.meta.url))
|
|
50
|
-
const src = readFileSync(join(here, '../src/voice-io.ts'), 'utf8')
|
|
51
|
-
|
|
52
|
-
// Find the non-commented active provider line in DIRECT_MODE_STT
|
|
53
|
-
const providerMatch = src.match(
|
|
54
|
-
/export const DIRECT_MODE_STT[\s\S]*?^\s+provider:\s*'([^']+)'/m
|
|
55
|
-
)
|
|
56
|
-
const modelMatch = src.match(
|
|
57
|
-
/export const DIRECT_MODE_STT[\s\S]*?model:\s*'([^']+)'/m
|
|
58
|
-
)
|
|
59
|
-
const langMatch = src.match(
|
|
60
|
-
/export const DIRECT_MODE_STT[\s\S]*?language:\s*'([^']+)'/m
|
|
61
|
-
)
|
|
62
|
-
DIRECT_MODE_STT_PROVIDER = providerMatch?.[1]
|
|
63
|
-
DIRECT_MODE_STT_MODEL = modelMatch?.[1]
|
|
64
|
-
DIRECT_MODE_STT_LANG = langMatch?.[1]
|
|
65
|
-
}
|
|
66
|
-
|
|
67
|
-
// ──────────────────────────────────────────────────────────────────────────────
|
|
68
|
-
// Tests
|
|
69
|
-
// ──────────────────────────────────────────────────────────────────────────────
|
|
70
|
-
|
|
71
|
-
let passed = 0
|
|
72
|
-
let failed = 0
|
|
73
|
-
|
|
74
|
-
function test(name: string, fn: () => void) {
|
|
75
|
-
try {
|
|
76
|
-
fn()
|
|
77
|
-
console.log(` ✅ ${name}`)
|
|
78
|
-
passed++
|
|
79
|
-
} catch (err: any) {
|
|
80
|
-
console.error(` ❌ ${name}`)
|
|
81
|
-
console.error(` ${err.message}`)
|
|
82
|
-
failed++
|
|
83
|
-
}
|
|
84
|
-
}
|
|
85
|
-
|
|
86
|
-
console.log('\n=== voice-io STT config regression tests ===\n')
|
|
87
|
-
|
|
88
|
-
// ── 1. Provider must be deepgram (not deepgram-flux) ──────────────────────────
|
|
89
|
-
test('DIRECT_MODE_STT.provider is "deepgram" (not deepgram-flux)', () => {
|
|
90
|
-
assert.equal(
|
|
91
|
-
DIRECT_MODE_STT_PROVIDER,
|
|
92
|
-
'deepgram',
|
|
93
|
-
`Expected provider "deepgram" but got "${DIRECT_MODE_STT_PROVIDER}". ` +
|
|
94
|
-
'deepgram-flux has a known keepalive timeout bug (silent ~30s kills the connection).'
|
|
95
|
-
)
|
|
96
|
-
})
|
|
97
|
-
|
|
98
|
-
test('DIRECT_MODE_STT.provider is NOT "deepgram-flux"', () => {
|
|
99
|
-
assert.notEqual(
|
|
100
|
-
DIRECT_MODE_STT_PROVIDER,
|
|
101
|
-
'deepgram-flux',
|
|
102
|
-
'deepgram-flux must not be the active provider — ' +
|
|
103
|
-
'Flux V2 has silent keepalive timeout bug (~30s silence kills connection).'
|
|
104
|
-
)
|
|
105
|
-
})
|
|
106
|
-
|
|
107
|
-
// ── 2. Model and language ──────────────────────────────────────────────────────
|
|
108
|
-
test('DIRECT_MODE_STT.model is "nova-3"', () => {
|
|
109
|
-
assert.equal(
|
|
110
|
-
DIRECT_MODE_STT_MODEL,
|
|
111
|
-
'nova-3',
|
|
112
|
-
`Expected model "nova-3" but got "${DIRECT_MODE_STT_MODEL}".`
|
|
113
|
-
)
|
|
114
|
-
})
|
|
115
|
-
|
|
116
|
-
test('DIRECT_MODE_STT.language is "en"', () => {
|
|
117
|
-
assert.equal(
|
|
118
|
-
DIRECT_MODE_STT_LANG,
|
|
119
|
-
'en',
|
|
120
|
-
`Expected language "en" but got "${DIRECT_MODE_STT_LANG}".`
|
|
121
|
-
)
|
|
122
|
-
})
|
|
123
|
-
|
|
124
|
-
// ── 3. STTConfig type-level: provider values are the expected set ─────────────
|
|
125
|
-
// (These are compile-time guarantees verified at build time, but we assert
|
|
126
|
-
// them at runtime too so future changes surface here.)
|
|
127
|
-
test('DIRECT_MODE_STT provider is one of the valid union members', () => {
|
|
128
|
-
const validProviders = ['deepgram', 'deepgram-flux', 'groq-whisper', 'openai-whisper']
|
|
129
|
-
assert.ok(
|
|
130
|
-
validProviders.includes(DIRECT_MODE_STT_PROVIDER!),
|
|
131
|
-
`provider "${DIRECT_MODE_STT_PROVIDER}" is not in valid set: ${validProviders.join(', ')}`
|
|
132
|
-
)
|
|
133
|
-
})
|
|
134
|
-
|
|
135
|
-
// ── 4. Backward-compat: createSTT deepgram path shape ────────────────────────
|
|
136
|
-
// We verify the source still has the deepgram case in createSTT so the switch
|
|
137
|
-
// won't hit the `default: throw` path at runtime.
|
|
138
|
-
test('createSTT source still contains deepgram case (backward compat)', async () => {
|
|
139
|
-
const { readFileSync } = await import('node:fs')
|
|
140
|
-
const { fileURLToPath } = await import('node:url')
|
|
141
|
-
const { dirname, join } = await import('node:path')
|
|
142
|
-
const here = dirname(fileURLToPath(import.meta.url))
|
|
143
|
-
const src = readFileSync(join(here, '../src/voice-io.ts'), 'utf8')
|
|
144
|
-
assert.ok(
|
|
145
|
-
src.includes("case 'deepgram':"),
|
|
146
|
-
"createSTT source must still contain case 'deepgram' — removing it would throw at runtime"
|
|
147
|
-
)
|
|
148
|
-
})
|
|
149
|
-
|
|
150
|
-
test('createSTT source still contains deepgram-flux case (backward compat)', async () => {
|
|
151
|
-
const { readFileSync } = await import('node:fs')
|
|
152
|
-
const { fileURLToPath } = await import('node:url')
|
|
153
|
-
const { dirname, join } = await import('node:path')
|
|
154
|
-
const here = dirname(fileURLToPath(import.meta.url))
|
|
155
|
-
const src = readFileSync(join(here, '../src/voice-io.ts'), 'utf8')
|
|
156
|
-
assert.ok(
|
|
157
|
-
src.includes("case 'deepgram-flux':"),
|
|
158
|
-
"createSTT source must still contain case 'deepgram-flux' — it must remain selectable via config"
|
|
159
|
-
)
|
|
160
|
-
})
|
|
161
|
-
|
|
162
|
-
// ── 5. deepgram endpointing value in createSTT ────────────────────────────────
|
|
163
|
-
test('createSTT deepgram case uses 550ms endpointing (mid-sentence fragment prevention)', async () => {
|
|
164
|
-
const { readFileSync } = await import('node:fs')
|
|
165
|
-
const { fileURLToPath } = await import('node:url')
|
|
166
|
-
const { dirname, join } = await import('node:path')
|
|
167
|
-
const here = dirname(fileURLToPath(import.meta.url))
|
|
168
|
-
const src = readFileSync(join(here, '../src/voice-io.ts'), 'utf8')
|
|
169
|
-
assert.ok(
|
|
170
|
-
src.includes('endpointing: 550'),
|
|
171
|
-
'createSTT deepgram case must use endpointing: 550ms to prevent mid-sentence transcript fragments'
|
|
172
|
-
)
|
|
173
|
-
})
|
|
174
|
-
|
|
175
|
-
// ── Summary ───────────────────────────────────────────────────────────────────
|
|
176
|
-
console.log(`\n--- ${passed + failed} tests: ${passed} passed, ${failed} failed ---\n`)
|
|
177
|
-
if (failed > 0) process.exit(1)
|