osborn 0.9.191 → 0.9.192

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/config.d.ts CHANGED
@@ -2,8 +2,8 @@ import type { McpServerConfig } from './claude-handler.js';
2
2
  export type VoiceMode = 'pipeline';
3
3
  export type EditMode = 'read-only' | 'edit';
4
4
  export type AgentMode = 'plan' | 'execute' | 'research';
5
- export type STTProvider = 'deepgram' | 'groq-whisper' | 'openai-whisper';
6
- export type TTSProvider = 'openai' | 'deepgram';
5
+ export type STTProvider = 'soniox' | 'deepgram' | 'groq-whisper' | 'openai-whisper';
6
+ export type TTSProvider = 'soniox' | 'openai' | 'deepgram';
7
7
  export interface DirectConfig {
8
8
  stt?: {
9
9
  provider?: STTProvider;
package/dist/config.js CHANGED
@@ -47,13 +47,29 @@ const DEFAULT_CONFIG = {
47
47
  voiceMode: 'pipeline',
48
48
  direct: {
49
49
  stt: {
50
- provider: 'deepgram',
51
- model: 'nova-3',
50
+ // Soniox stt-rt-v4 — semantic endpointing (ML, not silence), word timestamps,
51
+ // custom vocabulary via context.terms. ~60% cheaper than nova-3 ($0.0017 vs $0.0043/min).
52
+ // Needs SONIOX_API_KEY.
53
+ provider: 'soniox',
54
+ model: 'stt-rt-v4',
55
+ // Previous: Deepgram nova-3, silence-based endpointing (550ms configured in voice-io.ts).
56
+ // Switch back: provider: 'deepgram', model: 'nova-3'
52
57
  },
53
58
  tts: {
54
- provider: 'openai',
55
- model: 'tts-1-hd',
56
- voice: 'fable',
59
+ // Soniox tts-rt-v1 — real-time WebSocket streaming, speed control (0.7–1.3x),
60
+ // clean abort on interruption. Estimated ~$4–16/M chars vs OpenAI tts-1-hd $30/M.
61
+ // Pricing: $0.70/hr of generated speech (preview). Needs SONIOX_API_KEY.
62
+ provider: 'soniox',
63
+ model: 'tts-rt-v1',
64
+ voice: 'Maya',
65
+ // Previous: OpenAI tts-1-hd, voice fable — high quality, $30/M chars, ~500ms TTFB.
66
+ // Switch back: provider: 'openai', model: 'tts-1-hd', voice: 'fable'
67
+ // Other options already wired in voice-io.ts:
68
+ // Rime Mist v3: provider: 'rime', voice: 'cove' — 37ms TTFB, $30/M, WebSocket
69
+ // Fish Audio s2-pro: provider: 'fishaudio', voice: '<id>' — $15/M, voice cloning
70
+ // Groq Orpheus: provider: 'groq-orpheus', voice: 'autumn' — fast Groq chips, $22/M
71
+ // Deepgram Aura-2: provider: 'deepgram', model: 'aura-2-asteria-en' — $15/M, ~100ms TTFB
72
+ // OpenAI tts-1: provider: 'openai', model: 'tts-1', voice: 'fable' — $15/M
57
73
  },
58
74
  },
59
75
  mcpServers: {
@@ -5,8 +5,9 @@
5
5
  import * as deepgram from '@livekit/agents-plugin-deepgram';
6
6
  import * as openai from '@livekit/agents-plugin-openai';
7
7
  import * as silero from '@livekit/agents-plugin-silero';
8
+ import * as soniox from '@livekit/agents-plugin-soniox';
8
9
  export interface STTConfig {
9
- provider: 'deepgram' | 'deepgram-flux' | 'groq-whisper' | 'openai-whisper';
10
+ provider: 'soniox' | 'deepgram' | 'deepgram-flux' | 'groq-whisper' | 'openai-whisper';
10
11
  model?: string;
11
12
  language?: string;
12
13
  /** Deepgram Flux: end-of-turn confidence threshold (0.0-1.0, default 0.7) */
@@ -15,7 +16,7 @@ export interface STTConfig {
15
16
  eotTimeoutMs?: number;
16
17
  }
17
18
  export interface TTSConfig {
18
- provider: 'openai' | 'deepgram' | 'groq-orpheus' | 'fishaudio' | 'rime';
19
+ provider: 'soniox' | 'openai' | 'deepgram' | 'groq-orpheus' | 'fishaudio' | 'rime';
19
20
  voice?: string;
20
21
  model?: string;
21
22
  }
@@ -27,7 +28,7 @@ export interface VoiceIOConfig {
27
28
  * Create STT (Speech-to-Text) instance based on config
28
29
  * Note: Gemini STT is not available in Node.js, using Deepgram as default
29
30
  */
30
- export declare function createSTT(config: STTConfig): deepgram.STT | deepgram.STTv2 | openai.STT;
31
+ export declare function createSTT(config: STTConfig): soniox.STT | deepgram.STT | deepgram.STTv2 | openai.STT;
31
32
  /**
32
33
  * Create TTS (Text-to-Speech) instance based on config
33
34
  */
package/dist/voice-io.js CHANGED
@@ -7,13 +7,31 @@ import * as fishaudio from '@livekit/agents-plugin-fishaudio';
7
7
  import * as openai from '@livekit/agents-plugin-openai';
8
8
  import * as rime from '@livekit/agents-plugin-rime';
9
9
  import * as silero from '@livekit/agents-plugin-silero';
10
+ import * as soniox from '@livekit/agents-plugin-soniox';
10
11
  /**
11
12
  * Create STT (Speech-to-Text) instance based on config
12
13
  * Note: Gemini STT is not available in Node.js, using Deepgram as default
13
14
  */
14
15
  export function createSTT(config) {
15
16
  switch (config.provider) {
17
+ case 'soniox':
18
+ // Soniox stt-rt-v4 — semantic endpointing: ML model holds on incomplete thoughts,
19
+ // commits on natural sentence ends. maxEndpointDelayMs 500–3000ms (minimum = fastest).
20
+ // endpointLatencyAdjustmentLevel 0–3: higher = more aggressive latency reduction
21
+ // while keeping semantic smarts. context.terms biases recognition toward code vocab.
22
+ return new soniox.STT({
23
+ model: (config.model || 'stt-rt-v4'),
24
+ languageHints: config.language ? [config.language] : ['en'],
25
+ maxEndpointDelayMs: 1200, // give model room to decide on mid-thought pauses
26
+ endpointLatencyAdjustmentLevel: 2, // aggressive but not max — good for voice assistant
27
+ context: {
28
+ terms: ['Claude', 'TypeScript', 'LiveKit', 'Deepgram', 'npm', 'Railway', 'Fly.io'],
29
+ },
30
+ });
16
31
  case 'deepgram':
32
+ // Previous default. Silence-based endpointing (550ms configured = wait for 550ms quiet).
33
+ // Fast and reliable, but commits on any pause — doesn't understand mid-thought hesitations.
34
+ // Switch back: provider: 'deepgram', model: 'nova-3'
17
35
  return new deepgram.STT({
18
36
  model: (config.model || 'nova-3'),
19
37
  language: config.language || 'en',
@@ -49,13 +67,30 @@ export function createSTT(config) {
49
67
  export function createTTS(config) {
50
68
  let tts;
51
69
  switch (config.provider) {
70
+ case 'soniox':
71
+ // Soniox tts-rt-v1 — real-time WebSocket streaming, clean abort on interruption.
72
+ // Estimated ~$4–16/M chars ($0.70/hr of generated speech, preview pricing).
73
+ // speed: 0.7–1.3x. voices: Maya (female), others at soniox.com/docs/tts/voices.
74
+ // Previous TTS: OpenAI tts-1-hd (fable) — $30/M chars, ~500ms TTFB, HTTP chunked.
75
+ // Switch back: provider: 'openai', model: 'tts-1-hd', voice: 'fable'
76
+ tts = new soniox.TTS({
77
+ model: (config.model || 'tts-rt-v1'),
78
+ voice: config.voice || 'Maya',
79
+ speed: 1.0,
80
+ });
81
+ break;
52
82
  case 'openai':
83
+ // tts-1-hd: $30/M chars, ~500ms TTFB, 6 voices: alloy echo fable onyx nova shimmer.
84
+ // tts-1 (cheaper): $15/M chars, slightly lower quality, same voices.
53
85
  tts = new openai.TTS({
54
86
  voice: config.voice || 'alloy',
55
87
  model: config.model || 'tts-1',
56
88
  });
57
89
  break;
58
90
  case 'deepgram':
91
+ // Aura-2 voices: aura-2-asteria-en, aura-2-luna-en, aura-2-stella-en, aura-2-hera-en
92
+ // aura-2-orion-en, aura-2-arcas-en, aura-2-perseus-en, aura-2-angus-en, aura-2-orpheus-en
93
+ // ~$15/M chars, ~100ms TTFB.
59
94
  tts = new deepgram.TTS({
60
95
  model: (config.model || 'aura-2-asteria-en'),
61
96
  });
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.191",
3
+ "version": "0.9.192",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {
@@ -39,6 +39,7 @@
39
39
  "@livekit/agents-plugin-openai": "1.4.6",
40
40
  "@livekit/agents-plugin-rime": "1.4.6",
41
41
  "@livekit/agents-plugin-silero": "1.4.6",
42
+ "@livekit/agents-plugin-soniox": "^1.8.1",
42
43
  "@livekit/rtc-node": "0.13.29",
43
44
  "@modelcontextprotocol/sdk": "^1.29.0",
44
45
  "@smithery/api": "^0.48.0",