@tanstack/ai-client 0.6.0 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,180 @@
1
+ import type {
2
+ AnyClientTool,
3
+ AudioVisualization,
4
+ RealtimeEvent,
5
+ RealtimeEventHandler,
6
+ RealtimeMessage,
7
+ RealtimeMode,
8
+ RealtimeSessionConfig,
9
+ RealtimeStatus,
10
+ RealtimeToken,
11
+ } from '@tanstack/ai'
12
+
13
+ // ============================================================================
14
+ // Adapter Interface
15
+ // ============================================================================
16
+
17
+ /**
18
+ * Adapter interface for connecting to realtime providers.
19
+ * Each provider (OpenAI, ElevenLabs, etc.) implements this interface.
20
+ */
21
+ export interface RealtimeAdapter {
22
+ /** Provider identifier */
23
+ provider: string
24
+
25
+ /**
26
+ * Create a connection using the provided token
27
+ * @param token - The ephemeral token from the server
28
+ * @param clientTools - Optional client-side tools to register with the provider
29
+ * @returns A connection instance
30
+ */
31
+ connect: (
32
+ token: RealtimeToken,
33
+ clientTools?: ReadonlyArray<AnyClientTool>,
34
+ ) => Promise<RealtimeConnection>
35
+ }
36
+
37
+ /**
38
+ * Connection interface representing an active realtime session.
39
+ * Handles audio I/O, events, and session management.
40
+ */
41
+ export interface RealtimeConnection {
42
+ // Lifecycle
43
+ /** Disconnect from the realtime session */
44
+ disconnect: () => Promise<void>
45
+
46
+ // Audio I/O
47
+ /** Start capturing audio from the microphone */
48
+ startAudioCapture: () => Promise<void>
49
+ /** Stop capturing audio */
50
+ stopAudioCapture: () => void
51
+
52
+ // Text input
53
+ /** Send a text message (fallback for when voice isn't available) */
54
+ sendText: (text: string) => void
55
+
56
+ // Image input
57
+ /** Send an image to the conversation */
58
+ sendImage: (imageData: string, mimeType: string) => void
59
+
60
+ // Tool results
61
+ /** Send a tool execution result back to the provider */
62
+ sendToolResult: (callId: string, result: string) => void
63
+
64
+ // Session management
65
+ /** Update session configuration */
66
+ updateSession: (config: Partial<RealtimeSessionConfig>) => void
67
+ /** Interrupt the current response */
68
+ interrupt: () => void
69
+
70
+ // Events
71
+ /** Subscribe to connection events */
72
+ on: <TEvent extends RealtimeEvent>(
73
+ event: TEvent,
74
+ handler: RealtimeEventHandler<TEvent>,
75
+ ) => () => void
76
+
77
+ // Audio visualization
78
+ /** Get audio visualization data */
79
+ getAudioVisualization: () => AudioVisualization
80
+ }
81
+
82
+ // ============================================================================
83
+ // Client Options
84
+ // ============================================================================
85
+
86
+ /**
87
+ * Options for the RealtimeClient
88
+ */
89
+ export interface RealtimeClientOptions {
90
+ /**
91
+ * Function to fetch a realtime token from the server.
92
+ * Called on connect and when token needs refresh.
93
+ */
94
+ getToken: () => Promise<RealtimeToken>
95
+
96
+ /**
97
+ * The realtime adapter to use (e.g., openaiRealtime())
98
+ */
99
+ adapter: RealtimeAdapter
100
+
101
+ /**
102
+ * Client-side tools with execution logic
103
+ */
104
+ tools?: ReadonlyArray<AnyClientTool>
105
+
106
+ /**
107
+ * Auto-play assistant audio (default: true)
108
+ */
109
+ autoPlayback?: boolean
110
+
111
+ /**
112
+ * Request microphone access on connect (default: true)
113
+ */
114
+ autoCapture?: boolean
115
+
116
+ /**
117
+ * System instructions for the assistant
118
+ */
119
+ instructions?: string
120
+
121
+ /**
122
+ * Voice to use for audio output
123
+ */
124
+ voice?: string
125
+
126
+ /**
127
+ * Voice activity detection mode (default: 'server')
128
+ */
129
+ vadMode?: 'server' | 'semantic' | 'manual'
130
+
131
+ /**
132
+ * Output modalities for responses (e.g., ['audio', 'text'])
133
+ */
134
+ outputModalities?: Array<'audio' | 'text'>
135
+
136
+ /**
137
+ * Temperature for generation (provider-specific range)
138
+ */
139
+ temperature?: number
140
+
141
+ /**
142
+ * Maximum number of tokens in a response
143
+ */
144
+ maxOutputTokens?: number | 'inf'
145
+
146
+ /**
147
+ * Eagerness level for semantic VAD ('low', 'medium', 'high')
148
+ */
149
+ semanticEagerness?: 'low' | 'medium' | 'high'
150
+
151
+ // Callbacks
152
+ onStatusChange?: (status: RealtimeStatus) => void
153
+ onModeChange?: (mode: RealtimeMode) => void
154
+ onMessage?: (message: RealtimeMessage) => void
155
+ onError?: (error: Error) => void
156
+ onConnect?: () => void
157
+ onDisconnect?: () => void
158
+ onInterrupted?: () => void
159
+ }
160
+
161
+ // ============================================================================
162
+ // Client State
163
+ // ============================================================================
164
+
165
+ /**
166
+ * Internal state of the RealtimeClient
167
+ */
168
+ export interface RealtimeClientState {
169
+ status: RealtimeStatus
170
+ mode: RealtimeMode
171
+ messages: Array<RealtimeMessage>
172
+ pendingUserTranscript: string | null
173
+ pendingAssistantTranscript: string | null
174
+ error: Error | null
175
+ }
176
+
177
+ /**
178
+ * Callback type for state changes
179
+ */
180
+ export type RealtimeStateChangeCallback = (state: RealtimeClientState) => void