@crossworks/voice-client 0.230.43
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE.md +135 -0
- package/package.json +20 -0
- package/src/adapters/registry.ts +296 -0
- package/src/adapters/retry.ts +193 -0
- package/src/adapters/types.ts +866 -0
- package/src/audio-tags.test.ts +221 -0
- package/src/audio-tags.ts +191 -0
- package/src/catalog.test.ts +144 -0
- package/src/catalog.ts +237 -0
- package/src/catalogs/anthropic.ts +135 -0
- package/src/catalogs/assemblyai.ts +54 -0
- package/src/catalogs/copilot.ts +63 -0
- package/src/catalogs/deepgram.ts +61 -0
- package/src/catalogs/deepseek.ts +85 -0
- package/src/catalogs/elevenlabs.ts +244 -0
- package/src/catalogs/google.ts +332 -0
- package/src/catalogs/huggingface.ts +180 -0
- package/src/catalogs/openai-image.ts +62 -0
- package/src/catalogs/openai-vision.ts +53 -0
- package/src/catalogs/openrouter.ts +221 -0
- package/src/catalogs/xai.ts +330 -0
- package/src/index.ts +48 -0
- package/src/providers.test.ts +172 -0
- package/src/providers.ts +262 -0
- package/src/types.ts +137 -0
- package/tsconfig.json +4 -0
- package/tsconfig.tsbuildinfo +1 -0
|
@@ -0,0 +1,330 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* xAI (Grok) static catalog.
|
|
3
|
+
*
|
|
4
|
+
* The xAI docs don't officially publish a /v1/models programmatic
|
|
5
|
+
* listing — operators have to consult the console. We mirror what's
|
|
6
|
+
* documented at https://docs.x.ai/developers/models so the dropdown
|
|
7
|
+
* has rich descriptions and capability flags. The adapter still
|
|
8
|
+
* attempts a live `GET /v1/models` call (the API is OpenAI-compatible
|
|
9
|
+
* so it almost always implements it), and if that succeeds we
|
|
10
|
+
* intersect with this catalog to narrow to "what this key can use."
|
|
11
|
+
*
|
|
12
|
+
* Maintenance: when xAI ships a new Grok variant, add an entry here.
|
|
13
|
+
* Anything missing falls through to plain text-input ("Custom model
|
|
14
|
+
* id — make sure your key has access") rather than blocking the user.
|
|
15
|
+
* `pnpm -C server/web models:drift` reports what xAI serves that this
|
|
16
|
+
* file doesn't list — grok-4.5 went unlisted long enough to be found by
|
|
17
|
+
* reading @ai-sdk/xai's model union, which is what prompted that task.
|
|
18
|
+
*
|
|
19
|
+
* ⚠️ Pricing here is the SUB-200k-prompt tier. xAI doubles both rates
|
|
20
|
+
* once the prompt crosses 200k tokens, and `ChatModelInfo` carries one
|
|
21
|
+
* rate per direction, so a long-context turn costs twice what the cost
|
|
22
|
+
* dashboard attributes to it. Noted per-model where the window makes
|
|
23
|
+
* that reachable.
|
|
24
|
+
*
|
|
25
|
+
* Pricing reference (August 2026 docs, https://docs.x.ai/docs/models):
|
|
26
|
+
* grok-4.5 = $2.00/$6.00 per 1M, 500k context. grok-4.3 = $1.25/$2.50,
|
|
27
|
+
* 1M context. Older grok-3 variants redirect to grok-4.3 since May 15.
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
import type { ChatModelInfo } from '../adapters/types';
|
|
31
|
+
|
|
32
|
+
export const XAI_CHAT_MODELS: readonly ChatModelInfo[] = [
|
|
33
|
+
{
|
|
34
|
+
id: 'grok-4.5',
|
|
35
|
+
label: 'Grok 4.5',
|
|
36
|
+
description:
|
|
37
|
+
"xAI's most capable model. reasoning_effort steers depth here and cannot be disabled. Rates DOUBLE above a 200k-token prompt ($4.00/$12.00), which its 500k window makes easy to reach.",
|
|
38
|
+
contextTokens: 500_000,
|
|
39
|
+
capabilities: ['vision', 'function_calling', 'json_mode', 'reasoning'],
|
|
40
|
+
inputPricePer1M: 2.0,
|
|
41
|
+
outputPricePer1M: 6.0,
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
id: 'grok-4.3',
|
|
45
|
+
label: 'Grok 4.3',
|
|
46
|
+
description:
|
|
47
|
+
'Aliased from any deprecated grok-3/4 model id. Rates double above a 200k-token prompt ($2.50/$5.00).',
|
|
48
|
+
contextTokens: 1_000_000,
|
|
49
|
+
capabilities: ['vision', 'function_calling', 'json_mode'],
|
|
50
|
+
inputPricePer1M: 1.25,
|
|
51
|
+
outputPricePer1M: 2.5,
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
id: 'grok-4.20-0309-reasoning',
|
|
55
|
+
label: 'Grok 4.20 (reasoning)',
|
|
56
|
+
description:
|
|
57
|
+
'Reasoning variant — depth is fixed by the model id, not steerable with reasoning_effort.',
|
|
58
|
+
contextTokens: 1_000_000,
|
|
59
|
+
capabilities: ['reasoning', 'function_calling', 'json_mode'],
|
|
60
|
+
inputPricePer1M: 1.25,
|
|
61
|
+
outputPricePer1M: 2.5,
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
id: 'grok-4.20-0309-non-reasoning',
|
|
65
|
+
label: 'Grok 4.20 (no reasoning)',
|
|
66
|
+
description: 'Faster, cheaper variant of 4.20 without reasoning tokens.',
|
|
67
|
+
contextTokens: 1_000_000,
|
|
68
|
+
capabilities: ['function_calling', 'json_mode'],
|
|
69
|
+
inputPricePer1M: 1.25,
|
|
70
|
+
outputPricePer1M: 2.5,
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
id: 'grok-4.20-multi-agent-0309',
|
|
74
|
+
label: 'Grok 4.20 (multi-agent)',
|
|
75
|
+
description:
|
|
76
|
+
'2M context, designed for multi-agent workflows with shared state. reasoning_effort selects the agent COUNT here (4 or 16), not depth.',
|
|
77
|
+
contextTokens: 2_000_000,
|
|
78
|
+
capabilities: ['function_calling', 'json_mode'],
|
|
79
|
+
inputPricePer1M: 1.25,
|
|
80
|
+
outputPricePer1M: 2.5,
|
|
81
|
+
},
|
|
82
|
+
{
|
|
83
|
+
id: 'grok-3',
|
|
84
|
+
label: 'Grok 3 (alias)',
|
|
85
|
+
description:
|
|
86
|
+
'Alias — requests redirect to grok-4.3 and bill at grok-4.3 rates as of May 15, 2026.',
|
|
87
|
+
contextTokens: 1_000_000,
|
|
88
|
+
capabilities: ['vision', 'function_calling'],
|
|
89
|
+
inputPricePer1M: 1.25,
|
|
90
|
+
outputPricePer1M: 2.5,
|
|
91
|
+
},
|
|
92
|
+
];
|
|
93
|
+
|
|
94
|
+
export const XAI_BASE_URL = 'https://api.x.ai/v1';
|
|
95
|
+
|
|
96
|
+
// ─── xAI TTS (Grok voice) ────────────────────────────────────────────
|
|
97
|
+
//
|
|
98
|
+
// Endpoint: POST {XAI_BASE_URL}/tts
|
|
99
|
+
// Auth: Authorization: Bearer $XAI_API_KEY
|
|
100
|
+
// Body: {text, voice_id, language, output_format: {codec, sample_rate, bit_rate}}
|
|
101
|
+
//
|
|
102
|
+
// 5 voices, 20+ languages auto-detected, inline + wrapping speech tags.
|
|
103
|
+
|
|
104
|
+
import type { AudioTag, WrappingTag } from '../adapters/types';
|
|
105
|
+
|
|
106
|
+
/** Grok TTS model — xAI publishes "grok-voice-latest" as the alias. */
|
|
107
|
+
export const XAI_TTS_MODEL_ID = 'grok-voice-latest';
|
|
108
|
+
|
|
109
|
+
/** Voice catalog for Grok TTS. 5 voices, gender/character hints from
|
|
110
|
+
* the xAI launch blog. Operators see these in the worker form. */
|
|
111
|
+
export const XAI_TTS_VOICES = [
|
|
112
|
+
{ id: 'eve', description: 'female, warm — Grok default' },
|
|
113
|
+
{ id: 'ara', description: 'female, clear and bright' },
|
|
114
|
+
{ id: 'rex', description: 'male, deep and grounded' },
|
|
115
|
+
{ id: 'sal', description: 'male, neutral and friendly' },
|
|
116
|
+
{ id: 'leo', description: 'male, animated and energetic' },
|
|
117
|
+
] as const;
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* Inline speech tags Grok TTS honours. Square-bracket form, same shape
|
|
121
|
+
* as ElevenLabs's `[laughs]`. Pulled from
|
|
122
|
+
* https://docs.x.ai/developers/rest-api-reference/inference/voice
|
|
123
|
+
*
|
|
124
|
+
* Wrapping (span-styling) tags live separately — see
|
|
125
|
+
* {@link XAI_WRAPPING_TAGS} below.
|
|
126
|
+
*/
|
|
127
|
+
export const XAI_AUDIO_TAGS: readonly AudioTag[] = [
|
|
128
|
+
// Reactions / human sounds.
|
|
129
|
+
{ tag: '[laugh]', description: 'a hearty laugh', category: 'reaction' },
|
|
130
|
+
{ tag: '[chuckle]', description: 'a short, dry amusement', category: 'reaction' },
|
|
131
|
+
{ tag: '[giggle]', description: 'a light, playful laugh', category: 'reaction' },
|
|
132
|
+
{ tag: '[sigh]', description: 'a resigned or reflective exhale', category: 'reaction' },
|
|
133
|
+
{ tag: '[cry]', description: 'a sob or weeping; use rarely', category: 'reaction' },
|
|
134
|
+
{ tag: '[tsk]', description: 'a disapproving cluck', category: 'reaction' },
|
|
135
|
+
{
|
|
136
|
+
tag: '[tongue-click]',
|
|
137
|
+
description: 'a sharp tongue cluck — punctuation',
|
|
138
|
+
category: 'reaction',
|
|
139
|
+
},
|
|
140
|
+
{ tag: '[lip-smack]', description: 'a soft mouth sound; thoughtful beat', category: 'reaction' },
|
|
141
|
+
{ tag: '[hum-tune]', description: 'a short hummed melody', category: 'reaction' },
|
|
142
|
+
|
|
143
|
+
// Breath.
|
|
144
|
+
{ tag: '[breath]', description: 'a soft audible breath', category: 'reaction' },
|
|
145
|
+
{
|
|
146
|
+
tag: '[inhale]',
|
|
147
|
+
description: 'a sharp inhale; surprise or anticipation',
|
|
148
|
+
category: 'reaction',
|
|
149
|
+
},
|
|
150
|
+
{ tag: '[exhale]', description: 'a deliberate exhale; release', category: 'reaction' },
|
|
151
|
+
|
|
152
|
+
// Pacing.
|
|
153
|
+
{ tag: '[pause]', description: 'a short pause', category: 'cognitive' },
|
|
154
|
+
{ tag: '[long-pause]', description: 'a longer pause for emphasis', category: 'cognitive' },
|
|
155
|
+
];
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Returns the tag set for a given Grok TTS model. xAI publishes one
|
|
159
|
+
* voice model today (grok-voice-latest); future variants would branch
|
|
160
|
+
* here.
|
|
161
|
+
*/
|
|
162
|
+
export function audioTagsForXaiTtsModel(modelId: string): readonly AudioTag[] {
|
|
163
|
+
if (modelId === XAI_TTS_MODEL_ID || modelId === 'grok-voice') {
|
|
164
|
+
return XAI_AUDIO_TAGS;
|
|
165
|
+
}
|
|
166
|
+
return [];
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* Wrapping speech tags Grok TTS honours. Angle-bracket pairs that style
|
|
171
|
+
* the whole phrase they surround, e.g. `<whisper>it's a secret</whisper>`.
|
|
172
|
+
* Distinct from the inline `[bracket]` cues above — these apply a
|
|
173
|
+
* delivery style across a span rather than firing at a point.
|
|
174
|
+
*
|
|
175
|
+
* Source: https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
|
|
176
|
+
* (the "speech tags" section — wrapping tags). Keep this list in sync
|
|
177
|
+
* with the docs; the names map directly to `<name>…</name>`.
|
|
178
|
+
*/
|
|
179
|
+
export const XAI_WRAPPING_TAGS: readonly WrappingTag[] = [
|
|
180
|
+
// Volume.
|
|
181
|
+
{ name: 'loud', description: 'speak the wrapped phrase loudly', category: 'volume' },
|
|
182
|
+
{ name: 'soft', description: 'speak the wrapped phrase quietly and gently', category: 'volume' },
|
|
183
|
+
{
|
|
184
|
+
name: 'whisper',
|
|
185
|
+
description: 'whisper the wrapped phrase — for secrets / asides',
|
|
186
|
+
category: 'volume',
|
|
187
|
+
},
|
|
188
|
+
// Pitch.
|
|
189
|
+
{ name: 'high', description: 'raise the pitch across the wrapped phrase', category: 'pitch' },
|
|
190
|
+
// Pacing.
|
|
191
|
+
{ name: 'slow', description: 'slow the delivery of the wrapped phrase down', category: 'pacing' },
|
|
192
|
+
// Style.
|
|
193
|
+
{
|
|
194
|
+
name: 'singing',
|
|
195
|
+
description: 'sing the wrapped phrase rather than speak it; use rarely',
|
|
196
|
+
category: 'style',
|
|
197
|
+
},
|
|
198
|
+
];
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* Returns the wrapping-tag set for a given Grok TTS model. Mirrors
|
|
202
|
+
* {@link audioTagsForXaiTtsModel} — one voice model today; future
|
|
203
|
+
* variants branch here.
|
|
204
|
+
*/
|
|
205
|
+
export function wrappingTagsForXaiTtsModel(modelId: string): readonly WrappingTag[] {
|
|
206
|
+
if (modelId === XAI_TTS_MODEL_ID || modelId === 'grok-voice') {
|
|
207
|
+
return XAI_WRAPPING_TAGS;
|
|
208
|
+
}
|
|
209
|
+
return [];
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
// ─── xAI STT (Grok transcribe) ───────────────────────────────────────
|
|
213
|
+
//
|
|
214
|
+
// Endpoint: POST {XAI_BASE_URL}/stt
|
|
215
|
+
// Auth: Authorization: Bearer $XAI_API_KEY
|
|
216
|
+
// Body: multipart/form-data — `format`=true, `language`=<iso>, `file` LAST.
|
|
217
|
+
// Response: { text }
|
|
218
|
+
//
|
|
219
|
+
// Accepts WAV, MP3, WebM, OGG, M4A, MP4. 500 MB max per file.
|
|
220
|
+
// Docs: https://docs.x.ai/developers/model-capabilities/audio/speech-to-text
|
|
221
|
+
|
|
222
|
+
import type { SttModelInfo } from '../catalog';
|
|
223
|
+
|
|
224
|
+
/** xAI publishes a single transcription model today ("grok-stt"
|
|
225
|
+
* alias). When variants ship, add entries here — the adapter passes
|
|
226
|
+
* the model id through but xAI's STT API doesn't take a `model` field
|
|
227
|
+
* in the multipart body. We surface the id for traces only. */
|
|
228
|
+
export const XAI_STT_MODELS: readonly SttModelInfo[] = [
|
|
229
|
+
{
|
|
230
|
+
id: 'grok-stt',
|
|
231
|
+
label: 'grok-stt',
|
|
232
|
+
description:
|
|
233
|
+
'xAI Speech-to-Text. Accepts WAV/MP3/WebM/OGG/M4A/MP4 up to 500 MB, returns formatted text.',
|
|
234
|
+
supportsLanguageHint: true,
|
|
235
|
+
supportsTimestamps: false,
|
|
236
|
+
},
|
|
237
|
+
] as const;
|
|
238
|
+
|
|
239
|
+
// ─── xAI Image Generation ────────────────────────────────────────────
|
|
240
|
+
//
|
|
241
|
+
// Endpoint: POST {XAI_BASE_URL}/images/generations
|
|
242
|
+
// Auth: Bearer
|
|
243
|
+
// Body: OpenAI-compatible (prompt + model + n + response_format).
|
|
244
|
+
// xAI returns urls by default; we request response_format=
|
|
245
|
+
// 'b64_json' so the adapter gets bytes uniformly.
|
|
246
|
+
//
|
|
247
|
+
// Model history:
|
|
248
|
+
// - `grok-2-image-1212` — original launch model, DEPRECATED
|
|
249
|
+
// 2026-02-24. The API returns 404 "no longer accessible via the
|
|
250
|
+
// API. Please use grok-imagine-image instead." Removed from this
|
|
251
|
+
// catalog so the dropdown can't suggest it; the adapter
|
|
252
|
+
// translates lingering-config 404s into a migration hint.
|
|
253
|
+
// - `grok-imagine-image` — current model (Feb 2026 →). Same
|
|
254
|
+
// endpoint shape, slightly improved photo-realism.
|
|
255
|
+
|
|
256
|
+
import type { ImageGenModelInfo } from '../adapters/types';
|
|
257
|
+
|
|
258
|
+
export const XAI_IMAGE_MODELS: readonly ImageGenModelInfo[] = [
|
|
259
|
+
{
|
|
260
|
+
id: 'grok-imagine-image',
|
|
261
|
+
label: 'Grok Imagine',
|
|
262
|
+
description:
|
|
263
|
+
'xAI image generator (Feb 2026+). Photo-realistic and editorial styles. Reuses your chat key.',
|
|
264
|
+
// Grok Imagine steers by ASPECT RATIO, not pixel size (docs.x.ai, image
|
|
265
|
+
// generation, checked 2026-08). An earlier entry here claimed a fixed
|
|
266
|
+
// 1024x1024 and no steering at all, which is why the adapter forwarded
|
|
267
|
+
// nothing.
|
|
268
|
+
supportedAspectRatios: [
|
|
269
|
+
'auto',
|
|
270
|
+
'1:1',
|
|
271
|
+
'3:4',
|
|
272
|
+
'4:3',
|
|
273
|
+
'9:16',
|
|
274
|
+
'16:9',
|
|
275
|
+
'2:3',
|
|
276
|
+
'3:2',
|
|
277
|
+
'1:2',
|
|
278
|
+
'2:1',
|
|
279
|
+
],
|
|
280
|
+
pricePerImage: 0.07,
|
|
281
|
+
tier: 'balanced',
|
|
282
|
+
},
|
|
283
|
+
];
|
|
284
|
+
|
|
285
|
+
export const XAI_IMAGE_DEFAULT_MODEL = 'grok-imagine-image';
|
|
286
|
+
/** Old id we still recognise solely to rewrite its API-error
|
|
287
|
+
* message into a useful migration hint. The catalog no longer
|
|
288
|
+
* surfaces this id; only the adapter looks at it. */
|
|
289
|
+
export const XAI_IMAGE_DEPRECATED_MODELS: readonly string[] = ['grok-2-image-1212'];
|
|
290
|
+
|
|
291
|
+
// ─── xAI Vision ──────────────────────────────────────────────────────
|
|
292
|
+
//
|
|
293
|
+
// xAI's chat completions endpoint is OpenAI-compatible — vision is
|
|
294
|
+
// just messages[].content with image_url parts. As of May 2026 the
|
|
295
|
+
// grok-4.x line is multimodal; grok-3 was image-input-only for vision.
|
|
296
|
+
// Discovery cross-references against the chat /v1/models response,
|
|
297
|
+
// same as XAI_CHAT_MODELS.
|
|
298
|
+
|
|
299
|
+
import type { VisionModelInfo } from '../adapters/types';
|
|
300
|
+
|
|
301
|
+
export const XAI_VISION_MODELS: readonly VisionModelInfo[] = [
|
|
302
|
+
{
|
|
303
|
+
id: 'grok-4.3',
|
|
304
|
+
label: 'Grok 4.3',
|
|
305
|
+
description:
|
|
306
|
+
'Current default. Multimodal across text + image. Reuses your chat key. Recommended.',
|
|
307
|
+
contextTokens: 1_000_000,
|
|
308
|
+
inputPricePer1M: 1.25,
|
|
309
|
+
outputPricePer1M: 2.5,
|
|
310
|
+
tier: 'balanced',
|
|
311
|
+
},
|
|
312
|
+
{
|
|
313
|
+
id: 'grok-4.20-0309-non-reasoning',
|
|
314
|
+
label: 'Grok 4.20 (no reasoning)',
|
|
315
|
+
description:
|
|
316
|
+
'Faster, cheaper variant. Good when the task is plain OCR and you don’t need reasoning over the image.',
|
|
317
|
+
contextTokens: 1_000_000,
|
|
318
|
+
inputPricePer1M: 1.25,
|
|
319
|
+
outputPricePer1M: 2.5,
|
|
320
|
+
tier: 'fast',
|
|
321
|
+
},
|
|
322
|
+
{
|
|
323
|
+
id: 'grok-3',
|
|
324
|
+
label: 'Grok 3 (alias)',
|
|
325
|
+
description:
|
|
326
|
+
'Alias — redirects to grok-4.3. Listed so existing workers configured for it still resolve.',
|
|
327
|
+
contextTokens: 1_000_000,
|
|
328
|
+
tier: 'balanced',
|
|
329
|
+
},
|
|
330
|
+
];
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Browser-safe surface of @mantle/voice.
|
|
3
|
+
*
|
|
4
|
+
* The main barrel (`index.ts`) does `export * from './adapters'`, which pulls
|
|
5
|
+
* the network adapter layer (undici → node:crypto) — fine on the server, fatal
|
|
6
|
+
* in a client bundle ("Module parse failed" / node builtin in the browser).
|
|
7
|
+
* The UI settings pages only need *catalog data + provider metadata + pure
|
|
8
|
+
* helpers*, all of which live in adapter-free modules. Import those from HERE
|
|
9
|
+
* (`@mantle/voice/client`) instead of the barrel so the server adapters never
|
|
10
|
+
* reach the browser. Mirrors the `@mantle/content/contacts-format` leaf pattern.
|
|
11
|
+
*
|
|
12
|
+
* Keep this list adapter-free: only re-export modules whose value imports are
|
|
13
|
+
* empty (catalogs/*, catalog, providers, audio-tags) or pure metadata
|
|
14
|
+
* (adapters/registry's `isProviderWired` / `wiredCapabilitiesFor`).
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
// Provider catalogue + capability metadata (pure data + helpers).
|
|
18
|
+
export * from './providers';
|
|
19
|
+
|
|
20
|
+
// Voice/model catalogs + helpers.
|
|
21
|
+
export * from './catalog';
|
|
22
|
+
export * from './audio-tags';
|
|
23
|
+
|
|
24
|
+
// Per-provider model catalogs (data only).
|
|
25
|
+
export * from './catalogs/anthropic';
|
|
26
|
+
export * from './catalogs/google';
|
|
27
|
+
export * from './catalogs/xai';
|
|
28
|
+
export * from './catalogs/huggingface';
|
|
29
|
+
export * from './catalogs/openai-image';
|
|
30
|
+
export * from './catalogs/openai-vision';
|
|
31
|
+
export * from './catalogs/openrouter';
|
|
32
|
+
export * from './catalogs/elevenlabs';
|
|
33
|
+
export * from './catalogs/deepgram';
|
|
34
|
+
export * from './catalogs/assemblyai';
|
|
35
|
+
export * from './catalogs/deepseek';
|
|
36
|
+
|
|
37
|
+
// Pure wiring metadata (no adapter modules pulled — registry.ts only imports
|
|
38
|
+
// the retry helper, which is itself dependency-free).
|
|
39
|
+
export { isProviderWired, wiredCapabilitiesFor, WIRED_PROVIDERS } from './adapters/registry';
|
|
40
|
+
export type { WiredCapability } from './adapters/registry';
|
|
41
|
+
|
|
42
|
+
// Shared types.
|
|
43
|
+
export { TTS_VOICES } from './types';
|
|
44
|
+
export type { TtsVoice } from './types';
|
|
45
|
+
// adapters/types.ts is type-only (no value imports), so re-exporting its whole
|
|
46
|
+
// type surface is browser-safe and covers ChatModelInfo / VisionModelInfo /
|
|
47
|
+
// ImageGenModelInfo / AudioTag / WrappingTag that the worker form needs.
|
|
48
|
+
export type * from './adapters/types';
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for the canonical providers catalog.
|
|
3
|
+
*
|
|
4
|
+
* Why these tests exist: the `api_keys.service` column and
|
|
5
|
+
* `ai_workers.provider` column are free text that the runtime
|
|
6
|
+
* dispatch matches by exact string. Once a value is persisted, the
|
|
7
|
+
* UI catalog and the runtime have to agree forever — renaming
|
|
8
|
+
* 'openai' to 'open-ai' would orphan every saved key. These tests
|
|
9
|
+
* lock the ids so a rename is loud at PR time.
|
|
10
|
+
*
|
|
11
|
+
* Tests also catch the easy mistakes: capability mismatches (an STT
|
|
12
|
+
* provider without 'stt' in its capabilities), missing signup/docs
|
|
13
|
+
* URLs (the UI uses them as links), the kind→capability map
|
|
14
|
+
* accidentally drifting from the ai_workers.kind enum.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { describe, expect, it } from 'vitest';
|
|
18
|
+
import {
|
|
19
|
+
CAPABILITY_FOR_KIND,
|
|
20
|
+
SUPPORTED_PROVIDERS,
|
|
21
|
+
getProvider,
|
|
22
|
+
isProviderId,
|
|
23
|
+
providersForCapability,
|
|
24
|
+
} from './providers';
|
|
25
|
+
|
|
26
|
+
describe('SUPPORTED_PROVIDERS catalog', () => {
|
|
27
|
+
it('lists openrouter first (default chat path)', () => {
|
|
28
|
+
// Catalog order = dropdown order. OpenRouter is the entry-point
|
|
29
|
+
// for new users; if it slips down the list, the UX gets worse.
|
|
30
|
+
expect(SUPPORTED_PROVIDERS[0]?.id).toBe('openrouter');
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
it('lists openai second (required audio path)', () => {
|
|
34
|
+
// OpenAI is the only provider currently catalogued for TTS/STT, so it
|
|
35
|
+
// should be the second option new users see.
|
|
36
|
+
expect(SUPPORTED_PROVIDERS[1]?.id).toBe('openai');
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
it('every provider has a non-empty label, description, signupUrl, docsUrl', () => {
|
|
40
|
+
// Each of these is rendered in the UI. Missing values produce
|
|
41
|
+
// a broken-looking row, so we lock them down.
|
|
42
|
+
for (const p of SUPPORTED_PROVIDERS) {
|
|
43
|
+
expect(p.label.length, `${p.id}.label`).toBeGreaterThan(0);
|
|
44
|
+
expect(p.description.length, `${p.id}.description`).toBeGreaterThan(20);
|
|
45
|
+
expect(p.signupUrl.startsWith('http'), `${p.id}.signupUrl`).toBe(true);
|
|
46
|
+
expect(p.docsUrl.startsWith('http'), `${p.id}.docsUrl`).toBe(true);
|
|
47
|
+
}
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
it('every provider declares at least one capability', () => {
|
|
51
|
+
for (const p of SUPPORTED_PROVIDERS) {
|
|
52
|
+
expect(p.capabilities.length, `${p.id} should have ≥1 capability`).toBeGreaterThan(0);
|
|
53
|
+
}
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
it('ids are unique', () => {
|
|
57
|
+
// A duplicate id would silently shadow the earlier entry, breaking
|
|
58
|
+
// the canonical lookup. Lock it down.
|
|
59
|
+
const ids = SUPPORTED_PROVIDERS.map((p) => p.id);
|
|
60
|
+
expect(new Set(ids).size).toBe(ids.length);
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
it('the providers catalogued for audio (tts/stt) include openai', () => {
|
|
64
|
+
// OpenAI is the only currently-catalogued audio provider. If a refactor
|
|
65
|
+
// drops it from the catalogue, voice in/out goes dark.
|
|
66
|
+
expect(providersForCapability('tts').some((p) => p.id === 'openai')).toBe(true);
|
|
67
|
+
expect(providersForCapability('stt').some((p) => p.id === 'openai')).toBe(true);
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
it('openrouter covers chat + audio (tts/stt) + image_gen via its multimodal APIs', () => {
|
|
71
|
+
// OpenRouter now proxies audio (/audio/speech, /audio/transcriptions) and
|
|
72
|
+
// image generation (chat modalities) in addition to chat/embedding/vision —
|
|
73
|
+
// so one OpenRouter key powers every capability. The adapters
|
|
74
|
+
// (openrouter-tts/stt/image) back these.
|
|
75
|
+
const or = getProvider('openrouter')!;
|
|
76
|
+
expect(or.capabilities).toContain('chat');
|
|
77
|
+
expect(or.capabilities).toContain('tts');
|
|
78
|
+
expect(or.capabilities).toContain('stt');
|
|
79
|
+
expect(or.capabilities).toContain('image_gen');
|
|
80
|
+
});
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
describe('providersForCapability', () => {
|
|
84
|
+
it('returns providers in catalog order', () => {
|
|
85
|
+
// The dropdown reads in this order. OpenRouter is listed first in the
|
|
86
|
+
// catalog and now covers chat + tts + stt + image_gen + vision, so it sits
|
|
87
|
+
// at the top for each — the "one key powers everything" default.
|
|
88
|
+
const chat = providersForCapability('chat');
|
|
89
|
+
expect(chat[0]?.id).toBe('openrouter');
|
|
90
|
+
const tts = providersForCapability('tts');
|
|
91
|
+
expect(tts[0]?.id).toBe('openrouter');
|
|
92
|
+
const stt = providersForCapability('stt');
|
|
93
|
+
expect(stt[0]?.id).toBe('openrouter');
|
|
94
|
+
});
|
|
95
|
+
|
|
96
|
+
it('returns ONLY providers that declared the capability', () => {
|
|
97
|
+
// Defensive: catch any future provider that has the capability
|
|
98
|
+
// in its description but forgot to list it in `capabilities`.
|
|
99
|
+
for (const p of providersForCapability('tts')) {
|
|
100
|
+
expect(p.capabilities, `${p.id} listed as tts-capable`).toContain('tts');
|
|
101
|
+
}
|
|
102
|
+
for (const p of providersForCapability('image_gen')) {
|
|
103
|
+
expect(p.capabilities, `${p.id} listed as image_gen-capable`).toContain('image_gen');
|
|
104
|
+
}
|
|
105
|
+
});
|
|
106
|
+
});
|
|
107
|
+
|
|
108
|
+
describe('getProvider / isProviderId', () => {
|
|
109
|
+
it('looks up by id', () => {
|
|
110
|
+
expect(getProvider('openai')?.label).toBe('OpenAI');
|
|
111
|
+
expect(getProvider('openrouter')?.label).toBe('OpenRouter');
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
it('returns null for unknown ids (no nil-throwing)', () => {
|
|
115
|
+
// The api_keys.service column accepts any string, so legacy rows
|
|
116
|
+
// or hand-edited DBs can carry ids we don't know about. We must
|
|
117
|
+
// not crash the page on those.
|
|
118
|
+
expect(getProvider('made-up-provider')).toBeNull();
|
|
119
|
+
});
|
|
120
|
+
|
|
121
|
+
it('isProviderId narrows correctly', () => {
|
|
122
|
+
expect(isProviderId('openai')).toBe(true);
|
|
123
|
+
expect(isProviderId('openrouter')).toBe(true);
|
|
124
|
+
expect(isProviderId('made-up-provider')).toBe(false);
|
|
125
|
+
expect(isProviderId('')).toBe(false);
|
|
126
|
+
});
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
describe('CAPABILITY_FOR_KIND map', () => {
|
|
130
|
+
it('covers every ai_workers.kind value', () => {
|
|
131
|
+
// The ai_workers.kind enum (in @mantle/db) and this map must stay
|
|
132
|
+
// in sync — otherwise the worker form for a new kind has no
|
|
133
|
+
// provider dropdown filter and shows an EMPTY dropdown
|
|
134
|
+
// (`providersForCapability(undefined)` returns []). This list is
|
|
135
|
+
// hand-maintained because @mantle/voice doesn't depend on
|
|
136
|
+
// @mantle/db (and shouldn't, just for a test). When you add a new
|
|
137
|
+
// ai_worker_kind enum value, add it here AND to CAPABILITY_FOR_KIND
|
|
138
|
+
// in providers.ts. The original 'embedding' miss is why this test
|
|
139
|
+
// is louder about the contract.
|
|
140
|
+
const kinds = [
|
|
141
|
+
'reflector',
|
|
142
|
+
'extractor',
|
|
143
|
+
'summarizer',
|
|
144
|
+
'tts',
|
|
145
|
+
'stt',
|
|
146
|
+
'vision',
|
|
147
|
+
'image_gen',
|
|
148
|
+
'embedding',
|
|
149
|
+
'search',
|
|
150
|
+
'search_advanced',
|
|
151
|
+
'narrator',
|
|
152
|
+
'suggester',
|
|
153
|
+
];
|
|
154
|
+
for (const k of kinds) {
|
|
155
|
+
expect(CAPABILITY_FOR_KIND[k], `kind '${k}' has no capability mapping`).toBeTruthy();
|
|
156
|
+
}
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
it('chat-shaped workers map to the chat capability', () => {
|
|
160
|
+
// reflector, extractor, summarizer all make chat-completion calls.
|
|
161
|
+
expect(CAPABILITY_FOR_KIND.reflector).toBe('chat');
|
|
162
|
+
expect(CAPABILITY_FOR_KIND.extractor).toBe('chat');
|
|
163
|
+
expect(CAPABILITY_FOR_KIND.summarizer).toBe('chat');
|
|
164
|
+
});
|
|
165
|
+
|
|
166
|
+
it('media-shaped workers map to their own capability', () => {
|
|
167
|
+
expect(CAPABILITY_FOR_KIND.tts).toBe('tts');
|
|
168
|
+
expect(CAPABILITY_FOR_KIND.stt).toBe('stt');
|
|
169
|
+
expect(CAPABILITY_FOR_KIND.vision).toBe('vision');
|
|
170
|
+
expect(CAPABILITY_FOR_KIND.image_gen).toBe('image_gen');
|
|
171
|
+
});
|
|
172
|
+
});
|