@crossworks/voice-client 0.230.43
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE.md +135 -0
- package/package.json +20 -0
- package/src/adapters/registry.ts +296 -0
- package/src/adapters/retry.ts +193 -0
- package/src/adapters/types.ts +866 -0
- package/src/audio-tags.test.ts +221 -0
- package/src/audio-tags.ts +191 -0
- package/src/catalog.test.ts +144 -0
- package/src/catalog.ts +237 -0
- package/src/catalogs/anthropic.ts +135 -0
- package/src/catalogs/assemblyai.ts +54 -0
- package/src/catalogs/copilot.ts +63 -0
- package/src/catalogs/deepgram.ts +61 -0
- package/src/catalogs/deepseek.ts +85 -0
- package/src/catalogs/elevenlabs.ts +244 -0
- package/src/catalogs/google.ts +332 -0
- package/src/catalogs/huggingface.ts +180 -0
- package/src/catalogs/openai-image.ts +62 -0
- package/src/catalogs/openai-vision.ts +53 -0
- package/src/catalogs/openrouter.ts +221 -0
- package/src/catalogs/xai.ts +330 -0
- package/src/index.ts +48 -0
- package/src/providers.test.ts +172 -0
- package/src/providers.ts +262 -0
- package/src/types.ts +137 -0
- package/tsconfig.json +4 -0
- package/tsconfig.tsbuildinfo +1 -0
package/src/catalog.ts
ADDED
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Catalog of TTS / STT models + voices, per provider.
|
|
3
|
+
*
|
|
4
|
+
* Why a static catalog at all: OpenAI does NOT expose a programmatic
|
|
5
|
+
* endpoint to list voices. `/v1/audio/voices` doesn't exist; the voice
|
|
6
|
+
* list is documentation-only and changes when OpenAI ships new models.
|
|
7
|
+
* We keep the mapping in code so the UI can render a smart dropdown
|
|
8
|
+
* (model selected → voices for that model appear) without each page
|
|
9
|
+
* load having to scrape docs.
|
|
10
|
+
*
|
|
11
|
+
* What IS queryable: `/v1/models` returns the model ids the key has
|
|
12
|
+
* access to (alongside chat/embedding models). We cross-reference our
|
|
13
|
+
* catalog with that list to show ONLY the models the user can
|
|
14
|
+
* actually use — accounts on the free tier, or older keys, don't
|
|
15
|
+
* always have every model available.
|
|
16
|
+
*
|
|
17
|
+
* Maintenance: when OpenAI releases a new TTS model, add it here. The
|
|
18
|
+
* UI doesn't need a code change as long as the catalog is current.
|
|
19
|
+
*
|
|
20
|
+
* Other providers (ElevenLabs, Deepgram) get their own catalog entries
|
|
21
|
+
* when we implement them. ElevenLabs unlike OpenAI HAS a /v1/voices
|
|
22
|
+
* endpoint that returns the full list (including user-cloned voices),
|
|
23
|
+
* so for that provider we'd skip the static catalog and query live.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import type { TtsVoice } from './types';
|
|
27
|
+
|
|
28
|
+
/** Every voice OpenAI has ever shipped (across all TTS models). Kept
|
|
29
|
+
* as the source-of-truth string union so other modules don't have to
|
|
30
|
+
* reconcile names. */
|
|
31
|
+
export const ALL_OPENAI_VOICES = [
|
|
32
|
+
// Original 6 (tts-1, tts-1-hd).
|
|
33
|
+
'alloy',
|
|
34
|
+
'echo',
|
|
35
|
+
'fable',
|
|
36
|
+
'nova',
|
|
37
|
+
'onyx',
|
|
38
|
+
'shimmer',
|
|
39
|
+
// Added with later expansions — usable by tts-1, tts-1-hd, gpt-4o-mini-tts.
|
|
40
|
+
'ash',
|
|
41
|
+
'coral',
|
|
42
|
+
'sage',
|
|
43
|
+
// Newer voices — only on gpt-4o-mini-tts.
|
|
44
|
+
'ballad',
|
|
45
|
+
'verse',
|
|
46
|
+
// "Best quality" voices — gpt-4o-mini-tts only.
|
|
47
|
+
'marin',
|
|
48
|
+
'cedar',
|
|
49
|
+
] as const;
|
|
50
|
+
|
|
51
|
+
export type OpenAiVoice = (typeof ALL_OPENAI_VOICES)[number];
|
|
52
|
+
|
|
53
|
+
/** Short, human-readable description per voice. Used in the UI
|
|
54
|
+
* dropdown so the operator doesn't have to test all 13 to find the
|
|
55
|
+
* warm female voice. Pulled from OpenAI's published descriptions. */
|
|
56
|
+
export const VOICE_DESCRIPTIONS: Record<OpenAiVoice, string> = {
|
|
57
|
+
alloy: 'neutral, balanced',
|
|
58
|
+
ash: 'warm, expressive',
|
|
59
|
+
ballad: 'reflective, narrative',
|
|
60
|
+
cedar: 'high-quality, natural (recommended)',
|
|
61
|
+
coral: 'warm, friendly female',
|
|
62
|
+
echo: 'male, calm',
|
|
63
|
+
fable: 'British, warm',
|
|
64
|
+
marin: 'high-quality, natural (recommended)',
|
|
65
|
+
nova: 'warm, female (Saskia default)',
|
|
66
|
+
onyx: 'deep male, grounded',
|
|
67
|
+
sage: 'measured, thoughtful',
|
|
68
|
+
shimmer: 'soft, female',
|
|
69
|
+
verse: 'expressive, emotive',
|
|
70
|
+
};
|
|
71
|
+
|
|
72
|
+
/** A model entry: what it is, which voices it supports, and which
|
|
73
|
+
* feature flags are on. */
|
|
74
|
+
export type TtsModelInfo = {
|
|
75
|
+
id: string;
|
|
76
|
+
label: string;
|
|
77
|
+
description: string;
|
|
78
|
+
/** Voice ids the model accepts. OpenAI ships a narrow named union
|
|
79
|
+
* (alloy/nova/shimmer/…); xAI ships their own (eve/ara/rex/sal/leo);
|
|
80
|
+
* Gemini ships 30 (Kore/Puck/Zephyr/…); ElevenLabs ships UUIDs and
|
|
81
|
+
* per-account clones. We type this as `readonly string[]` so each
|
|
82
|
+
* provider's adapter can return its own list without union-widening
|
|
83
|
+
* the framework. Consumers treating these as `OpenAiVoice` for
|
|
84
|
+
* type-narrowing have always known to runtime-check anyway. */
|
|
85
|
+
voices: readonly string[];
|
|
86
|
+
/** Whether the model accepts a free-form `instructions` parameter for
|
|
87
|
+
* style steering ("speak warmly", "be calm"). Only true for
|
|
88
|
+
* gpt-4o-mini-tts at the moment. */
|
|
89
|
+
supportsInstructions: boolean;
|
|
90
|
+
/** Tier hint for cost display. */
|
|
91
|
+
tier: 'low-latency' | 'high-quality' | 'steerable';
|
|
92
|
+
};
|
|
93
|
+
|
|
94
|
+
/** TTS catalog. Order matters — list shown to users in this order in
|
|
95
|
+
* the dropdown. */
|
|
96
|
+
export const OPENAI_TTS_MODELS: readonly TtsModelInfo[] = [
|
|
97
|
+
{
|
|
98
|
+
id: 'gpt-4o-mini-tts',
|
|
99
|
+
label: 'gpt-4o-mini-tts',
|
|
100
|
+
description:
|
|
101
|
+
'Newest TTS model. 13 voices, accepts style instructions ("speak warmly"). Recommended.',
|
|
102
|
+
voices: [
|
|
103
|
+
'alloy',
|
|
104
|
+
'ash',
|
|
105
|
+
'ballad',
|
|
106
|
+
'coral',
|
|
107
|
+
'echo',
|
|
108
|
+
'fable',
|
|
109
|
+
'nova',
|
|
110
|
+
'onyx',
|
|
111
|
+
'sage',
|
|
112
|
+
'shimmer',
|
|
113
|
+
'verse',
|
|
114
|
+
'marin',
|
|
115
|
+
'cedar',
|
|
116
|
+
],
|
|
117
|
+
supportsInstructions: true,
|
|
118
|
+
tier: 'steerable',
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
id: 'tts-1',
|
|
122
|
+
label: 'tts-1',
|
|
123
|
+
description: 'Original TTS model. 9 voices, low latency. Cheaper than gpt-4o-mini-tts.',
|
|
124
|
+
voices: ['alloy', 'ash', 'coral', 'echo', 'fable', 'nova', 'onyx', 'sage', 'shimmer'],
|
|
125
|
+
supportsInstructions: false,
|
|
126
|
+
tier: 'low-latency',
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
id: 'tts-1-hd',
|
|
130
|
+
label: 'tts-1-hd',
|
|
131
|
+
description: 'Higher-quality variant of tts-1. Same 9 voices, higher fidelity, ~2× cost.',
|
|
132
|
+
voices: ['alloy', 'ash', 'coral', 'echo', 'fable', 'nova', 'onyx', 'sage', 'shimmer'],
|
|
133
|
+
supportsInstructions: false,
|
|
134
|
+
tier: 'high-quality',
|
|
135
|
+
},
|
|
136
|
+
] as const;
|
|
137
|
+
|
|
138
|
+
export type SttModelInfo = {
|
|
139
|
+
id: string;
|
|
140
|
+
label: string;
|
|
141
|
+
description: string;
|
|
142
|
+
/** Whether this model accepts the `language` hint param. */
|
|
143
|
+
supportsLanguageHint: boolean;
|
|
144
|
+
/** Whether this model returns word-level timestamps. */
|
|
145
|
+
supportsTimestamps: boolean;
|
|
146
|
+
};
|
|
147
|
+
|
|
148
|
+
/** STT catalog. As of May 2026 OpenAI ships whisper-1 alongside the
|
|
149
|
+
* newer gpt-4o-mini-transcribe / gpt-4o-transcribe variants which
|
|
150
|
+
* are higher-quality but cost more. */
|
|
151
|
+
export const OPENAI_STT_MODELS: readonly SttModelInfo[] = [
|
|
152
|
+
{
|
|
153
|
+
id: 'whisper-1',
|
|
154
|
+
label: 'whisper-1',
|
|
155
|
+
description: 'Stable, cheap. Excellent multilingual support including Afrikaans.',
|
|
156
|
+
supportsLanguageHint: true,
|
|
157
|
+
supportsTimestamps: true,
|
|
158
|
+
},
|
|
159
|
+
{
|
|
160
|
+
id: 'gpt-4o-mini-transcribe',
|
|
161
|
+
label: 'gpt-4o-mini-transcribe',
|
|
162
|
+
description: 'Newer transcription model. Better accuracy than whisper-1, similar price.',
|
|
163
|
+
supportsLanguageHint: true,
|
|
164
|
+
supportsTimestamps: false,
|
|
165
|
+
},
|
|
166
|
+
{
|
|
167
|
+
id: 'gpt-4o-transcribe',
|
|
168
|
+
label: 'gpt-4o-transcribe',
|
|
169
|
+
description: 'Highest-accuracy transcription. Costs more; best for difficult audio.',
|
|
170
|
+
supportsLanguageHint: true,
|
|
171
|
+
supportsTimestamps: false,
|
|
172
|
+
},
|
|
173
|
+
] as const;
|
|
174
|
+
|
|
175
|
+
/** Look up a TTS model by id. Returns null if we don't know about it
|
|
176
|
+
* (e.g. user typed a custom model name in the form). Callers should
|
|
177
|
+
* fall back to a permissive default voice list in that case. */
|
|
178
|
+
export function getTtsModel(id: string): TtsModelInfo | null {
|
|
179
|
+
return OPENAI_TTS_MODELS.find((m) => m.id === id) ?? null;
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
export function getSttModel(id: string): SttModelInfo | null {
|
|
183
|
+
return OPENAI_STT_MODELS.find((m) => m.id === id) ?? null;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/** Type-narrowing helper: is this voice valid for OpenAI? Used by the
|
|
187
|
+
* form validator and the synth call. */
|
|
188
|
+
export function isOpenAiVoice(v: string): v is OpenAiVoice {
|
|
189
|
+
return (ALL_OPENAI_VOICES as readonly string[]).includes(v);
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/** All voices, with descriptions, for a given model. Returned in a
|
|
193
|
+
* stable order matching the catalog. If the model isn't in the
|
|
194
|
+
* catalog, returns an empty array (caller decides whether to fall
|
|
195
|
+
* back to the legacy 6-voice list from @mantle/voice/types). */
|
|
196
|
+
export function voicesForModel(modelId: string): Array<{ id: OpenAiVoice; description: string }> {
|
|
197
|
+
const model = getTtsModel(modelId);
|
|
198
|
+
if (!model) return [];
|
|
199
|
+
// OPENAI_TTS_MODELS only lists OpenAI voice ids by construction —
|
|
200
|
+
// the runtime cast is safe because we narrow via getTtsModel which
|
|
201
|
+
// only returns OpenAI catalog entries.
|
|
202
|
+
return model.voices.map((v) => ({
|
|
203
|
+
id: v as OpenAiVoice,
|
|
204
|
+
description: VOICE_DESCRIPTIONS[v as OpenAiVoice] ?? '',
|
|
205
|
+
}));
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
// Bridge to the existing TtsVoice union in types.ts — keeps backward
|
|
209
|
+
// compat with code already using that narrower type. New code should
|
|
210
|
+
// prefer OpenAiVoice for forward-compat with the expanded voice set.
|
|
211
|
+
void ([] as TtsVoice[]);
|
|
212
|
+
|
|
213
|
+
export type DiscoveryResult<T> = {
|
|
214
|
+
/** The catalog entries the key can actually use, in catalog order. */
|
|
215
|
+
available: T[];
|
|
216
|
+
/** True if the live filter succeeded; false if we fell back to the
|
|
217
|
+
* full catalog because the API call failed. The UI surfaces a hint
|
|
218
|
+
* in that case ("couldn't verify; showing everything"). */
|
|
219
|
+
filtered: boolean;
|
|
220
|
+
/** When `filtered=false`, the reason. Null on success. */
|
|
221
|
+
error: string | null;
|
|
222
|
+
/**
|
|
223
|
+
* EVERY model id the provider reported, before we intersected it with our
|
|
224
|
+
* catalog. Optional: only the adapters that curate a static catalog and
|
|
225
|
+
* filter it need to set this.
|
|
226
|
+
*
|
|
227
|
+
* It exists because `available` answers "which of OUR models are live?" and
|
|
228
|
+
* deliberately discards the other half of the answer — "which of THEIRS
|
|
229
|
+
* aren't ours?" That discarded half is the only automatic signal that a
|
|
230
|
+
* catalog has gone stale, and we were computing it on every discovery call
|
|
231
|
+
* and dropping it. grok-4.5 shipped, our dropdown never mentioned it, and
|
|
232
|
+
* nothing anywhere could have said so.
|
|
233
|
+
*
|
|
234
|
+
* Consumed by `pnpm -C server/web models:drift`. See {@link catalogDrift}.
|
|
235
|
+
*/
|
|
236
|
+
liveIds?: string[];
|
|
237
|
+
};
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Anthropic (Claude) static catalog.
|
|
3
|
+
*
|
|
4
|
+
* Anthropic's API is NOT OpenAI-compatible. It uses its own message
|
|
5
|
+
* format (system is a separate top-level field, messages contain only
|
|
6
|
+
* user/assistant roles) at `https://api.anthropic.com/v1/messages`.
|
|
7
|
+
* The adapter handles translation to/from the unified ChatDispatcher
|
|
8
|
+
* interface so callers don't have to care.
|
|
9
|
+
*
|
|
10
|
+
* Auth: `x-api-key` header + a required `anthropic-version` header
|
|
11
|
+
* (we pin to 2023-06-01 since that's the stable API). The Models API
|
|
12
|
+
* at GET /v1/messages/../v1/models returns the live model list with
|
|
13
|
+
* capabilities — we use it for discovery but the catalog below is
|
|
14
|
+
* the source of truth for rich UI metadata.
|
|
15
|
+
*
|
|
16
|
+
* Maintenance: when Anthropic ships a new Claude generation (e.g.
|
|
17
|
+
* 4.8 or 5.0), add it here. Anthropic uses dateless aliases for
|
|
18
|
+
* 4.6+ (no more 'claude-3-5-sonnet-20241022' style), so model ids
|
|
19
|
+
* stay readable.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import type { ChatModelInfo, VisionModelInfo } from '../adapters/types';
|
|
23
|
+
|
|
24
|
+
export const ANTHROPIC_BASE_URL = 'https://api.anthropic.com';
|
|
25
|
+
/** Required header value on all Anthropic API calls. */
|
|
26
|
+
export const ANTHROPIC_API_VERSION = '2023-06-01';
|
|
27
|
+
|
|
28
|
+
export const ANTHROPIC_CHAT_MODELS: readonly ChatModelInfo[] = [
|
|
29
|
+
// ── Current generation (4.6/4.7) ─────────────────────────────────
|
|
30
|
+
{
|
|
31
|
+
id: 'claude-opus-4-7',
|
|
32
|
+
label: 'Claude Opus 4.7',
|
|
33
|
+
description: 'Anthropic flagship. Best for complex reasoning + agentic coding. 1M context.',
|
|
34
|
+
contextTokens: 1_000_000,
|
|
35
|
+
capabilities: ['vision', 'reasoning', 'function_calling'],
|
|
36
|
+
inputPricePer1M: 5,
|
|
37
|
+
outputPricePer1M: 25,
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
id: 'claude-sonnet-5',
|
|
41
|
+
label: 'Claude Sonnet 5',
|
|
42
|
+
description:
|
|
43
|
+
'Best speed/intelligence balance — newer and cheaper than 4.6. 1M context, supports extended thinking. Default choice.',
|
|
44
|
+
contextTokens: 1_000_000,
|
|
45
|
+
capabilities: ['vision', 'reasoning', 'function_calling'],
|
|
46
|
+
inputPricePer1M: 2,
|
|
47
|
+
outputPricePer1M: 10,
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
id: 'claude-sonnet-4-6',
|
|
51
|
+
label: 'Claude Sonnet 4.6',
|
|
52
|
+
description: 'Previous-generation Sonnet. 1M context, supports extended thinking.',
|
|
53
|
+
contextTokens: 1_000_000,
|
|
54
|
+
capabilities: ['vision', 'reasoning', 'function_calling'],
|
|
55
|
+
inputPricePer1M: 3,
|
|
56
|
+
outputPricePer1M: 15,
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
id: 'claude-haiku-4-5',
|
|
60
|
+
label: 'Claude Haiku 4.5',
|
|
61
|
+
description:
|
|
62
|
+
'Fastest model with near-frontier intelligence. 200k context. Great for cheap, fast jobs.',
|
|
63
|
+
contextTokens: 200_000,
|
|
64
|
+
capabilities: ['vision', 'reasoning', 'function_calling'],
|
|
65
|
+
inputPricePer1M: 1,
|
|
66
|
+
outputPricePer1M: 5,
|
|
67
|
+
},
|
|
68
|
+
|
|
69
|
+
// ── Legacy (still available, slightly cheaper or different trade-offs) ──
|
|
70
|
+
{
|
|
71
|
+
id: 'claude-opus-4-6',
|
|
72
|
+
label: 'Claude Opus 4.6 (legacy)',
|
|
73
|
+
description: 'Previous Opus generation. Migrate to 4.7 for best results.',
|
|
74
|
+
contextTokens: 1_000_000,
|
|
75
|
+
capabilities: ['vision', 'reasoning', 'function_calling'],
|
|
76
|
+
inputPricePer1M: 5,
|
|
77
|
+
outputPricePer1M: 25,
|
|
78
|
+
},
|
|
79
|
+
{
|
|
80
|
+
id: 'claude-sonnet-4-5',
|
|
81
|
+
label: 'Claude Sonnet 4.5 (legacy)',
|
|
82
|
+
description: 'Previous Sonnet generation. Still capable; 200k context.',
|
|
83
|
+
contextTokens: 200_000,
|
|
84
|
+
capabilities: ['vision', 'reasoning', 'function_calling'],
|
|
85
|
+
inputPricePer1M: 3,
|
|
86
|
+
outputPricePer1M: 15,
|
|
87
|
+
},
|
|
88
|
+
];
|
|
89
|
+
|
|
90
|
+
/** Vision-capable Claude models. Every current Claude model accepts
|
|
91
|
+
* image content blocks, but we surface a smaller curated set for the
|
|
92
|
+
* vision-worker dropdown — Haiku for cheap high-volume OCR, Sonnet
|
|
93
|
+
* for balanced quality, Opus when accuracy on hard handwriting
|
|
94
|
+
* matters more than cost. Opus is over-the-top for printed text. */
|
|
95
|
+
export const ANTHROPIC_VISION_MODELS: readonly VisionModelInfo[] = [
|
|
96
|
+
{
|
|
97
|
+
id: 'claude-haiku-4-5',
|
|
98
|
+
label: 'Claude Haiku 4.5',
|
|
99
|
+
description:
|
|
100
|
+
'Cheap, fast vision. Solid on printed text and clean handwriting. Default choice for high-volume OCR.',
|
|
101
|
+
contextTokens: 200_000,
|
|
102
|
+
inputPricePer1M: 1,
|
|
103
|
+
outputPricePer1M: 5,
|
|
104
|
+
tier: 'fast',
|
|
105
|
+
},
|
|
106
|
+
{
|
|
107
|
+
id: 'claude-sonnet-5',
|
|
108
|
+
label: 'Claude Sonnet 5',
|
|
109
|
+
description:
|
|
110
|
+
'Best balance. Strong on messy handwriting + multi-column layouts. Recommended for note photos.',
|
|
111
|
+
contextTokens: 1_000_000,
|
|
112
|
+
inputPricePer1M: 2,
|
|
113
|
+
outputPricePer1M: 10,
|
|
114
|
+
tier: 'balanced',
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
id: 'claude-sonnet-4-6',
|
|
118
|
+
label: 'Claude Sonnet 4.6',
|
|
119
|
+
description: 'Previous-generation Sonnet vision. Still strong on documents.',
|
|
120
|
+
contextTokens: 1_000_000,
|
|
121
|
+
inputPricePer1M: 3,
|
|
122
|
+
outputPricePer1M: 15,
|
|
123
|
+
tier: 'balanced',
|
|
124
|
+
},
|
|
125
|
+
{
|
|
126
|
+
id: 'claude-opus-4-7',
|
|
127
|
+
label: 'Claude Opus 4.7',
|
|
128
|
+
description:
|
|
129
|
+
'Frontier. Use when Sonnet misreads cursive or you need diagram comprehension alongside text. Pricey.',
|
|
130
|
+
contextTokens: 1_000_000,
|
|
131
|
+
inputPricePer1M: 5,
|
|
132
|
+
outputPricePer1M: 25,
|
|
133
|
+
tier: 'quality',
|
|
134
|
+
},
|
|
135
|
+
];
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* AssemblyAI static catalog.
|
|
3
|
+
*
|
|
4
|
+
* AssemblyAI's transcription API is a two-step async job:
|
|
5
|
+
* 1. POST /v2/upload — binary body, returns a temporary upload URL.
|
|
6
|
+
* 2. POST /v2/transcript with `{audio_url, speech_model, language_code}`
|
|
7
|
+
* — returns a transcript id with status='queued' or 'processing'.
|
|
8
|
+
* 3. Poll GET /v2/transcript/{id} until status='completed' (or 'error').
|
|
9
|
+
*
|
|
10
|
+
* Because of step 3, AssemblyAI is unsuitable for ultra-low-latency
|
|
11
|
+
* use cases — the round trip for a short voice note is 2-5 seconds
|
|
12
|
+
* even when the model itself is fast. We use it when the operator
|
|
13
|
+
* specifically wants diarization or sentiment, which AssemblyAI
|
|
14
|
+
* exposes alongside the base transcript.
|
|
15
|
+
*
|
|
16
|
+
* Speech models (May 2026): the "universal" tier is the current
|
|
17
|
+
* default; "best" and "nano" are documented but billing tiers vary by
|
|
18
|
+
* account. We surface the documented values; the adapter passes the
|
|
19
|
+
* id through to AssemblyAI as `speech_model` in the request body.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import type { SttModelInfo } from '../catalog';
|
|
23
|
+
|
|
24
|
+
export const ASSEMBLYAI_BASE_URL = 'https://api.assemblyai.com';
|
|
25
|
+
|
|
26
|
+
/** Hard cap on polling — if the transcript isn't done after this many
|
|
27
|
+
* seconds, the adapter gives up rather than blocking the test action
|
|
28
|
+
* forever. AssemblyAI's docs say 95% of jobs finish in under 35s for
|
|
29
|
+
* audio under 5 minutes, so 60s is a generous upper bound. */
|
|
30
|
+
export const ASSEMBLYAI_POLL_TIMEOUT_SECONDS = 60;
|
|
31
|
+
|
|
32
|
+
export const ASSEMBLYAI_STT_MODELS: readonly SttModelInfo[] = [
|
|
33
|
+
{
|
|
34
|
+
id: 'universal',
|
|
35
|
+
label: 'Universal',
|
|
36
|
+
description: 'Default tier. Balanced accuracy + speed, 99 languages.',
|
|
37
|
+
supportsLanguageHint: true,
|
|
38
|
+
supportsTimestamps: true,
|
|
39
|
+
},
|
|
40
|
+
{
|
|
41
|
+
id: 'best',
|
|
42
|
+
label: 'Best',
|
|
43
|
+
description: 'Highest-accuracy tier. Slower + pricier than universal; good for hard audio.',
|
|
44
|
+
supportsLanguageHint: true,
|
|
45
|
+
supportsTimestamps: true,
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
id: 'nano',
|
|
49
|
+
label: 'Nano',
|
|
50
|
+
description: 'Cheap, fast, lower accuracy. Use for clean audio at scale.',
|
|
51
|
+
supportsLanguageHint: true,
|
|
52
|
+
supportsTimestamps: true,
|
|
53
|
+
},
|
|
54
|
+
] as const;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* GitHub Copilot static catalog.
|
|
3
|
+
*
|
|
4
|
+
* Copilot exposes an OpenAI-compatible `/chat/completions` endpoint at
|
|
5
|
+
* `api.githubcopilot.com` that fronts a rotating roster of frontier models
|
|
6
|
+
* (GPT, Claude, Gemini, o-series) under one subscription. The adapter at
|
|
7
|
+
* `../adapters/copilot-chat.ts` reuses the shared openai-compat helpers and
|
|
8
|
+
* adds Copilot's token exchange + editor headers (see `copilot-auth.ts`).
|
|
9
|
+
*
|
|
10
|
+
* Auth: the worker's "API key" is a GitHub OAuth token (the Copilot device-flow
|
|
11
|
+
* token, `gho_…`); the adapter exchanges it for a short-lived Copilot token.
|
|
12
|
+
*
|
|
13
|
+
* The roster changes often and is gated by the account's Copilot plan, so live
|
|
14
|
+
* discovery (`GET /models`) is authoritative — this list is just what the model
|
|
15
|
+
* dropdown shows before discovery returns. Reasoning models carry the
|
|
16
|
+
* 'reasoning' capability; the adapter requests reasoning via `reasoning_effort`
|
|
17
|
+
* when the thinking budget is set. Prices are omitted (Copilot bills by
|
|
18
|
+
* subscription / request quota, not per-token).
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import type { ChatModelInfo } from '../adapters/types';
|
|
22
|
+
|
|
23
|
+
export const COPILOT_BASE_URL = 'https://api.githubcopilot.com';
|
|
24
|
+
|
|
25
|
+
export const COPILOT_CHAT_MODELS: readonly ChatModelInfo[] = [
|
|
26
|
+
{
|
|
27
|
+
id: 'gpt-5',
|
|
28
|
+
label: 'GPT-5 (Copilot)',
|
|
29
|
+
description:
|
|
30
|
+
'OpenAI GPT-5 via GitHub Copilot. Reasoning model — depth set by reasoning_effort.',
|
|
31
|
+
contextTokens: 264_000,
|
|
32
|
+
capabilities: ['reasoning', 'function_calling', 'vision'],
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
id: 'gpt-5-mini',
|
|
36
|
+
label: 'GPT-5 mini (Copilot)',
|
|
37
|
+
description: 'Faster, cheaper GPT-5 tier via Copilot. Reasoning-capable.',
|
|
38
|
+
contextTokens: 264_000,
|
|
39
|
+
capabilities: ['reasoning', 'function_calling'],
|
|
40
|
+
},
|
|
41
|
+
{
|
|
42
|
+
id: 'claude-sonnet-4.5',
|
|
43
|
+
label: 'Claude Sonnet 4.5 (Copilot)',
|
|
44
|
+
description:
|
|
45
|
+
'Anthropic Claude Sonnet 4.5 via Copilot. Strong agentic + tool use, reasoning-capable.',
|
|
46
|
+
contextTokens: 200_000,
|
|
47
|
+
capabilities: ['reasoning', 'function_calling', 'vision'],
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
id: 'o4-mini',
|
|
51
|
+
label: 'o4-mini (Copilot)',
|
|
52
|
+
description: 'OpenAI o4-mini reasoning model via Copilot. Fast reasoning for tool-heavy work.',
|
|
53
|
+
contextTokens: 200_000,
|
|
54
|
+
capabilities: ['reasoning', 'function_calling'],
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
id: 'gemini-2.5-pro',
|
|
58
|
+
label: 'Gemini 2.5 Pro (Copilot)',
|
|
59
|
+
description: 'Google Gemini 2.5 Pro via Copilot. Large context, reasoning-capable.',
|
|
60
|
+
contextTokens: 1_000_000,
|
|
61
|
+
capabilities: ['reasoning', 'function_calling', 'vision'],
|
|
62
|
+
},
|
|
63
|
+
];
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deepgram static catalog.
|
|
3
|
+
*
|
|
4
|
+
* Deepgram's transcription API is shaped quite differently from
|
|
5
|
+
* OpenAI's:
|
|
6
|
+
* - Endpoint: POST https://api.deepgram.com/v1/listen
|
|
7
|
+
* - Auth: `Authorization: Token <api-key>` (not Bearer)
|
|
8
|
+
* - Body: raw audio bytes (NOT multipart) — Content-Type header
|
|
9
|
+
* carries the mime so Deepgram knows the codec.
|
|
10
|
+
* - Model + language + features go as URL query params, not body
|
|
11
|
+
* fields. E.g. `?model=nova-3&language=en&smart_format=true`.
|
|
12
|
+
*
|
|
13
|
+
* Models below mirror the documented "general" line as of May 2026.
|
|
14
|
+
* Nova-3 is the current flagship. Older nova-2 is kept around for
|
|
15
|
+
* cost-sensitive callers; enhanced/base are legacy tiers some
|
|
16
|
+
* accounts still default to.
|
|
17
|
+
*
|
|
18
|
+
* Discovery: Deepgram has a `/v1/projects/{id}/models` endpoint, but
|
|
19
|
+
* it requires the project id alongside the API key — and the project
|
|
20
|
+
* id isn't carried by api_keys today (just the key string). We skip
|
|
21
|
+
* live discovery and ship the static catalog. Operators can type a
|
|
22
|
+
* custom model id in the form if a new variant ships before the
|
|
23
|
+
* catalog is updated.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import type { SttModelInfo } from '../catalog';
|
|
27
|
+
|
|
28
|
+
export const DEEPGRAM_BASE_URL = 'https://api.deepgram.com';
|
|
29
|
+
|
|
30
|
+
export const DEEPGRAM_STT_MODELS: readonly SttModelInfo[] = [
|
|
31
|
+
{
|
|
32
|
+
id: 'nova-3',
|
|
33
|
+
label: 'Nova 3',
|
|
34
|
+
description:
|
|
35
|
+
'Current flagship. Best accuracy across 36 languages, lowest latency in the lineup.',
|
|
36
|
+
supportsLanguageHint: true,
|
|
37
|
+
supportsTimestamps: true,
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
id: 'nova-2',
|
|
41
|
+
label: 'Nova 2',
|
|
42
|
+
description:
|
|
43
|
+
'Previous-gen flagship. Slightly lower accuracy than Nova 3 but cheaper per minute.',
|
|
44
|
+
supportsLanguageHint: true,
|
|
45
|
+
supportsTimestamps: true,
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
id: 'enhanced',
|
|
49
|
+
label: 'Enhanced',
|
|
50
|
+
description: 'Legacy tier. Use Nova 3 unless you have a contract reason to pin this.',
|
|
51
|
+
supportsLanguageHint: true,
|
|
52
|
+
supportsTimestamps: true,
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
id: 'base',
|
|
56
|
+
label: 'Base',
|
|
57
|
+
description: 'Cheapest tier; lowest accuracy. Fine for clean studio audio.',
|
|
58
|
+
supportsLanguageHint: true,
|
|
59
|
+
supportsTimestamps: true,
|
|
60
|
+
},
|
|
61
|
+
] as const;
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DeepSeek static catalog.
|
|
3
|
+
*
|
|
4
|
+
* DeepSeek's API is OpenAI-compatible — same `/chat/completions` shape,
|
|
5
|
+
* snake_case fields, tool_calls + tool_call_id, image_url for vision.
|
|
6
|
+
* The adapter at `../adapters/deepseek-chat.ts` reuses the shared
|
|
7
|
+
* `toOpenAICompatMessages` / `extractOpenAICompatToolCalls` helpers
|
|
8
|
+
* from openai-compat.ts.
|
|
9
|
+
*
|
|
10
|
+
* **One non-standard quirk worth flagging here:** DeepSeek surfaces
|
|
11
|
+
* prompt-cache hits as TOP-LEVEL `usage.prompt_cache_hit_tokens` and
|
|
12
|
+
* `usage.prompt_cache_miss_tokens` fields — NOT the OpenAI-compat
|
|
13
|
+
* `usage.prompt_tokens_details.cached_tokens` shape that xAI / HF
|
|
14
|
+
* sub-providers use. The adapter handles this in its own usage
|
|
15
|
+
* extraction; can't reuse a shared helper for this part. Reference:
|
|
16
|
+
* https://api-docs.deepseek.com/guides/kv_cache.
|
|
17
|
+
*
|
|
18
|
+
* Caching is AUTOMATIC: no cache_control markers needed. Prefix
|
|
19
|
+
* matches trigger cache hits server-side, billed at ~2% of the
|
|
20
|
+
* fresh-input rate (an enormous discount vs. Anthropic's ~10%).
|
|
21
|
+
*
|
|
22
|
+
* Maintenance: when DeepSeek deprecates `deepseek-chat` /
|
|
23
|
+
* `deepseek-reasoner` (announced for 2026-07-24), drop them from this
|
|
24
|
+
* list — they're kept here for now so existing workers that picked
|
|
25
|
+
* them keep showing in the dropdown.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
import type { ChatModelInfo } from '../adapters/types';
|
|
29
|
+
|
|
30
|
+
export const DEEPSEEK_BASE_URL = 'https://api.deepseek.com';
|
|
31
|
+
|
|
32
|
+
export const DEEPSEEK_CHAT_MODELS: readonly ChatModelInfo[] = [
|
|
33
|
+
// ── Current generation (V4) ──────────────────────────────────────
|
|
34
|
+
{
|
|
35
|
+
id: 'deepseek-v4-pro',
|
|
36
|
+
label: 'DeepSeek V4 Pro',
|
|
37
|
+
description:
|
|
38
|
+
'DeepSeek flagship. Strong reasoning + tool use, 1M context. ' +
|
|
39
|
+
'Automatic prompt caching surfaces ~2% cache-hit rate (extreme ' +
|
|
40
|
+
'savings on re-sent prefixes). Promo pricing through 2026-05-31.',
|
|
41
|
+
contextTokens: 1_000_000,
|
|
42
|
+
capabilities: ['reasoning', 'function_calling'],
|
|
43
|
+
inputPricePer1M: 0.435,
|
|
44
|
+
outputPricePer1M: 0.87,
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
id: 'deepseek-v4-flash',
|
|
48
|
+
label: 'DeepSeek V4 Flash',
|
|
49
|
+
description:
|
|
50
|
+
'DeepSeek fast/cheap. 1M context, automatic prompt caching. ' +
|
|
51
|
+
'Bargain pricing — fits well for high-volume extractor / ' +
|
|
52
|
+
'summarizer workloads.',
|
|
53
|
+
contextTokens: 1_000_000,
|
|
54
|
+
capabilities: ['function_calling'],
|
|
55
|
+
inputPricePer1M: 0.14,
|
|
56
|
+
outputPricePer1M: 0.28,
|
|
57
|
+
},
|
|
58
|
+
// ── Legacy aliases (deprecated 2026-07-24) ───────────────────────
|
|
59
|
+
// Kept so existing workers configured for these slugs keep working
|
|
60
|
+
// until the deprecation date. New workers should pick a V4 model
|
|
61
|
+
// above instead.
|
|
62
|
+
{
|
|
63
|
+
id: 'deepseek-chat',
|
|
64
|
+
label: 'DeepSeek Chat (legacy)',
|
|
65
|
+
description:
|
|
66
|
+
'Legacy alias for deepseek-v4-flash non-thinking mode. ' +
|
|
67
|
+
'Deprecated 2026-07-24 — migrate to deepseek-v4-flash.',
|
|
68
|
+
contextTokens: 1_000_000,
|
|
69
|
+
capabilities: ['function_calling'],
|
|
70
|
+
inputPricePer1M: 0.14,
|
|
71
|
+
outputPricePer1M: 0.28,
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
id: 'deepseek-reasoner',
|
|
75
|
+
label: 'DeepSeek Reasoner (legacy)',
|
|
76
|
+
description:
|
|
77
|
+
'Legacy alias for deepseek-v4-flash thinking mode. ' +
|
|
78
|
+
'Deprecated 2026-07-24 — migrate to deepseek-v4-pro for ' +
|
|
79
|
+
'reasoning workloads.',
|
|
80
|
+
contextTokens: 1_000_000,
|
|
81
|
+
capabilities: ['reasoning', 'function_calling'],
|
|
82
|
+
inputPricePer1M: 0.14,
|
|
83
|
+
outputPricePer1M: 0.28,
|
|
84
|
+
},
|
|
85
|
+
];
|