@crossworks/voice-client 0.230.43

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/catalog.ts ADDED
@@ -0,0 +1,237 @@
1
+ /**
2
+ * Catalog of TTS / STT models + voices, per provider.
3
+ *
4
+ * Why a static catalog at all: OpenAI does NOT expose a programmatic
5
+ * endpoint to list voices. `/v1/audio/voices` doesn't exist; the voice
6
+ * list is documentation-only and changes when OpenAI ships new models.
7
+ * We keep the mapping in code so the UI can render a smart dropdown
8
+ * (model selected → voices for that model appear) without each page
9
+ * load having to scrape docs.
10
+ *
11
+ * What IS queryable: `/v1/models` returns the model ids the key has
12
+ * access to (alongside chat/embedding models). We cross-reference our
13
+ * catalog with that list to show ONLY the models the user can
14
+ * actually use — accounts on the free tier, or older keys, don't
15
+ * always have every model available.
16
+ *
17
+ * Maintenance: when OpenAI releases a new TTS model, add it here. The
18
+ * UI doesn't need a code change as long as the catalog is current.
19
+ *
20
+ * Other providers (ElevenLabs, Deepgram) get their own catalog entries
21
+ * when we implement them. ElevenLabs unlike OpenAI HAS a /v1/voices
22
+ * endpoint that returns the full list (including user-cloned voices),
23
+ * so for that provider we'd skip the static catalog and query live.
24
+ */
25
+
26
+ import type { TtsVoice } from './types';
27
+
28
+ /** Every voice OpenAI has ever shipped (across all TTS models). Kept
29
+ * as the source-of-truth string union so other modules don't have to
30
+ * reconcile names. */
31
+ export const ALL_OPENAI_VOICES = [
32
+ // Original 6 (tts-1, tts-1-hd).
33
+ 'alloy',
34
+ 'echo',
35
+ 'fable',
36
+ 'nova',
37
+ 'onyx',
38
+ 'shimmer',
39
+ // Added with later expansions — usable by tts-1, tts-1-hd, gpt-4o-mini-tts.
40
+ 'ash',
41
+ 'coral',
42
+ 'sage',
43
+ // Newer voices — only on gpt-4o-mini-tts.
44
+ 'ballad',
45
+ 'verse',
46
+ // "Best quality" voices — gpt-4o-mini-tts only.
47
+ 'marin',
48
+ 'cedar',
49
+ ] as const;
50
+
51
+ export type OpenAiVoice = (typeof ALL_OPENAI_VOICES)[number];
52
+
53
+ /** Short, human-readable description per voice. Used in the UI
54
+ * dropdown so the operator doesn't have to test all 13 to find the
55
+ * warm female voice. Pulled from OpenAI's published descriptions. */
56
+ export const VOICE_DESCRIPTIONS: Record<OpenAiVoice, string> = {
57
+ alloy: 'neutral, balanced',
58
+ ash: 'warm, expressive',
59
+ ballad: 'reflective, narrative',
60
+ cedar: 'high-quality, natural (recommended)',
61
+ coral: 'warm, friendly female',
62
+ echo: 'male, calm',
63
+ fable: 'British, warm',
64
+ marin: 'high-quality, natural (recommended)',
65
+ nova: 'warm, female (Saskia default)',
66
+ onyx: 'deep male, grounded',
67
+ sage: 'measured, thoughtful',
68
+ shimmer: 'soft, female',
69
+ verse: 'expressive, emotive',
70
+ };
71
+
72
+ /** A model entry: what it is, which voices it supports, and which
73
+ * feature flags are on. */
74
+ export type TtsModelInfo = {
75
+ id: string;
76
+ label: string;
77
+ description: string;
78
+ /** Voice ids the model accepts. OpenAI ships a narrow named union
79
+ * (alloy/nova/shimmer/…); xAI ships their own (eve/ara/rex/sal/leo);
80
+ * Gemini ships 30 (Kore/Puck/Zephyr/…); ElevenLabs ships UUIDs and
81
+ * per-account clones. We type this as `readonly string[]` so each
82
+ * provider's adapter can return its own list without union-widening
83
+ * the framework. Consumers treating these as `OpenAiVoice` for
84
+ * type-narrowing have always known to runtime-check anyway. */
85
+ voices: readonly string[];
86
+ /** Whether the model accepts a free-form `instructions` parameter for
87
+ * style steering ("speak warmly", "be calm"). Only true for
88
+ * gpt-4o-mini-tts at the moment. */
89
+ supportsInstructions: boolean;
90
+ /** Tier hint for cost display. */
91
+ tier: 'low-latency' | 'high-quality' | 'steerable';
92
+ };
93
+
94
+ /** TTS catalog. Order matters — list shown to users in this order in
95
+ * the dropdown. */
96
+ export const OPENAI_TTS_MODELS: readonly TtsModelInfo[] = [
97
+ {
98
+ id: 'gpt-4o-mini-tts',
99
+ label: 'gpt-4o-mini-tts',
100
+ description:
101
+ 'Newest TTS model. 13 voices, accepts style instructions ("speak warmly"). Recommended.',
102
+ voices: [
103
+ 'alloy',
104
+ 'ash',
105
+ 'ballad',
106
+ 'coral',
107
+ 'echo',
108
+ 'fable',
109
+ 'nova',
110
+ 'onyx',
111
+ 'sage',
112
+ 'shimmer',
113
+ 'verse',
114
+ 'marin',
115
+ 'cedar',
116
+ ],
117
+ supportsInstructions: true,
118
+ tier: 'steerable',
119
+ },
120
+ {
121
+ id: 'tts-1',
122
+ label: 'tts-1',
123
+ description: 'Original TTS model. 9 voices, low latency. Cheaper than gpt-4o-mini-tts.',
124
+ voices: ['alloy', 'ash', 'coral', 'echo', 'fable', 'nova', 'onyx', 'sage', 'shimmer'],
125
+ supportsInstructions: false,
126
+ tier: 'low-latency',
127
+ },
128
+ {
129
+ id: 'tts-1-hd',
130
+ label: 'tts-1-hd',
131
+ description: 'Higher-quality variant of tts-1. Same 9 voices, higher fidelity, ~2× cost.',
132
+ voices: ['alloy', 'ash', 'coral', 'echo', 'fable', 'nova', 'onyx', 'sage', 'shimmer'],
133
+ supportsInstructions: false,
134
+ tier: 'high-quality',
135
+ },
136
+ ] as const;
137
+
138
+ export type SttModelInfo = {
139
+ id: string;
140
+ label: string;
141
+ description: string;
142
+ /** Whether this model accepts the `language` hint param. */
143
+ supportsLanguageHint: boolean;
144
+ /** Whether this model returns word-level timestamps. */
145
+ supportsTimestamps: boolean;
146
+ };
147
+
148
+ /** STT catalog. As of May 2026 OpenAI ships whisper-1 alongside the
149
+ * newer gpt-4o-mini-transcribe / gpt-4o-transcribe variants which
150
+ * are higher-quality but cost more. */
151
+ export const OPENAI_STT_MODELS: readonly SttModelInfo[] = [
152
+ {
153
+ id: 'whisper-1',
154
+ label: 'whisper-1',
155
+ description: 'Stable, cheap. Excellent multilingual support including Afrikaans.',
156
+ supportsLanguageHint: true,
157
+ supportsTimestamps: true,
158
+ },
159
+ {
160
+ id: 'gpt-4o-mini-transcribe',
161
+ label: 'gpt-4o-mini-transcribe',
162
+ description: 'Newer transcription model. Better accuracy than whisper-1, similar price.',
163
+ supportsLanguageHint: true,
164
+ supportsTimestamps: false,
165
+ },
166
+ {
167
+ id: 'gpt-4o-transcribe',
168
+ label: 'gpt-4o-transcribe',
169
+ description: 'Highest-accuracy transcription. Costs more; best for difficult audio.',
170
+ supportsLanguageHint: true,
171
+ supportsTimestamps: false,
172
+ },
173
+ ] as const;
174
+
175
+ /** Look up a TTS model by id. Returns null if we don't know about it
176
+ * (e.g. user typed a custom model name in the form). Callers should
177
+ * fall back to a permissive default voice list in that case. */
178
+ export function getTtsModel(id: string): TtsModelInfo | null {
179
+ return OPENAI_TTS_MODELS.find((m) => m.id === id) ?? null;
180
+ }
181
+
182
+ export function getSttModel(id: string): SttModelInfo | null {
183
+ return OPENAI_STT_MODELS.find((m) => m.id === id) ?? null;
184
+ }
185
+
186
+ /** Type-narrowing helper: is this voice valid for OpenAI? Used by the
187
+ * form validator and the synth call. */
188
+ export function isOpenAiVoice(v: string): v is OpenAiVoice {
189
+ return (ALL_OPENAI_VOICES as readonly string[]).includes(v);
190
+ }
191
+
192
+ /** All voices, with descriptions, for a given model. Returned in a
193
+ * stable order matching the catalog. If the model isn't in the
194
+ * catalog, returns an empty array (caller decides whether to fall
195
+ * back to the legacy 6-voice list from @mantle/voice/types). */
196
+ export function voicesForModel(modelId: string): Array<{ id: OpenAiVoice; description: string }> {
197
+ const model = getTtsModel(modelId);
198
+ if (!model) return [];
199
+ // OPENAI_TTS_MODELS only lists OpenAI voice ids by construction —
200
+ // the runtime cast is safe because we narrow via getTtsModel which
201
+ // only returns OpenAI catalog entries.
202
+ return model.voices.map((v) => ({
203
+ id: v as OpenAiVoice,
204
+ description: VOICE_DESCRIPTIONS[v as OpenAiVoice] ?? '',
205
+ }));
206
+ }
207
+
208
+ // Bridge to the existing TtsVoice union in types.ts — keeps backward
209
+ // compat with code already using that narrower type. New code should
210
+ // prefer OpenAiVoice for forward-compat with the expanded voice set.
211
+ void ([] as TtsVoice[]);
212
+
213
+ export type DiscoveryResult<T> = {
214
+ /** The catalog entries the key can actually use, in catalog order. */
215
+ available: T[];
216
+ /** True if the live filter succeeded; false if we fell back to the
217
+ * full catalog because the API call failed. The UI surfaces a hint
218
+ * in that case ("couldn't verify; showing everything"). */
219
+ filtered: boolean;
220
+ /** When `filtered=false`, the reason. Null on success. */
221
+ error: string | null;
222
+ /**
223
+ * EVERY model id the provider reported, before we intersected it with our
224
+ * catalog. Optional: only the adapters that curate a static catalog and
225
+ * filter it need to set this.
226
+ *
227
+ * It exists because `available` answers "which of OUR models are live?" and
228
+ * deliberately discards the other half of the answer — "which of THEIRS
229
+ * aren't ours?" That discarded half is the only automatic signal that a
230
+ * catalog has gone stale, and we were computing it on every discovery call
231
+ * and dropping it. grok-4.5 shipped, our dropdown never mentioned it, and
232
+ * nothing anywhere could have said so.
233
+ *
234
+ * Consumed by `pnpm -C server/web models:drift`. See {@link catalogDrift}.
235
+ */
236
+ liveIds?: string[];
237
+ };
@@ -0,0 +1,135 @@
1
+ /**
2
+ * Anthropic (Claude) static catalog.
3
+ *
4
+ * Anthropic's API is NOT OpenAI-compatible. It uses its own message
5
+ * format (system is a separate top-level field, messages contain only
6
+ * user/assistant roles) at `https://api.anthropic.com/v1/messages`.
7
+ * The adapter handles translation to/from the unified ChatDispatcher
8
+ * interface so callers don't have to care.
9
+ *
10
+ * Auth: `x-api-key` header + a required `anthropic-version` header
11
+ * (we pin to 2023-06-01 since that's the stable API). The Models API
12
+ * at GET /v1/messages/../v1/models returns the live model list with
13
+ * capabilities — we use it for discovery but the catalog below is
14
+ * the source of truth for rich UI metadata.
15
+ *
16
+ * Maintenance: when Anthropic ships a new Claude generation (e.g.
17
+ * 4.8 or 5.0), add it here. Anthropic uses dateless aliases for
18
+ * 4.6+ (no more 'claude-3-5-sonnet-20241022' style), so model ids
19
+ * stay readable.
20
+ */
21
+
22
+ import type { ChatModelInfo, VisionModelInfo } from '../adapters/types';
23
+
24
+ export const ANTHROPIC_BASE_URL = 'https://api.anthropic.com';
25
+ /** Required header value on all Anthropic API calls. */
26
+ export const ANTHROPIC_API_VERSION = '2023-06-01';
27
+
28
+ export const ANTHROPIC_CHAT_MODELS: readonly ChatModelInfo[] = [
29
+ // ── Current generation (4.6/4.7) ─────────────────────────────────
30
+ {
31
+ id: 'claude-opus-4-7',
32
+ label: 'Claude Opus 4.7',
33
+ description: 'Anthropic flagship. Best for complex reasoning + agentic coding. 1M context.',
34
+ contextTokens: 1_000_000,
35
+ capabilities: ['vision', 'reasoning', 'function_calling'],
36
+ inputPricePer1M: 5,
37
+ outputPricePer1M: 25,
38
+ },
39
+ {
40
+ id: 'claude-sonnet-5',
41
+ label: 'Claude Sonnet 5',
42
+ description:
43
+ 'Best speed/intelligence balance — newer and cheaper than 4.6. 1M context, supports extended thinking. Default choice.',
44
+ contextTokens: 1_000_000,
45
+ capabilities: ['vision', 'reasoning', 'function_calling'],
46
+ inputPricePer1M: 2,
47
+ outputPricePer1M: 10,
48
+ },
49
+ {
50
+ id: 'claude-sonnet-4-6',
51
+ label: 'Claude Sonnet 4.6',
52
+ description: 'Previous-generation Sonnet. 1M context, supports extended thinking.',
53
+ contextTokens: 1_000_000,
54
+ capabilities: ['vision', 'reasoning', 'function_calling'],
55
+ inputPricePer1M: 3,
56
+ outputPricePer1M: 15,
57
+ },
58
+ {
59
+ id: 'claude-haiku-4-5',
60
+ label: 'Claude Haiku 4.5',
61
+ description:
62
+ 'Fastest model with near-frontier intelligence. 200k context. Great for cheap, fast jobs.',
63
+ contextTokens: 200_000,
64
+ capabilities: ['vision', 'reasoning', 'function_calling'],
65
+ inputPricePer1M: 1,
66
+ outputPricePer1M: 5,
67
+ },
68
+
69
+ // ── Legacy (still available, slightly cheaper or different trade-offs) ──
70
+ {
71
+ id: 'claude-opus-4-6',
72
+ label: 'Claude Opus 4.6 (legacy)',
73
+ description: 'Previous Opus generation. Migrate to 4.7 for best results.',
74
+ contextTokens: 1_000_000,
75
+ capabilities: ['vision', 'reasoning', 'function_calling'],
76
+ inputPricePer1M: 5,
77
+ outputPricePer1M: 25,
78
+ },
79
+ {
80
+ id: 'claude-sonnet-4-5',
81
+ label: 'Claude Sonnet 4.5 (legacy)',
82
+ description: 'Previous Sonnet generation. Still capable; 200k context.',
83
+ contextTokens: 200_000,
84
+ capabilities: ['vision', 'reasoning', 'function_calling'],
85
+ inputPricePer1M: 3,
86
+ outputPricePer1M: 15,
87
+ },
88
+ ];
89
+
90
+ /** Vision-capable Claude models. Every current Claude model accepts
91
+ * image content blocks, but we surface a smaller curated set for the
92
+ * vision-worker dropdown — Haiku for cheap high-volume OCR, Sonnet
93
+ * for balanced quality, Opus when accuracy on hard handwriting
94
+ * matters more than cost. Opus is over-the-top for printed text. */
95
+ export const ANTHROPIC_VISION_MODELS: readonly VisionModelInfo[] = [
96
+ {
97
+ id: 'claude-haiku-4-5',
98
+ label: 'Claude Haiku 4.5',
99
+ description:
100
+ 'Cheap, fast vision. Solid on printed text and clean handwriting. Default choice for high-volume OCR.',
101
+ contextTokens: 200_000,
102
+ inputPricePer1M: 1,
103
+ outputPricePer1M: 5,
104
+ tier: 'fast',
105
+ },
106
+ {
107
+ id: 'claude-sonnet-5',
108
+ label: 'Claude Sonnet 5',
109
+ description:
110
+ 'Best balance. Strong on messy handwriting + multi-column layouts. Recommended for note photos.',
111
+ contextTokens: 1_000_000,
112
+ inputPricePer1M: 2,
113
+ outputPricePer1M: 10,
114
+ tier: 'balanced',
115
+ },
116
+ {
117
+ id: 'claude-sonnet-4-6',
118
+ label: 'Claude Sonnet 4.6',
119
+ description: 'Previous-generation Sonnet vision. Still strong on documents.',
120
+ contextTokens: 1_000_000,
121
+ inputPricePer1M: 3,
122
+ outputPricePer1M: 15,
123
+ tier: 'balanced',
124
+ },
125
+ {
126
+ id: 'claude-opus-4-7',
127
+ label: 'Claude Opus 4.7',
128
+ description:
129
+ 'Frontier. Use when Sonnet misreads cursive or you need diagram comprehension alongside text. Pricey.',
130
+ contextTokens: 1_000_000,
131
+ inputPricePer1M: 5,
132
+ outputPricePer1M: 25,
133
+ tier: 'quality',
134
+ },
135
+ ];
@@ -0,0 +1,54 @@
1
+ /**
2
+ * AssemblyAI static catalog.
3
+ *
4
+ * AssemblyAI's transcription API is a two-step async job:
5
+ * 1. POST /v2/upload — binary body, returns a temporary upload URL.
6
+ * 2. POST /v2/transcript with `{audio_url, speech_model, language_code}`
7
+ * — returns a transcript id with status='queued' or 'processing'.
8
+ * 3. Poll GET /v2/transcript/{id} until status='completed' (or 'error').
9
+ *
10
+ * Because of step 3, AssemblyAI is unsuitable for ultra-low-latency
11
+ * use cases — the round trip for a short voice note is 2-5 seconds
12
+ * even when the model itself is fast. We use it when the operator
13
+ * specifically wants diarization or sentiment, which AssemblyAI
14
+ * exposes alongside the base transcript.
15
+ *
16
+ * Speech models (May 2026): the "universal" tier is the current
17
+ * default; "best" and "nano" are documented but billing tiers vary by
18
+ * account. We surface the documented values; the adapter passes the
19
+ * id through to AssemblyAI as `speech_model` in the request body.
20
+ */
21
+
22
+ import type { SttModelInfo } from '../catalog';
23
+
24
+ export const ASSEMBLYAI_BASE_URL = 'https://api.assemblyai.com';
25
+
26
+ /** Hard cap on polling — if the transcript isn't done after this many
27
+ * seconds, the adapter gives up rather than blocking the test action
28
+ * forever. AssemblyAI's docs say 95% of jobs finish in under 35s for
29
+ * audio under 5 minutes, so 60s is a generous upper bound. */
30
+ export const ASSEMBLYAI_POLL_TIMEOUT_SECONDS = 60;
31
+
32
+ export const ASSEMBLYAI_STT_MODELS: readonly SttModelInfo[] = [
33
+ {
34
+ id: 'universal',
35
+ label: 'Universal',
36
+ description: 'Default tier. Balanced accuracy + speed, 99 languages.',
37
+ supportsLanguageHint: true,
38
+ supportsTimestamps: true,
39
+ },
40
+ {
41
+ id: 'best',
42
+ label: 'Best',
43
+ description: 'Highest-accuracy tier. Slower + pricier than universal; good for hard audio.',
44
+ supportsLanguageHint: true,
45
+ supportsTimestamps: true,
46
+ },
47
+ {
48
+ id: 'nano',
49
+ label: 'Nano',
50
+ description: 'Cheap, fast, lower accuracy. Use for clean audio at scale.',
51
+ supportsLanguageHint: true,
52
+ supportsTimestamps: true,
53
+ },
54
+ ] as const;
@@ -0,0 +1,63 @@
1
+ /**
2
+ * GitHub Copilot static catalog.
3
+ *
4
+ * Copilot exposes an OpenAI-compatible `/chat/completions` endpoint at
5
+ * `api.githubcopilot.com` that fronts a rotating roster of frontier models
6
+ * (GPT, Claude, Gemini, o-series) under one subscription. The adapter at
7
+ * `../adapters/copilot-chat.ts` reuses the shared openai-compat helpers and
8
+ * adds Copilot's token exchange + editor headers (see `copilot-auth.ts`).
9
+ *
10
+ * Auth: the worker's "API key" is a GitHub OAuth token (the Copilot device-flow
11
+ * token, `gho_…`); the adapter exchanges it for a short-lived Copilot token.
12
+ *
13
+ * The roster changes often and is gated by the account's Copilot plan, so live
14
+ * discovery (`GET /models`) is authoritative — this list is just what the model
15
+ * dropdown shows before discovery returns. Reasoning models carry the
16
+ * 'reasoning' capability; the adapter requests reasoning via `reasoning_effort`
17
+ * when the thinking budget is set. Prices are omitted (Copilot bills by
18
+ * subscription / request quota, not per-token).
19
+ */
20
+
21
+ import type { ChatModelInfo } from '../adapters/types';
22
+
23
+ export const COPILOT_BASE_URL = 'https://api.githubcopilot.com';
24
+
25
+ export const COPILOT_CHAT_MODELS: readonly ChatModelInfo[] = [
26
+ {
27
+ id: 'gpt-5',
28
+ label: 'GPT-5 (Copilot)',
29
+ description:
30
+ 'OpenAI GPT-5 via GitHub Copilot. Reasoning model — depth set by reasoning_effort.',
31
+ contextTokens: 264_000,
32
+ capabilities: ['reasoning', 'function_calling', 'vision'],
33
+ },
34
+ {
35
+ id: 'gpt-5-mini',
36
+ label: 'GPT-5 mini (Copilot)',
37
+ description: 'Faster, cheaper GPT-5 tier via Copilot. Reasoning-capable.',
38
+ contextTokens: 264_000,
39
+ capabilities: ['reasoning', 'function_calling'],
40
+ },
41
+ {
42
+ id: 'claude-sonnet-4.5',
43
+ label: 'Claude Sonnet 4.5 (Copilot)',
44
+ description:
45
+ 'Anthropic Claude Sonnet 4.5 via Copilot. Strong agentic + tool use, reasoning-capable.',
46
+ contextTokens: 200_000,
47
+ capabilities: ['reasoning', 'function_calling', 'vision'],
48
+ },
49
+ {
50
+ id: 'o4-mini',
51
+ label: 'o4-mini (Copilot)',
52
+ description: 'OpenAI o4-mini reasoning model via Copilot. Fast reasoning for tool-heavy work.',
53
+ contextTokens: 200_000,
54
+ capabilities: ['reasoning', 'function_calling'],
55
+ },
56
+ {
57
+ id: 'gemini-2.5-pro',
58
+ label: 'Gemini 2.5 Pro (Copilot)',
59
+ description: 'Google Gemini 2.5 Pro via Copilot. Large context, reasoning-capable.',
60
+ contextTokens: 1_000_000,
61
+ capabilities: ['reasoning', 'function_calling', 'vision'],
62
+ },
63
+ ];
@@ -0,0 +1,61 @@
1
+ /**
2
+ * Deepgram static catalog.
3
+ *
4
+ * Deepgram's transcription API is shaped quite differently from
5
+ * OpenAI's:
6
+ * - Endpoint: POST https://api.deepgram.com/v1/listen
7
+ * - Auth: `Authorization: Token <api-key>` (not Bearer)
8
+ * - Body: raw audio bytes (NOT multipart) — Content-Type header
9
+ * carries the mime so Deepgram knows the codec.
10
+ * - Model + language + features go as URL query params, not body
11
+ * fields. E.g. `?model=nova-3&language=en&smart_format=true`.
12
+ *
13
+ * Models below mirror the documented "general" line as of May 2026.
14
+ * Nova-3 is the current flagship. Older nova-2 is kept around for
15
+ * cost-sensitive callers; enhanced/base are legacy tiers some
16
+ * accounts still default to.
17
+ *
18
+ * Discovery: Deepgram has a `/v1/projects/{id}/models` endpoint, but
19
+ * it requires the project id alongside the API key — and the project
20
+ * id isn't carried by api_keys today (just the key string). We skip
21
+ * live discovery and ship the static catalog. Operators can type a
22
+ * custom model id in the form if a new variant ships before the
23
+ * catalog is updated.
24
+ */
25
+
26
+ import type { SttModelInfo } from '../catalog';
27
+
28
+ export const DEEPGRAM_BASE_URL = 'https://api.deepgram.com';
29
+
30
+ export const DEEPGRAM_STT_MODELS: readonly SttModelInfo[] = [
31
+ {
32
+ id: 'nova-3',
33
+ label: 'Nova 3',
34
+ description:
35
+ 'Current flagship. Best accuracy across 36 languages, lowest latency in the lineup.',
36
+ supportsLanguageHint: true,
37
+ supportsTimestamps: true,
38
+ },
39
+ {
40
+ id: 'nova-2',
41
+ label: 'Nova 2',
42
+ description:
43
+ 'Previous-gen flagship. Slightly lower accuracy than Nova 3 but cheaper per minute.',
44
+ supportsLanguageHint: true,
45
+ supportsTimestamps: true,
46
+ },
47
+ {
48
+ id: 'enhanced',
49
+ label: 'Enhanced',
50
+ description: 'Legacy tier. Use Nova 3 unless you have a contract reason to pin this.',
51
+ supportsLanguageHint: true,
52
+ supportsTimestamps: true,
53
+ },
54
+ {
55
+ id: 'base',
56
+ label: 'Base',
57
+ description: 'Cheapest tier; lowest accuracy. Fine for clean studio audio.',
58
+ supportsLanguageHint: true,
59
+ supportsTimestamps: true,
60
+ },
61
+ ] as const;
@@ -0,0 +1,85 @@
1
+ /**
2
+ * DeepSeek static catalog.
3
+ *
4
+ * DeepSeek's API is OpenAI-compatible — same `/chat/completions` shape,
5
+ * snake_case fields, tool_calls + tool_call_id, image_url for vision.
6
+ * The adapter at `../adapters/deepseek-chat.ts` reuses the shared
7
+ * `toOpenAICompatMessages` / `extractOpenAICompatToolCalls` helpers
8
+ * from openai-compat.ts.
9
+ *
10
+ * **One non-standard quirk worth flagging here:** DeepSeek surfaces
11
+ * prompt-cache hits as TOP-LEVEL `usage.prompt_cache_hit_tokens` and
12
+ * `usage.prompt_cache_miss_tokens` fields — NOT the OpenAI-compat
13
+ * `usage.prompt_tokens_details.cached_tokens` shape that xAI / HF
14
+ * sub-providers use. The adapter handles this in its own usage
15
+ * extraction; can't reuse a shared helper for this part. Reference:
16
+ * https://api-docs.deepseek.com/guides/kv_cache.
17
+ *
18
+ * Caching is AUTOMATIC: no cache_control markers needed. Prefix
19
+ * matches trigger cache hits server-side, billed at ~2% of the
20
+ * fresh-input rate (an enormous discount vs. Anthropic's ~10%).
21
+ *
22
+ * Maintenance: when DeepSeek deprecates `deepseek-chat` /
23
+ * `deepseek-reasoner` (announced for 2026-07-24), drop them from this
24
+ * list — they're kept here for now so existing workers that picked
25
+ * them keep showing in the dropdown.
26
+ */
27
+
28
+ import type { ChatModelInfo } from '../adapters/types';
29
+
30
+ export const DEEPSEEK_BASE_URL = 'https://api.deepseek.com';
31
+
32
+ export const DEEPSEEK_CHAT_MODELS: readonly ChatModelInfo[] = [
33
+ // ── Current generation (V4) ──────────────────────────────────────
34
+ {
35
+ id: 'deepseek-v4-pro',
36
+ label: 'DeepSeek V4 Pro',
37
+ description:
38
+ 'DeepSeek flagship. Strong reasoning + tool use, 1M context. ' +
39
+ 'Automatic prompt caching surfaces ~2% cache-hit rate (extreme ' +
40
+ 'savings on re-sent prefixes). Promo pricing through 2026-05-31.',
41
+ contextTokens: 1_000_000,
42
+ capabilities: ['reasoning', 'function_calling'],
43
+ inputPricePer1M: 0.435,
44
+ outputPricePer1M: 0.87,
45
+ },
46
+ {
47
+ id: 'deepseek-v4-flash',
48
+ label: 'DeepSeek V4 Flash',
49
+ description:
50
+ 'DeepSeek fast/cheap. 1M context, automatic prompt caching. ' +
51
+ 'Bargain pricing — fits well for high-volume extractor / ' +
52
+ 'summarizer workloads.',
53
+ contextTokens: 1_000_000,
54
+ capabilities: ['function_calling'],
55
+ inputPricePer1M: 0.14,
56
+ outputPricePer1M: 0.28,
57
+ },
58
+ // ── Legacy aliases (deprecated 2026-07-24) ───────────────────────
59
+ // Kept so existing workers configured for these slugs keep working
60
+ // until the deprecation date. New workers should pick a V4 model
61
+ // above instead.
62
+ {
63
+ id: 'deepseek-chat',
64
+ label: 'DeepSeek Chat (legacy)',
65
+ description:
66
+ 'Legacy alias for deepseek-v4-flash non-thinking mode. ' +
67
+ 'Deprecated 2026-07-24 — migrate to deepseek-v4-flash.',
68
+ contextTokens: 1_000_000,
69
+ capabilities: ['function_calling'],
70
+ inputPricePer1M: 0.14,
71
+ outputPricePer1M: 0.28,
72
+ },
73
+ {
74
+ id: 'deepseek-reasoner',
75
+ label: 'DeepSeek Reasoner (legacy)',
76
+ description:
77
+ 'Legacy alias for deepseek-v4-flash thinking mode. ' +
78
+ 'Deprecated 2026-07-24 — migrate to deepseek-v4-pro for ' +
79
+ 'reasoning workloads.',
80
+ contextTokens: 1_000_000,
81
+ capabilities: ['reasoning', 'function_calling'],
82
+ inputPricePer1M: 0.14,
83
+ outputPricePer1M: 0.28,
84
+ },
85
+ ];