@crossworks/voice-client 0.230.43

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,244 @@
1
+ /**
2
+ * ElevenLabs static catalog.
3
+ *
4
+ * ElevenLabs's TTS API is shaped very differently from OpenAI:
5
+ * - Voice id lives in the URL: POST /v1/text-to-speech/{voice_id}
6
+ * - Model id is in the request body (`model_id`)
7
+ * - Output format is a QUERY param (`output_format=opus_48000_64`)
8
+ * - Voices include user-cloned ones (returned by /v1/voices)
9
+ *
10
+ * Models below are the publicly-documented voice synthesis models.
11
+ * Discovery via GET /v1/models returns the live list including any
12
+ * preview/beta models the key has access to.
13
+ *
14
+ * Voice strategy: ElevenLabs has hundreds of voices (premade + cloned).
15
+ * The adapter's `voicesForModel` queries `/v1/voices` live to return
16
+ * every voice the key can use — including the user's clones — rather
17
+ * than hardcoding a list. The static fallback below is the small set
18
+ * of "premade" voices everyone gets, used only when discovery fails.
19
+ *
20
+ * Output format mapping for Mantle's Telegram-voice use case:
21
+ * `opus_48000_64` → 'audio/ogg', plays as a Telegram voice-note bubble.
22
+ * Other formats are available for non-Telegram surfaces.
23
+ */
24
+
25
+ import type { AudioTag } from '../adapters/types';
26
+
27
+ export const ELEVENLABS_BASE_URL = 'https://api.elevenlabs.io';
28
+
29
+ /**
30
+ * Documented v3 audio tags. ElevenLabs publishes these in five
31
+ * conceptual buckets (their docs use slightly different names; we
32
+ * normalise to the categories on the AudioTag type). Saskia gets a
33
+ * paragraph in her prompt listing the supported tags when the active
34
+ * TTS worker is configured for ElevenLabs v3.
35
+ *
36
+ * Tags are case-insensitive in the API but we render them lowercase
37
+ * for consistency. Pulled from
38
+ * https://elevenlabs.io/blog/v3-audiotags
39
+ * https://elevenlabs.io/blog/eleven-v3-audio-tags-expressing-emotional-context-in-speech
40
+ *
41
+ * Not exhaustive — ElevenLabs's docs say "many more effective tags
42
+ * beyond the listed examples." This list covers the ones with
43
+ * documented, stable behaviour.
44
+ */
45
+ export const ELEVENLABS_V3_AUDIO_TAGS: readonly AudioTag[] = [
46
+ // Human reactions — the most useful for conversational warmth.
47
+ {
48
+ tag: '[laughs]',
49
+ description: 'a warm chuckle; use for genuine amusement',
50
+ category: 'reaction',
51
+ },
52
+ { tag: '[laughs softly]', description: 'a quiet, intimate chuckle', category: 'reaction' },
53
+ { tag: '[chuckles]', description: 'short, dry amusement', category: 'reaction' },
54
+ { tag: '[snorts]', description: 'a short, derisive or surprised exhale', category: 'reaction' },
55
+ { tag: '[sighs]', description: 'resigned exhale; reflective or weary', category: 'reaction' },
56
+ { tag: '[gasps]', description: 'sharp inhale of surprise', category: 'reaction' },
57
+ {
58
+ tag: '[clears throat]',
59
+ description: 'transitional beat, often before a serious point',
60
+ category: 'reaction',
61
+ },
62
+
63
+ // Delivery / performance.
64
+ {
65
+ tag: '[whispers]',
66
+ description: 'intimate, lowered voice for secrets or asides',
67
+ category: 'delivery',
68
+ },
69
+ { tag: '[shouts]', description: 'raised voice; use sparingly', category: 'delivery' },
70
+
71
+ // Cognitive / pacing beats.
72
+ { tag: '[pauses]', description: 'a deliberate beat of silence', category: 'cognitive' },
73
+ { tag: '[hesitates]', description: 'briefly stalls, as if thinking', category: 'cognitive' },
74
+ { tag: '[stammers]', description: 'broken cadence; nerves or surprise', category: 'cognitive' },
75
+
76
+ // Emotional states (modify the line that follows).
77
+ { tag: '[excited]', description: 'warm, energetic delivery', category: 'emotion' },
78
+ { tag: '[curious]', description: 'rising-inflection, inquisitive', category: 'emotion' },
79
+ { tag: '[happy]', description: 'bright, smiling tone', category: 'emotion' },
80
+ { tag: '[sad]', description: 'slower, weighted', category: 'emotion' },
81
+ { tag: '[nervous]', description: 'slightly faster, breathier', category: 'emotion' },
82
+ { tag: '[frustrated]', description: 'tight, clipped', category: 'emotion' },
83
+ { tag: '[calm]', description: 'steady, unhurried', category: 'emotion' },
84
+ { tag: '[sorrowful]', description: 'deeper sadness; for genuine grief', category: 'emotion' },
85
+ { tag: '[mischievously]', description: 'playful, with a hint of trouble', category: 'emotion' },
86
+ { tag: '[crying]', description: 'distressed delivery; rare', category: 'emotion' },
87
+
88
+ // Tone cues.
89
+ { tag: '[cheerfully]', description: 'lift and brightness throughout', category: 'tone' },
90
+ { tag: '[flatly]', description: 'unaffected, monotone', category: 'tone' },
91
+ { tag: '[deadpan]', description: 'expressionless; great for dry humour', category: 'tone' },
92
+ { tag: '[playfully]', description: 'light, teasing', category: 'tone' },
93
+ {
94
+ tag: '[resigned tone]',
95
+ description: 'accepting the inevitable; soft sigh implicit',
96
+ category: 'tone',
97
+ },
98
+ ];
99
+
100
+ /** TTS model metadata. ElevenLabs swaps the OpenAI {model,voice}
101
+ * separation so each entry represents the TTS engine, not the voice. */
102
+ export type ElevenLabsTtsModel = {
103
+ id: string;
104
+ label: string;
105
+ description: string;
106
+ /** True if the model supports >29 languages. */
107
+ multilingual: boolean;
108
+ /** Approximate output speed tier ('fast' / 'balanced' / 'quality'). */
109
+ speed: 'fast' | 'balanced' | 'quality';
110
+ /** Whether this model honours inline audio tags. Only v3 has the
111
+ * full vocabulary; older models render bracketed tags as literal
112
+ * text (which is why we strip them defensively before send). */
113
+ supportsAudioTags?: boolean;
114
+ };
115
+
116
+ export const ELEVENLABS_TTS_MODELS: readonly ElevenLabsTtsModel[] = [
117
+ {
118
+ id: 'eleven_v3',
119
+ label: 'Eleven v3',
120
+ description:
121
+ 'Newest, highest-quality generation. Honours the full inline audio-tag vocabulary (laughs, whispers, sighs, emotion cues).',
122
+ multilingual: true,
123
+ speed: 'quality',
124
+ supportsAudioTags: true,
125
+ },
126
+ {
127
+ id: 'eleven_multilingual_v2',
128
+ label: 'Multilingual v2',
129
+ description:
130
+ 'Stable default. 29 languages, balanced quality + speed. Inline audio tags NOT honoured — they get rendered as literal text, so the adapter strips them.',
131
+ multilingual: true,
132
+ speed: 'balanced',
133
+ supportsAudioTags: false,
134
+ },
135
+ {
136
+ id: 'eleven_turbo_v2_5',
137
+ label: 'Turbo v2.5',
138
+ description:
139
+ 'Lower latency, 32 languages. Use when speed matters more than perfection. No inline audio tags.',
140
+ multilingual: true,
141
+ speed: 'fast',
142
+ supportsAudioTags: false,
143
+ },
144
+ {
145
+ id: 'eleven_flash_v2_5',
146
+ label: 'Flash v2.5',
147
+ description: 'Lowest latency (~75ms). 32 languages. Best for streaming. No inline audio tags.',
148
+ multilingual: true,
149
+ speed: 'fast',
150
+ supportsAudioTags: false,
151
+ },
152
+ {
153
+ id: 'eleven_monolingual_v1',
154
+ label: 'Monolingual v1',
155
+ description: 'English-only legacy model. Cheap; consider v3 instead.',
156
+ multilingual: false,
157
+ speed: 'balanced',
158
+ supportsAudioTags: false,
159
+ },
160
+ ];
161
+
162
+ /**
163
+ * Audio-tag lookup per model. Returns the documented tag set for
164
+ * models that support them, empty list otherwise. Used by the
165
+ * ElevenLabs adapter's `supportedAudioTags` to gate which tags are
166
+ * advertised to the LLM in Saskia's prompt.
167
+ */
168
+ export function audioTagsForElevenLabsModel(modelId: string): readonly AudioTag[] {
169
+ const m = ELEVENLABS_TTS_MODELS.find((x) => x.id === modelId);
170
+ if (!m || !m.supportsAudioTags) return [];
171
+ // Today only v3 has documented tags. When ElevenLabs publishes a
172
+ // tag set for a future model, branch here on modelId rather than
173
+ // returning the same list for every supporting model.
174
+ return ELEVENLABS_V3_AUDIO_TAGS;
175
+ }
176
+
177
+ // ─── ElevenLabs STT (Scribe) ─────────────────────────────────────────
178
+ //
179
+ // Endpoint: POST {ELEVENLABS_BASE_URL}/v1/speech-to-text
180
+ // Auth: xi-api-key header
181
+ // Body: multipart/form-data — `model_id` + `file` + optional
182
+ // `language_code`, `diarize`, `timestamps_granularity`.
183
+ // Response: { text, language_code, language_probability, words?: [...] }
184
+ //
185
+ // Scribe v1 supports 99 languages, returns word-level timing when
186
+ // `timestamps_granularity=word`. Pricing is per-minute; cheaper than
187
+ // OpenAI Whisper for most language tiers.
188
+
189
+ import type { SttModelInfo } from '../catalog';
190
+
191
+ export const ELEVENLABS_STT_MODELS: readonly SttModelInfo[] = [
192
+ {
193
+ id: 'scribe_v1',
194
+ label: 'Scribe v1',
195
+ description:
196
+ 'ElevenLabs Scribe transcription. 99 languages, word-level timestamps, diarization optional.',
197
+ supportsLanguageHint: true,
198
+ supportsTimestamps: true,
199
+ },
200
+ ] as const;
201
+
202
+ /** Output format query-param values. We pick `opus_48000_64` for
203
+ * Telegram (Telegram-native voice notes are OGG/Opus), but other
204
+ * surfaces may want different containers. */
205
+ export const ELEVENLABS_OUTPUT_FORMATS = [
206
+ 'opus_48000_64',
207
+ 'opus_48000_128',
208
+ 'mp3_44100_128',
209
+ 'mp3_44100_192',
210
+ 'mp3_22050_32',
211
+ 'pcm_16000',
212
+ 'pcm_44100',
213
+ 'wav_44100',
214
+ ] as const;
215
+ export type ElevenLabsOutputFormat = (typeof ELEVENLABS_OUTPUT_FORMATS)[number];
216
+
217
+ /** MIME for a given ElevenLabs output_format. Used when handing the
218
+ * audio bytes off to Telegram or to the browser <audio> element. */
219
+ export function mimeForElevenLabsFormat(format: string): string {
220
+ if (format.startsWith('opus_')) return 'audio/ogg';
221
+ if (format.startsWith('mp3_')) return 'audio/mpeg';
222
+ if (format.startsWith('wav_')) return 'audio/wav';
223
+ if (format.startsWith('pcm_')) return 'audio/pcm';
224
+ if (format.startsWith('ulaw_')) return 'audio/basic';
225
+ if (format.startsWith('alaw_')) return 'audio/basic';
226
+ return 'application/octet-stream';
227
+ }
228
+
229
+ /**
230
+ * Default premade voice ids. Used as a static fallback when the
231
+ * /v1/voices discovery call fails. These ids are stable across
232
+ * accounts — ElevenLabs ships them with every free + paid plan.
233
+ */
234
+ export const ELEVENLABS_PREMADE_VOICES: readonly { id: string; description: string }[] = [
235
+ { id: '21m00Tcm4TlvDq8ikWAM', description: 'Rachel — calm, narrative female' },
236
+ { id: 'AZnzlk1XvdvUeBnXmlld', description: 'Domi — strong, confident female' },
237
+ { id: 'EXAVITQu4vr4xnSDxMaL', description: 'Bella — soft, gentle female' },
238
+ { id: 'ErXwobaYiN019PkySvjV', description: 'Antoni — warm male' },
239
+ { id: 'MF3mGyEYCl7XYWbV9V6O', description: 'Elli — emotional, expressive female' },
240
+ { id: 'TxGEqnHWrfWFTfGW9XjX', description: 'Josh — deep male' },
241
+ { id: 'VR6AewLTigWG4xSOukaG', description: 'Arnold — crisp male' },
242
+ { id: 'pNInz6obpgDQGcFmaJgB', description: 'Adam — narrative male' },
243
+ { id: 'yoZ06aMxZJJ28mfd3POQ', description: 'Sam — neutral male' },
244
+ ];
@@ -0,0 +1,332 @@
1
+ /**
2
+ * Google (Gemini) static catalog.
3
+ *
4
+ * Gemini's API is NOT OpenAI-compatible. It uses `contents` (with
5
+ * `parts`) instead of `messages`, `systemInstruction` as a separate
6
+ * top-level field, and roles 'user' / 'model' (not 'assistant'). The
7
+ * adapter handles the translation.
8
+ *
9
+ * Endpoint: POST /v1beta/models/{model}:generateContent
10
+ * Auth: `x-goog-api-key` header
11
+ * Models endpoint: GET /v1beta/models?key=...
12
+ *
13
+ * Notable Gemini quirks:
14
+ * - Huge context windows (1M-2M tokens) for the 3.x models.
15
+ * - 3.x is preview-tagged but production-stable for most uses.
16
+ * - Gemini also ships TTS and embedding models — we cover chat here;
17
+ * a separate google-tts.ts / google-embed.ts can land later.
18
+ */
19
+
20
+ import type {
21
+ AudioTag,
22
+ ChatModelInfo,
23
+ ImageGenModelInfo,
24
+ VisionModelInfo,
25
+ } from '../adapters/types';
26
+
27
+ export const GOOGLE_BASE_URL = 'https://generativelanguage.googleapis.com/v1beta';
28
+
29
+ // ─── Gemini TTS ──────────────────────────────────────────────────────
30
+ //
31
+ // Endpoint: POST {GOOGLE_BASE_URL}/models/{model}:generateContent
32
+ // Auth: x-goog-api-key header
33
+ // Body: contents (text), generationConfig {
34
+ // responseModalities: ['AUDIO'],
35
+ // speechConfig: { voiceConfig: { prebuiltVoiceConfig: { voiceName } } }
36
+ // }
37
+ // Output: The audio comes back inline as inlineData (base64 PCM).
38
+ //
39
+ // Gemini TTS supports BOTH inline audio tags ([whispers], [laughs])
40
+ // AND natural-language style steering inside the text itself ("Say
41
+ // excitedly: ..."). We expose tags via the adapter framework; the
42
+ // natural-language steering option remains available to operators
43
+ // who put it in the worker's system prompt.
44
+
45
+ /** Gemini TTS model ids. Two variants — Flash for low-latency / cost
46
+ * and Pro for studio-quality. Both are preview-tagged but production-
47
+ * stable for most uses. */
48
+ export const GOOGLE_TTS_MODELS = [
49
+ 'gemini-2.5-flash-preview-tts',
50
+ 'gemini-2.5-pro-preview-tts',
51
+ ] as const;
52
+ export type GoogleTtsModelId = (typeof GOOGLE_TTS_MODELS)[number];
53
+
54
+ /** Gemini publishes 30 prebuilt voices. Names come from Greek/myth
55
+ * references; gender/character notes from the Gemini docs and
56
+ * cookbook samples. Operators see these in the worker form's voice
57
+ * dropdown. */
58
+ export const GOOGLE_TTS_VOICES = [
59
+ // Most-used / recommended.
60
+ { id: 'Kore', description: 'female, balanced — Gemini default' },
61
+ { id: 'Puck', description: 'male, expressive' },
62
+ { id: 'Zephyr', description: 'male, light and airy' },
63
+ { id: 'Charon', description: 'male, grounded and warm' },
64
+ { id: 'Fenrir', description: 'male, deep' },
65
+ { id: 'Leda', description: 'female, soft' },
66
+ { id: 'Aoede', description: 'female, melodic' },
67
+ { id: 'Orus', description: 'male, neutral' },
68
+ // The remaining 22 — left as id-only so the dropdown isn't bloated
69
+ // with guessed descriptions. Live discovery surfaces all 30.
70
+ { id: 'Callirrhoe', description: 'female' },
71
+ { id: 'Autonoe', description: 'female' },
72
+ { id: 'Enceladus', description: 'male' },
73
+ { id: 'Iapetus', description: 'male' },
74
+ { id: 'Umbriel', description: 'male' },
75
+ { id: 'Algieba', description: 'male' },
76
+ { id: 'Despina', description: 'female' },
77
+ { id: 'Erinome', description: 'female' },
78
+ { id: 'Algenib', description: 'male' },
79
+ { id: 'Rasalgethi', description: 'male' },
80
+ { id: 'Laomedeia', description: 'female' },
81
+ { id: 'Achernar', description: 'female' },
82
+ { id: 'Alnilam', description: 'male' },
83
+ { id: 'Schedar', description: 'male' },
84
+ { id: 'Gacrux', description: 'female' },
85
+ { id: 'Pulcherrima', description: 'female' },
86
+ { id: 'Achird', description: 'male' },
87
+ { id: 'Zubenelgenubi', description: 'male' },
88
+ { id: 'Vindemiatrix', description: 'female' },
89
+ { id: 'Sadachbia', description: 'male' },
90
+ { id: 'Sadaltager', description: 'male' },
91
+ { id: 'Sulafat', description: 'female' },
92
+ ] as const;
93
+
94
+ /**
95
+ * Inline audio tags Gemini TTS interprets. The Gemini docs note that
96
+ * tags "like [whispers] or [laughs]" are honoured but don't publish a
97
+ * canonical exhaustive list — they describe a more open vocabulary
98
+ * driven by natural-language understanding. We ship the documented
99
+ * examples plus the well-known ElevenLabs-shaped ones since Gemini
100
+ * tends to understand them too.
101
+ *
102
+ * If a tag in this list doesn't render perfectly on Gemini, fall
103
+ * back to natural-language steering in the worker's system prompt
104
+ * ("Speak softly here:", "She laughs as she says:") — Gemini handles
105
+ * that path well.
106
+ */
107
+ export const GOOGLE_AUDIO_TAGS: readonly AudioTag[] = [
108
+ // Reactions — documented examples in Gemini's docs.
109
+ { tag: '[laughs]', description: 'a warm laugh', category: 'reaction' },
110
+ { tag: '[chuckles]', description: 'short amusement', category: 'reaction' },
111
+ { tag: '[sighs]', description: 'a resigned exhale', category: 'reaction' },
112
+ { tag: '[gasps]', description: 'sharp inhale of surprise', category: 'reaction' },
113
+ { tag: '[clears throat]', description: 'transitional beat', category: 'reaction' },
114
+
115
+ // Delivery — documented.
116
+ { tag: '[whispers]', description: 'intimate lowered voice', category: 'delivery' },
117
+ { tag: '[shouts]', description: 'raised voice; use sparingly', category: 'delivery' },
118
+
119
+ // Cognitive — documented.
120
+ { tag: '[pauses]', description: 'a deliberate beat of silence', category: 'cognitive' },
121
+
122
+ // Emotion / tone — Gemini's natural-language steering means these
123
+ // work reliably in bracket form too.
124
+ { tag: '[excited]', description: 'warm, energetic delivery', category: 'emotion' },
125
+ { tag: '[curious]', description: 'rising-inflection, inquisitive', category: 'emotion' },
126
+ { tag: '[happy]', description: 'bright, smiling tone', category: 'emotion' },
127
+ { tag: '[sad]', description: 'slower, weighted', category: 'emotion' },
128
+ { tag: '[serious]', description: 'measured, weighty', category: 'emotion' },
129
+ { tag: '[calm]', description: 'steady, unhurried', category: 'emotion' },
130
+ { tag: '[playfully]', description: 'light, teasing', category: 'tone' },
131
+ { tag: '[deadpan]', description: 'expressionless; dry humour', category: 'tone' },
132
+ { tag: '[cheerfully]', description: 'lift and brightness', category: 'tone' },
133
+ ];
134
+
135
+ /**
136
+ * Audio-tag lookup per Gemini TTS model. Both Flash and Pro honour
137
+ * the same vocabulary; the difference is fidelity, not steering.
138
+ */
139
+ export function audioTagsForGoogleTtsModel(modelId: string): readonly AudioTag[] {
140
+ if ((GOOGLE_TTS_MODELS as readonly string[]).includes(modelId)) {
141
+ return GOOGLE_AUDIO_TAGS;
142
+ }
143
+ return [];
144
+ }
145
+
146
+ export const GOOGLE_CHAT_MODELS: readonly ChatModelInfo[] = [
147
+ // ── Gemini 3 series (current) ────────────────────────────────────
148
+ {
149
+ id: 'gemini-3.1-pro-preview',
150
+ label: 'Gemini 3.1 Pro (preview)',
151
+ description:
152
+ 'Latest flagship. Advanced reasoning, multimodal, 2M context. Recommended default.',
153
+ contextTokens: 2_000_000,
154
+ capabilities: ['vision', 'reasoning', 'function_calling', 'json_mode'],
155
+ },
156
+ {
157
+ id: 'gemini-3-flash-preview',
158
+ label: 'Gemini 3 Flash (preview)',
159
+ description: 'Frontier-class performance at low cost. Fast multimodal.',
160
+ contextTokens: 1_000_000,
161
+ capabilities: ['vision', 'function_calling', 'json_mode'],
162
+ },
163
+ {
164
+ id: 'gemini-3.1-flash-lite',
165
+ label: 'Gemini 3.1 Flash Lite',
166
+ description: 'Stable Flash-Lite tier. Cheapest in the 3.x family.',
167
+ contextTokens: 1_000_000,
168
+ capabilities: ['vision', 'function_calling'],
169
+ },
170
+
171
+ // ── Gemini 2.5 series (stable, widely available) ─────────────────
172
+ {
173
+ id: 'gemini-2.5-pro',
174
+ label: 'Gemini 2.5 Pro',
175
+ description: 'Stable Pro tier. 2M context, deep reasoning, multimodal.',
176
+ contextTokens: 2_000_000,
177
+ capabilities: ['vision', 'reasoning', 'function_calling', 'json_mode'],
178
+ },
179
+ {
180
+ id: 'gemini-2.5-flash',
181
+ label: 'Gemini 2.5 Flash',
182
+ description: 'Best price/perf in the 2.5 family. Multimodal.',
183
+ contextTokens: 1_000_000,
184
+ capabilities: ['vision', 'function_calling', 'json_mode'],
185
+ },
186
+ {
187
+ id: 'gemini-2.5-flash-lite',
188
+ label: 'Gemini 2.5 Flash Lite',
189
+ description: 'Fastest and most budget-friendly multimodal model.',
190
+ contextTokens: 1_000_000,
191
+ capabilities: ['vision', 'function_calling'],
192
+ },
193
+ ];
194
+
195
+ // ─── Gemini as STT ───────────────────────────────────────────────────
196
+ //
197
+ // Unlike OpenAI / xAI / Deepgram / AssemblyAI which all expose a
198
+ // dedicated `/transcriptions` endpoint, Google ships transcription as
199
+ // "ask Gemini to transcribe this audio." The shape is the same
200
+ // generateContent call used for chat — we pass an inline audio part
201
+ // and a system prompt telling the model to output just the transcript.
202
+ //
203
+ // The catalog here only lists models that actually accept audio input.
204
+ // Older 1.5 / pre-multimodal models don't, and the adapter rejects them
205
+ // with a clear error rather than silently dropping the audio.
206
+
207
+ import type { SttModelInfo } from '../catalog';
208
+
209
+ /** Models that accept audio parts in generateContent and can return a
210
+ * transcript when prompted to. Per Google's docs the multimodal-input
211
+ * surface covers the 3.x and 2.5 lines. */
212
+ export const GOOGLE_STT_MODELS: readonly SttModelInfo[] = [
213
+ {
214
+ id: 'gemini-2.5-flash',
215
+ label: 'Gemini 2.5 Flash',
216
+ description:
217
+ 'Cheapest multimodal model. Recommended default for transcription — Gemini Pro adds little for voice.',
218
+ supportsLanguageHint: true,
219
+ supportsTimestamps: false,
220
+ },
221
+ {
222
+ id: 'gemini-2.5-flash-lite',
223
+ label: 'Gemini 2.5 Flash Lite',
224
+ description: 'Fastest, lowest-cost option for short clips.',
225
+ supportsLanguageHint: true,
226
+ supportsTimestamps: false,
227
+ },
228
+ {
229
+ id: 'gemini-2.5-pro',
230
+ label: 'Gemini 2.5 Pro',
231
+ description: 'Higher accuracy on noisy or accented audio. More expensive.',
232
+ supportsLanguageHint: true,
233
+ supportsTimestamps: false,
234
+ },
235
+ {
236
+ id: 'gemini-3-flash-preview',
237
+ label: 'Gemini 3 Flash (preview)',
238
+ description: 'Latest Flash with stronger multilingual coverage.',
239
+ supportsLanguageHint: true,
240
+ supportsTimestamps: false,
241
+ },
242
+ ] as const;
243
+
244
+ // ─── Gemini Vision ───────────────────────────────────────────────────
245
+ //
246
+ // Every modern Gemini model (2.5+ and 3.x) is multimodal — same
247
+ // endpoint as chat, just an inlineData `image/jpeg` part instead of
248
+ // (or alongside) text. We surface the practical picks: Flash-Lite for
249
+ // the cheap default, Flash for the balanced choice, Pro when an image
250
+ // has dense text or diagrams worth paying more for.
251
+
252
+ export const GOOGLE_VISION_MODELS: readonly VisionModelInfo[] = [
253
+ {
254
+ id: 'gemini-2.5-flash-lite',
255
+ label: 'Gemini 2.5 Flash Lite',
256
+ description: 'Cheapest, fastest vision. Great for clean printed text and bulk receipt OCR.',
257
+ contextTokens: 1_000_000,
258
+ tier: 'fast',
259
+ },
260
+ {
261
+ id: 'gemini-2.5-flash',
262
+ label: 'Gemini 2.5 Flash',
263
+ description:
264
+ 'Best price/perf. Handles handwritten notes and multi-column layouts. Recommended default.',
265
+ contextTokens: 1_000_000,
266
+ tier: 'balanced',
267
+ },
268
+ {
269
+ id: 'gemini-2.5-pro',
270
+ label: 'Gemini 2.5 Pro',
271
+ description: 'Strongest accuracy on hard handwriting + diagrams. 2M context.',
272
+ contextTokens: 2_000_000,
273
+ tier: 'quality',
274
+ },
275
+ {
276
+ id: 'gemini-3-flash-preview',
277
+ label: 'Gemini 3 Flash (preview)',
278
+ description: 'Latest Flash. Higher multilingual fidelity; preview-tagged but production-ok.',
279
+ contextTokens: 1_000_000,
280
+ tier: 'balanced',
281
+ },
282
+ ];
283
+
284
+ // ─── Google Imagen (image generation) ────────────────────────────────
285
+ //
286
+ // Endpoint: POST {GOOGLE_BASE_URL}/models/{model}:predict
287
+ // NOTE: this is a different shape from chat's `generateContent`.
288
+ // Body is { instances: [{prompt}], parameters: { ... } }.
289
+ // Auth: `x-goog-api-key` header (same as the rest of Google).
290
+ // Response: { predictions: [{ bytesBase64Encoded, mimeType }] }
291
+ //
292
+ // Imagen sits behind a separate billing/quota gate from Gemini chat —
293
+ // confirm at console.cloud.google.com that the project has Imagen API
294
+ // enabled before pointing a worker here. Auth failures surface as a
295
+ // confusing 404 ("model not found") rather than 401; the adapter
296
+ // translates these into a clearer hint.
297
+ //
298
+ // Sizes accept Imagen's aspect-ratio strings ('1:1', '16:9', etc.)
299
+ // rather than NxN pixels. The adapter maps the standard 'NNNNxNNNN'
300
+ // format used elsewhere to the closest aspect ratio.
301
+
302
+ export const GOOGLE_IMAGE_MODELS: readonly ImageGenModelInfo[] = [
303
+ {
304
+ id: 'imagen-4.0-generate-001',
305
+ label: 'Imagen 4',
306
+ description: 'Current Imagen flagship. Strong on detail + composition. Recommended.',
307
+ supportedSizes: ['1024x1024', '1408x768', '768x1408'],
308
+ supportedAspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
309
+ pricePerImage: 0.04,
310
+ tier: 'quality',
311
+ },
312
+ {
313
+ id: 'imagen-4.0-fast-generate-001',
314
+ label: 'Imagen 4 Fast',
315
+ description: 'Faster + cheaper Imagen 4 variant. Lower fidelity, higher throughput.',
316
+ supportedSizes: ['1024x1024', '1408x768', '768x1408'],
317
+ supportedAspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
318
+ pricePerImage: 0.02,
319
+ tier: 'fast',
320
+ },
321
+ {
322
+ id: 'imagen-3.0-generate-002',
323
+ label: 'Imagen 3',
324
+ description: 'Previous-generation Imagen. Still capable + widely available.',
325
+ supportedSizes: ['1024x1024', '1408x768', '768x1408'],
326
+ supportedAspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
327
+ pricePerImage: 0.04,
328
+ tier: 'balanced',
329
+ },
330
+ ];
331
+
332
+ export const GOOGLE_IMAGE_DEFAULT_MODEL = 'imagen-4.0-generate-001';