@crossworks/voice-client 0.230.43
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE.md +135 -0
- package/package.json +20 -0
- package/src/adapters/registry.ts +296 -0
- package/src/adapters/retry.ts +193 -0
- package/src/adapters/types.ts +866 -0
- package/src/audio-tags.test.ts +221 -0
- package/src/audio-tags.ts +191 -0
- package/src/catalog.test.ts +144 -0
- package/src/catalog.ts +237 -0
- package/src/catalogs/anthropic.ts +135 -0
- package/src/catalogs/assemblyai.ts +54 -0
- package/src/catalogs/copilot.ts +63 -0
- package/src/catalogs/deepgram.ts +61 -0
- package/src/catalogs/deepseek.ts +85 -0
- package/src/catalogs/elevenlabs.ts +244 -0
- package/src/catalogs/google.ts +332 -0
- package/src/catalogs/huggingface.ts +180 -0
- package/src/catalogs/openai-image.ts +62 -0
- package/src/catalogs/openai-vision.ts +53 -0
- package/src/catalogs/openrouter.ts +221 -0
- package/src/catalogs/xai.ts +330 -0
- package/src/index.ts +48 -0
- package/src/providers.test.ts +172 -0
- package/src/providers.ts +262 -0
- package/src/types.ts +137 -0
- package/tsconfig.json +4 -0
- package/tsconfig.tsbuildinfo +1 -0
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Hugging Face Inference Providers static catalog (chat models only).
|
|
3
|
+
*
|
|
4
|
+
* HF's API is fundamentally different from OpenAI's — it's a *router*
|
|
5
|
+
* that proxies to many sub-providers (Cerebras, Groq, Together,
|
|
6
|
+
* SambaNova, Fireworks, etc.) under one OpenAI-compatible endpoint
|
|
7
|
+
* at `router.huggingface.co/v1/chat/completions`. You can append
|
|
8
|
+
* routing-policy suffixes to the model id:
|
|
9
|
+
*
|
|
10
|
+
* openai/gpt-oss-120b:fastest ← default; HF picks lowest latency
|
|
11
|
+
* openai/gpt-oss-120b:cheapest ← lowest cost-per-output-token
|
|
12
|
+
* openai/gpt-oss-120b:preferred ← honour user's preference list
|
|
13
|
+
* openai/gpt-oss-120b:groq ← pin to a specific provider
|
|
14
|
+
*
|
|
15
|
+
* The :fastest / :cheapest / :preferred / :<provider> suffix is
|
|
16
|
+
* handled server-side — we just include or omit it on the model
|
|
17
|
+
* string we pass to /v1/chat/completions. The adapter exposes a
|
|
18
|
+
* `routing` knob on the worker params so operators can pick.
|
|
19
|
+
*
|
|
20
|
+
* The catalog below picks notable, broadly-available open-weight
|
|
21
|
+
* models. HF lists thousands; we curate the ones worth defaulting to.
|
|
22
|
+
* The `GET /v1/models` endpoint returns the live list with per-provider
|
|
23
|
+
* pricing/throughput, so discovery fills in the rest.
|
|
24
|
+
*
|
|
25
|
+
* Capabilities: HF Inference Providers cover chat, vision (VLM),
|
|
26
|
+
* embedding, text-to-image, text-to-video, and speech-to-text via
|
|
27
|
+
* specialty endpoints, but ONLY the chat endpoint is OpenAI-compatible.
|
|
28
|
+
* Non-chat tasks need custom call shapes — we ship chat now and add
|
|
29
|
+
* other tasks as adapters when we want them.
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
import type { ChatModelInfo, ImageGenModelInfo } from '../adapters/types';
|
|
33
|
+
|
|
34
|
+
export const HUGGINGFACE_BASE_URL = 'https://router.huggingface.co/v1';
|
|
35
|
+
/** Image-gen inference uses a separate, repo-id-based endpoint. The
|
|
36
|
+
* router doesn't proxy image tasks today — calls go straight to
|
|
37
|
+
* api-inference for the per-model URL. */
|
|
38
|
+
export const HUGGINGFACE_INFERENCE_BASE_URL = 'https://api-inference.huggingface.co';
|
|
39
|
+
|
|
40
|
+
/** Suffixes the HF router accepts on the model id. The adapter
|
|
41
|
+
* appends one of these to whatever model the user picks. */
|
|
42
|
+
export const HUGGINGFACE_ROUTING_POLICIES = ['fastest', 'cheapest', 'preferred'] as const;
|
|
43
|
+
export type HuggingfaceRoutingPolicy = (typeof HUGGINGFACE_ROUTING_POLICIES)[number];
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Curated list of notable open-weight chat models reliably available
|
|
47
|
+
* through HF's router. Live `/v1/models` returns more — these are
|
|
48
|
+
* useful defaults for the dropdown.
|
|
49
|
+
*/
|
|
50
|
+
export const HUGGINGFACE_CHAT_MODELS: readonly ChatModelInfo[] = [
|
|
51
|
+
// ── OpenAI's open-weights releases ──────────────────────────────
|
|
52
|
+
{
|
|
53
|
+
id: 'openai/gpt-oss-120b',
|
|
54
|
+
label: 'gpt-oss 120B (OpenAI open)',
|
|
55
|
+
description: 'Large open-weights model from OpenAI. Solid general-purpose chat.',
|
|
56
|
+
contextTokens: 128_000,
|
|
57
|
+
capabilities: ['function_calling'],
|
|
58
|
+
},
|
|
59
|
+
{
|
|
60
|
+
id: 'openai/gpt-oss-20b',
|
|
61
|
+
label: 'gpt-oss 20B (OpenAI open)',
|
|
62
|
+
description: 'Lighter open-weights model. Fast, cheaper, weaker reasoning.',
|
|
63
|
+
contextTokens: 128_000,
|
|
64
|
+
capabilities: ['function_calling'],
|
|
65
|
+
},
|
|
66
|
+
|
|
67
|
+
// ── DeepSeek (strong reasoning at low cost) ─────────────────────
|
|
68
|
+
{
|
|
69
|
+
id: 'deepseek-ai/DeepSeek-R1',
|
|
70
|
+
label: 'DeepSeek R1',
|
|
71
|
+
description: 'Open reasoning model. Strong on math/code; verbose chain-of-thought.',
|
|
72
|
+
contextTokens: 64_000,
|
|
73
|
+
capabilities: ['reasoning'],
|
|
74
|
+
},
|
|
75
|
+
{
|
|
76
|
+
id: 'deepseek-ai/DeepSeek-V3',
|
|
77
|
+
label: 'DeepSeek V3',
|
|
78
|
+
description: 'General-purpose open chat. Cheap; solid coding ability.',
|
|
79
|
+
contextTokens: 64_000,
|
|
80
|
+
capabilities: ['function_calling'],
|
|
81
|
+
},
|
|
82
|
+
|
|
83
|
+
// ── Llama family (Meta open-weights) ────────────────────────────
|
|
84
|
+
{
|
|
85
|
+
id: 'meta-llama/Llama-3.3-70B-Instruct',
|
|
86
|
+
label: 'Llama 3.3 70B Instruct',
|
|
87
|
+
description: 'Latest open Llama instruction-tuned. Good balance of cost vs. quality.',
|
|
88
|
+
contextTokens: 128_000,
|
|
89
|
+
capabilities: ['function_calling'],
|
|
90
|
+
},
|
|
91
|
+
{
|
|
92
|
+
id: 'meta-llama/Llama-3.2-11B-Vision-Instruct',
|
|
93
|
+
label: 'Llama 3.2 11B Vision',
|
|
94
|
+
description: 'Llama with image understanding. Whiteboard / receipt OCR candidate.',
|
|
95
|
+
contextTokens: 128_000,
|
|
96
|
+
capabilities: ['vision', 'function_calling'],
|
|
97
|
+
},
|
|
98
|
+
|
|
99
|
+
// ── Mistral open weights ────────────────────────────────────────
|
|
100
|
+
{
|
|
101
|
+
id: 'mistralai/Mistral-Large-Instruct-2411',
|
|
102
|
+
label: 'Mistral Large',
|
|
103
|
+
description: 'Mistral flagship open chat model.',
|
|
104
|
+
contextTokens: 128_000,
|
|
105
|
+
capabilities: ['function_calling'],
|
|
106
|
+
},
|
|
107
|
+
{
|
|
108
|
+
id: 'mistralai/Mixtral-8x22B-Instruct-v0.1',
|
|
109
|
+
label: 'Mixtral 8×22B Instruct',
|
|
110
|
+
description: 'Mixture-of-experts. Strong multilingual; cheaper than dense large models.',
|
|
111
|
+
contextTokens: 64_000,
|
|
112
|
+
capabilities: ['function_calling'],
|
|
113
|
+
},
|
|
114
|
+
|
|
115
|
+
// ── Qwen (Alibaba — strong on code) ─────────────────────────────
|
|
116
|
+
{
|
|
117
|
+
id: 'Qwen/Qwen2.5-72B-Instruct',
|
|
118
|
+
label: 'Qwen 2.5 72B',
|
|
119
|
+
description: 'Strong general-purpose chat; particularly good at code.',
|
|
120
|
+
contextTokens: 128_000,
|
|
121
|
+
capabilities: ['function_calling'],
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
id: 'Qwen/Qwen2.5-Coder-32B-Instruct',
|
|
125
|
+
label: 'Qwen 2.5 Coder 32B',
|
|
126
|
+
description: 'Code-specialist. Pair-programming / refactoring duty.',
|
|
127
|
+
contextTokens: 128_000,
|
|
128
|
+
capabilities: ['function_calling'],
|
|
129
|
+
},
|
|
130
|
+
{
|
|
131
|
+
id: 'Qwen/Qwen2-VL-72B-Instruct',
|
|
132
|
+
label: 'Qwen 2 VL 72B (vision)',
|
|
133
|
+
description: 'Vision-language model. Image understanding plus chat.',
|
|
134
|
+
contextTokens: 32_000,
|
|
135
|
+
capabilities: ['vision'],
|
|
136
|
+
},
|
|
137
|
+
];
|
|
138
|
+
|
|
139
|
+
// ─── Hugging Face image generation ───────────────────────────────────
|
|
140
|
+
//
|
|
141
|
+
// Endpoint: POST {HUGGINGFACE_INFERENCE_BASE_URL}/models/<repo>
|
|
142
|
+
// Auth: Bearer
|
|
143
|
+
// Body: { inputs: <prompt>, parameters: { negative_prompt, seed,
|
|
144
|
+
// num_inference_steps, guidance_scale, width, height } }
|
|
145
|
+
// Response: raw image bytes (Content-Type image/png or image/jpeg).
|
|
146
|
+
//
|
|
147
|
+
// HF's catalogue is huge; we surface only the proven defaults so the
|
|
148
|
+
// dropdown isn't 200 items long. Operators can still type any repo id
|
|
149
|
+
// into the worker's `model` field for custom selections (the adapter
|
|
150
|
+
// passes it through verbatim).
|
|
151
|
+
|
|
152
|
+
export const HUGGINGFACE_IMAGE_MODELS: readonly ImageGenModelInfo[] = [
|
|
153
|
+
{
|
|
154
|
+
id: 'black-forest-labs/FLUX.1-dev',
|
|
155
|
+
label: 'FLUX.1 dev',
|
|
156
|
+
description:
|
|
157
|
+
'Black Forest Labs FLUX.1 dev. Best open-source image quality today. Slower than Schnell.',
|
|
158
|
+
tier: 'quality',
|
|
159
|
+
},
|
|
160
|
+
{
|
|
161
|
+
id: 'black-forest-labs/FLUX.1-schnell',
|
|
162
|
+
label: 'FLUX.1 schnell',
|
|
163
|
+
description: 'Distilled FLUX. 4-step inference, 5-10x faster than dev. Slight quality drop.',
|
|
164
|
+
tier: 'fast',
|
|
165
|
+
},
|
|
166
|
+
{
|
|
167
|
+
id: 'stabilityai/stable-diffusion-xl-base-1.0',
|
|
168
|
+
label: 'SDXL Base 1.0',
|
|
169
|
+
description: 'Stability AI SDXL. Older baseline; widely supported by community LoRAs.',
|
|
170
|
+
tier: 'balanced',
|
|
171
|
+
},
|
|
172
|
+
{
|
|
173
|
+
id: 'stabilityai/stable-diffusion-3.5-large',
|
|
174
|
+
label: 'Stable Diffusion 3.5 Large',
|
|
175
|
+
description: 'Stability AI SD 3.5 Large. Higher quality than SDXL, more compute per image.',
|
|
176
|
+
tier: 'quality',
|
|
177
|
+
},
|
|
178
|
+
];
|
|
179
|
+
|
|
180
|
+
export const HUGGINGFACE_IMAGE_DEFAULT_MODEL = 'black-forest-labs/FLUX.1-schnell';
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OpenAI image-generation catalog.
|
|
3
|
+
*
|
|
4
|
+
* Endpoint: POST https://api.openai.com/v1/images/generations
|
|
5
|
+
* Auth: Bearer
|
|
6
|
+
*
|
|
7
|
+
* Models we surface in the worker form (May 2026 reality):
|
|
8
|
+
*
|
|
9
|
+
* - `gpt-image-1`: the current default. Native multimodal model,
|
|
10
|
+
* returns base64 by default (no `response_format` switch needed).
|
|
11
|
+
* Better instruction-following than DALL-E for text-in-image.
|
|
12
|
+
* ~$0.04-0.17 per image depending on size/quality.
|
|
13
|
+
* - `dall-e-3`: still around but on the way out. Use when you
|
|
14
|
+
* specifically need its style-steering ('vivid'/'natural').
|
|
15
|
+
* Returns URLs by default; the adapter requests b64_json so we
|
|
16
|
+
* get bytes uniformly.
|
|
17
|
+
* - `dall-e-2`: legacy. Cheap (~$0.02 per image), lower quality.
|
|
18
|
+
* Useful for high-volume placeholder generation.
|
|
19
|
+
*
|
|
20
|
+
* Why no live discovery: OpenAI's /v1/models endpoint returns ALL
|
|
21
|
+
* models the key can use (chat, embeddings, etc.) — not just image
|
|
22
|
+
* ones. We could filter by name prefix but the catalog stays simpler
|
|
23
|
+
* + clearer if we ship the curated list and skip discovery for image
|
|
24
|
+
* gen.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import type { ImageGenModelInfo } from '../adapters/types';
|
|
28
|
+
|
|
29
|
+
export const OPENAI_IMAGE_MODELS: readonly ImageGenModelInfo[] = [
|
|
30
|
+
{
|
|
31
|
+
id: 'gpt-image-1',
|
|
32
|
+
label: 'gpt-image-1',
|
|
33
|
+
description:
|
|
34
|
+
'Current OpenAI default. Best text-in-image rendering, strong instruction-following. Recommended.',
|
|
35
|
+
supportedSizes: ['1024x1024', '1024x1536', '1536x1024'],
|
|
36
|
+
supportedQualities: ['low', 'medium', 'high', 'auto'],
|
|
37
|
+
pricePerImage: 0.07,
|
|
38
|
+
tier: 'balanced',
|
|
39
|
+
},
|
|
40
|
+
{
|
|
41
|
+
id: 'dall-e-3',
|
|
42
|
+
label: 'DALL-E 3',
|
|
43
|
+
description:
|
|
44
|
+
'Steerable with style=vivid|natural. Good for stylised art; gpt-image-1 beats it for realism + text.',
|
|
45
|
+
supportedSizes: ['1024x1024', '1024x1792', '1792x1024'],
|
|
46
|
+
supportedStyles: ['vivid', 'natural'],
|
|
47
|
+
supportedQualities: ['standard', 'hd'],
|
|
48
|
+
pricePerImage: 0.08,
|
|
49
|
+
tier: 'quality',
|
|
50
|
+
},
|
|
51
|
+
{
|
|
52
|
+
id: 'dall-e-2',
|
|
53
|
+
label: 'DALL-E 2 (legacy)',
|
|
54
|
+
description: 'Older, cheaper. ~$0.02 per image. Use when quality matters less than throughput.',
|
|
55
|
+
supportedSizes: ['256x256', '512x512', '1024x1024'],
|
|
56
|
+
pricePerImage: 0.02,
|
|
57
|
+
tier: 'fast',
|
|
58
|
+
},
|
|
59
|
+
];
|
|
60
|
+
|
|
61
|
+
/** Default model for openai-image when no worker.model is set. */
|
|
62
|
+
export const OPENAI_IMAGE_DEFAULT_MODEL = 'gpt-image-1';
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OpenAI vision model catalog.
|
|
3
|
+
*
|
|
4
|
+
* Every gpt-4o family model accepts image_url content parts; older
|
|
5
|
+
* gpt-4-vision-preview is still around but deprecated. We list the
|
|
6
|
+
* three operators are most likely to pick:
|
|
7
|
+
*
|
|
8
|
+
* - gpt-4o-mini: ~10x cheaper than gpt-4o. Strong on printed text
|
|
9
|
+
* and clean handwriting. Good default for the photo-of-notes
|
|
10
|
+
* case the user mentioned.
|
|
11
|
+
* - gpt-4o: full quality. Better on cursive, faint pencil, and
|
|
12
|
+
* multi-column layouts. Use when mini misses too much.
|
|
13
|
+
* - gpt-4-vision-preview: legacy. Don't pick this unless you have
|
|
14
|
+
* a specific reason — discoverModels still surfaces it if the
|
|
15
|
+
* account has access.
|
|
16
|
+
*
|
|
17
|
+
* Endpoint is shared with chat: POST /v1/chat/completions. The image
|
|
18
|
+
* goes in messages[].content as `{type: 'image_url', image_url: {url:
|
|
19
|
+
* 'data:image/jpeg;base64,...'}}`. Adapter handles the encoding.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import type { VisionModelInfo } from '../adapters/types';
|
|
23
|
+
|
|
24
|
+
export const OPENAI_VISION_MODELS: readonly VisionModelInfo[] = [
|
|
25
|
+
{
|
|
26
|
+
id: 'gpt-4o-mini',
|
|
27
|
+
label: 'gpt-4o-mini',
|
|
28
|
+
description:
|
|
29
|
+
'Cheap, fast vision. Recommended default for high-volume OCR of notes, receipts, screenshots.',
|
|
30
|
+
contextTokens: 128_000,
|
|
31
|
+
inputPricePer1M: 0.15,
|
|
32
|
+
outputPricePer1M: 0.6,
|
|
33
|
+
tier: 'fast',
|
|
34
|
+
},
|
|
35
|
+
{
|
|
36
|
+
id: 'gpt-4o',
|
|
37
|
+
label: 'gpt-4o',
|
|
38
|
+
description:
|
|
39
|
+
'Full-quality vision. Use when gpt-4o-mini misreads cursive or you need diagram/chart comprehension alongside text.',
|
|
40
|
+
contextTokens: 128_000,
|
|
41
|
+
inputPricePer1M: 2.5,
|
|
42
|
+
outputPricePer1M: 10,
|
|
43
|
+
tier: 'balanced',
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
id: 'gpt-4-vision-preview',
|
|
47
|
+
label: 'gpt-4-vision-preview (legacy)',
|
|
48
|
+
description:
|
|
49
|
+
'Deprecated. Kept here only so existing workers configured for it still appear in the dropdown.',
|
|
50
|
+
contextTokens: 128_000,
|
|
51
|
+
tier: 'quality',
|
|
52
|
+
},
|
|
53
|
+
];
|
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Curated OpenRouter chat catalog — Mantle's headline picks.
|
|
3
|
+
*
|
|
4
|
+
* OpenRouter serves 300+ chat models across every major provider. We
|
|
5
|
+
* don't try to mirror that — the full live catalog is fetched at
|
|
6
|
+
* runtime by `discoverModels` against `https://openrouter.ai/api/v1/models`
|
|
7
|
+
* (keyless). This static list is what the worker form's model dropdown
|
|
8
|
+
* renders BEFORE discovery completes, and the fallback when discovery
|
|
9
|
+
* fails (rare; OR's /models endpoint is among the most reliable in
|
|
10
|
+
* the catalogue).
|
|
11
|
+
*
|
|
12
|
+
* Pick criteria — what makes it onto this list:
|
|
13
|
+
* - The model the project's default workers ship pointing at, OR
|
|
14
|
+
* - A current top-tier headline from each major lab (Anthropic,
|
|
15
|
+
* OpenAI, Google, xAI), OR
|
|
16
|
+
* - A notable open-weights model that's frequently asked for
|
|
17
|
+
* (DeepSeek's reasoning lines, Llama 4).
|
|
18
|
+
*
|
|
19
|
+
* Capabilities + pricing here mirror the source-of-truth pricing
|
|
20
|
+
* table in [packages/tracing/src/pricing.ts](../../tracing/src/pricing.ts).
|
|
21
|
+
* Keep them aligned when models shift — the chat-adapter UI reads
|
|
22
|
+
* these for the dropdown hints, and tracing reads pricing.ts for the
|
|
23
|
+
* fallback cost calculation. Diverging would let a model show one
|
|
24
|
+
* price in the dropdown and bill at a different one.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import type { ChatModelInfo, VisionModelInfo } from '../adapters/types';
|
|
28
|
+
|
|
29
|
+
/** OpenRouter API base. The SDK constructs this internally too; we
|
|
30
|
+
* expose it for the catalogue + discovery so both paths agree. */
|
|
31
|
+
export const OPENROUTER_BASE_URL = 'https://openrouter.ai/api/v1';
|
|
32
|
+
|
|
33
|
+
/** A small curated set of OpenRouter chat models we vouch for as
|
|
34
|
+
* defaults. The full live catalog comes from discovery; this is the
|
|
35
|
+
* "before the network call returns" picture. */
|
|
36
|
+
export const OPENROUTER_CHAT_MODELS: readonly ChatModelInfo[] = [
|
|
37
|
+
// Anthropic — the responder's default lives here.
|
|
38
|
+
{
|
|
39
|
+
id: 'anthropic/claude-sonnet-5',
|
|
40
|
+
label: 'Claude Sonnet 5',
|
|
41
|
+
description:
|
|
42
|
+
'Anthropic flagship. Strong reasoning + tool use, 1M context. Default for the responder and Saskia. Supports prompt caching (~10× cheaper on cache hits).',
|
|
43
|
+
contextTokens: 1_000_000,
|
|
44
|
+
capabilities: ['reasoning', 'function_calling', 'vision'],
|
|
45
|
+
inputPricePer1M: 2,
|
|
46
|
+
outputPricePer1M: 10,
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
id: 'anthropic/claude-sonnet-4.6',
|
|
50
|
+
label: 'Claude Sonnet 4.6',
|
|
51
|
+
description: 'Previous-generation Sonnet. 1M context; supports prompt caching.',
|
|
52
|
+
contextTokens: 1_000_000,
|
|
53
|
+
capabilities: ['reasoning', 'function_calling', 'vision'],
|
|
54
|
+
inputPricePer1M: 3,
|
|
55
|
+
outputPricePer1M: 15,
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
id: 'anthropic/claude-haiku-4.5',
|
|
59
|
+
label: 'Claude Haiku 4.5',
|
|
60
|
+
description:
|
|
61
|
+
'Anthropic small/fast. Cheap, fast, 200K context. Default for the extractor + summarizer. Supports prompt caching.',
|
|
62
|
+
contextTokens: 200_000,
|
|
63
|
+
capabilities: ['function_calling', 'vision'],
|
|
64
|
+
inputPricePer1M: 0.8,
|
|
65
|
+
outputPricePer1M: 4,
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
id: 'anthropic/claude-opus-4.7',
|
|
69
|
+
label: 'Claude Opus 4.7',
|
|
70
|
+
description:
|
|
71
|
+
'Anthropic top-tier. Highest capability, highest price. 1M context. Use when the responder needs to chew on something hard.',
|
|
72
|
+
contextTokens: 1_000_000,
|
|
73
|
+
capabilities: ['reasoning', 'function_calling', 'vision'],
|
|
74
|
+
inputPricePer1M: 15,
|
|
75
|
+
outputPricePer1M: 75,
|
|
76
|
+
},
|
|
77
|
+
// OpenAI — headline picks.
|
|
78
|
+
{
|
|
79
|
+
id: 'openai/gpt-5',
|
|
80
|
+
label: 'GPT-5',
|
|
81
|
+
description: 'OpenAI flagship. Strong all-rounder, native vision + tool use. 256K context.',
|
|
82
|
+
contextTokens: 256_000,
|
|
83
|
+
capabilities: ['reasoning', 'function_calling', 'vision'],
|
|
84
|
+
inputPricePer1M: 5,
|
|
85
|
+
outputPricePer1M: 15,
|
|
86
|
+
},
|
|
87
|
+
{
|
|
88
|
+
id: 'openai/gpt-4o',
|
|
89
|
+
label: 'GPT-4o',
|
|
90
|
+
description: 'OpenAI multimodal workhorse. Still widely used for cost reasons. 128K context.',
|
|
91
|
+
contextTokens: 128_000,
|
|
92
|
+
capabilities: ['function_calling', 'vision'],
|
|
93
|
+
inputPricePer1M: 2.5,
|
|
94
|
+
outputPricePer1M: 10,
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
id: 'openai/gpt-4o-mini',
|
|
98
|
+
label: 'GPT-4o Mini',
|
|
99
|
+
description:
|
|
100
|
+
'OpenAI cheap fast model. Good extractor candidate when keeping costs low matters more than precision.',
|
|
101
|
+
contextTokens: 128_000,
|
|
102
|
+
capabilities: ['function_calling', 'vision'],
|
|
103
|
+
inputPricePer1M: 0.15,
|
|
104
|
+
outputPricePer1M: 0.6,
|
|
105
|
+
},
|
|
106
|
+
// Google — Gemini 2.5+.
|
|
107
|
+
{
|
|
108
|
+
id: 'google/gemini-2.5-pro',
|
|
109
|
+
label: 'Gemini 2.5 Pro',
|
|
110
|
+
description:
|
|
111
|
+
'Google top-tier with 2M context. Strong long-document handling; implicit prompt caching on Gemini 2.5+.',
|
|
112
|
+
contextTokens: 2_000_000,
|
|
113
|
+
capabilities: ['reasoning', 'function_calling', 'vision'],
|
|
114
|
+
inputPricePer1M: 1.25,
|
|
115
|
+
outputPricePer1M: 10,
|
|
116
|
+
},
|
|
117
|
+
{
|
|
118
|
+
id: 'google/gemini-2.5-flash',
|
|
119
|
+
label: 'Gemini 2.5 Flash',
|
|
120
|
+
description:
|
|
121
|
+
'Google fast/cheap. 1M context, implicit caching. Good fit for extractor + summarizer when not using Anthropic.',
|
|
122
|
+
contextTokens: 1_000_000,
|
|
123
|
+
capabilities: ['function_calling', 'vision'],
|
|
124
|
+
inputPricePer1M: 0.3,
|
|
125
|
+
outputPricePer1M: 2.5,
|
|
126
|
+
},
|
|
127
|
+
// xAI.
|
|
128
|
+
{
|
|
129
|
+
id: 'x-ai/grok-4',
|
|
130
|
+
label: 'Grok 4',
|
|
131
|
+
description:
|
|
132
|
+
'xAI flagship. 256K context, vision-capable. Real-time web access on some variants. Automatic prompt caching server-side.',
|
|
133
|
+
contextTokens: 256_000,
|
|
134
|
+
capabilities: ['reasoning', 'function_calling', 'vision'],
|
|
135
|
+
inputPricePer1M: 3,
|
|
136
|
+
outputPricePer1M: 15,
|
|
137
|
+
},
|
|
138
|
+
// DeepSeek — open-weights reasoning.
|
|
139
|
+
{
|
|
140
|
+
id: 'deepseek/deepseek-v3.1',
|
|
141
|
+
label: 'DeepSeek V3.1',
|
|
142
|
+
description:
|
|
143
|
+
'Open-weights MoE. Strong reasoning at low price; popular for cost-sensitive workloads.',
|
|
144
|
+
contextTokens: 128_000,
|
|
145
|
+
capabilities: ['function_calling'],
|
|
146
|
+
inputPricePer1M: 0.27,
|
|
147
|
+
outputPricePer1M: 1.1,
|
|
148
|
+
},
|
|
149
|
+
// Meta — Llama 4 if/when added; placeholder pricing.
|
|
150
|
+
{
|
|
151
|
+
id: 'meta-llama/llama-4-maverick',
|
|
152
|
+
label: 'Llama 4 Maverick',
|
|
153
|
+
description:
|
|
154
|
+
'Open-weights from Meta. Reasonable cost, broad capability. Pricing varies by sub-provider OR routes to.',
|
|
155
|
+
contextTokens: 256_000,
|
|
156
|
+
capabilities: ['function_calling', 'vision'],
|
|
157
|
+
inputPricePer1M: 0.5,
|
|
158
|
+
outputPricePer1M: 1.5,
|
|
159
|
+
},
|
|
160
|
+
];
|
|
161
|
+
|
|
162
|
+
/** Curated vision-capable OpenRouter routes for the Vision + Document worker
|
|
163
|
+
* dropdowns (before discovery returns the live image-input list). The
|
|
164
|
+
* Anthropic + Gemini routes also support native PDF, so they back the
|
|
165
|
+
* Document worker's `extractDocument` path; gpt-4o is image-only here. */
|
|
166
|
+
export const OPENROUTER_VISION_MODELS: readonly VisionModelInfo[] = [
|
|
167
|
+
{
|
|
168
|
+
id: 'anthropic/claude-sonnet-5',
|
|
169
|
+
label: 'Claude Sonnet 5',
|
|
170
|
+
description: 'Strong document + table reading; native PDF. Great default for invoices.',
|
|
171
|
+
contextTokens: 1_000_000,
|
|
172
|
+
inputPricePer1M: 2,
|
|
173
|
+
outputPricePer1M: 10,
|
|
174
|
+
tier: 'balanced',
|
|
175
|
+
},
|
|
176
|
+
{
|
|
177
|
+
id: 'anthropic/claude-sonnet-4.6',
|
|
178
|
+
label: 'Claude Sonnet 4.6',
|
|
179
|
+
description: 'Previous-generation Sonnet vision; native PDF.',
|
|
180
|
+
contextTokens: 1_000_000,
|
|
181
|
+
inputPricePer1M: 3,
|
|
182
|
+
outputPricePer1M: 15,
|
|
183
|
+
tier: 'balanced',
|
|
184
|
+
},
|
|
185
|
+
{
|
|
186
|
+
id: 'anthropic/claude-opus-4.7',
|
|
187
|
+
label: 'Claude Opus 4.7',
|
|
188
|
+
description: 'Top-tier vision/document fidelity; native PDF. Highest cost.',
|
|
189
|
+
contextTokens: 1_000_000,
|
|
190
|
+
inputPricePer1M: 15,
|
|
191
|
+
outputPricePer1M: 75,
|
|
192
|
+
tier: 'quality',
|
|
193
|
+
},
|
|
194
|
+
{
|
|
195
|
+
id: 'anthropic/claude-haiku-4.5',
|
|
196
|
+
label: 'Claude Haiku 4.5',
|
|
197
|
+
description: 'Cheap + fast vision; native PDF. Good for high-volume describe/OCR.',
|
|
198
|
+
contextTokens: 200_000,
|
|
199
|
+
inputPricePer1M: 0.8,
|
|
200
|
+
outputPricePer1M: 4,
|
|
201
|
+
tier: 'fast',
|
|
202
|
+
},
|
|
203
|
+
{
|
|
204
|
+
id: 'google/gemini-2.5-pro',
|
|
205
|
+
label: 'Gemini 2.5 Pro',
|
|
206
|
+
description: 'Google multimodal; native PDF. Strong on dense layouts.',
|
|
207
|
+
contextTokens: 1_000_000,
|
|
208
|
+
inputPricePer1M: 1.25,
|
|
209
|
+
outputPricePer1M: 10,
|
|
210
|
+
tier: 'balanced',
|
|
211
|
+
},
|
|
212
|
+
{
|
|
213
|
+
id: 'openai/gpt-4o',
|
|
214
|
+
label: 'GPT-4o',
|
|
215
|
+
description: 'OpenAI multimodal (images). PDFs fall back to page OCR via OpenRouter.',
|
|
216
|
+
contextTokens: 128_000,
|
|
217
|
+
inputPricePer1M: 2.5,
|
|
218
|
+
outputPricePer1M: 10,
|
|
219
|
+
tier: 'balanced',
|
|
220
|
+
},
|
|
221
|
+
];
|