@molecule/api-resource-ai-models 1.0.0 → 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +672 -0
- package/dist/models.d.ts +15 -3
- package/dist/models.d.ts.map +1 -1
- package/dist/models.js +85 -3
- package/package.json +9 -8
package/README.md
ADDED
|
@@ -0,0 +1,672 @@
|
|
|
1
|
+
<!--
|
|
2
|
+
AUTO-GENERATED — DO NOT EDIT THIS FILE.
|
|
3
|
+
Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
|
|
4
|
+
Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
|
|
5
|
+
To change this document, edit the module-level JSDoc in src/index.ts.
|
|
6
|
+
Generated: 2026-08-06T19:42:52.531Z
|
|
7
|
+
-->
|
|
8
|
+
|
|
9
|
+
# @molecule/api-resource-ai-models
|
|
10
|
+
|
|
11
|
+
> **Auto-generated, AI-first package reference** for the [molecule.dev](https://molecule.dev) ecosystem.
|
|
12
|
+
> It is written to be read by coding agents as much as by people, and is generated from this
|
|
13
|
+
> package's source — edit `src/index.ts` JSDoc, not this file.
|
|
14
|
+
|
|
15
|
+
AI model catalog resource.
|
|
16
|
+
|
|
17
|
+
Server-side source of truth for available AI models plus an
|
|
18
|
+
authentication-gated discovery endpoint (`GET /ai/models`). Server consumers
|
|
19
|
+
(chat handler, compaction) import `MODELS` / `getModel` / `MODEL_IDS`
|
|
20
|
+
directly; authenticated clients fetch the filtered projection over HTTP. The
|
|
21
|
+
`list` handler enforces the session check itself and fails closed with `401`,
|
|
22
|
+
so the configured-model catalog is never disclosed to an unauthenticated
|
|
23
|
+
caller even if the route's `'authenticate'` middleware is stripped by codegen.
|
|
24
|
+
|
|
25
|
+
## Quick Start
|
|
26
|
+
|
|
27
|
+
```typescript
|
|
28
|
+
import { getModel, MODEL_IDS } from '@molecule/api-resource-ai-models'
|
|
29
|
+
|
|
30
|
+
// Server-side validation of a client-selected model id:
|
|
31
|
+
if (!MODEL_IDS.has(requestedId)) {
|
|
32
|
+
throw new Error('Unknown or retired model')
|
|
33
|
+
}
|
|
34
|
+
const model = getModel(requestedId)! // full definition (pricing, effort levels)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Type
|
|
38
|
+
|
|
39
|
+
`resource`
|
|
40
|
+
|
|
41
|
+
## Installation
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
npm install @molecule/api-resource-ai-models @molecule/api-bond @molecule/api-i18n @molecule/api-resource
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## API
|
|
48
|
+
|
|
49
|
+
### Interfaces
|
|
50
|
+
|
|
51
|
+
#### `ListModelsResponse`
|
|
52
|
+
|
|
53
|
+
Response shape returned by `GET /ai/models`.
|
|
54
|
+
|
|
55
|
+
```typescript
|
|
56
|
+
interface ListModelsResponse {
|
|
57
|
+
models: ModelDefinition[]
|
|
58
|
+
/**
|
|
59
|
+
* Per-mode server default model ids for the requester's tier. Optional —
|
|
60
|
+
* servers that don't compute tier-aware defaults omit it, and clients fall
|
|
61
|
+
* back to generic "default" labeling.
|
|
62
|
+
*/
|
|
63
|
+
defaults?: ModeModelDefaults
|
|
64
|
+
}
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
#### `ModelDefinition`
|
|
68
|
+
|
|
69
|
+
Full server-side metadata for an AI model. Consumed directly by the chat
|
|
70
|
+
handler, compaction, and any other server-side cost / budget logic.
|
|
71
|
+
|
|
72
|
+
```typescript
|
|
73
|
+
interface ModelDefinition {
|
|
74
|
+
/** API model ID (e.g. `'claude-sonnet-4-6'`, `'gpt-5.4'`). */
|
|
75
|
+
id: string
|
|
76
|
+
/** Which AI provider serves this model. */
|
|
77
|
+
provider: AIProviderID
|
|
78
|
+
/** Human-readable label (e.g. `'Claude Sonnet 4.6'`). */
|
|
79
|
+
label: string
|
|
80
|
+
/** Short description for UI display. */
|
|
81
|
+
description: string
|
|
82
|
+
/** Maximum input context window in tokens. */
|
|
83
|
+
contextWindow: number
|
|
84
|
+
/** Maximum output tokens per response. */
|
|
85
|
+
maxOutputTokens: number
|
|
86
|
+
/** Whether the model supports extended thinking / chain-of-thought. */
|
|
87
|
+
supportsThinking: boolean
|
|
88
|
+
/** Default thinking budget in tokens (only relevant when `supportsThinking` is true). */
|
|
89
|
+
thinkingBudgetTokens: number
|
|
90
|
+
/**
|
|
91
|
+
* Whether the thinking budget can be controlled via API params.
|
|
92
|
+
* When false, the model always reasons but does not accept a thinking / reasoning_effort param.
|
|
93
|
+
*/
|
|
94
|
+
thinkingConfigurable: boolean
|
|
95
|
+
/**
|
|
96
|
+
* The model's OWN reasoning-effort levels, ordered ascending (least → most
|
|
97
|
+
* effort) — the exact values a user picks, that get persisted, and that the
|
|
98
|
+
* `/effort` command offers. There is NO abstract scale: these are the model's
|
|
99
|
+
* real levels.
|
|
100
|
+
*
|
|
101
|
+
* - **Native-effort models** (Anthropic `output_config.effort`, OpenAI
|
|
102
|
+
* `reasoning_effort`, Gemini `thinking_level`, …) list their provider values
|
|
103
|
+
* verbatim, e.g. `['low', 'high', 'xhigh', 'max']` — each value is sent as
|
|
104
|
+
* the provider's effort param as-is.
|
|
105
|
+
* - **Budget-configurable models** (a raw thinking-token budget, no native
|
|
106
|
+
* level names — e.g. Claude Haiku 4.5, Qwen3.7) list scaled-budget LABELS,
|
|
107
|
+
* e.g. `['4K', '8K', '16K', '32K']`, with {@link effortBudgetTokens} mapping
|
|
108
|
+
* each label to the actual token budget sent.
|
|
109
|
+
* - **Fixed-reasoning models** (DeepSeek executors, Kimi, …) omit this field
|
|
110
|
+
* entirely — reasoning depth can't be tuned, so there is nothing to pick.
|
|
111
|
+
*
|
|
112
|
+
* A persisted value outside the active model's set degrades to the nearest one
|
|
113
|
+
* (`model-selection.ts` `resolveEffortForModel`). Absent → no effort choice.
|
|
114
|
+
*/
|
|
115
|
+
supportedEffortLevels?: EffortLevel[]
|
|
116
|
+
/**
|
|
117
|
+
* The model's default effort value — the one used when the user hasn't chosen.
|
|
118
|
+
* MUST be a member of {@link supportedEffortLevels}. Absent only when the model
|
|
119
|
+
* has no effort levels (fixed reasoning).
|
|
120
|
+
*/
|
|
121
|
+
defaultEffortLevel?: EffortLevel
|
|
122
|
+
/**
|
|
123
|
+
* For budget-configurable models ONLY: maps each label in
|
|
124
|
+
* {@link supportedEffortLevels} to the thinking-token budget it sends
|
|
125
|
+
* (e.g. `{ '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 }`). Its
|
|
126
|
+
* presence is what marks a model as budget-driven rather than native-effort:
|
|
127
|
+
* a model WITH this map sends `budget_tokens`; a model WITHOUT it sends its
|
|
128
|
+
* chosen level as the provider's native effort param.
|
|
129
|
+
*
|
|
130
|
+
* CRITICAL for Anthropic 4.6+ models (Fable 5, Opus 4.8/4.6, Sonnet 5 / 4.6):
|
|
131
|
+
* these must NOT carry this map — they are native-effort models, and sending
|
|
132
|
+
* `budget_tokens` returns a 400 on Fable 5 / Opus 4.8 / Sonnet 5.
|
|
133
|
+
*/
|
|
134
|
+
effortBudgetTokens?: Record<string, number>
|
|
135
|
+
/** Whether the model supports vision (images, documents, etc.). */
|
|
136
|
+
supportsVision: boolean
|
|
137
|
+
/** Whether the model supports prompt caching. */
|
|
138
|
+
supportsPromptCaching: boolean
|
|
139
|
+
/** Whether the model supports tool use / function calling. */
|
|
140
|
+
supportsTools: boolean
|
|
141
|
+
/**
|
|
142
|
+
* The model cannot combine function tools with ANY reasoning on the provider's
|
|
143
|
+
* chat-completions endpoint, so a request carrying tools must pin reasoning
|
|
144
|
+
* OFF or it is rejected outright.
|
|
145
|
+
*
|
|
146
|
+
* Set for the gpt-5.6 family, which answers a tools request with:
|
|
147
|
+
* `Function tools with reasoning_effort are not supported for <model> in
|
|
148
|
+
* /v1/chat/completions. To use function tools, use /v1/responses or set
|
|
149
|
+
* reasoning_effort to 'none'.` (400 — verified live, 2026-07-30). Omitting the
|
|
150
|
+
* effort field entirely does NOT help: the model applies its own default and
|
|
151
|
+
* still 400s. Only an explicit `'none'` works.
|
|
152
|
+
*
|
|
153
|
+
* This is a per-model API fact, so it lives in the catalogue rather than as a
|
|
154
|
+
* model-name branch inside a bond.
|
|
155
|
+
*
|
|
156
|
+
* **This is a workaround, not the fix.** Pinning reasoning off means an agentic
|
|
157
|
+
* caller — which always carries tools — never gets reasoning from these models.
|
|
158
|
+
* The real fix is migrating the OpenAI bond to `/v1/responses`, which supports
|
|
159
|
+
* both together; until then, working-without-reasoning beats 400.
|
|
160
|
+
*/
|
|
161
|
+
toolsRequireReasoningOff?: boolean
|
|
162
|
+
/**
|
|
163
|
+
* Provider-specific server tool type for web search (e.g. `'web_search_20250305'`).
|
|
164
|
+
* When set, the chat handler sends this as a ServerTool alongside custom tools.
|
|
165
|
+
* Omit if the model / provider does not support native web search.
|
|
166
|
+
*/
|
|
167
|
+
webSearchToolType?: string
|
|
168
|
+
/**
|
|
169
|
+
* Provider-specific server tool type for code execution (e.g. `'code_execution_20250825'`).
|
|
170
|
+
* Omit if the model / provider does not support native code execution.
|
|
171
|
+
*/
|
|
172
|
+
codeExecutionToolType?: string
|
|
173
|
+
/**
|
|
174
|
+
* Provider-specific server tool type for web fetch / URL context (e.g. `'web_fetch_20260209'`).
|
|
175
|
+
* Omit if the model / provider does not support native web fetch.
|
|
176
|
+
*/
|
|
177
|
+
webFetchToolType?: string
|
|
178
|
+
/** Whether this model is available on the free tier (only one model should be true). */
|
|
179
|
+
freeTier?: boolean
|
|
180
|
+
/**
|
|
181
|
+
* Regions in which this model is free-tier selectable even though the model
|
|
182
|
+
* as a whole is not `freeTier` — for models whose regional hosts price very
|
|
183
|
+
* differently (e.g. a cheap native host powering the free planner while its
|
|
184
|
+
* ~3× re-host stays paid-only). Ignored when `freeTier` is true (all regions
|
|
185
|
+
* free); omitted → no free-tier access outside `freeTier`.
|
|
186
|
+
*/
|
|
187
|
+
freeTierRegions?: string[]
|
|
188
|
+
/**
|
|
189
|
+
* Processing regions this model can run in, as arbitrary region codes; the
|
|
190
|
+
* FIRST entry is the model's default region. Omit for `['us']` (the platform
|
|
191
|
+
* default — a single-region US model). A single-entry list pins the model to
|
|
192
|
+
* that region regardless of the user's per-model choice (e.g. `['cn']` for a
|
|
193
|
+
* model with no US re-host). Dispatch resolves a region to the
|
|
194
|
+
* `<provider>-<region>` named bond (`'us'` → the bare `<provider>` bond).
|
|
195
|
+
*/
|
|
196
|
+
regions?: string[]
|
|
197
|
+
/**
|
|
198
|
+
* Per-region price overrides in USD per MTok, keyed by region code, for
|
|
199
|
+
* regions whose host bills differently from the base rates (e.g. a US
|
|
200
|
+
* re-host of a Chinese-origin model). The BASE `*PricePerMTok` fields always
|
|
201
|
+
* carry the native provider's list prices (what models.dev / the freshness
|
|
202
|
+
* gate verify); a region with no entry here bills at the base rates. Omitted
|
|
203
|
+
* cache fields fall back to the region's `inputPricePerMTok` (hosts with no
|
|
204
|
+
* cache discount / no write premium).
|
|
205
|
+
*/
|
|
206
|
+
regionPricing?: Record<
|
|
207
|
+
string,
|
|
208
|
+
{
|
|
209
|
+
/** Region input price per million uncached tokens in USD. */
|
|
210
|
+
inputPricePerMTok: number
|
|
211
|
+
/** Region output price per million tokens in USD. */
|
|
212
|
+
outputPricePerMTok: number
|
|
213
|
+
/** Region prompt-cache read price per million tokens in USD. */
|
|
214
|
+
cacheReadPricePerMTok?: number
|
|
215
|
+
/** Region prompt-cache write price per million tokens in USD. */
|
|
216
|
+
cacheWritePricePerMTok?: number
|
|
217
|
+
}
|
|
218
|
+
>
|
|
219
|
+
/** Input price per million *uncached* (fresh) input tokens in USD. */
|
|
220
|
+
inputPricePerMTok: number
|
|
221
|
+
/** Output price per million tokens in USD. */
|
|
222
|
+
outputPricePerMTok: number
|
|
223
|
+
/**
|
|
224
|
+
* Price per million prompt-cache *read* (cache-hit) input tokens in USD.
|
|
225
|
+
*
|
|
226
|
+
* REQUIRED — never omit. Prompt caching is enabled for the agentic loop, so
|
|
227
|
+
* for a long conversation the cache-read tokens are the DOMINANT input
|
|
228
|
+
* category. Pricing them at `0` (the bug this field fixes) systematically
|
|
229
|
+
* under-measures real upstream spend and lets cost-gated budgets be blown
|
|
230
|
+
* past their caps. Conventionally a steep discount on `inputPricePerMTok`
|
|
231
|
+
* (e.g. Anthropic / OpenAI / DeepSeek bill cache reads at ~0.1×). MUST be
|
|
232
|
+
* `<= inputPricePerMTok` — a cache hit is never more expensive than fresh
|
|
233
|
+
* input.
|
|
234
|
+
*/
|
|
235
|
+
cacheReadPricePerMTok: number
|
|
236
|
+
/**
|
|
237
|
+
* Price per million prompt-cache *write* (cache-creation) input tokens in USD.
|
|
238
|
+
*
|
|
239
|
+
* REQUIRED — never omit. The first time a prefix is cached the provider may
|
|
240
|
+
* charge a premium (Anthropic's 5-minute cache write is ~1.25× input);
|
|
241
|
+
* providers that auto-cache at no extra charge (OpenAI, DeepSeek) set this
|
|
242
|
+
* equal to `inputPricePerMTok`. MUST be `>= inputPricePerMTok` — a cache
|
|
243
|
+
* write is never cheaper than fresh input. Only the Anthropic bond currently
|
|
244
|
+
* emits `cacheCreationInputTokens`, but every model declares this so a new
|
|
245
|
+
* cache-emitting bond can never silently bill cache writes at `0`.
|
|
246
|
+
*/
|
|
247
|
+
cacheWritePricePerMTok: number
|
|
248
|
+
/**
|
|
249
|
+
* Optional provider peak-hour pricing: during the listed UTC windows, ALL of
|
|
250
|
+
* this model's token prices (input, output, cache read/write) bill at
|
|
251
|
+
* `multiplier × ` the listed rates. Metering MUST price each request by its
|
|
252
|
+
* own timestamp via `priceMultiplierAt()` — never assume the flat rate — or
|
|
253
|
+
* peak-hour usage is under-metered and the platform eats the difference
|
|
254
|
+
* (e.g. DeepSeek's announced 2× Beijing-business-hours pricing).
|
|
255
|
+
*
|
|
256
|
+
* Windows are minutes-since-midnight UTC, half-open `[start, end)`; a window
|
|
257
|
+
* may wrap midnight (`start > end`).
|
|
258
|
+
*/
|
|
259
|
+
peakPricing?: {
|
|
260
|
+
windows: { startMinuteUtc: number; endMinuteUtc: number }[]
|
|
261
|
+
multiplier: number
|
|
262
|
+
}
|
|
263
|
+
/**
|
|
264
|
+
* Fast-mode ("priority speed") pricing — the per-MTok rates billed when a
|
|
265
|
+
* request runs with the provider's fast/priority tier (e.g. Anthropic's
|
|
266
|
+
* `speed: "fast"` research preview: same model, up to ~2.5× output speed, at
|
|
267
|
+
* premium pricing). PRESENCE of this field is the capability flag: a model
|
|
268
|
+
* without it does not support fast mode, and metering/UI/dispatch all key off
|
|
269
|
+
* that. All four fields are required for the same never-under-meter reasons
|
|
270
|
+
* as the base rates. Metering MUST price a turn by the speed the provider
|
|
271
|
+
* REPORTS it ran at (`TokenUsage.speed`), not the speed requested — a
|
|
272
|
+
* fast-mode 429 that falls back to standard must not bill 2×.
|
|
273
|
+
*/
|
|
274
|
+
fastPricing?: {
|
|
275
|
+
/** Fast-mode input price per million uncached tokens in USD. */
|
|
276
|
+
inputPricePerMTok: number
|
|
277
|
+
/** Fast-mode output price per million tokens in USD. */
|
|
278
|
+
outputPricePerMTok: number
|
|
279
|
+
/** Fast-mode prompt-cache read price per million tokens in USD. */
|
|
280
|
+
cacheReadPricePerMTok: number
|
|
281
|
+
/** Fast-mode prompt-cache write price per million tokens in USD. */
|
|
282
|
+
cacheWritePricePerMTok: number
|
|
283
|
+
}
|
|
284
|
+
/** Reliable knowledge cutoff date (YYYY-MM-DD). */
|
|
285
|
+
knowledgeCutoff: string
|
|
286
|
+
/**
|
|
287
|
+
* When the model was (or will be) deprecated (YYYY-MM-DD).
|
|
288
|
+
*
|
|
289
|
+
* Past dates: still selectable, but the picker tucks them into an "Older
|
|
290
|
+
* models" section so newcomers default to current entries. Saved selections
|
|
291
|
+
* (a user previously picked this) keep working. Future dates: still treated
|
|
292
|
+
* as current — useful for scheduling a deprecation in advance.
|
|
293
|
+
*
|
|
294
|
+
* Omit entirely for current models.
|
|
295
|
+
*/
|
|
296
|
+
deprecatedAt?: string
|
|
297
|
+
/**
|
|
298
|
+
* Whether this model is fully disabled — removed from selection and the
|
|
299
|
+
* public listing while remaining priceable for historical usage.
|
|
300
|
+
*
|
|
301
|
+
* Stronger than {@link deprecatedAt}: a deprecated model is still selectable
|
|
302
|
+
* (the picker just tucks it into an "Older models" section), whereas a
|
|
303
|
+
* disabled model vanishes from every *exposure* surface — it is excluded from
|
|
304
|
+
* `MODEL_IDS`, `getAvailableModels()`, the `GET /ai/models` listing, and the
|
|
305
|
+
* client-side free-tier / deprecation-partition helpers. Use it for a model
|
|
306
|
+
* the provider has retired or deprecated (e.g. `grok-code-fast-1`).
|
|
307
|
+
*
|
|
308
|
+
* `getModel(id)` STILL returns a disabled entry so a saved selection or a
|
|
309
|
+
* historical usage row can always be priced — NEVER delete a disabled model,
|
|
310
|
+
* or its past usage silently meters as free. Omit entirely for active models.
|
|
311
|
+
*/
|
|
312
|
+
disabled?: boolean
|
|
313
|
+
}
|
|
314
|
+
```
|
|
315
|
+
|
|
316
|
+
#### `ModelTokenRates`
|
|
317
|
+
|
|
318
|
+
The four per-MTok token rates a turn bills at (USD).
|
|
319
|
+
|
|
320
|
+
```typescript
|
|
321
|
+
interface ModelTokenRates {
|
|
322
|
+
/** Input price per million uncached tokens in USD. */
|
|
323
|
+
inputPricePerMTok: number
|
|
324
|
+
/** Output price per million tokens in USD. */
|
|
325
|
+
outputPricePerMTok: number
|
|
326
|
+
/** Prompt-cache read price per million tokens in USD. */
|
|
327
|
+
cacheReadPricePerMTok: number
|
|
328
|
+
/** Prompt-cache write price per million tokens in USD. */
|
|
329
|
+
cacheWritePricePerMTok: number
|
|
330
|
+
}
|
|
331
|
+
```
|
|
332
|
+
|
|
333
|
+
#### `ModeModelDefaults`
|
|
334
|
+
|
|
335
|
+
The model ids the SERVER falls back to per mode/job when the user hasn't
|
|
336
|
+
picked one — already resolved for the requester's tier (a free-tier caller
|
|
337
|
+
sees the free-tier clamp ids, a paid caller the paid defaults). Lets the
|
|
338
|
+
client label an unset per-mode picker "Default (<model>)" instead of a
|
|
339
|
+
vague "default" the user can't decode.
|
|
340
|
+
|
|
341
|
+
```typescript
|
|
342
|
+
interface ModeModelDefaults {
|
|
343
|
+
/** Model id used in plan mode when nothing is configured. */
|
|
344
|
+
plan: string
|
|
345
|
+
/** Model id used in execute mode when nothing is configured. */
|
|
346
|
+
execute: string
|
|
347
|
+
/** Model id used for commit-message generation when nothing is configured. */
|
|
348
|
+
commit: string
|
|
349
|
+
/** Model id used for conversation compaction when nothing is configured. */
|
|
350
|
+
compact: string
|
|
351
|
+
}
|
|
352
|
+
```
|
|
353
|
+
|
|
354
|
+
### Types
|
|
355
|
+
|
|
356
|
+
#### `AIProviderID`
|
|
357
|
+
|
|
358
|
+
AI provider identifier.
|
|
359
|
+
|
|
360
|
+
Maps to the bond category used at runtime (e.g. `bond('ai', 'anthropic', anthropicProvider)`).
|
|
361
|
+
Adding a new provider here means a corresponding AI bond package must exist.
|
|
362
|
+
|
|
363
|
+
```typescript
|
|
364
|
+
type AIProviderID =
|
|
365
|
+
| 'anthropic'
|
|
366
|
+
| 'openai'
|
|
367
|
+
| 'google'
|
|
368
|
+
| 'xai'
|
|
369
|
+
| 'deepseek'
|
|
370
|
+
| 'meta'
|
|
371
|
+
| 'moonshot'
|
|
372
|
+
| 'minimax'
|
|
373
|
+
| 'alibaba'
|
|
374
|
+
| 'zhipu'
|
|
375
|
+
/**
|
|
376
|
+
* A model served by a USER-configured endpoint + key (bring-your-own AI)
|
|
377
|
+
* rather than a platform bond. Never appears in the static catalog — hosts
|
|
378
|
+
* synthesize these definitions at runtime from per-project provider config,
|
|
379
|
+
* with all prices 0 (the user pays their own provider directly).
|
|
380
|
+
*/
|
|
381
|
+
| 'custom'
|
|
382
|
+
```
|
|
383
|
+
|
|
384
|
+
#### `EffortLevel`
|
|
385
|
+
|
|
386
|
+
A reasoning-effort value — a model's OWN native effort level.
|
|
387
|
+
|
|
388
|
+
There is no abstract cross-model scale: the value stored on a project and
|
|
389
|
+
sent to the provider IS the model's real level (e.g. `'high'` / `'xhigh'` /
|
|
390
|
+
`'max'` on current Claude models, `'medium'` on Grok, or a scaled
|
|
391
|
+
thinking-budget label like `'16K'` on budget-configurable models). Each model
|
|
392
|
+
declares its own ordered {@link ModelDefinition.supportedEffortLevels}; a
|
|
393
|
+
value that isn't in the active model's set degrades to the nearest one (see
|
|
394
|
+
`model-selection.ts`). Mirrored by the client-side `EffortLevel` in
|
|
395
|
+
`@molecule/app-ai-models`; keep the two in sync. Re-declared (rather than
|
|
396
|
+
imported) by the ide-react and molecule-dev consumers per the cross-stack
|
|
397
|
+
rule, but this catalog is the canonical home.
|
|
398
|
+
|
|
399
|
+
```typescript
|
|
400
|
+
type EffortLevel = string
|
|
401
|
+
```
|
|
402
|
+
|
|
403
|
+
### Functions
|
|
404
|
+
|
|
405
|
+
#### `effectiveModelRegion(modelDef, requested)`
|
|
406
|
+
|
|
407
|
+
Resolve a model's effective processing region: the requested region when the
|
|
408
|
+
model's {@link ModelDefinition.regions} list offers it, else the model's
|
|
409
|
+
DEFAULT region (the first listed; `'us'` when the catalog omits regions,
|
|
410
|
+
which also covers unknown model ids). A single-entry `regions` list pins the
|
|
411
|
+
model regardless of the request (e.g. a model with no US re-host).
|
|
412
|
+
|
|
413
|
+
```typescript
|
|
414
|
+
function effectiveModelRegion(modelDef: ModelDefinition | undefined, requested?: string): string
|
|
415
|
+
```
|
|
416
|
+
|
|
417
|
+
- `modelDef` — The model definition (or undefined for unknown ids).
|
|
418
|
+
- `requested` — The user's per-model region choice, if any.
|
|
419
|
+
|
|
420
|
+
**Returns:** The effective region code.
|
|
421
|
+
|
|
422
|
+
#### `getAvailableModels(availableProviders)`
|
|
423
|
+
|
|
424
|
+
Get models that are currently usable — filtered to only providers that are available.
|
|
425
|
+
|
|
426
|
+
The caller passes in which provider IDs are active (i.e. have a bond wired).
|
|
427
|
+
`disabled` models are excluded — they are never offered for selection.
|
|
428
|
+
|
|
429
|
+
```typescript
|
|
430
|
+
function getAvailableModels(
|
|
431
|
+
availableProviders: ReadonlySet<AIProviderID> | readonly AIProviderID[],
|
|
432
|
+
): readonly ModelDefinition[]
|
|
433
|
+
```
|
|
434
|
+
|
|
435
|
+
- `availableProviders` — Set or array of provider IDs that have active bonds.
|
|
436
|
+
|
|
437
|
+
**Returns:** Non-disabled models whose provider is in the available set.
|
|
438
|
+
|
|
439
|
+
#### `getModel(id)`
|
|
440
|
+
|
|
441
|
+
Look up a model definition by ID.
|
|
442
|
+
|
|
443
|
+
Returns `disabled` models too: a saved selection or a historical usage row
|
|
444
|
+
may reference a since-retired model, and it must stay priceable. Use
|
|
445
|
+
{@link MODEL_IDS} / {@link getAvailableModels} (which exclude disabled
|
|
446
|
+
models) to decide what is _selectable_.
|
|
447
|
+
|
|
448
|
+
```typescript
|
|
449
|
+
function getModel(id: string): ModelDefinition | undefined
|
|
450
|
+
```
|
|
451
|
+
|
|
452
|
+
- `id` — The API model ID.
|
|
453
|
+
|
|
454
|
+
**Returns:** The model definition, or `undefined` if not found.
|
|
455
|
+
|
|
456
|
+
#### `getModelsByProvider(provider)`
|
|
457
|
+
|
|
458
|
+
Get all models for a specific provider.
|
|
459
|
+
|
|
460
|
+
```typescript
|
|
461
|
+
function getModelsByProvider(provider: AIProviderID): readonly ModelDefinition[]
|
|
462
|
+
```
|
|
463
|
+
|
|
464
|
+
- `provider` — The provider ID.
|
|
465
|
+
|
|
466
|
+
**Returns:** Array of model definitions for that provider.
|
|
467
|
+
|
|
468
|
+
#### `list(_req, res)`
|
|
469
|
+
|
|
470
|
+
Returns models whose `provider` has a bond registered under the `'ai'`
|
|
471
|
+
category. When no AI providers are bonded the response is `{ models: [] }`,
|
|
472
|
+
which signals a misconfigured server rather than masking the issue.
|
|
473
|
+
|
|
474
|
+
Fails closed with `401` when there is no authenticated session, so the model
|
|
475
|
+
catalog is never disclosed to an unauthenticated caller even if the route's
|
|
476
|
+
`'authenticate'` middleware is dropped by codegen.
|
|
477
|
+
|
|
478
|
+
```typescript
|
|
479
|
+
function list(_req: MoleculeRequest, res: MoleculeResponse): Promise<void>
|
|
480
|
+
```
|
|
481
|
+
|
|
482
|
+
- `_req` — The request object (unused).
|
|
483
|
+
- `res` — The response object.
|
|
484
|
+
|
|
485
|
+
#### `modelRegionRates(modelDef, requested)`
|
|
486
|
+
|
|
487
|
+
The token rates for a model in a given processing region: the model's
|
|
488
|
+
{@link ModelDefinition.regionPricing} override for the region when one
|
|
489
|
+
exists, else the base rates (the native provider's list prices). Omitted
|
|
490
|
+
cache fields in an override fall back to the override's input price (hosts
|
|
491
|
+
with no cache discount / no write premium). The region is resolved via
|
|
492
|
+
{@link effectiveModelRegion}, so callers may pass the raw user choice.
|
|
493
|
+
|
|
494
|
+
```typescript
|
|
495
|
+
function modelRegionRates(modelDef: ModelDefinition, requested?: string): ModelTokenRates
|
|
496
|
+
```
|
|
497
|
+
|
|
498
|
+
- `modelDef` — The model definition.
|
|
499
|
+
- `requested` — The user's per-model region choice, if any.
|
|
500
|
+
|
|
501
|
+
**Returns:** The region-effective rates.
|
|
502
|
+
|
|
503
|
+
#### `priceMultiplierAt(modelDef, at)`
|
|
504
|
+
|
|
505
|
+
The price multiplier in effect for a model at a given instant.
|
|
506
|
+
|
|
507
|
+
Consults the model's {@link ModelDefinition.peakPricing} windows (UTC,
|
|
508
|
+
half-open, may wrap midnight). Metering MUST call this with each request's
|
|
509
|
+
own timestamp so peak-hour usage bills at the provider's real rate — pricing
|
|
510
|
+
everything at the flat rate silently under-meters peak traffic.
|
|
511
|
+
|
|
512
|
+
```typescript
|
|
513
|
+
function priceMultiplierAt(modelDef: ModelDefinition | undefined, at: Date): number
|
|
514
|
+
```
|
|
515
|
+
|
|
516
|
+
- `modelDef` — The model definition (or undefined).
|
|
517
|
+
- `at` — The instant the request was made.
|
|
518
|
+
|
|
519
|
+
**Returns:** The multiplier (`1` outside peak windows or when none are declared).
|
|
520
|
+
|
|
521
|
+
### Constants
|
|
522
|
+
|
|
523
|
+
#### `MODEL_IDS`
|
|
524
|
+
|
|
525
|
+
Set of _selectable_ model IDs for fast validation.
|
|
526
|
+
|
|
527
|
+
Excludes `disabled` models so a retired model (e.g. `grok-code-fast-1`) can
|
|
528
|
+
never be chosen for a new chat, while {@link getModel} still resolves it for
|
|
529
|
+
historical pricing.
|
|
530
|
+
|
|
531
|
+
```typescript
|
|
532
|
+
const MODEL_IDS: ReadonlySet<string>
|
|
533
|
+
```
|
|
534
|
+
|
|
535
|
+
#### `MODELS`
|
|
536
|
+
|
|
537
|
+
All available AI models, grouped by provider, ordered from most to least capable.
|
|
538
|
+
|
|
539
|
+
To add or remove a model, edit this array. Both the server-side validation
|
|
540
|
+
and the public discovery endpoint will update automatically.
|
|
541
|
+
|
|
542
|
+
Effort is each model's OWN native value — there is no abstract scale (see
|
|
543
|
+
{@link ModelDefinition.supportedEffortLevels}):
|
|
544
|
+
|
|
545
|
+
- A model driven by a provider-native effort/level param lists its provider
|
|
546
|
+
values verbatim in `supportedEffortLevels` (ascending), with
|
|
547
|
+
`defaultEffortLevel` = the provider's default/recommended level for agentic
|
|
548
|
+
coding. NO `effortBudgetTokens`.
|
|
549
|
+
- A model with a controllable token budget but no native level names (e.g.
|
|
550
|
+
Claude Haiku 4.5's `budget_tokens`, Qwen's `thinking_budget`) lists
|
|
551
|
+
scaled-budget labels (`['4K', '8K', '16K', '32K']`) with `effortBudgetTokens`
|
|
552
|
+
mapping each label to the token budget it sends.
|
|
553
|
+
- A model whose reasoning is fixed (always-on or on/off only, no depth
|
|
554
|
+
control) carries `thinkingConfigurable: false` and OMITS both fields —
|
|
555
|
+
there is nothing to tune.
|
|
556
|
+
|
|
557
|
+
Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
558
|
+
2026-07-30 GPT-5.6 repricing — cross-check prices against models.dev with
|
|
559
|
+
`npm run check:model-freshness` from the workspace root):
|
|
560
|
+
|
|
561
|
+
- Anthropic: https://platform.claude.com/docs/en/about-claude/models/overview
|
|
562
|
+
- /docs/en/build-with-claude/effort (fable-5 / opus-5 / sonnet-5 current;
|
|
563
|
+
opus-4-8 superseded by opus-5 at identical pricing but still served — it is
|
|
564
|
+
the recommended refusal-fallback model; effort ladder on all three current
|
|
565
|
+
models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
|
|
566
|
+
(re-verified 2026-08-06: added opus-4-7 — legacy but Active, $5/$25,
|
|
567
|
+
cache $0.50/$6.25, 1M ctx / 128K out per the overview + pricing pages;
|
|
568
|
+
models.dev first listed it 2026-08-06)
|
|
569
|
+
- OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
|
|
570
|
+
2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
|
|
571
|
+
20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
|
|
572
|
+
gpt-5.4 still listed as current; long-context 2× variants exist upstream —
|
|
573
|
+
not modeled, same as the Gemini/Grok tiers; Sol "Fast mode" 2.5× speed at
|
|
574
|
+
2× price announced 2026-07-30 — not yet modeled)
|
|
575
|
+
- Google: https://ai.google.dev/gemini-api/docs/pricing (gemini-3.6-flash GA
|
|
576
|
+
2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
577
|
+
gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
578
|
+
of 2026-07-28 despite the coming-soon badge; do not add until it has an id)
|
|
579
|
+
- xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
580
|
+
(grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
581
|
+
not modeled; reasoning_effort low|medium|high default high, image input;
|
|
582
|
+
grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
583
|
+
grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
584
|
+
- DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (unchanged V4
|
|
585
|
+
Pro/Flash pricing; legacy deepseek-chat/-reasoner ids fully retired
|
|
586
|
+
2026-07-24 — never in this catalog; the announced peak-hour 2× surcharge is
|
|
587
|
+
still NOT active as of 2026-07-28, see the entries)
|
|
588
|
+
- Moonshot: https://platform.kimi.ai/docs/models (kimi-k3 flagship 2026-07-16
|
|
589
|
+
— 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
590
|
+
reasoning_content that must be replayed through tool loops, the same
|
|
591
|
+
constraint that keeps kimi-k2.7-code out; add BOTH once the moonshot bond
|
|
592
|
+
supports preserved thinking + reasoning_effort low|high|max. kimi-k2.6
|
|
593
|
+
remains the newest model the bond can run correctly.)
|
|
594
|
+
- MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
595
|
+
minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
596
|
+
- Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
597
|
+
(qwen3.8-max GA'd 2026-08-03 on the pay-as-you-go international API and is
|
|
598
|
+
IN the catalog — verified 2026-08-04: flat $2/$6 per MTok on
|
|
599
|
+
help.aliyun.com/en/model-studio/model-pricing (Singapore International
|
|
600
|
+
CNY 14.988/44.965 at the same fixed conversion that maps qwen3.7-max's
|
|
601
|
+
CNY 18.736/56.207 to its $2.50/$7.50 list), 1M ctx, hybrid thinking,
|
|
602
|
+
tools. Cache rates come from the ZH context-cache doc
|
|
603
|
+
(help.aliyun.com/zh/model-studio/context-cache), which lists qwen3.8-max
|
|
604
|
+
as supported in every region under the unconditional standard table
|
|
605
|
+
(implicit: hit 20% of input, creation 100%; explicit: hit 10%, creation
|
|
606
|
+
125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
|
|
607
|
+
which an earlier pass misread as "excepted/console-only". qwen3.7-max
|
|
608
|
+
still runs its 50%-off promo — billed here at list, $2.50/$7.50)
|
|
609
|
+
- Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
|
|
610
|
+
the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
|
|
611
|
+
|
|
612
|
+
Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
613
|
+
where the provider doesn't publish one; the provider sources above verify
|
|
614
|
+
id / pricing / context window.
|
|
615
|
+
|
|
616
|
+
```typescript
|
|
617
|
+
const MODELS: readonly ModelDefinition[]
|
|
618
|
+
```
|
|
619
|
+
|
|
620
|
+
#### `requestHandlerMap`
|
|
621
|
+
|
|
622
|
+
Map of request handlers for the AI model catalog routes.
|
|
623
|
+
|
|
624
|
+
```typescript
|
|
625
|
+
const requestHandlerMap: { readonly list: typeof list }
|
|
626
|
+
```
|
|
627
|
+
|
|
628
|
+
#### `routes`
|
|
629
|
+
|
|
630
|
+
Route array for the AI model catalog: GET list of available models.
|
|
631
|
+
|
|
632
|
+
```typescript
|
|
633
|
+
const routes: readonly [
|
|
634
|
+
{
|
|
635
|
+
readonly method: 'get'
|
|
636
|
+
readonly path: '/ai/models'
|
|
637
|
+
readonly handler: 'list'
|
|
638
|
+
readonly middlewares: readonly ['authenticate']
|
|
639
|
+
},
|
|
640
|
+
]
|
|
641
|
+
```
|
|
642
|
+
|
|
643
|
+
## Injection Notes
|
|
644
|
+
|
|
645
|
+
### Requirements
|
|
646
|
+
|
|
647
|
+
Peer dependencies:
|
|
648
|
+
|
|
649
|
+
- `@molecule/api-bond` ^1.0.1
|
|
650
|
+
- `@molecule/api-i18n` ^1.0.1
|
|
651
|
+
- `@molecule/api-resource` ^1.0.1
|
|
652
|
+
|
|
653
|
+
### Runtime Dependencies
|
|
654
|
+
|
|
655
|
+
- `@molecule/api-bond`
|
|
656
|
+
- `@molecule/api-i18n`
|
|
657
|
+
- `@molecule/api-resource`
|
|
658
|
+
|
|
659
|
+
- **No database, no migration — the catalog is code.** Add/retire models by
|
|
660
|
+
editing `models.ts`; validation (`MODEL_IDS`) and the discovery endpoint
|
|
661
|
+
update automatically.
|
|
662
|
+
- **`GET /ai/models` only lists models whose provider is BONDED.** The handler
|
|
663
|
+
intersects `MODELS` with the names registered under the `'ai'` bond category,
|
|
664
|
+
so `bond('ai', '<name>', provider)` names must equal
|
|
665
|
+
`ModelDefinition.provider` ids. An empty `{ models: [] }` response means no
|
|
666
|
+
AI bond is wired — fix the wiring; never hardcode a model list client-side.
|
|
667
|
+
- **Disabled models stay resolvable on purpose.** `getModel(id)` returns
|
|
668
|
+
retired models so historical usage still prices correctly; gate what a user
|
|
669
|
+
may SELECT with `MODEL_IDS` / `getAvailableModels()`, never with `getModel()`.
|
|
670
|
+
- The list handler enforces authentication in-handler (fails closed 401) — if
|
|
671
|
+
you fork it, keep that check; route middleware alone can be stripped by
|
|
672
|
+
codegen.
|
package/dist/models.d.ts
CHANGED
|
@@ -36,6 +36,9 @@ import type { ModelDefinition } from './types.js';
|
|
|
36
36
|
* opus-4-8 superseded by opus-5 at identical pricing but still served — it is
|
|
37
37
|
* the recommended refusal-fallback model; effort ladder on all three current
|
|
38
38
|
* models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
|
|
39
|
+
* (re-verified 2026-08-06: added opus-4-7 — legacy but Active, $5/$25,
|
|
40
|
+
* cache $0.50/$6.25, 1M ctx / 128K out per the overview + pricing pages;
|
|
41
|
+
* models.dev first listed it 2026-08-06)
|
|
39
42
|
* - OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
|
|
40
43
|
* 2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
|
|
41
44
|
* 20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
|
|
@@ -64,9 +67,18 @@ import type { ModelDefinition } from './types.js';
|
|
|
64
67
|
* - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
65
68
|
* minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
66
69
|
* - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
67
|
-
* (
|
|
68
|
-
*
|
|
69
|
-
*
|
|
70
|
+
* (qwen3.8-max GA'd 2026-08-03 on the pay-as-you-go international API and is
|
|
71
|
+
* IN the catalog — verified 2026-08-04: flat $2/$6 per MTok on
|
|
72
|
+
* help.aliyun.com/en/model-studio/model-pricing (Singapore International
|
|
73
|
+
* CNY 14.988/44.965 at the same fixed conversion that maps qwen3.7-max's
|
|
74
|
+
* CNY 18.736/56.207 to its $2.50/$7.50 list), 1M ctx, hybrid thinking,
|
|
75
|
+
* tools. Cache rates come from the ZH context-cache doc
|
|
76
|
+
* (help.aliyun.com/zh/model-studio/context-cache), which lists qwen3.8-max
|
|
77
|
+
* as supported in every region under the unconditional standard table
|
|
78
|
+
* (implicit: hit 20% of input, creation 100%; explicit: hit 10%, creation
|
|
79
|
+
* 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
|
|
80
|
+
* which an earlier pass misread as "excepted/console-only". qwen3.7-max
|
|
81
|
+
* still runs its 50%-off promo — billed here at list, $2.50/$7.50)
|
|
70
82
|
* - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
|
|
71
83
|
* the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
|
|
72
84
|
*
|
package/dist/models.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6EG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAmtCnC,CAAA"}
|
package/dist/models.js
CHANGED
|
@@ -35,6 +35,9 @@
|
|
|
35
35
|
* opus-4-8 superseded by opus-5 at identical pricing but still served — it is
|
|
36
36
|
* the recommended refusal-fallback model; effort ladder on all three current
|
|
37
37
|
* models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
|
|
38
|
+
* (re-verified 2026-08-06: added opus-4-7 — legacy but Active, $5/$25,
|
|
39
|
+
* cache $0.50/$6.25, 1M ctx / 128K out per the overview + pricing pages;
|
|
40
|
+
* models.dev first listed it 2026-08-06)
|
|
38
41
|
* - OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
|
|
39
42
|
* 2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
|
|
40
43
|
* 20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
|
|
@@ -63,9 +66,18 @@
|
|
|
63
66
|
* - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
64
67
|
* minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
65
68
|
* - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
66
|
-
* (
|
|
67
|
-
*
|
|
68
|
-
*
|
|
69
|
+
* (qwen3.8-max GA'd 2026-08-03 on the pay-as-you-go international API and is
|
|
70
|
+
* IN the catalog — verified 2026-08-04: flat $2/$6 per MTok on
|
|
71
|
+
* help.aliyun.com/en/model-studio/model-pricing (Singapore International
|
|
72
|
+
* CNY 14.988/44.965 at the same fixed conversion that maps qwen3.7-max's
|
|
73
|
+
* CNY 18.736/56.207 to its $2.50/$7.50 list), 1M ctx, hybrid thinking,
|
|
74
|
+
* tools. Cache rates come from the ZH context-cache doc
|
|
75
|
+
* (help.aliyun.com/zh/model-studio/context-cache), which lists qwen3.8-max
|
|
76
|
+
* as supported in every region under the unconditional standard table
|
|
77
|
+
* (implicit: hit 20% of input, creation 100%; explicit: hit 10%, creation
|
|
78
|
+
* 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
|
|
79
|
+
* which an earlier pass misread as "excepted/console-only". qwen3.7-max
|
|
80
|
+
* still runs its 50%-off promo — billed here at list, $2.50/$7.50)
|
|
69
81
|
* - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
|
|
70
82
|
* the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
|
|
71
83
|
*
|
|
@@ -213,6 +225,40 @@ export const MODELS = [
|
|
|
213
225
|
cacheWritePricePerMTok: 3.75,
|
|
214
226
|
knowledgeCutoff: '2026-01-01',
|
|
215
227
|
},
|
|
228
|
+
{
|
|
229
|
+
id: 'claude-opus-4-7',
|
|
230
|
+
provider: 'anthropic',
|
|
231
|
+
label: 'Claude Opus 4.7',
|
|
232
|
+
description: 'Older Opus — long-horizon agentic work, knowledge work & vision',
|
|
233
|
+
contextWindow: 1_000_000,
|
|
234
|
+
maxOutputTokens: 128_000,
|
|
235
|
+
supportsThinking: true,
|
|
236
|
+
thinkingBudgetTokens: 16_000,
|
|
237
|
+
thinkingConfigurable: true,
|
|
238
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
|
|
239
|
+
defaultEffortLevel: 'high',
|
|
240
|
+
// Adaptive thinking only — budget_tokens is REJECTED (400); xhigh effort
|
|
241
|
+
// debuted on this model. Unlike opus-5, omitting `thinking` runs WITHOUT
|
|
242
|
+
// thinking — set {type:"adaptive"} explicitly. First model on the new
|
|
243
|
+
// tokenizer (~30% more tokens than 4.6 for the same text).
|
|
244
|
+
supportsVision: true,
|
|
245
|
+
supportsPromptCaching: true,
|
|
246
|
+
supportsTools: true,
|
|
247
|
+
webSearchToolType: 'web_search_20260209',
|
|
248
|
+
codeExecutionToolType: 'code_execution_20250825',
|
|
249
|
+
webFetchToolType: 'web_fetch_20260209',
|
|
250
|
+
inputPricePerMTok: 5,
|
|
251
|
+
outputPricePerMTok: 25,
|
|
252
|
+
// Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
|
|
253
|
+
cacheReadPricePerMTok: 0.5,
|
|
254
|
+
cacheWritePricePerMTok: 6.25,
|
|
255
|
+
knowledgeCutoff: '2026-01-01',
|
|
256
|
+
// Superseded by claude-opus-4-8 (launched 2026-05-28) at identical pricing;
|
|
257
|
+
// still Active upstream (deprecations page 2026-08-06: retires no sooner
|
|
258
|
+
// than 2027-04-16). Selectable under "Older models". NO fast mode —
|
|
259
|
+
// speed:"fast" on 4.7 returns an error (pricing page, fast-mode section).
|
|
260
|
+
deprecatedAt: '2026-05-28',
|
|
261
|
+
},
|
|
216
262
|
{
|
|
217
263
|
id: 'claude-opus-4-6',
|
|
218
264
|
provider: 'anthropic',
|
|
@@ -1097,6 +1143,42 @@ export const MODELS = [
|
|
|
1097
1143
|
// Prices are DashScope international list rates (the bond calls DashScope,
|
|
1098
1144
|
// not OpenRouter; a 50%-off promo currently applies — billed at list).
|
|
1099
1145
|
// ---------------------------------------------------------------------------
|
|
1146
|
+
// qwen3.8-max (GA 2026-08-03) succeeds qwen3.7-max as the agentic flagship,
|
|
1147
|
+
// priced BELOW it at $2/$6 (intl CNY 14.988/44.965, same fixed conversion).
|
|
1148
|
+
// Same hybrid thinking mechanism as the 3.7 series (enable_thinking default
|
|
1149
|
+
// ON + thinking_budget; preserve_thinking supported). Context cache uses the
|
|
1150
|
+
// standard implicit rates — see the Sources block for the ZH-doc citation.
|
|
1151
|
+
{
|
|
1152
|
+
id: 'qwen3.8-max',
|
|
1153
|
+
provider: 'alibaba',
|
|
1154
|
+
label: 'Qwen3.8 Max',
|
|
1155
|
+
description: 'Alibaba agentic flagship — 1M context, hybrid thinking',
|
|
1156
|
+
contextWindow: 1_000_000,
|
|
1157
|
+
// Alibaba's public pages don't state a max-output figure; models.dev says
|
|
1158
|
+
// 131,072 — kept at the 3.7-max figure until the provider publishes one
|
|
1159
|
+
// (understating only shortens completions; overstating would error).
|
|
1160
|
+
maxOutputTokens: 65_536,
|
|
1161
|
+
supportsThinking: true,
|
|
1162
|
+
thinkingBudgetTokens: 8_000,
|
|
1163
|
+
thinkingConfigurable: true,
|
|
1164
|
+
supportedEffortLevels: ['4K', '8K', '16K', '32K'],
|
|
1165
|
+
defaultEffortLevel: '8K',
|
|
1166
|
+
effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
|
|
1167
|
+
// models.dev claims image+video input, but Alibaba's own model catalog
|
|
1168
|
+
// lists qwen3.8-max under text generation (VL remains a separate line) —
|
|
1169
|
+
// false until the provider's page says otherwise.
|
|
1170
|
+
supportsVision: false,
|
|
1171
|
+
supportsPromptCaching: true,
|
|
1172
|
+
supportsTools: true,
|
|
1173
|
+
inputPricePerMTok: 2,
|
|
1174
|
+
outputPricePerMTok: 6,
|
|
1175
|
+
// Implicit context cache: read = 20% of input, no write premium.
|
|
1176
|
+
cacheReadPricePerMTok: 0.4,
|
|
1177
|
+
cacheWritePricePerMTok: 2,
|
|
1178
|
+
regions: ['us', 'cn'],
|
|
1179
|
+
// Not published by Alibaba — best-effort estimate.
|
|
1180
|
+
knowledgeCutoff: '2026-04-01',
|
|
1181
|
+
},
|
|
1100
1182
|
{
|
|
1101
1183
|
id: 'qwen3.7-max',
|
|
1102
1184
|
provider: 'alibaba',
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@molecule/api-resource-ai-models",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.2",
|
|
4
4
|
"description": "AI model catalog — server-side source of truth plus an authentication-gated discovery endpoint",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -17,7 +17,8 @@
|
|
|
17
17
|
}
|
|
18
18
|
},
|
|
19
19
|
"files": [
|
|
20
|
-
"dist"
|
|
20
|
+
"dist",
|
|
21
|
+
"README.md"
|
|
21
22
|
],
|
|
22
23
|
"keywords": [
|
|
23
24
|
"molecule",
|
|
@@ -26,17 +27,17 @@
|
|
|
26
27
|
],
|
|
27
28
|
"license": "Apache-2.0",
|
|
28
29
|
"devDependencies": {
|
|
29
|
-
"@molecule/api-bond": "1.0.
|
|
30
|
-
"@molecule/api-i18n": "1.0.
|
|
31
|
-
"@molecule/api-resource": "1.0.
|
|
30
|
+
"@molecule/api-bond": "1.0.1",
|
|
31
|
+
"@molecule/api-i18n": "1.0.1",
|
|
32
|
+
"@molecule/api-resource": "1.0.1",
|
|
32
33
|
"@types/node": "26.1.2",
|
|
33
34
|
"typescript": "6.0.3",
|
|
34
35
|
"vitest": "4.1.10"
|
|
35
36
|
},
|
|
36
37
|
"peerDependencies": {
|
|
37
|
-
"@molecule/api-bond": "^1.0.
|
|
38
|
-
"@molecule/api-i18n": "^1.0.
|
|
39
|
-
"@molecule/api-resource": "^1.0.
|
|
38
|
+
"@molecule/api-bond": "^1.0.1",
|
|
39
|
+
"@molecule/api-i18n": "^1.0.1",
|
|
40
|
+
"@molecule/api-resource": "^1.0.1"
|
|
40
41
|
},
|
|
41
42
|
"peerDependenciesMeta": {
|
|
42
43
|
"@molecule/api-i18n": {
|