@molecule/api-resource-ai-models 1.2.5 → 1.2.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -3
- package/dist/models.d.ts +40 -2
- package/dist/models.d.ts.map +1 -1
- package/dist/models.js +233 -33
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
|
|
|
3
3
|
Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
|
|
4
4
|
Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
|
|
5
5
|
To change this document, edit the module-level JSDoc in src/index.ts.
|
|
6
|
-
Generated: 2026-08-
|
|
6
|
+
Generated: 2026-08-28T08:14:50.991Z
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
# @molecule/api-resource-ai-models
|
|
@@ -728,6 +728,16 @@ All available AI models, grouped by provider, ordered from most to least capable
|
|
|
728
728
|
To add or remove a model, edit this array. Both the server-side validation
|
|
729
729
|
and the public discovery endpoint will update automatically.
|
|
730
730
|
|
|
731
|
+
EVERY entry field that shapes the outbound request (`webSearchToolType`,
|
|
732
|
+
`supportedEffortLevels`/`defaultEffortLevel`/`effortBudgetTokens`,
|
|
733
|
+
`maxOutputTokens`, `regions`) must hold for the model's ACTUAL serving
|
|
734
|
+
host(s) in every region — not just look right in a docs table. The
|
|
735
|
+
pre-commit hook enforces this with a LIVE Synthase-shaped probe of each
|
|
736
|
+
added/changed entry (molecule-dev `verify:model-dispatch --staged-catalog`);
|
|
737
|
+
a value the host rejects fails the whole request for every user
|
|
738
|
+
(glm-5.3-flash 2026-08-28: a `webSearchToolType` both zhipu hosts reject
|
|
739
|
+
made every Synthase turn 400/422 while the model itself worked fine).
|
|
740
|
+
|
|
731
741
|
Effort is each model's OWN native value — there is no abstract scale (see
|
|
732
742
|
{@link ModelDefinition.supportedEffortLevels}):
|
|
733
743
|
|
|
@@ -773,6 +783,12 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
773
783
|
gpt-5.4 still listed as current; long-context 2× variants exist upstream —
|
|
774
784
|
not modeled, same as the Gemini/Grok tiers; Sol "Fast mode" 2.5× speed at
|
|
775
785
|
2× price announced 2026-07-30 — not yet modeled)
|
|
786
|
+
(re-verified 2026-08-25: -sol on a >20% PROMO at $4/$20 (cache read $0.40,
|
|
787
|
+
cache write $5) since 2026-08-22, footnoted on that page as "available at
|
|
788
|
+
least through November 21, 2026" — catalog holds LIST $5/$30 so metering
|
|
789
|
+
never under-charges when it lapses, per KNOWN_DIVERGENCES; -terra and
|
|
790
|
+
-luna unchanged and matching the page; cache-read 0.1× and cache-write
|
|
791
|
+
1.25× are now published first-party for all three tiers)
|
|
776
792
|
- Google: https://ai.google.dev/gemini-api/docs/pricing (gemini-3.6-flash GA
|
|
777
793
|
2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
778
794
|
gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
@@ -783,6 +799,14 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
783
799
|
2026-12-31 billed here at list; specs from /docs/models/gemini-3.7-flash:
|
|
784
800
|
1M ctx / 65,536 out, thinking low|medium|high (no minimal), vision, tools,
|
|
785
801
|
caching, search grounding, code execution, url context)
|
|
802
|
+
(re-verified 2026-08-21: the pricing page shows that SAME launch promo on
|
|
803
|
+
gemini-3.6-flash too, verbatim "$0.75 through December 31, 2026. $1.50
|
|
804
|
+
starting January 1, 2027." for input, "$3.75 … $7.50" for output and
|
|
805
|
+
"$0.075 … $0.15" for caching read — i.e. the promo is flash-tier-wide, not
|
|
806
|
+
3.7-only as the 2026-08-13 note assumed. Both flash entries therefore stay
|
|
807
|
+
at the post-promo list rate ($1.50/$7.50, cache read $0.15) under the same
|
|
808
|
+
never-under-charge policy, each with a KNOWN_DIVERGENCES entry expiring
|
|
809
|
+
2026-12-31. gemini-3.5-flash carries no promo: still $1.50/$9.00, $0.15)
|
|
786
810
|
- xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
787
811
|
(grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
788
812
|
not modeled; reasoning_effort low|medium|high default high, image input;
|
|
@@ -827,8 +851,22 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
827
851
|
$1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
828
852
|
put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
829
853
|
publishes it as a distinct first-party DashScope model id.)
|
|
830
|
-
- Zhipu: https://docs.z.ai/guides/overview/pricing
|
|
831
|
-
|
|
854
|
+
- Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
855
|
+
glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
856
|
+
catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
|
857
|
+
128K out, text-only, reasoning_effort low|high|max with reasoning no longer
|
|
858
|
+
disableable. It supersedes glm-5.2: same tier, same base weights, gains are
|
|
859
|
+
post-training only. NOT on DeepInfra — api.deepinfra.com/models/zai-org/
|
|
860
|
+
GLM-5.3 returns "model not found", so it is cn-region-only for now and the
|
|
861
|
+
ZHIPU_US_MODEL_MAP needs no entry.
|
|
862
|
+
glm-5.3-flash (2026-08-26, natively multimodal, 1M ctx, 128K out) added
|
|
863
|
+
2026-08-27 at the LIST card ($0.15/$0.50, cached $0.03 — verified on the
|
|
864
|
+
Z.ai pricing page) so metering never under-charges during the 50%-off
|
|
865
|
+
launch promo ($0.075/$0.25) that runs to 2026-09-09 24:00 UTC+8; the promo
|
|
866
|
+
is a KNOWN_DIVERGENCES entry in check-model-freshness. US path is the
|
|
867
|
+
DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
868
|
+
($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
869
|
+
in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
832
870
|
|
|
833
871
|
Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
834
872
|
where the provider doesn't publish one; the provider sources above verify
|
package/dist/models.d.ts
CHANGED
|
@@ -14,6 +14,16 @@ import type { ModelDefinition } from './types.js';
|
|
|
14
14
|
* To add or remove a model, edit this array. Both the server-side validation
|
|
15
15
|
* and the public discovery endpoint will update automatically.
|
|
16
16
|
*
|
|
17
|
+
* EVERY entry field that shapes the outbound request (`webSearchToolType`,
|
|
18
|
+
* `supportedEffortLevels`/`defaultEffortLevel`/`effortBudgetTokens`,
|
|
19
|
+
* `maxOutputTokens`, `regions`) must hold for the model's ACTUAL serving
|
|
20
|
+
* host(s) in every region — not just look right in a docs table. The
|
|
21
|
+
* pre-commit hook enforces this with a LIVE Synthase-shaped probe of each
|
|
22
|
+
* added/changed entry (molecule-dev `verify:model-dispatch --staged-catalog`);
|
|
23
|
+
* a value the host rejects fails the whole request for every user
|
|
24
|
+
* (glm-5.3-flash 2026-08-28: a `webSearchToolType` both zhipu hosts reject
|
|
25
|
+
* made every Synthase turn 400/422 while the model itself worked fine).
|
|
26
|
+
*
|
|
17
27
|
* Effort is each model's OWN native value — there is no abstract scale (see
|
|
18
28
|
* {@link ModelDefinition.supportedEffortLevels}):
|
|
19
29
|
* - A model driven by a provider-native effort/level param lists its provider
|
|
@@ -57,6 +67,12 @@ import type { ModelDefinition } from './types.js';
|
|
|
57
67
|
* gpt-5.4 still listed as current; long-context 2× variants exist upstream —
|
|
58
68
|
* not modeled, same as the Gemini/Grok tiers; Sol "Fast mode" 2.5× speed at
|
|
59
69
|
* 2× price announced 2026-07-30 — not yet modeled)
|
|
70
|
+
* (re-verified 2026-08-25: -sol on a >20% PROMO at $4/$20 (cache read $0.40,
|
|
71
|
+
* cache write $5) since 2026-08-22, footnoted on that page as "available at
|
|
72
|
+
* least through November 21, 2026" — catalog holds LIST $5/$30 so metering
|
|
73
|
+
* never under-charges when it lapses, per KNOWN_DIVERGENCES; -terra and
|
|
74
|
+
* -luna unchanged and matching the page; cache-read 0.1× and cache-write
|
|
75
|
+
* 1.25× are now published first-party for all three tiers)
|
|
60
76
|
* - Google: https://ai.google.dev/gemini-api/docs/pricing (gemini-3.6-flash GA
|
|
61
77
|
* 2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
62
78
|
* gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
@@ -67,6 +83,14 @@ import type { ModelDefinition } from './types.js';
|
|
|
67
83
|
* 2026-12-31 billed here at list; specs from /docs/models/gemini-3.7-flash:
|
|
68
84
|
* 1M ctx / 65,536 out, thinking low|medium|high (no minimal), vision, tools,
|
|
69
85
|
* caching, search grounding, code execution, url context)
|
|
86
|
+
* (re-verified 2026-08-21: the pricing page shows that SAME launch promo on
|
|
87
|
+
* gemini-3.6-flash too, verbatim "$0.75 through December 31, 2026. $1.50
|
|
88
|
+
* starting January 1, 2027." for input, "$3.75 … $7.50" for output and
|
|
89
|
+
* "$0.075 … $0.15" for caching read — i.e. the promo is flash-tier-wide, not
|
|
90
|
+
* 3.7-only as the 2026-08-13 note assumed. Both flash entries therefore stay
|
|
91
|
+
* at the post-promo list rate ($1.50/$7.50, cache read $0.15) under the same
|
|
92
|
+
* never-under-charge policy, each with a KNOWN_DIVERGENCES entry expiring
|
|
93
|
+
* 2026-12-31. gemini-3.5-flash carries no promo: still $1.50/$9.00, $0.15)
|
|
70
94
|
* - xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
71
95
|
* (grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
72
96
|
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
@@ -111,8 +135,22 @@ import type { ModelDefinition } from './types.js';
|
|
|
111
135
|
* $1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
112
136
|
* put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
113
137
|
* publishes it as a distinct first-party DashScope model id.)
|
|
114
|
-
* - Zhipu: https://docs.z.ai/guides/overview/pricing
|
|
115
|
-
*
|
|
138
|
+
* - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
139
|
+
* glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
140
|
+
* catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
|
141
|
+
* 128K out, text-only, reasoning_effort low|high|max with reasoning no longer
|
|
142
|
+
* disableable. It supersedes glm-5.2: same tier, same base weights, gains are
|
|
143
|
+
* post-training only. NOT on DeepInfra — api.deepinfra.com/models/zai-org/
|
|
144
|
+
* GLM-5.3 returns "model not found", so it is cn-region-only for now and the
|
|
145
|
+
* ZHIPU_US_MODEL_MAP needs no entry.
|
|
146
|
+
* glm-5.3-flash (2026-08-26, natively multimodal, 1M ctx, 128K out) added
|
|
147
|
+
* 2026-08-27 at the LIST card ($0.15/$0.50, cached $0.03 — verified on the
|
|
148
|
+
* Z.ai pricing page) so metering never under-charges during the 50%-off
|
|
149
|
+
* launch promo ($0.075/$0.25) that runs to 2026-09-09 24:00 UTC+8; the promo
|
|
150
|
+
* is a KNOWN_DIVERGENCES entry in check-model-freshness. US path is the
|
|
151
|
+
* DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
152
|
+
* ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
153
|
+
* in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
116
154
|
*
|
|
117
155
|
* Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
118
156
|
* where the provider doesn't publish one; the provider sources above verify
|
package/dist/models.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAmJG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAsjDnC,CAAA"}
|
package/dist/models.js
CHANGED
|
@@ -13,6 +13,16 @@
|
|
|
13
13
|
* To add or remove a model, edit this array. Both the server-side validation
|
|
14
14
|
* and the public discovery endpoint will update automatically.
|
|
15
15
|
*
|
|
16
|
+
* EVERY entry field that shapes the outbound request (`webSearchToolType`,
|
|
17
|
+
* `supportedEffortLevels`/`defaultEffortLevel`/`effortBudgetTokens`,
|
|
18
|
+
* `maxOutputTokens`, `regions`) must hold for the model's ACTUAL serving
|
|
19
|
+
* host(s) in every region — not just look right in a docs table. The
|
|
20
|
+
* pre-commit hook enforces this with a LIVE Synthase-shaped probe of each
|
|
21
|
+
* added/changed entry (molecule-dev `verify:model-dispatch --staged-catalog`);
|
|
22
|
+
* a value the host rejects fails the whole request for every user
|
|
23
|
+
* (glm-5.3-flash 2026-08-28: a `webSearchToolType` both zhipu hosts reject
|
|
24
|
+
* made every Synthase turn 400/422 while the model itself worked fine).
|
|
25
|
+
*
|
|
16
26
|
* Effort is each model's OWN native value — there is no abstract scale (see
|
|
17
27
|
* {@link ModelDefinition.supportedEffortLevels}):
|
|
18
28
|
* - A model driven by a provider-native effort/level param lists its provider
|
|
@@ -56,6 +66,12 @@
|
|
|
56
66
|
* gpt-5.4 still listed as current; long-context 2× variants exist upstream —
|
|
57
67
|
* not modeled, same as the Gemini/Grok tiers; Sol "Fast mode" 2.5× speed at
|
|
58
68
|
* 2× price announced 2026-07-30 — not yet modeled)
|
|
69
|
+
* (re-verified 2026-08-25: -sol on a >20% PROMO at $4/$20 (cache read $0.40,
|
|
70
|
+
* cache write $5) since 2026-08-22, footnoted on that page as "available at
|
|
71
|
+
* least through November 21, 2026" — catalog holds LIST $5/$30 so metering
|
|
72
|
+
* never under-charges when it lapses, per KNOWN_DIVERGENCES; -terra and
|
|
73
|
+
* -luna unchanged and matching the page; cache-read 0.1× and cache-write
|
|
74
|
+
* 1.25× are now published first-party for all three tiers)
|
|
59
75
|
* - Google: https://ai.google.dev/gemini-api/docs/pricing (gemini-3.6-flash GA
|
|
60
76
|
* 2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
61
77
|
* gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
@@ -66,6 +82,14 @@
|
|
|
66
82
|
* 2026-12-31 billed here at list; specs from /docs/models/gemini-3.7-flash:
|
|
67
83
|
* 1M ctx / 65,536 out, thinking low|medium|high (no minimal), vision, tools,
|
|
68
84
|
* caching, search grounding, code execution, url context)
|
|
85
|
+
* (re-verified 2026-08-21: the pricing page shows that SAME launch promo on
|
|
86
|
+
* gemini-3.6-flash too, verbatim "$0.75 through December 31, 2026. $1.50
|
|
87
|
+
* starting January 1, 2027." for input, "$3.75 … $7.50" for output and
|
|
88
|
+
* "$0.075 … $0.15" for caching read — i.e. the promo is flash-tier-wide, not
|
|
89
|
+
* 3.7-only as the 2026-08-13 note assumed. Both flash entries therefore stay
|
|
90
|
+
* at the post-promo list rate ($1.50/$7.50, cache read $0.15) under the same
|
|
91
|
+
* never-under-charge policy, each with a KNOWN_DIVERGENCES entry expiring
|
|
92
|
+
* 2026-12-31. gemini-3.5-flash carries no promo: still $1.50/$9.00, $0.15)
|
|
69
93
|
* - xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
70
94
|
* (grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
71
95
|
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
@@ -110,8 +134,22 @@
|
|
|
110
134
|
* $1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
111
135
|
* put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
112
136
|
* publishes it as a distinct first-party DashScope model id.)
|
|
113
|
-
* - Zhipu: https://docs.z.ai/guides/overview/pricing
|
|
114
|
-
*
|
|
137
|
+
* - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
138
|
+
* glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
139
|
+
* catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
|
140
|
+
* 128K out, text-only, reasoning_effort low|high|max with reasoning no longer
|
|
141
|
+
* disableable. It supersedes glm-5.2: same tier, same base weights, gains are
|
|
142
|
+
* post-training only. NOT on DeepInfra — api.deepinfra.com/models/zai-org/
|
|
143
|
+
* GLM-5.3 returns "model not found", so it is cn-region-only for now and the
|
|
144
|
+
* ZHIPU_US_MODEL_MAP needs no entry.
|
|
145
|
+
* glm-5.3-flash (2026-08-26, natively multimodal, 1M ctx, 128K out) added
|
|
146
|
+
* 2026-08-27 at the LIST card ($0.15/$0.50, cached $0.03 — verified on the
|
|
147
|
+
* Z.ai pricing page) so metering never under-charges during the 50%-off
|
|
148
|
+
* launch promo ($0.075/$0.25) that runs to 2026-09-09 24:00 UTC+8; the promo
|
|
149
|
+
* is a KNOWN_DIVERGENCES entry in check-model-freshness. US path is the
|
|
150
|
+
* DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
151
|
+
* ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
152
|
+
* in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
115
153
|
*
|
|
116
154
|
* Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
117
155
|
* where the provider doesn't publish one; the provider sources above verify
|
|
@@ -146,7 +184,11 @@ export const MODELS = [
|
|
|
146
184
|
supportsPromptCaching: true,
|
|
147
185
|
supportsTools: true,
|
|
148
186
|
webSearchToolType: 'web_search_20260209',
|
|
149
|
-
|
|
187
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
188
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
189
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
190
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
191
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
150
192
|
webFetchToolType: 'web_fetch_20260209',
|
|
151
193
|
inputPricePerMTok: 10,
|
|
152
194
|
outputPricePerMTok: 50,
|
|
@@ -186,7 +228,11 @@ export const MODELS = [
|
|
|
186
228
|
supportsPromptCaching: true,
|
|
187
229
|
supportsTools: true,
|
|
188
230
|
webSearchToolType: 'web_search_20260209',
|
|
189
|
-
|
|
231
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
232
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
233
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
234
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
235
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
190
236
|
webFetchToolType: 'web_fetch_20260209',
|
|
191
237
|
// Drop-in successor to Opus 4.8 at identical pricing.
|
|
192
238
|
inputPricePerMTok: 5,
|
|
@@ -215,7 +261,11 @@ export const MODELS = [
|
|
|
215
261
|
supportsPromptCaching: true,
|
|
216
262
|
supportsTools: true,
|
|
217
263
|
webSearchToolType: 'web_search_20260209',
|
|
218
|
-
|
|
264
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
265
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
266
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
267
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
268
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
219
269
|
webFetchToolType: 'web_fetch_20260209',
|
|
220
270
|
inputPricePerMTok: 5,
|
|
221
271
|
outputPricePerMTok: 25,
|
|
@@ -248,7 +298,11 @@ export const MODELS = [
|
|
|
248
298
|
supportsPromptCaching: true,
|
|
249
299
|
supportsTools: true,
|
|
250
300
|
webSearchToolType: 'web_search_20260209',
|
|
251
|
-
|
|
301
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
302
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
303
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
304
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
305
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
252
306
|
webFetchToolType: 'web_fetch_20260209',
|
|
253
307
|
// Standard pricing. Intro pricing ($2/$10) applies through 2026-08-31 —
|
|
254
308
|
// billed here at standard so metering never under-charges; revisit after.
|
|
@@ -279,7 +333,11 @@ export const MODELS = [
|
|
|
279
333
|
supportsPromptCaching: true,
|
|
280
334
|
supportsTools: true,
|
|
281
335
|
webSearchToolType: 'web_search_20260209',
|
|
282
|
-
|
|
336
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
337
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
338
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
339
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
340
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
283
341
|
webFetchToolType: 'web_fetch_20260209',
|
|
284
342
|
inputPricePerMTok: 5,
|
|
285
343
|
outputPricePerMTok: 25,
|
|
@@ -314,7 +372,11 @@ export const MODELS = [
|
|
|
314
372
|
supportsPromptCaching: true,
|
|
315
373
|
supportsTools: true,
|
|
316
374
|
webSearchToolType: 'web_search_20260209',
|
|
317
|
-
|
|
375
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
376
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
377
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
378
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
379
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
318
380
|
webFetchToolType: 'web_fetch_20260209',
|
|
319
381
|
inputPricePerMTok: 5,
|
|
320
382
|
outputPricePerMTok: 25,
|
|
@@ -345,7 +407,11 @@ export const MODELS = [
|
|
|
345
407
|
supportsPromptCaching: true,
|
|
346
408
|
supportsTools: true,
|
|
347
409
|
webSearchToolType: 'web_search_20260209',
|
|
348
|
-
|
|
410
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
411
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
412
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
413
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
414
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
349
415
|
webFetchToolType: 'web_fetch_20260209',
|
|
350
416
|
inputPricePerMTok: 3,
|
|
351
417
|
outputPricePerMTok: 15,
|
|
@@ -387,16 +453,17 @@ export const MODELS = [
|
|
|
387
453
|
},
|
|
388
454
|
// ---------------------------------------------------------------------------
|
|
389
455
|
// OpenAI
|
|
390
|
-
// Verified: https://developers.openai.com/api/docs/pricing (2026-
|
|
456
|
+
// Verified: https://developers.openai.com/api/docs/pricing (2026-08-25)
|
|
391
457
|
// GPT-5.6 family GA 2026-07-09: gpt-5.6 is an alias for gpt-5.6-sol; Sol is
|
|
392
458
|
// the frontier tier, Terra the balanced tier, Luna the cheap/fast tier.
|
|
393
|
-
// Cached input is billed at 0.1× input
|
|
394
|
-
//
|
|
395
|
-
//
|
|
396
|
-
//
|
|
397
|
-
// not yet on the docs page — carried over from gpt-5.5 (low|medium|
|
|
398
|
-
// xhigh, default medium); re-verify. Long-context 2× price variants
|
|
399
|
-
// upstream — not modeled (same as the Gemini/Grok >200K tiers)
|
|
459
|
+
// Cached input is billed at 0.1× input, cache WRITES at 1.25× input — both
|
|
460
|
+
// now published first-party (the page grew explicit "Cached Input"/"Cache
|
|
461
|
+
// Writes" columns; the 1.25× the catalog had been carrying from third-party
|
|
462
|
+
// trackers is confirmed across all three tiers). reasoning_effort values for
|
|
463
|
+
// 5.6 are not yet on the docs page — carried over from gpt-5.5 (low|medium|
|
|
464
|
+
// high|xhigh, default medium); re-verify. Long-context 2× price variants
|
|
465
|
+
// exist upstream — not modeled (same as the Gemini/Grok >200K tiers), and
|
|
466
|
+
// neither are the Batch (0.5×) or Sol "Fast mode" (2×) cards.
|
|
400
467
|
// ---------------------------------------------------------------------------
|
|
401
468
|
{
|
|
402
469
|
id: 'gpt-5.6-sol',
|
|
@@ -417,11 +484,21 @@ export const MODELS = [
|
|
|
417
484
|
supportsTools: true,
|
|
418
485
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
419
486
|
toolsRequireReasoningOff: true,
|
|
420
|
-
webSearchToolType:
|
|
487
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
488
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
489
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
490
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
491
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
421
492
|
codeExecutionToolType: 'code_interpreter',
|
|
493
|
+
// LIST price. OpenAI ran a >20% PROMO from 2026-08-22 ($4/$20, cache read
|
|
494
|
+
// $0.40, cache write $5) — "GPT-5.6 Sol's promotional pricing is available
|
|
495
|
+
// at least through November 21, 2026" per the pricing page's own footnote.
|
|
496
|
+
// Billed here at list so metering never under-charges when it lapses; same
|
|
497
|
+
// policy as the gemini-flash and claude-sonnet-5 promos, and recorded in
|
|
498
|
+
// check-model-freshness.mjs KNOWN_DIVERGENCES (expires 2026-11-21).
|
|
422
499
|
inputPricePerMTok: 5,
|
|
423
500
|
outputPricePerMTok: 30,
|
|
424
|
-
// Cached input 0.1× input
|
|
501
|
+
// Cached input 0.1× input, cache write 1.25× input (see section note).
|
|
425
502
|
cacheReadPricePerMTok: 0.5,
|
|
426
503
|
cacheWritePricePerMTok: 6.25,
|
|
427
504
|
// Not published — best-effort estimate.
|
|
@@ -444,12 +521,16 @@ export const MODELS = [
|
|
|
444
521
|
supportsTools: true,
|
|
445
522
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
446
523
|
toolsRequireReasoningOff: true,
|
|
447
|
-
webSearchToolType:
|
|
524
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
525
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
526
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
527
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
528
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
448
529
|
codeExecutionToolType: 'code_interpreter',
|
|
449
530
|
// Repriced 2026-07-30 (20% cut from $2.50/$15).
|
|
450
531
|
inputPricePerMTok: 2,
|
|
451
532
|
outputPricePerMTok: 12,
|
|
452
|
-
// Cached input 0.1× input
|
|
533
|
+
// Cached input 0.1× input, cache write 1.25× input (see section note).
|
|
453
534
|
cacheReadPricePerMTok: 0.2,
|
|
454
535
|
cacheWritePricePerMTok: 2.5,
|
|
455
536
|
// Not published — best-effort estimate.
|
|
@@ -472,12 +553,16 @@ export const MODELS = [
|
|
|
472
553
|
supportsTools: true,
|
|
473
554
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
474
555
|
toolsRequireReasoningOff: true,
|
|
475
|
-
webSearchToolType:
|
|
556
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
557
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
558
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
559
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
560
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
476
561
|
codeExecutionToolType: 'code_interpreter',
|
|
477
562
|
// Repriced 2026-07-30 (80% cut from $1/$6).
|
|
478
563
|
inputPricePerMTok: 0.2,
|
|
479
564
|
outputPricePerMTok: 1.2,
|
|
480
|
-
// Cached input 0.1× input
|
|
565
|
+
// Cached input 0.1× input, cache write 1.25× input (see section note).
|
|
481
566
|
cacheReadPricePerMTok: 0.02,
|
|
482
567
|
cacheWritePricePerMTok: 0.25,
|
|
483
568
|
// The free-tier PLAN default (2026-08-18) — us-only OpenAI, so this carve-out
|
|
@@ -507,7 +592,11 @@ export const MODELS = [
|
|
|
507
592
|
supportsTools: true,
|
|
508
593
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
509
594
|
toolsRequireReasoningOff: true,
|
|
510
|
-
webSearchToolType:
|
|
595
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
596
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
597
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
598
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
599
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
511
600
|
codeExecutionToolType: 'code_interpreter',
|
|
512
601
|
inputPricePerMTok: 5,
|
|
513
602
|
outputPricePerMTok: 30,
|
|
@@ -537,7 +626,11 @@ export const MODELS = [
|
|
|
537
626
|
supportsTools: true,
|
|
538
627
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
539
628
|
toolsRequireReasoningOff: true,
|
|
540
|
-
webSearchToolType:
|
|
629
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
630
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
631
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
632
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
633
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
541
634
|
codeExecutionToolType: 'code_interpreter',
|
|
542
635
|
inputPricePerMTok: 2.5,
|
|
543
636
|
outputPricePerMTok: 15,
|
|
@@ -569,7 +662,11 @@ export const MODELS = [
|
|
|
569
662
|
supportsTools: true,
|
|
570
663
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
571
664
|
toolsRequireReasoningOff: true,
|
|
572
|
-
webSearchToolType:
|
|
665
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
666
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
667
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
668
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
669
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
573
670
|
codeExecutionToolType: 'code_interpreter',
|
|
574
671
|
inputPricePerMTok: 0.75,
|
|
575
672
|
outputPricePerMTok: 4.5,
|
|
@@ -654,6 +751,12 @@ export const MODELS = [
|
|
|
654
751
|
codeExecutionToolType: 'code_execution',
|
|
655
752
|
webFetchToolType: 'url_context',
|
|
656
753
|
// GA 2026-07-21 — same input price as 3.5-flash, CHEAPER output ($7.50 vs $9).
|
|
754
|
+
// LIST price $1.50/$7.50. Verified 2026-08-21: the pricing page runs the
|
|
755
|
+
// SAME flash-tier launch promo as 3.7-flash on this id ($0.75/$3.75, cache
|
|
756
|
+
// read $0.075) through 2026-12-31, reverting to list on 2027-01-01; billed
|
|
757
|
+
// here at list so metering never under-charges (identical policy and
|
|
758
|
+
// expiry to the gemini-3.7-flash entry above — see its matching
|
|
759
|
+
// KNOWN_DIVERGENCES entry in scripts/check-model-freshness.mjs).
|
|
657
760
|
inputPricePerMTok: 1.5,
|
|
658
761
|
outputPricePerMTok: 7.5,
|
|
659
762
|
// Gemini context cache: read $0.15/M (0.1× input), no write premium
|
|
@@ -1467,12 +1570,92 @@ export const MODELS = [
|
|
|
1467
1570
|
// Zhipu (GLM)
|
|
1468
1571
|
// Verified: https://docs.z.ai/guides/overview/pricing
|
|
1469
1572
|
// https://docs.z.ai/api-reference/llm/chat-completion
|
|
1470
|
-
// glm-5.
|
|
1471
|
-
//
|
|
1472
|
-
//
|
|
1473
|
-
//
|
|
1573
|
+
// glm-5.3 (2026-08-14) is the flagship — same base weights as glm-5.2, all
|
|
1574
|
+
// gains from post-training, same rate card ($1.40/$4.40, cached $0.26).
|
|
1575
|
+
// glm-5.2 took reasoning_effort minimal|none|low|medium|high|xhigh|max
|
|
1576
|
+
// (low/medium coerce to high, xhigh to max; minimal/none skip thinking).
|
|
1577
|
+
// glm-5.3 NARROWS that: low|high|max only, default max, and disabling
|
|
1578
|
+
// reasoning is no longer supported — so no minimal/none tier.
|
|
1474
1579
|
// glm-5 has thinking on/off only and was REPRICED (was $0.72/$2.30).
|
|
1475
1580
|
// ---------------------------------------------------------------------------
|
|
1581
|
+
{
|
|
1582
|
+
id: 'glm-5.3',
|
|
1583
|
+
provider: 'zhipu',
|
|
1584
|
+
label: 'GLM-5.3',
|
|
1585
|
+
description: 'Open-source SOTA agentic — 1M context',
|
|
1586
|
+
contextWindow: 1_048_576,
|
|
1587
|
+
maxOutputTokens: 131_072,
|
|
1588
|
+
supportsThinking: true,
|
|
1589
|
+
thinkingBudgetTokens: 8_000,
|
|
1590
|
+
thinkingConfigurable: true,
|
|
1591
|
+
supportedEffortLevels: ['low', 'high', 'max'],
|
|
1592
|
+
defaultEffortLevel: 'high',
|
|
1593
|
+
// Z.ai's own default is max (deep reasoning); high is the balanced tier we
|
|
1594
|
+
// default to, same call as glm-5.2. Thinking can NOT be turned off here.
|
|
1595
|
+
supportsVision: false,
|
|
1596
|
+
supportsPromptCaching: true,
|
|
1597
|
+
supportsTools: true,
|
|
1598
|
+
// The chat-completion reference gates web_search per model only in its
|
|
1599
|
+
// VISION section; for text models the tool is listed unconditionally, as it
|
|
1600
|
+
// was when glm-5.2 was cataloged.
|
|
1601
|
+
webSearchToolType: 'web_search',
|
|
1602
|
+
inputPricePerMTok: 1.4,
|
|
1603
|
+
outputPricePerMTok: 4.4,
|
|
1604
|
+
// GLM context cache: read ≈0.19× input, no write premium.
|
|
1605
|
+
cacheReadPricePerMTok: 0.26,
|
|
1606
|
+
cacheWritePricePerMTok: 1.4,
|
|
1607
|
+
// No US re-host exists — DeepInfra serves GLM-5.2 and GLM-5.3-Flash but
|
|
1608
|
+
// returns "model not found" for zai-org/GLM-5.3 (checked 2026-08-26), so
|
|
1609
|
+
// this is pinned to the native host and bills the list card above.
|
|
1610
|
+
regions: ['cn'],
|
|
1611
|
+
// Same base weights as glm-5.2, so the same best-effort estimate — Z.ai
|
|
1612
|
+
// publishes no cutoff.
|
|
1613
|
+
knowledgeCutoff: '2025-06-01',
|
|
1614
|
+
},
|
|
1615
|
+
{
|
|
1616
|
+
id: 'glm-5.3-flash',
|
|
1617
|
+
provider: 'zhipu',
|
|
1618
|
+
label: 'GLM-5.3 Flash',
|
|
1619
|
+
description: 'Native multimodal, efficient coding + agents — 1M context',
|
|
1620
|
+
contextWindow: 1_048_576,
|
|
1621
|
+
// "GLM-5.3-Flash ... support[s] a maximum output length of 128K" —
|
|
1622
|
+
// chat-completion API reference, checked 2026-08-27.
|
|
1623
|
+
maxOutputTokens: 131_072,
|
|
1624
|
+
supportsThinking: true,
|
|
1625
|
+
thinkingBudgetTokens: 8_000,
|
|
1626
|
+
thinkingConfigurable: true,
|
|
1627
|
+
// Same narrowed surface as glm-5.3: low|high|max only, thinking cannot be
|
|
1628
|
+
// disabled. Z.ai recommends max; high is the balanced tier we default to,
|
|
1629
|
+
// same call as glm-5.3.
|
|
1630
|
+
supportedEffortLevels: ['low', 'high', 'max'],
|
|
1631
|
+
defaultEffortLevel: 'high',
|
|
1632
|
+
// Native multimodal input: text, images, video, files.
|
|
1633
|
+
supportsVision: true,
|
|
1634
|
+
supportsPromptCaching: true,
|
|
1635
|
+
supportsTools: true,
|
|
1636
|
+
webSearchToolType: 'web_search',
|
|
1637
|
+
// LIST card. A 50%-off launch promo ($0.075/$0.25, cached $0.015) runs to
|
|
1638
|
+
// 2026-09-09 24:00 UTC+8; billed at list so metering never under-charges —
|
|
1639
|
+
// the promo is a KNOWN_DIVERGENCES entry in check-model-freshness.
|
|
1640
|
+
inputPricePerMTok: 0.15,
|
|
1641
|
+
outputPricePerMTok: 0.5,
|
|
1642
|
+
// GLM context cache: read 0.2× input, no write premium.
|
|
1643
|
+
cacheReadPricePerMTok: 0.03,
|
|
1644
|
+
cacheWritePricePerMTok: 0.15,
|
|
1645
|
+
// US default: DeepInfra (zai-org/GLM-5.3-Flash) bills exactly the list
|
|
1646
|
+
// card — $0.15/$0.50, cache read 0.2× = $0.03. Verified live 2026-08-27.
|
|
1647
|
+
regions: ['us', 'cn'],
|
|
1648
|
+
// NB: DeepInfra returns NO cached-token usage (verified live 2026-08-28:
|
|
1649
|
+
// an identical ~2.6k-token prefix twice reported full input both passes,
|
|
1650
|
+
// no prompt_tokens_details) — the us cacheRead rate below is informational
|
|
1651
|
+
// only; metering always bills full input there. Native z.ai reports and
|
|
1652
|
+
// discounts cached tokens correctly.
|
|
1653
|
+
regionPricing: {
|
|
1654
|
+
us: { inputPricePerMTok: 0.15, outputPricePerMTok: 0.5, cacheReadPricePerMTok: 0.03 },
|
|
1655
|
+
},
|
|
1656
|
+
// Not published by Z.ai — best-effort estimate for the 5.3 generation.
|
|
1657
|
+
knowledgeCutoff: '2025-06-01',
|
|
1658
|
+
},
|
|
1476
1659
|
{
|
|
1477
1660
|
id: 'glm-5.2',
|
|
1478
1661
|
provider: 'zhipu',
|
|
@@ -1498,11 +1681,22 @@ export const MODELS = [
|
|
|
1498
1681
|
cacheWritePricePerMTok: 1.4,
|
|
1499
1682
|
// US default (DeepInfra bills ~half native). Verified 2026-08-01.
|
|
1500
1683
|
regions: ['us', 'cn'],
|
|
1684
|
+
// NB: DeepInfra returns NO cached-token usage (verified live 2026-08-28:
|
|
1685
|
+
// an identical ~2.6k-token prefix twice reported full input both passes,
|
|
1686
|
+
// no prompt_tokens_details) — the us cacheRead rate below is informational
|
|
1687
|
+
// only; metering always bills full input there. Native z.ai reports and
|
|
1688
|
+
// discounts cached tokens correctly.
|
|
1501
1689
|
regionPricing: {
|
|
1502
1690
|
us: { inputPricePerMTok: 0.75, outputPricePerMTok: 2.4, cacheReadPricePerMTok: 0.14 },
|
|
1503
1691
|
},
|
|
1504
1692
|
// Not published by Z.ai — best-effort estimate.
|
|
1505
1693
|
knowledgeCutoff: '2025-06-01',
|
|
1694
|
+
// Superseded by glm-5.3 (same tier, same base weights, same rate card) —
|
|
1695
|
+
// kept priceable. Its DeepInfra US re-host was the cheaper way to run this
|
|
1696
|
+
// tier ($0.75/$2.40); glm-5.3 is not on that host yet, so the zhipu slot is
|
|
1697
|
+
// native-only until it is.
|
|
1698
|
+
deprecatedAt: '2026-08-26',
|
|
1699
|
+
supersededBy: 'glm-5.3',
|
|
1506
1700
|
},
|
|
1507
1701
|
{
|
|
1508
1702
|
id: 'glm-5',
|
|
@@ -1528,13 +1722,19 @@ export const MODELS = [
|
|
|
1528
1722
|
cacheWritePricePerMTok: 1,
|
|
1529
1723
|
// US default (DeepInfra bills below native). Verified 2026-08-01.
|
|
1530
1724
|
regions: ['us', 'cn'],
|
|
1725
|
+
// NB: DeepInfra returns NO cached-token usage (verified live 2026-08-28:
|
|
1726
|
+
// an identical ~2.6k-token prefix twice reported full input both passes,
|
|
1727
|
+
// no prompt_tokens_details) — the us cacheRead rate below is informational
|
|
1728
|
+
// only; metering always bills full input there. Native z.ai reports and
|
|
1729
|
+
// discounts cached tokens correctly.
|
|
1531
1730
|
regionPricing: {
|
|
1532
1731
|
us: { inputPricePerMTok: 0.6, outputPricePerMTok: 2.08, cacheReadPricePerMTok: 0.12 },
|
|
1533
1732
|
},
|
|
1534
1733
|
knowledgeCutoff: '2025-01-01',
|
|
1535
|
-
// Superseded by glm-5.
|
|
1536
|
-
// priceable.
|
|
1734
|
+
// Superseded by glm-5.3 (same line, bigger window, reasoning_effort) — kept
|
|
1735
|
+
// priceable. Points past glm-5.2, which is itself superseded: supersededBy
|
|
1736
|
+
// must name a SELECTABLE model so a saved selection resolves in one hop.
|
|
1537
1737
|
deprecatedAt: '2026-07-28',
|
|
1538
|
-
supersededBy: 'glm-5.
|
|
1738
|
+
supersededBy: 'glm-5.3',
|
|
1539
1739
|
},
|
|
1540
1740
|
];
|
package/package.json
CHANGED