@molecule/api-resource-ai-models 1.2.4 → 1.2.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +31 -3
- package/dist/models.d.ts +30 -2
- package/dist/models.d.ts.map +1 -1
- package/dist/models.js +143 -20
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
|
|
|
3
3
|
Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
|
|
4
4
|
Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
|
|
5
5
|
To change this document, edit the module-level JSDoc in src/index.ts.
|
|
6
|
-
Generated: 2026-08-
|
|
6
|
+
Generated: 2026-08-27T00:37:24.634Z
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
# @molecule/api-resource-ai-models
|
|
@@ -773,6 +773,12 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
773
773
|
gpt-5.4 still listed as current; long-context 2× variants exist upstream —
|
|
774
774
|
not modeled, same as the Gemini/Grok tiers; Sol "Fast mode" 2.5× speed at
|
|
775
775
|
2× price announced 2026-07-30 — not yet modeled)
|
|
776
|
+
(re-verified 2026-08-25: -sol on a >20% PROMO at $4/$20 (cache read $0.40,
|
|
777
|
+
cache write $5) since 2026-08-22, footnoted on that page as "available at
|
|
778
|
+
least through November 21, 2026" — catalog holds LIST $5/$30 so metering
|
|
779
|
+
never under-charges when it lapses, per KNOWN_DIVERGENCES; -terra and
|
|
780
|
+
-luna unchanged and matching the page; cache-read 0.1× and cache-write
|
|
781
|
+
1.25× are now published first-party for all three tiers)
|
|
776
782
|
- Google: https://ai.google.dev/gemini-api/docs/pricing (gemini-3.6-flash GA
|
|
777
783
|
2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
778
784
|
gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
@@ -783,6 +789,14 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
783
789
|
2026-12-31 billed here at list; specs from /docs/models/gemini-3.7-flash:
|
|
784
790
|
1M ctx / 65,536 out, thinking low|medium|high (no minimal), vision, tools,
|
|
785
791
|
caching, search grounding, code execution, url context)
|
|
792
|
+
(re-verified 2026-08-21: the pricing page shows that SAME launch promo on
|
|
793
|
+
gemini-3.6-flash too, verbatim "$0.75 through December 31, 2026. $1.50
|
|
794
|
+
starting January 1, 2027." for input, "$3.75 … $7.50" for output and
|
|
795
|
+
"$0.075 … $0.15" for caching read — i.e. the promo is flash-tier-wide, not
|
|
796
|
+
3.7-only as the 2026-08-13 note assumed. Both flash entries therefore stay
|
|
797
|
+
at the post-promo list rate ($1.50/$7.50, cache read $0.15) under the same
|
|
798
|
+
never-under-charge policy, each with a KNOWN_DIVERGENCES entry expiring
|
|
799
|
+
2026-12-31. gemini-3.5-flash carries no promo: still $1.50/$9.00, $0.15)
|
|
786
800
|
- xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
787
801
|
(grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
788
802
|
not modeled; reasoning_effort low|medium|high default high, image input;
|
|
@@ -827,8 +841,22 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
827
841
|
$1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
828
842
|
put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
829
843
|
publishes it as a distinct first-party DashScope model id.)
|
|
830
|
-
- Zhipu: https://docs.z.ai/guides/overview/pricing
|
|
831
|
-
|
|
844
|
+
- Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
845
|
+
glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
846
|
+
catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
|
847
|
+
128K out, text-only, reasoning_effort low|high|max with reasoning no longer
|
|
848
|
+
disableable. It supersedes glm-5.2: same tier, same base weights, gains are
|
|
849
|
+
post-training only. NOT on DeepInfra — api.deepinfra.com/models/zai-org/
|
|
850
|
+
GLM-5.3 returns "model not found", so it is cn-region-only for now and the
|
|
851
|
+
ZHIPU_US_MODEL_MAP needs no entry.
|
|
852
|
+
glm-5.3-flash (2026-08-26, natively multimodal, 1M ctx, 128K out) added
|
|
853
|
+
2026-08-27 at the LIST card ($0.15/$0.50, cached $0.03 — verified on the
|
|
854
|
+
Z.ai pricing page) so metering never under-charges during the 50%-off
|
|
855
|
+
launch promo ($0.075/$0.25) that runs to 2026-09-09 24:00 UTC+8; the promo
|
|
856
|
+
is a KNOWN_DIVERGENCES entry in check-model-freshness. US path is the
|
|
857
|
+
DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
858
|
+
($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
859
|
+
in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
832
860
|
|
|
833
861
|
Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
834
862
|
where the provider doesn't publish one; the provider sources above verify
|
package/dist/models.d.ts
CHANGED
|
@@ -57,6 +57,12 @@ import type { ModelDefinition } from './types.js';
|
|
|
57
57
|
* gpt-5.4 still listed as current; long-context 2× variants exist upstream —
|
|
58
58
|
* not modeled, same as the Gemini/Grok tiers; Sol "Fast mode" 2.5× speed at
|
|
59
59
|
* 2× price announced 2026-07-30 — not yet modeled)
|
|
60
|
+
* (re-verified 2026-08-25: -sol on a >20% PROMO at $4/$20 (cache read $0.40,
|
|
61
|
+
* cache write $5) since 2026-08-22, footnoted on that page as "available at
|
|
62
|
+
* least through November 21, 2026" — catalog holds LIST $5/$30 so metering
|
|
63
|
+
* never under-charges when it lapses, per KNOWN_DIVERGENCES; -terra and
|
|
64
|
+
* -luna unchanged and matching the page; cache-read 0.1× and cache-write
|
|
65
|
+
* 1.25× are now published first-party for all three tiers)
|
|
60
66
|
* - Google: https://ai.google.dev/gemini-api/docs/pricing (gemini-3.6-flash GA
|
|
61
67
|
* 2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
62
68
|
* gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
@@ -67,6 +73,14 @@ import type { ModelDefinition } from './types.js';
|
|
|
67
73
|
* 2026-12-31 billed here at list; specs from /docs/models/gemini-3.7-flash:
|
|
68
74
|
* 1M ctx / 65,536 out, thinking low|medium|high (no minimal), vision, tools,
|
|
69
75
|
* caching, search grounding, code execution, url context)
|
|
76
|
+
* (re-verified 2026-08-21: the pricing page shows that SAME launch promo on
|
|
77
|
+
* gemini-3.6-flash too, verbatim "$0.75 through December 31, 2026. $1.50
|
|
78
|
+
* starting January 1, 2027." for input, "$3.75 … $7.50" for output and
|
|
79
|
+
* "$0.075 … $0.15" for caching read — i.e. the promo is flash-tier-wide, not
|
|
80
|
+
* 3.7-only as the 2026-08-13 note assumed. Both flash entries therefore stay
|
|
81
|
+
* at the post-promo list rate ($1.50/$7.50, cache read $0.15) under the same
|
|
82
|
+
* never-under-charge policy, each with a KNOWN_DIVERGENCES entry expiring
|
|
83
|
+
* 2026-12-31. gemini-3.5-flash carries no promo: still $1.50/$9.00, $0.15)
|
|
70
84
|
* - xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
71
85
|
* (grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
72
86
|
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
@@ -111,8 +125,22 @@ import type { ModelDefinition } from './types.js';
|
|
|
111
125
|
* $1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
112
126
|
* put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
113
127
|
* publishes it as a distinct first-party DashScope model id.)
|
|
114
|
-
* - Zhipu: https://docs.z.ai/guides/overview/pricing
|
|
115
|
-
*
|
|
128
|
+
* - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
129
|
+
* glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
130
|
+
* catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
|
131
|
+
* 128K out, text-only, reasoning_effort low|high|max with reasoning no longer
|
|
132
|
+
* disableable. It supersedes glm-5.2: same tier, same base weights, gains are
|
|
133
|
+
* post-training only. NOT on DeepInfra — api.deepinfra.com/models/zai-org/
|
|
134
|
+
* GLM-5.3 returns "model not found", so it is cn-region-only for now and the
|
|
135
|
+
* ZHIPU_US_MODEL_MAP needs no entry.
|
|
136
|
+
* glm-5.3-flash (2026-08-26, natively multimodal, 1M ctx, 128K out) added
|
|
137
|
+
* 2026-08-27 at the LIST card ($0.15/$0.50, cached $0.03 — verified on the
|
|
138
|
+
* Z.ai pricing page) so metering never under-charges during the 50%-off
|
|
139
|
+
* launch promo ($0.075/$0.25) that runs to 2026-09-09 24:00 UTC+8; the promo
|
|
140
|
+
* is a KNOWN_DIVERGENCES entry in check-model-freshness. US path is the
|
|
141
|
+
* DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
142
|
+
* ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
143
|
+
* in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
116
144
|
*
|
|
117
145
|
* Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
118
146
|
* where the provider doesn't publish one; the provider sources above verify
|
package/dist/models.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAyIG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAm/CnC,CAAA"}
|
package/dist/models.js
CHANGED
|
@@ -56,6 +56,12 @@
|
|
|
56
56
|
* gpt-5.4 still listed as current; long-context 2× variants exist upstream —
|
|
57
57
|
* not modeled, same as the Gemini/Grok tiers; Sol "Fast mode" 2.5× speed at
|
|
58
58
|
* 2× price announced 2026-07-30 — not yet modeled)
|
|
59
|
+
* (re-verified 2026-08-25: -sol on a >20% PROMO at $4/$20 (cache read $0.40,
|
|
60
|
+
* cache write $5) since 2026-08-22, footnoted on that page as "available at
|
|
61
|
+
* least through November 21, 2026" — catalog holds LIST $5/$30 so metering
|
|
62
|
+
* never under-charges when it lapses, per KNOWN_DIVERGENCES; -terra and
|
|
63
|
+
* -luna unchanged and matching the page; cache-read 0.1× and cache-write
|
|
64
|
+
* 1.25× are now published first-party for all three tiers)
|
|
59
65
|
* - Google: https://ai.google.dev/gemini-api/docs/pricing (gemini-3.6-flash GA
|
|
60
66
|
* 2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
61
67
|
* gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
@@ -66,6 +72,14 @@
|
|
|
66
72
|
* 2026-12-31 billed here at list; specs from /docs/models/gemini-3.7-flash:
|
|
67
73
|
* 1M ctx / 65,536 out, thinking low|medium|high (no minimal), vision, tools,
|
|
68
74
|
* caching, search grounding, code execution, url context)
|
|
75
|
+
* (re-verified 2026-08-21: the pricing page shows that SAME launch promo on
|
|
76
|
+
* gemini-3.6-flash too, verbatim "$0.75 through December 31, 2026. $1.50
|
|
77
|
+
* starting January 1, 2027." for input, "$3.75 … $7.50" for output and
|
|
78
|
+
* "$0.075 … $0.15" for caching read — i.e. the promo is flash-tier-wide, not
|
|
79
|
+
* 3.7-only as the 2026-08-13 note assumed. Both flash entries therefore stay
|
|
80
|
+
* at the post-promo list rate ($1.50/$7.50, cache read $0.15) under the same
|
|
81
|
+
* never-under-charge policy, each with a KNOWN_DIVERGENCES entry expiring
|
|
82
|
+
* 2026-12-31. gemini-3.5-flash carries no promo: still $1.50/$9.00, $0.15)
|
|
69
83
|
* - xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
70
84
|
* (grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
71
85
|
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
@@ -110,8 +124,22 @@
|
|
|
110
124
|
* $1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
111
125
|
* put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
112
126
|
* publishes it as a distinct first-party DashScope model id.)
|
|
113
|
-
* - Zhipu: https://docs.z.ai/guides/overview/pricing
|
|
114
|
-
*
|
|
127
|
+
* - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
128
|
+
* glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
129
|
+
* catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
|
130
|
+
* 128K out, text-only, reasoning_effort low|high|max with reasoning no longer
|
|
131
|
+
* disableable. It supersedes glm-5.2: same tier, same base weights, gains are
|
|
132
|
+
* post-training only. NOT on DeepInfra — api.deepinfra.com/models/zai-org/
|
|
133
|
+
* GLM-5.3 returns "model not found", so it is cn-region-only for now and the
|
|
134
|
+
* ZHIPU_US_MODEL_MAP needs no entry.
|
|
135
|
+
* glm-5.3-flash (2026-08-26, natively multimodal, 1M ctx, 128K out) added
|
|
136
|
+
* 2026-08-27 at the LIST card ($0.15/$0.50, cached $0.03 — verified on the
|
|
137
|
+
* Z.ai pricing page) so metering never under-charges during the 50%-off
|
|
138
|
+
* launch promo ($0.075/$0.25) that runs to 2026-09-09 24:00 UTC+8; the promo
|
|
139
|
+
* is a KNOWN_DIVERGENCES entry in check-model-freshness. US path is the
|
|
140
|
+
* DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
141
|
+
* ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
142
|
+
* in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
115
143
|
*
|
|
116
144
|
* Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
117
145
|
* where the provider doesn't publish one; the provider sources above verify
|
|
@@ -387,16 +415,17 @@ export const MODELS = [
|
|
|
387
415
|
},
|
|
388
416
|
// ---------------------------------------------------------------------------
|
|
389
417
|
// OpenAI
|
|
390
|
-
// Verified: https://developers.openai.com/api/docs/pricing (2026-
|
|
418
|
+
// Verified: https://developers.openai.com/api/docs/pricing (2026-08-25)
|
|
391
419
|
// GPT-5.6 family GA 2026-07-09: gpt-5.6 is an alias for gpt-5.6-sol; Sol is
|
|
392
420
|
// the frontier tier, Terra the balanced tier, Luna the cheap/fast tier.
|
|
393
|
-
// Cached input is billed at 0.1× input
|
|
394
|
-
//
|
|
395
|
-
//
|
|
396
|
-
//
|
|
397
|
-
// not yet on the docs page — carried over from gpt-5.5 (low|medium|
|
|
398
|
-
// xhigh, default medium); re-verify. Long-context 2× price variants
|
|
399
|
-
// upstream — not modeled (same as the Gemini/Grok >200K tiers)
|
|
421
|
+
// Cached input is billed at 0.1× input, cache WRITES at 1.25× input — both
|
|
422
|
+
// now published first-party (the page grew explicit "Cached Input"/"Cache
|
|
423
|
+
// Writes" columns; the 1.25× the catalog had been carrying from third-party
|
|
424
|
+
// trackers is confirmed across all three tiers). reasoning_effort values for
|
|
425
|
+
// 5.6 are not yet on the docs page — carried over from gpt-5.5 (low|medium|
|
|
426
|
+
// high|xhigh, default medium); re-verify. Long-context 2× price variants
|
|
427
|
+
// exist upstream — not modeled (same as the Gemini/Grok >200K tiers), and
|
|
428
|
+
// neither are the Batch (0.5×) or Sol "Fast mode" (2×) cards.
|
|
400
429
|
// ---------------------------------------------------------------------------
|
|
401
430
|
{
|
|
402
431
|
id: 'gpt-5.6-sol',
|
|
@@ -419,9 +448,15 @@ export const MODELS = [
|
|
|
419
448
|
toolsRequireReasoningOff: true,
|
|
420
449
|
webSearchToolType: 'web_search',
|
|
421
450
|
codeExecutionToolType: 'code_interpreter',
|
|
451
|
+
// LIST price. OpenAI ran a >20% PROMO from 2026-08-22 ($4/$20, cache read
|
|
452
|
+
// $0.40, cache write $5) — "GPT-5.6 Sol's promotional pricing is available
|
|
453
|
+
// at least through November 21, 2026" per the pricing page's own footnote.
|
|
454
|
+
// Billed here at list so metering never under-charges when it lapses; same
|
|
455
|
+
// policy as the gemini-flash and claude-sonnet-5 promos, and recorded in
|
|
456
|
+
// check-model-freshness.mjs KNOWN_DIVERGENCES (expires 2026-11-21).
|
|
422
457
|
inputPricePerMTok: 5,
|
|
423
458
|
outputPricePerMTok: 30,
|
|
424
|
-
// Cached input 0.1× input
|
|
459
|
+
// Cached input 0.1× input, cache write 1.25× input (see section note).
|
|
425
460
|
cacheReadPricePerMTok: 0.5,
|
|
426
461
|
cacheWritePricePerMTok: 6.25,
|
|
427
462
|
// Not published — best-effort estimate.
|
|
@@ -449,7 +484,7 @@ export const MODELS = [
|
|
|
449
484
|
// Repriced 2026-07-30 (20% cut from $2.50/$15).
|
|
450
485
|
inputPricePerMTok: 2,
|
|
451
486
|
outputPricePerMTok: 12,
|
|
452
|
-
// Cached input 0.1× input
|
|
487
|
+
// Cached input 0.1× input, cache write 1.25× input (see section note).
|
|
453
488
|
cacheReadPricePerMTok: 0.2,
|
|
454
489
|
cacheWritePricePerMTok: 2.5,
|
|
455
490
|
// Not published — best-effort estimate.
|
|
@@ -477,7 +512,7 @@ export const MODELS = [
|
|
|
477
512
|
// Repriced 2026-07-30 (80% cut from $1/$6).
|
|
478
513
|
inputPricePerMTok: 0.2,
|
|
479
514
|
outputPricePerMTok: 1.2,
|
|
480
|
-
// Cached input 0.1× input
|
|
515
|
+
// Cached input 0.1× input, cache write 1.25× input (see section note).
|
|
481
516
|
cacheReadPricePerMTok: 0.02,
|
|
482
517
|
cacheWritePricePerMTok: 0.25,
|
|
483
518
|
// The free-tier PLAN default (2026-08-18) — us-only OpenAI, so this carve-out
|
|
@@ -654,6 +689,12 @@ export const MODELS = [
|
|
|
654
689
|
codeExecutionToolType: 'code_execution',
|
|
655
690
|
webFetchToolType: 'url_context',
|
|
656
691
|
// GA 2026-07-21 — same input price as 3.5-flash, CHEAPER output ($7.50 vs $9).
|
|
692
|
+
// LIST price $1.50/$7.50. Verified 2026-08-21: the pricing page runs the
|
|
693
|
+
// SAME flash-tier launch promo as 3.7-flash on this id ($0.75/$3.75, cache
|
|
694
|
+
// read $0.075) through 2026-12-31, reverting to list on 2027-01-01; billed
|
|
695
|
+
// here at list so metering never under-charges (identical policy and
|
|
696
|
+
// expiry to the gemini-3.7-flash entry above — see its matching
|
|
697
|
+
// KNOWN_DIVERGENCES entry in scripts/check-model-freshness.mjs).
|
|
657
698
|
inputPricePerMTok: 1.5,
|
|
658
699
|
outputPricePerMTok: 7.5,
|
|
659
700
|
// Gemini context cache: read $0.15/M (0.1× input), no write premium
|
|
@@ -1467,12 +1508,87 @@ export const MODELS = [
|
|
|
1467
1508
|
// Zhipu (GLM)
|
|
1468
1509
|
// Verified: https://docs.z.ai/guides/overview/pricing
|
|
1469
1510
|
// https://docs.z.ai/api-reference/llm/chat-completion
|
|
1470
|
-
// glm-5.
|
|
1471
|
-
//
|
|
1472
|
-
//
|
|
1473
|
-
//
|
|
1511
|
+
// glm-5.3 (2026-08-14) is the flagship — same base weights as glm-5.2, all
|
|
1512
|
+
// gains from post-training, same rate card ($1.40/$4.40, cached $0.26).
|
|
1513
|
+
// glm-5.2 took reasoning_effort minimal|none|low|medium|high|xhigh|max
|
|
1514
|
+
// (low/medium coerce to high, xhigh to max; minimal/none skip thinking).
|
|
1515
|
+
// glm-5.3 NARROWS that: low|high|max only, default max, and disabling
|
|
1516
|
+
// reasoning is no longer supported — so no minimal/none tier.
|
|
1474
1517
|
// glm-5 has thinking on/off only and was REPRICED (was $0.72/$2.30).
|
|
1475
1518
|
// ---------------------------------------------------------------------------
|
|
1519
|
+
{
|
|
1520
|
+
id: 'glm-5.3',
|
|
1521
|
+
provider: 'zhipu',
|
|
1522
|
+
label: 'GLM-5.3',
|
|
1523
|
+
description: 'Open-source SOTA agentic — 1M context',
|
|
1524
|
+
contextWindow: 1_048_576,
|
|
1525
|
+
maxOutputTokens: 131_072,
|
|
1526
|
+
supportsThinking: true,
|
|
1527
|
+
thinkingBudgetTokens: 8_000,
|
|
1528
|
+
thinkingConfigurable: true,
|
|
1529
|
+
supportedEffortLevels: ['low', 'high', 'max'],
|
|
1530
|
+
defaultEffortLevel: 'high',
|
|
1531
|
+
// Z.ai's own default is max (deep reasoning); high is the balanced tier we
|
|
1532
|
+
// default to, same call as glm-5.2. Thinking can NOT be turned off here.
|
|
1533
|
+
supportsVision: false,
|
|
1534
|
+
supportsPromptCaching: true,
|
|
1535
|
+
supportsTools: true,
|
|
1536
|
+
// The chat-completion reference gates web_search per model only in its
|
|
1537
|
+
// VISION section; for text models the tool is listed unconditionally, as it
|
|
1538
|
+
// was when glm-5.2 was cataloged.
|
|
1539
|
+
webSearchToolType: 'web_search',
|
|
1540
|
+
inputPricePerMTok: 1.4,
|
|
1541
|
+
outputPricePerMTok: 4.4,
|
|
1542
|
+
// GLM context cache: read ≈0.19× input, no write premium.
|
|
1543
|
+
cacheReadPricePerMTok: 0.26,
|
|
1544
|
+
cacheWritePricePerMTok: 1.4,
|
|
1545
|
+
// No US re-host exists — DeepInfra serves GLM-5.2 and GLM-5.3-Flash but
|
|
1546
|
+
// returns "model not found" for zai-org/GLM-5.3 (checked 2026-08-26), so
|
|
1547
|
+
// this is pinned to the native host and bills the list card above.
|
|
1548
|
+
regions: ['cn'],
|
|
1549
|
+
// Same base weights as glm-5.2, so the same best-effort estimate — Z.ai
|
|
1550
|
+
// publishes no cutoff.
|
|
1551
|
+
knowledgeCutoff: '2025-06-01',
|
|
1552
|
+
},
|
|
1553
|
+
{
|
|
1554
|
+
id: 'glm-5.3-flash',
|
|
1555
|
+
provider: 'zhipu',
|
|
1556
|
+
label: 'GLM-5.3 Flash',
|
|
1557
|
+
description: 'Native multimodal, efficient coding + agents — 1M context',
|
|
1558
|
+
contextWindow: 1_048_576,
|
|
1559
|
+
// "GLM-5.3-Flash ... support[s] a maximum output length of 128K" —
|
|
1560
|
+
// chat-completion API reference, checked 2026-08-27.
|
|
1561
|
+
maxOutputTokens: 131_072,
|
|
1562
|
+
supportsThinking: true,
|
|
1563
|
+
thinkingBudgetTokens: 8_000,
|
|
1564
|
+
thinkingConfigurable: true,
|
|
1565
|
+
// Same narrowed surface as glm-5.3: low|high|max only, thinking cannot be
|
|
1566
|
+
// disabled. Z.ai recommends max; high is the balanced tier we default to,
|
|
1567
|
+
// same call as glm-5.3.
|
|
1568
|
+
supportedEffortLevels: ['low', 'high', 'max'],
|
|
1569
|
+
defaultEffortLevel: 'high',
|
|
1570
|
+
// Native multimodal input: text, images, video, files.
|
|
1571
|
+
supportsVision: true,
|
|
1572
|
+
supportsPromptCaching: true,
|
|
1573
|
+
supportsTools: true,
|
|
1574
|
+
webSearchToolType: 'web_search',
|
|
1575
|
+
// LIST card. A 50%-off launch promo ($0.075/$0.25, cached $0.015) runs to
|
|
1576
|
+
// 2026-09-09 24:00 UTC+8; billed at list so metering never under-charges —
|
|
1577
|
+
// the promo is a KNOWN_DIVERGENCES entry in check-model-freshness.
|
|
1578
|
+
inputPricePerMTok: 0.15,
|
|
1579
|
+
outputPricePerMTok: 0.5,
|
|
1580
|
+
// GLM context cache: read 0.2× input, no write premium.
|
|
1581
|
+
cacheReadPricePerMTok: 0.03,
|
|
1582
|
+
cacheWritePricePerMTok: 0.15,
|
|
1583
|
+
// US default: DeepInfra (zai-org/GLM-5.3-Flash) bills exactly the list
|
|
1584
|
+
// card — $0.15/$0.50, cache read 0.2× = $0.03. Verified live 2026-08-27.
|
|
1585
|
+
regions: ['us', 'cn'],
|
|
1586
|
+
regionPricing: {
|
|
1587
|
+
us: { inputPricePerMTok: 0.15, outputPricePerMTok: 0.5, cacheReadPricePerMTok: 0.03 },
|
|
1588
|
+
},
|
|
1589
|
+
// Not published by Z.ai — best-effort estimate for the 5.3 generation.
|
|
1590
|
+
knowledgeCutoff: '2025-06-01',
|
|
1591
|
+
},
|
|
1476
1592
|
{
|
|
1477
1593
|
id: 'glm-5.2',
|
|
1478
1594
|
provider: 'zhipu',
|
|
@@ -1503,6 +1619,12 @@ export const MODELS = [
|
|
|
1503
1619
|
},
|
|
1504
1620
|
// Not published by Z.ai — best-effort estimate.
|
|
1505
1621
|
knowledgeCutoff: '2025-06-01',
|
|
1622
|
+
// Superseded by glm-5.3 (same tier, same base weights, same rate card) —
|
|
1623
|
+
// kept priceable. Its DeepInfra US re-host was the cheaper way to run this
|
|
1624
|
+
// tier ($0.75/$2.40); glm-5.3 is not on that host yet, so the zhipu slot is
|
|
1625
|
+
// native-only until it is.
|
|
1626
|
+
deprecatedAt: '2026-08-26',
|
|
1627
|
+
supersededBy: 'glm-5.3',
|
|
1506
1628
|
},
|
|
1507
1629
|
{
|
|
1508
1630
|
id: 'glm-5',
|
|
@@ -1532,9 +1654,10 @@ export const MODELS = [
|
|
|
1532
1654
|
us: { inputPricePerMTok: 0.6, outputPricePerMTok: 2.08, cacheReadPricePerMTok: 0.12 },
|
|
1533
1655
|
},
|
|
1534
1656
|
knowledgeCutoff: '2025-01-01',
|
|
1535
|
-
// Superseded by glm-5.
|
|
1536
|
-
// priceable.
|
|
1657
|
+
// Superseded by glm-5.3 (same line, bigger window, reasoning_effort) — kept
|
|
1658
|
+
// priceable. Points past glm-5.2, which is itself superseded: supersededBy
|
|
1659
|
+
// must name a SELECTABLE model so a saved selection resolves in one hop.
|
|
1537
1660
|
deprecatedAt: '2026-07-28',
|
|
1538
|
-
supersededBy: 'glm-5.
|
|
1661
|
+
supersededBy: 'glm-5.3',
|
|
1539
1662
|
},
|
|
1540
1663
|
];
|
package/package.json
CHANGED