@yanlinglabs/winter-provider-catalog 0.0.23 → 0.0.24

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -205,7 +205,9 @@
205
205
  "confidence": "declared",
206
206
  "observedAt": "2026-09-05T00:00:00Z"
207
207
  },
208
- "unsupportedParameters": [],
208
+ "unsupportedParameters": [
209
+ "thinking.type.adaptive"
210
+ ],
209
211
  "status": "candidate"
210
212
  },
211
213
  {
@@ -345,6 +347,15 @@
345
347
  "sourceRef": "continuity report §4.4, BOTH halves. Own-state acceptance: the model's own `thinking` blocks are replayed to it unchanged, in order, with their signatures intact (§4.4's replay rules and its worked request). Why the domain is NARROW: \"when changing Claude models, prior `thinking` and `redacted_thinking` blocks should be stripped because they are tied to the model that produced them. Therefore 'same provider' is not automatically 'same continuation domain.'\" The domain is this model alone, never the Anthropic provider.",
346
348
  "confidence": "declared",
347
349
  "observedAt": "2026-09-05T00:00:00Z"
350
+ },
351
+ "effortRequest": {
352
+ "value": {
353
+ "field": "output_config.effort"
354
+ },
355
+ "source": "official-doc",
356
+ "confidence": "declared",
357
+ "observedAt": "2026-09-25T12:30:00Z",
358
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
348
359
  }
349
360
  },
350
361
  "pricing": {
@@ -362,7 +373,8 @@
362
373
  "unsupportedParameters": [
363
374
  "temperature",
364
375
  "top_p",
365
- "top_k"
376
+ "top_k",
377
+ "thinking.type.enabled"
366
378
  ],
367
379
  "status": "candidate"
368
380
  },
@@ -505,6 +517,15 @@
505
517
  "sourceRef": "continuity report §4.4, BOTH halves. Own-state acceptance: §4.4 requires the model's own assistant blocks — `thinking` with its signature, `redacted_thinking` — replayed unchanged and in order across a tool loop. Why the domain is NARROW: Anthropic documents model switching as a boundary at which those blocks are STRIPPED, because they are tied to the producing model; \"same provider\" is therefore not \"same continuation domain\", and this domain is the model alone.",
506
518
  "confidence": "declared",
507
519
  "observedAt": "2026-09-05T00:00:00Z"
520
+ },
521
+ "effortRequest": {
522
+ "value": {
523
+ "field": "output_config.effort"
524
+ },
525
+ "source": "official-doc",
526
+ "confidence": "declared",
527
+ "observedAt": "2026-09-25T12:30:00Z",
528
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
508
529
  }
509
530
  },
510
531
  "pricing": {
@@ -522,7 +543,8 @@
522
543
  "unsupportedParameters": [
523
544
  "temperature",
524
545
  "top_p",
525
- "top_k"
546
+ "top_k",
547
+ "thinking.type.enabled"
526
548
  ],
527
549
  "status": "candidate"
528
550
  },
@@ -620,7 +642,16 @@
620
642
  "max"
621
643
  ],
622
644
  "continuation": "opaque-provider-state",
623
- "defaultEffort": "high"
645
+ "defaultEffort": "high",
646
+ "effortRequest": {
647
+ "value": {
648
+ "field": "output_config.effort"
649
+ },
650
+ "source": "official-doc",
651
+ "confidence": "declared",
652
+ "observedAt": "2026-09-25T12:30:00Z",
653
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
654
+ }
624
655
  },
625
656
  "pricing": {
626
657
  "value": {
@@ -637,7 +668,9 @@
637
668
  "unsupportedParameters": [
638
669
  "temperature",
639
670
  "top_p",
640
- "top_k"
671
+ "top_k",
672
+ "thinking.type.enabled",
673
+ "thinking.type.disabled"
641
674
  ],
642
675
  "status": "candidate"
643
676
  },
@@ -733,7 +766,9 @@
733
766
  "confidence": "declared",
734
767
  "observedAt": "2026-09-19T00:00:00Z"
735
768
  },
736
- "unsupportedParameters": [],
769
+ "unsupportedParameters": [
770
+ "thinking.type.adaptive"
771
+ ],
737
772
  "status": "candidate"
738
773
  },
739
774
  {
@@ -831,7 +866,16 @@
831
866
  "high"
832
867
  ],
833
868
  "continuation": "opaque-provider-state",
834
- "defaultEffort": "high"
869
+ "defaultEffort": "high",
870
+ "effortRequest": {
871
+ "value": {
872
+ "field": "output_config.effort"
873
+ },
874
+ "source": "official-doc",
875
+ "confidence": "declared",
876
+ "observedAt": "2026-09-25T12:30:00Z",
877
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
878
+ }
835
879
  },
836
880
  "pricing": {
837
881
  "value": {
@@ -845,7 +889,9 @@
845
889
  "confidence": "declared",
846
890
  "observedAt": "2026-09-19T00:00:00Z"
847
891
  },
848
- "unsupportedParameters": [],
892
+ "unsupportedParameters": [
893
+ "thinking.type.adaptive"
894
+ ],
849
895
  "status": "candidate"
850
896
  },
851
897
  {
@@ -941,7 +987,16 @@
941
987
  "max"
942
988
  ],
943
989
  "continuation": "opaque-provider-state",
944
- "defaultEffort": "high"
990
+ "defaultEffort": "high",
991
+ "effortRequest": {
992
+ "value": {
993
+ "field": "output_config.effort"
994
+ },
995
+ "source": "official-doc",
996
+ "confidence": "declared",
997
+ "observedAt": "2026-09-25T12:30:00Z",
998
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
999
+ }
945
1000
  },
946
1001
  "pricing": {
947
1002
  "value": {
@@ -1052,7 +1107,16 @@
1052
1107
  "max"
1053
1108
  ],
1054
1109
  "continuation": "opaque-provider-state",
1055
- "defaultEffort": "high"
1110
+ "defaultEffort": "high",
1111
+ "effortRequest": {
1112
+ "value": {
1113
+ "field": "output_config.effort"
1114
+ },
1115
+ "source": "official-doc",
1116
+ "confidence": "declared",
1117
+ "observedAt": "2026-09-25T12:30:00Z",
1118
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
1119
+ }
1056
1120
  },
1057
1121
  "pricing": {
1058
1122
  "value": {
@@ -1069,7 +1133,8 @@
1069
1133
  "unsupportedParameters": [
1070
1134
  "temperature",
1071
1135
  "top_p",
1072
- "top_k"
1136
+ "top_k",
1137
+ "thinking.type.enabled"
1073
1138
  ],
1074
1139
  "status": "candidate"
1075
1140
  },
@@ -1167,7 +1232,16 @@
1167
1232
  "max"
1168
1233
  ],
1169
1234
  "continuation": "opaque-provider-state",
1170
- "defaultEffort": "high"
1235
+ "defaultEffort": "high",
1236
+ "effortRequest": {
1237
+ "value": {
1238
+ "field": "output_config.effort"
1239
+ },
1240
+ "source": "official-doc",
1241
+ "confidence": "declared",
1242
+ "observedAt": "2026-09-25T12:30:00Z",
1243
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
1244
+ }
1171
1245
  },
1172
1246
  "pricing": {
1173
1247
  "value": {
@@ -1184,7 +1258,8 @@
1184
1258
  "unsupportedParameters": [
1185
1259
  "temperature",
1186
1260
  "top_p",
1187
- "top_k"
1261
+ "top_k",
1262
+ "thinking.type.enabled"
1188
1263
  ],
1189
1264
  "status": "candidate"
1190
1265
  },
@@ -1297,7 +1372,9 @@
1297
1372
  "confidence": "declared",
1298
1373
  "observedAt": "2026-09-19T00:00:00Z"
1299
1374
  },
1300
- "unsupportedParameters": [],
1375
+ "unsupportedParameters": [
1376
+ "thinking.type.adaptive"
1377
+ ],
1301
1378
  "status": "candidate"
1302
1379
  },
1303
1380
  {
@@ -1393,7 +1470,16 @@
1393
1470
  "max"
1394
1471
  ],
1395
1472
  "continuation": "opaque-provider-state",
1396
- "defaultEffort": "high"
1473
+ "defaultEffort": "high",
1474
+ "effortRequest": {
1475
+ "value": {
1476
+ "field": "output_config.effort"
1477
+ },
1478
+ "source": "official-doc",
1479
+ "confidence": "declared",
1480
+ "observedAt": "2026-09-25T12:30:00Z",
1481
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
1482
+ }
1397
1483
  },
1398
1484
  "pricing": {
1399
1485
  "value": {
@@ -2352,7 +2438,7 @@
2352
2438
  "deepseek-v4-flash",
2353
2439
  "deepseek-v4-flash-vision-exp"
2354
2440
  ],
2355
- "$comment": "DeepSeek's official model list now names `deepseek-flash` as V4.1 Flash. The legacy aliases are temporarily routed to it. Winter drives its OpenAI Chat Completions surface, whose documented `reasoning_content` replay contract is therefore carried only on this direct-dialect row. — 2026-09-25 refresh: Chat uses thinking.type enabled/disabled plus reasoning_effort none/low/high/max (none disables); Responses uses reasoning.effort. https://api-docs.deepseek.com/guides/responses_api/ guide table still says only Flash but https://api-docs.deepseek.com/api/create-response/ model enum and https://api-docs.deepseek.com/quick_start/pricing/ feature table both list Pro too. Responses stateless: no previous_response_id. Temperature/presence_penalty/frequency_penalty ignored in thinking mode, not rejected. Legacy V4 Flash IDs are retired and route to V4.1 Flash. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
2441
+ "$comment": "DeepSeek's official model list now names `deepseek-flash` as V4.1 Flash. The legacy aliases are temporarily routed to it. Winter drives its OpenAI Chat Completions surface, whose documented `reasoning_content` replay contract is therefore carried only on this direct-dialect row. — 2026-09-25 refresh: Chat uses thinking.type enabled/disabled plus reasoning_effort none/low/high/max (none disables); Responses uses reasoning.effort. https://api-docs.deepseek.com/guides/responses_api/ guide table still says only Flash but https://api-docs.deepseek.com/api/create-response/ model enum and https://api-docs.deepseek.com/quick_start/pricing/ feature table both list Pro too. Responses stateless: no previous_response_id. Temperature/presence_penalty/frequency_penalty ignored in thinking mode, not rejected. Legacy V4 Flash IDs are retired and route to V4.1 Flash. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://api-docs.deepseek.com/guides/thinking_mode/ — Chat Completions reasoning_effort is low/high/max; thinking.type=disabled is the separate off switch; none belongs to Responses reasoning.effort; https://api-docs.deepseek.com/guides/thinking_mode/ — correct per-dialect thinking evidence",
2356
2442
  "endpoints": [
2357
2443
  "chat"
2358
2444
  ],
@@ -2428,12 +2514,11 @@
2428
2514
  "supported": {
2429
2515
  "value": true,
2430
2516
  "source": "official-doc",
2431
- "sourceRef": "https://api-docs.deepseek.com/guides/thinking_mode/ — Chat reasoning_effort and Responses reasoning.effort allow none/low/high/max; default high; none disables",
2517
+ "sourceRef": "https://api-docs.deepseek.com/guides/thinking_mode/ — Chat Completions thinking toggle uses thinking.type enabled/disabled; reasoning_effort low/high/max; default high. Responses reasoning.effort additionally accepts none",
2432
2518
  "confidence": "declared",
2433
- "observedAt": "2026-09-25T08:45:47Z"
2519
+ "observedAt": "2026-09-25T12:31:56.381Z"
2434
2520
  },
2435
2521
  "efforts": [
2436
- "none",
2437
2522
  "low",
2438
2523
  "high",
2439
2524
  "max"
@@ -2482,7 +2567,7 @@
2482
2567
  "upstreamId": "deepseek-v4-pro",
2483
2568
  "displayName": "DeepSeek V4 Pro",
2484
2569
  "aliases": [],
2485
- "$comment": "DeepSeek's current model list retains `deepseek-v4-pro`. Winter drives the documented OpenAI Chat Completions surface, rather than the stale upstream extraction's Responses endpoint, because its readable reasoning and tool-loop replay contract are documented on Chat Completions. — 2026-09-25 refresh: Chat uses thinking.type enabled/disabled plus reasoning_effort none/low/high/max (none disables); Responses uses reasoning.effort. https://api-docs.deepseek.com/guides/responses_api/ guide table still says only Flash but https://api-docs.deepseek.com/api/create-response/ model enum and https://api-docs.deepseek.com/quick_start/pricing/ feature table both list Pro too. Responses stateless: no previous_response_id. Temperature/presence_penalty/frequency_penalty ignored in thinking mode, not rejected. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
2570
+ "$comment": "DeepSeek's current model list retains `deepseek-v4-pro`. Winter drives the documented OpenAI Chat Completions surface, rather than the stale upstream extraction's Responses endpoint, because its readable reasoning and tool-loop replay contract are documented on Chat Completions. — 2026-09-25 refresh: Chat uses thinking.type enabled/disabled plus reasoning_effort none/low/high/max (none disables); Responses uses reasoning.effort. https://api-docs.deepseek.com/guides/responses_api/ guide table still says only Flash but https://api-docs.deepseek.com/api/create-response/ model enum and https://api-docs.deepseek.com/quick_start/pricing/ feature table both list Pro too. Responses stateless: no previous_response_id. Temperature/presence_penalty/frequency_penalty ignored in thinking mode, not rejected. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://api-docs.deepseek.com/guides/thinking_mode/ — Chat Completions reasoning_effort is low/high/max; thinking.type=disabled is the separate off switch; none belongs to Responses reasoning.effort; https://api-docs.deepseek.com/guides/thinking_mode/ — correct per-dialect thinking evidence",
2486
2571
  "endpoints": [
2487
2572
  "chat"
2488
2573
  ],
@@ -2557,12 +2642,11 @@
2557
2642
  "supported": {
2558
2643
  "value": true,
2559
2644
  "source": "official-doc",
2560
- "sourceRef": "https://api-docs.deepseek.com/guides/thinking_mode/ — Chat reasoning_effort and Responses reasoning.effort allow none/low/high/max; default high; none disables",
2645
+ "sourceRef": "https://api-docs.deepseek.com/guides/thinking_mode/ — Chat Completions thinking toggle uses thinking.type enabled/disabled; reasoning_effort low/high/max; default high. Responses reasoning.effort additionally accepts none",
2561
2646
  "confidence": "declared",
2562
- "observedAt": "2026-09-25T08:45:47Z"
2647
+ "observedAt": "2026-09-25T12:31:56.381Z"
2563
2648
  },
2564
2649
  "efforts": [
2565
- "none",
2566
2650
  "low",
2567
2651
  "high",
2568
2652
  "max"
@@ -4173,6 +4257,7 @@
4173
4257
  "high"
4174
4258
  ],
4175
4259
  "continuation": "opaque-provider-state",
4260
+ "defaultEffort": "medium",
4176
4261
  "readableState": {
4177
4262
  "value": "summary",
4178
4263
  "source": "official-doc",
@@ -5141,7 +5226,7 @@
5141
5226
  "aliases": [
5142
5227
  "grok-4.3-latest"
5143
5228
  ],
5144
- "$comment": "Read from the pinned entry's own accepted literals — only the provider's `executor` is unrepresentable. The two `targetFormat: \"openai-responses\"` siblings are omitted rather than reshaped; see the provider row's comment. — 2026-09-25 refresh: Model and aliases documented at https://docs.x.ai/developers/models/grok-4.3. The catalog's xai provider currently declares Chat Completions; xAI also offers Responses. No model-specific max output token cap was found. xAI's models page says logprobs/top_logprobs on grok-4.20 and newer are silently ignored, not rejected (https://docs.x.ai/developers/models). Legacy Anthropic-compatible POST /v1/messages remains documented but fully deprecated (https://docs.x.ai/developers/rest-api-reference/inference/legacy); unauthenticated POST today returned 401 after validating the request, so route exists but successful inference was not verified. POST /anthropic returned 404. Detail page lists none/low/medium/high/xhigh, default low. Its prose only names none/low/medium/high; xhigh appears in the detailed effort list. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
5229
+ "$comment": "Read from the pinned entry's own accepted literals — only the provider's `executor` is unrepresentable. The two `targetFormat: \"openai-responses\"` siblings are omitted rather than reshaped; see the provider row's comment. — 2026-09-25 refresh: Model and aliases documented at https://docs.x.ai/developers/models/grok-4.3. The catalog's xai provider currently declares Chat Completions; xAI also offers Responses. No model-specific max output token cap was found. xAI's models page says logprobs/top_logprobs on grok-4.20 and newer are silently ignored, not rejected (https://docs.x.ai/developers/models). Legacy Anthropic-compatible POST /v1/messages remains documented but fully deprecated (https://docs.x.ai/developers/rest-api-reference/inference/legacy); unauthenticated POST today returned 401 after validating the request, so route exists but successful inference was not verified. POST /anthropic returned 404. Detail page lists none/low/medium/high/xhigh, default low. Its prose only names none/low/medium/high; xhigh appears in the detailed effort list. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://docs.x.ai/developers/pricing — US regional availability currently only grok-4.7 and grok-4.6",
5145
5230
  "endpoints": [
5146
5231
  "chat"
5147
5232
  ],
@@ -5224,9 +5309,9 @@
5224
5309
  "cacheReadPerMTokUsd": 0.2
5225
5310
  },
5226
5311
  "source": "official-doc",
5227
- "sourceRef": "https://docs.x.ai/developers/pricing — current global Standard Text API price table for grok-4.3; it charges the listed long-context rate for every token after the prompt reaches 200K and applies a 10% US-regional premium. This catalog records the standard short-context rate (retrieved 2026-09-19).",
5312
+ "sourceRef": "https://docs.x.ai/developers/pricing — grok-4.3 Standard global short-context rate: $1.25 input/$2.5 output/$0.2 cached per million; long-context threshold 200K. US-regional 1.1x applies only to models currently on that endpoint, listed as grok-4.7 and grok-4.6; not grok-4.3.",
5228
5313
  "confidence": "declared",
5229
- "observedAt": "2026-09-19T00:00:00Z"
5314
+ "observedAt": "2026-09-25T12:31:56.381Z"
5230
5315
  },
5231
5316
  "unsupportedParameters": [],
5232
5317
  "status": "candidate"
@@ -5241,7 +5326,7 @@
5241
5326
  "grok-code-fast",
5242
5327
  "grok-code-fast-1-0825"
5243
5328
  ],
5244
- "$comment": "Read from the pinned entry's own accepted literals — only the provider's `executor` is unrepresentable. The two `targetFormat: \"openai-responses\"` siblings are omitted rather than reshaped; see the provider row's comment. — 2026-09-25 refresh: Model and aliases documented at https://docs.x.ai/developers/models/grok-build-0.1. The catalog's xai provider currently declares Chat Completions; xAI also offers Responses. No model-specific max output token cap was found. xAI's models page says logprobs/top_logprobs on grok-4.20 and newer are silently ignored, not rejected (https://docs.x.ai/developers/models). Legacy Anthropic-compatible POST /v1/messages remains documented but fully deprecated (https://docs.x.ai/developers/rest-api-reference/inference/legacy); unauthenticated POST today returned 401 after validating the request, so route exists but successful inference was not verified. POST /anthropic returned 404. xAI's May 15 retirement page says grok-code-fast-1 redirects to grok-4.3, whereas this model page lists it as a grok-build-0.1 alias; live resolution of that alias needs checking (https://docs.x.ai/developers/migration/may-15-retirement). Model page does not document an effort vocabulary for this exact model, so efforts is empty. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
5329
+ "$comment": "Read from the pinned entry's own accepted literals — only the provider's `executor` is unrepresentable. The two `targetFormat: \"openai-responses\"` siblings are omitted rather than reshaped; see the provider row's comment. — 2026-09-25 refresh: Model and aliases documented at https://docs.x.ai/developers/models/grok-build-0.1. The catalog's xai provider currently declares Chat Completions; xAI also offers Responses. No model-specific max output token cap was found. xAI's models page says logprobs/top_logprobs on grok-4.20 and newer are silently ignored, not rejected (https://docs.x.ai/developers/models). Legacy Anthropic-compatible POST /v1/messages remains documented but fully deprecated (https://docs.x.ai/developers/rest-api-reference/inference/legacy); unauthenticated POST today returned 401 after validating the request, so route exists but successful inference was not verified. POST /anthropic returned 404. xAI's May 15 retirement page says grok-code-fast-1 redirects to grok-4.3, whereas this model page lists it as a grok-build-0.1 alias; live resolution of that alias needs checking (https://docs.x.ai/developers/migration/may-15-retirement). Model page does not document an effort vocabulary for this exact model, so efforts is empty. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://docs.x.ai/developers/pricing — US regional availability currently only grok-4.7 and grok-4.6",
5245
5330
  "endpoints": [
5246
5331
  "chat"
5247
5332
  ],
@@ -5317,9 +5402,9 @@
5317
5402
  "cacheReadPerMTokUsd": 0.2
5318
5403
  },
5319
5404
  "source": "official-doc",
5320
- "sourceRef": "https://docs.x.ai/developers/pricing — current global Standard Text API price table for grok-build-0.1; it charges the listed long-context rate for every token after the prompt reaches 200K and applies a 10% US-regional premium. This catalog records the standard short-context rate (retrieved 2026-09-19).",
5405
+ "sourceRef": "https://docs.x.ai/developers/pricing — grok-build-0.1 Standard global short-context rate: $1 input/$2 output/$0.2 cached per million; long-context threshold 200K. US-regional 1.1x applies only to models currently on that endpoint, listed as grok-4.7 and grok-4.6; not grok-build-0.1.",
5321
5406
  "confidence": "declared",
5322
- "observedAt": "2026-09-19T00:00:00Z"
5407
+ "observedAt": "2026-09-25T12:31:56.381Z"
5323
5408
  },
5324
5409
  "unsupportedParameters": [],
5325
5410
  "status": "candidate"
@@ -5330,10 +5415,17 @@
5330
5415
  "upstreamId": "glm-4.7",
5331
5416
  "displayName": "GLM-4.7",
5332
5417
  "aliases": [],
5333
- "$comment": "R6b-5: duplicated from the `zai-anthropic` sibling's upstream facts with its own key. `toolCalling` fails CLOSED because upstream states no flag for these rows — see WS-13 §8.1. Reasoning evidence added for SDK 0.0.11 (R-10b-8/W18-16, router review findings): GLM exposes readable `reasoning_content` (full-exposed, like DeepSeek), so this row must never be treated as a model with nothing to lose, and its own carriage (as a SOURCE, cross-family) must render `kind=\"exposed\"`, not `kind=\"summary\"`. `continuation` is `plaintext`: Z.ai's own Preserved Thinking / Interleaved Thinking docs (docs.z.ai/guides/capabilities/thinking-mode, and `ChatThinking.clear_thinking` on docs.z.ai/api-reference/llm/chat-completion) document forwarding the full historical `reasoning_content` back in `messages` verbatim and in order — the textual passback contract the schema's `plaintext` value exists for, exactly as `deepseek/deepseek-reasoner` already carries. This ALSO fixes same-domain carriage: `capabilitiesFrom`'s `readableState` gates the OpenAI chat-completions adapter's own `captureExposedReasoning` flag (`chat-completions.ts`), which — with `reasoning: null` — was silently disabling GLM's own tool-loop reasoning replay, contrary to Z.ai's default-on Interleaved Thinking. Passback is opt-in (`clear_thinking` defaults `true` = cleared on the standard API endpoint; the Coding Plan endpoint defaults it `false`) — this row states what the documented contract IS once engaged, not that Winter enables it by default. `toolLoopRequirement` is `silent-degradation`, not DeepSeek's `hard-error`: Z.ai's own wording is “may degrade performance ... or prevent the feature from taking effect”, never a stated error response; nothing in this codebase reads `toolLoopRequirement` for behavior today, so this is evidentiary only. No `zai-anthropic/*` sibling gets this: no official Z.ai documentation of an Anthropic-dialect (`/v1/messages`) thinking contract was found (docs.z.ai's own site index carries no such page as of 2026-09-14) — WS-13 §8.2 forbids assuming one dialect's contract on another, so those rows stay untouched. — 2026-09-25 refresh: OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: no documented off switch; core parameters describe GLM-4.7 forced thinking (Flash variants inferred). Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs.",
5418
+ "$comment": "R6b-5: duplicated from the `zai-anthropic` sibling's upstream facts with its own key. `toolCalling` fails CLOSED because upstream states no flag for these rows — see WS-13 §8.1. Reasoning evidence added for SDK 0.0.11 (R-10b-8/W18-16, router review findings): GLM exposes readable `reasoning_content` (full-exposed, like DeepSeek), so this row must never be treated as a model with nothing to lose, and its own carriage (as a SOURCE, cross-family) must render `kind=\"exposed\"`, not `kind=\"summary\"`. `continuation` is `plaintext`: Z.ai's own Preserved Thinking / Interleaved Thinking docs (docs.z.ai/guides/capabilities/thinking-mode, and `ChatThinking.clear_thinking` on docs.z.ai/api-reference/llm/chat-completion) document forwarding the full historical `reasoning_content` back in `messages` verbatim and in order — the textual passback contract the schema's `plaintext` value exists for, exactly as `deepseek/deepseek-reasoner` already carries. This ALSO fixes same-domain carriage: `capabilitiesFrom`'s `readableState` gates the OpenAI chat-completions adapter's own `captureExposedReasoning` flag (`chat-completions.ts`), which — with `reasoning: null` — was silently disabling GLM's own tool-loop reasoning replay, contrary to Z.ai's default-on Interleaved Thinking. Passback is opt-in (`clear_thinking` defaults `true` = cleared on the standard API endpoint; the Coding Plan endpoint defaults it `false`) — this row states what the documented contract IS once engaged, not that Winter enables it by default. `toolLoopRequirement` is `silent-degradation`, not DeepSeek's `hard-error`: Z.ai's own wording is “may degrade performance ... or prevent the feature from taking effect”, never a stated error response; nothing in this codebase reads `toolLoopRequirement` for behavior today, so this is evidentiary only. No `zai-anthropic/*` sibling gets this: no official Z.ai documentation of an Anthropic-dialect (`/v1/messages`) thinking contract was found (docs.z.ai's own site index carries no such page as of 2026-09-14) — WS-13 §8.2 forbids assuming one dialect's contract on another, so those rows stay untouched. — 2026-09-25 refresh: OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: no documented off switch; core parameters describe GLM-4.7 forced thinking (Flash variants inferred). Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs. — 2026-09-25 audit fix (0.0.24): https://docs.z.ai/guides/llm/glm-4.7 — glm-4.7 context length 200K",
5334
5419
  "endpoints": [
5335
5420
  "chat"
5336
5421
  ],
5422
+ "contextWindow": {
5423
+ "value": 200000,
5424
+ "source": "official-doc",
5425
+ "sourceRef": "https://docs.z.ai/guides/llm/glm-4.7 — glm-4.7 model guide/table gives 200K context length; K expanded as 1,000 tokens",
5426
+ "confidence": "inferred",
5427
+ "observedAt": "2026-09-25T12:31:56.381Z"
5428
+ },
5337
5429
  "maxOutputTokens": {
5338
5430
  "value": 131072,
5339
5431
  "source": "official-doc",
@@ -5432,10 +5524,17 @@
5432
5524
  "upstreamId": "glm-4.7-flash",
5433
5525
  "displayName": "GLM-4.7-FLASH",
5434
5526
  "aliases": [],
5435
- "$comment": "R6b-5: duplicated from the `zai-anthropic` sibling's upstream facts with its own key. `toolCalling` fails CLOSED because upstream states no flag for these rows — see WS-13 §8.1. Reasoning evidence added for SDK 0.0.11 (R-10b-8/W18-16, router review findings): GLM exposes readable `reasoning_content` (full-exposed, like DeepSeek), so this row must never be treated as a model with nothing to lose, and its own carriage (as a SOURCE, cross-family) must render `kind=\"exposed\"`, not `kind=\"summary\"`. `continuation` is `plaintext`: Z.ai's own Preserved Thinking / Interleaved Thinking docs (docs.z.ai/guides/capabilities/thinking-mode, and `ChatThinking.clear_thinking` on docs.z.ai/api-reference/llm/chat-completion) document forwarding the full historical `reasoning_content` back in `messages` verbatim and in order — the textual passback contract the schema's `plaintext` value exists for, exactly as `deepseek/deepseek-reasoner` already carries. This ALSO fixes same-domain carriage: `capabilitiesFrom`'s `readableState` gates the OpenAI chat-completions adapter's own `captureExposedReasoning` flag (`chat-completions.ts`), which — with `reasoning: null` — was silently disabling GLM's own tool-loop reasoning replay, contrary to Z.ai's default-on Interleaved Thinking. Passback is opt-in (`clear_thinking` defaults `true` = cleared on the standard API endpoint; the Coding Plan endpoint defaults it `false`) — this row states what the documented contract IS once engaged, not that Winter enables it by default. `toolLoopRequirement` is `silent-degradation`, not DeepSeek's `hard-error`: Z.ai's own wording is “may degrade performance ... or prevent the feature from taking effect”, never a stated error response; nothing in this codebase reads `toolLoopRequirement` for behavior today, so this is evidentiary only. No `zai-anthropic/*` sibling gets this: no official Z.ai documentation of an Anthropic-dialect (`/v1/messages`) thinking contract was found (docs.z.ai's own site index carries no such page as of 2026-09-14) — WS-13 §8.2 forbids assuming one dialect's contract on another, so those rows stay untouched. — 2026-09-25 refresh: OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: no documented off switch; core parameters describe GLM-4.7 forced thinking (Flash variants inferred). Effort control: thinking.type toggle only; no documented graded effort. Free on current price page; pricing omitted, not encoded as zero. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs.",
5527
+ "$comment": "R6b-5: duplicated from the `zai-anthropic` sibling's upstream facts with its own key. `toolCalling` fails CLOSED because upstream states no flag for these rows — see WS-13 §8.1. Reasoning evidence added for SDK 0.0.11 (R-10b-8/W18-16, router review findings): GLM exposes readable `reasoning_content` (full-exposed, like DeepSeek), so this row must never be treated as a model with nothing to lose, and its own carriage (as a SOURCE, cross-family) must render `kind=\"exposed\"`, not `kind=\"summary\"`. `continuation` is `plaintext`: Z.ai's own Preserved Thinking / Interleaved Thinking docs (docs.z.ai/guides/capabilities/thinking-mode, and `ChatThinking.clear_thinking` on docs.z.ai/api-reference/llm/chat-completion) document forwarding the full historical `reasoning_content` back in `messages` verbatim and in order — the textual passback contract the schema's `plaintext` value exists for, exactly as `deepseek/deepseek-reasoner` already carries. This ALSO fixes same-domain carriage: `capabilitiesFrom`'s `readableState` gates the OpenAI chat-completions adapter's own `captureExposedReasoning` flag (`chat-completions.ts`), which — with `reasoning: null` — was silently disabling GLM's own tool-loop reasoning replay, contrary to Z.ai's default-on Interleaved Thinking. Passback is opt-in (`clear_thinking` defaults `true` = cleared on the standard API endpoint; the Coding Plan endpoint defaults it `false`) — this row states what the documented contract IS once engaged, not that Winter enables it by default. `toolLoopRequirement` is `silent-degradation`, not DeepSeek's `hard-error`: Z.ai's own wording is “may degrade performance ... or prevent the feature from taking effect”, never a stated error response; nothing in this codebase reads `toolLoopRequirement` for behavior today, so this is evidentiary only. No `zai-anthropic/*` sibling gets this: no official Z.ai documentation of an Anthropic-dialect (`/v1/messages`) thinking contract was found (docs.z.ai's own site index carries no such page as of 2026-09-14) — WS-13 §8.2 forbids assuming one dialect's contract on another, so those rows stay untouched. — 2026-09-25 refresh: OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: no documented off switch; core parameters describe GLM-4.7 forced thinking (Flash variants inferred). Effort control: thinking.type toggle only; no documented graded effort. Free on current price page; pricing omitted, not encoded as zero. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs. — 2026-09-25 audit fix (0.0.24): https://docs.z.ai/llms-full.txt — glm-4.7-flash context length 200K",
5436
5528
  "endpoints": [
5437
5529
  "chat"
5438
5530
  ],
5531
+ "contextWindow": {
5532
+ "value": 200000,
5533
+ "source": "official-doc",
5534
+ "sourceRef": "https://docs.z.ai/llms-full.txt — glm-4.7-flash model guide/table gives 200K context length; K expanded as 1,000 tokens",
5535
+ "confidence": "inferred",
5536
+ "observedAt": "2026-09-25T12:31:56.381Z"
5537
+ },
5439
5538
  "inputModalities": {
5440
5539
  "value": [
5441
5540
  "text"
@@ -5520,10 +5619,17 @@
5520
5619
  "upstreamId": "glm-5",
5521
5620
  "displayName": "GLM-5",
5522
5621
  "aliases": [],
5523
- "$comment": "R6b-5: duplicated from the `zai-anthropic` sibling's upstream facts with its own key. `toolCalling` fails CLOSED because upstream states no flag for these rows — see WS-13 §8.1. Reasoning evidence added for SDK 0.0.11 (R-10b-8/W18-16, router review findings): GLM exposes readable `reasoning_content` (full-exposed, like DeepSeek), so this row must never be treated as a model with nothing to lose, and its own carriage (as a SOURCE, cross-family) must render `kind=\"exposed\"`, not `kind=\"summary\"`. `continuation` is `plaintext`: Z.ai's own Preserved Thinking / Interleaved Thinking docs (docs.z.ai/guides/capabilities/thinking-mode, and `ChatThinking.clear_thinking` on docs.z.ai/api-reference/llm/chat-completion) document forwarding the full historical `reasoning_content` back in `messages` verbatim and in order — the textual passback contract the schema's `plaintext` value exists for, exactly as `deepseek/deepseek-reasoner` already carries. This ALSO fixes same-domain carriage: `capabilitiesFrom`'s `readableState` gates the OpenAI chat-completions adapter's own `captureExposedReasoning` flag (`chat-completions.ts`), which — with `reasoning: null` — was silently disabling GLM's own tool-loop reasoning replay, contrary to Z.ai's default-on Interleaved Thinking. Passback is opt-in (`clear_thinking` defaults `true` = cleared on the standard API endpoint; the Coding Plan endpoint defaults it `false`) — this row states what the documented contract IS once engaged, not that Winter enables it by default. `toolLoopRequirement` is `silent-degradation`, not DeepSeek's `hard-error`: Z.ai's own wording is “may degrade performance ... or prevent the feature from taking effect”, never a stated error response; nothing in this codebase reads `toolLoopRequirement` for behavior today, so this is evidentiary only. No `zai-anthropic/*` sibling gets this: no official Z.ai documentation of an Anthropic-dialect (`/v1/messages`) thinking contract was found (docs.z.ai's own site index carries no such page as of 2026-09-14) — WS-13 §8.2 forbids assuming one dialect's contract on another, so those rows stay untouched. — 2026-09-25 refresh: OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: yes; thinking.type=disabled. Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs.",
5622
+ "$comment": "R6b-5: duplicated from the `zai-anthropic` sibling's upstream facts with its own key. `toolCalling` fails CLOSED because upstream states no flag for these rows — see WS-13 §8.1. Reasoning evidence added for SDK 0.0.11 (R-10b-8/W18-16, router review findings): GLM exposes readable `reasoning_content` (full-exposed, like DeepSeek), so this row must never be treated as a model with nothing to lose, and its own carriage (as a SOURCE, cross-family) must render `kind=\"exposed\"`, not `kind=\"summary\"`. `continuation` is `plaintext`: Z.ai's own Preserved Thinking / Interleaved Thinking docs (docs.z.ai/guides/capabilities/thinking-mode, and `ChatThinking.clear_thinking` on docs.z.ai/api-reference/llm/chat-completion) document forwarding the full historical `reasoning_content` back in `messages` verbatim and in order — the textual passback contract the schema's `plaintext` value exists for, exactly as `deepseek/deepseek-reasoner` already carries. This ALSO fixes same-domain carriage: `capabilitiesFrom`'s `readableState` gates the OpenAI chat-completions adapter's own `captureExposedReasoning` flag (`chat-completions.ts`), which — with `reasoning: null` — was silently disabling GLM's own tool-loop reasoning replay, contrary to Z.ai's default-on Interleaved Thinking. Passback is opt-in (`clear_thinking` defaults `true` = cleared on the standard API endpoint; the Coding Plan endpoint defaults it `false`) — this row states what the documented contract IS once engaged, not that Winter enables it by default. `toolLoopRequirement` is `silent-degradation`, not DeepSeek's `hard-error`: Z.ai's own wording is “may degrade performance ... or prevent the feature from taking effect”, never a stated error response; nothing in this codebase reads `toolLoopRequirement` for behavior today, so this is evidentiary only. No `zai-anthropic/*` sibling gets this: no official Z.ai documentation of an Anthropic-dialect (`/v1/messages`) thinking contract was found (docs.z.ai's own site index carries no such page as of 2026-09-14) — WS-13 §8.2 forbids assuming one dialect's contract on another, so those rows stay untouched. — 2026-09-25 refresh: OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: yes; thinking.type=disabled. Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs. — 2026-09-25 audit fix (0.0.24): https://docs.z.ai/guides/llm/glm-5 — glm-5 context length 200K",
5524
5623
  "endpoints": [
5525
5624
  "chat"
5526
5625
  ],
5626
+ "contextWindow": {
5627
+ "value": 200000,
5628
+ "source": "official-doc",
5629
+ "sourceRef": "https://docs.z.ai/guides/llm/glm-5 — glm-5 model guide/table gives 200K context length; K expanded as 1,000 tokens",
5630
+ "confidence": "inferred",
5631
+ "observedAt": "2026-09-25T12:31:56.381Z"
5632
+ },
5527
5633
  "maxOutputTokens": {
5528
5634
  "value": 131072,
5529
5635
  "source": "official-doc",
@@ -5622,10 +5728,17 @@
5622
5728
  "upstreamId": "glm-5-turbo",
5623
5729
  "displayName": "GLM 5 Turbo (OpenAI dialect)",
5624
5730
  "aliases": [],
5625
- "$comment": "R6b-5: duplicated from the `zai-anthropic` sibling's upstream facts with its own key. `toolCalling` fails CLOSED because upstream states no flag for these rows — see WS-13 §8.1. Reasoning evidence added for SDK 0.0.11 (R-10b-8/W18-16, router review findings): GLM exposes readable `reasoning_content` (full-exposed, like DeepSeek), so this row must never be treated as a model with nothing to lose, and its own carriage (as a SOURCE, cross-family) must render `kind=\"exposed\"`, not `kind=\"summary\"`. `continuation` is `plaintext`: Z.ai's own Preserved Thinking / Interleaved Thinking docs (docs.z.ai/guides/capabilities/thinking-mode, and `ChatThinking.clear_thinking` on docs.z.ai/api-reference/llm/chat-completion) document forwarding the full historical `reasoning_content` back in `messages` verbatim and in order — the textual passback contract the schema's `plaintext` value exists for, exactly as `deepseek/deepseek-reasoner` already carries. This ALSO fixes same-domain carriage: `capabilitiesFrom`'s `readableState` gates the OpenAI chat-completions adapter's own `captureExposedReasoning` flag (`chat-completions.ts`), which — with `reasoning: null` — was silently disabling GLM's own tool-loop reasoning replay, contrary to Z.ai's default-on Interleaved Thinking. Passback is opt-in (`clear_thinking` defaults `true` = cleared on the standard API endpoint; the Coding Plan endpoint defaults it `false`) — this row states what the documented contract IS once engaged, not that Winter enables it by default. `toolLoopRequirement` is `silent-degradation`, not DeepSeek's `hard-error`: Z.ai's own wording is “may degrade performance ... or prevent the feature from taking effect”, never a stated error response; nothing in this codebase reads `toolLoopRequirement` for behavior today, so this is evidentiary only. No `zai-anthropic/*` sibling gets this: no official Z.ai documentation of an Anthropic-dialect (`/v1/messages`) thinking contract was found (docs.z.ai's own site index carries no such page as of 2026-09-14) — WS-13 §8.2 forbids assuming one dialect's contract on another, so those rows stay untouched. — DEPRECATED 2026-09-25 refresh: absent from Z.AI's current pricing page and chat model enum (https://docs.z.ai/guides/overview/pricing, checked 2026-09-25); Z.AI names no successor or retirement date — its generation is served by glm-5.1/5.2/5.3.",
5731
+ "$comment": "R6b-5: duplicated from the `zai-anthropic` sibling's upstream facts with its own key. `toolCalling` fails CLOSED because upstream states no flag for these rows — see WS-13 §8.1. Reasoning evidence added for SDK 0.0.11 (R-10b-8/W18-16, router review findings): GLM exposes readable `reasoning_content` (full-exposed, like DeepSeek), so this row must never be treated as a model with nothing to lose, and its own carriage (as a SOURCE, cross-family) must render `kind=\"exposed\"`, not `kind=\"summary\"`. `continuation` is `plaintext`: Z.ai's own Preserved Thinking / Interleaved Thinking docs (docs.z.ai/guides/capabilities/thinking-mode, and `ChatThinking.clear_thinking` on docs.z.ai/api-reference/llm/chat-completion) document forwarding the full historical `reasoning_content` back in `messages` verbatim and in order — the textual passback contract the schema's `plaintext` value exists for, exactly as `deepseek/deepseek-reasoner` already carries. This ALSO fixes same-domain carriage: `capabilitiesFrom`'s `readableState` gates the OpenAI chat-completions adapter's own `captureExposedReasoning` flag (`chat-completions.ts`), which — with `reasoning: null` — was silently disabling GLM's own tool-loop reasoning replay, contrary to Z.ai's default-on Interleaved Thinking. Passback is opt-in (`clear_thinking` defaults `true` = cleared on the standard API endpoint; the Coding Plan endpoint defaults it `false`) — this row states what the documented contract IS once engaged, not that Winter enables it by default. `toolLoopRequirement` is `silent-degradation`, not DeepSeek's `hard-error`: Z.ai's own wording is “may degrade performance ... or prevent the feature from taking effect”, never a stated error response; nothing in this codebase reads `toolLoopRequirement` for behavior today, so this is evidentiary only. No `zai-anthropic/*` sibling gets this: no official Z.ai documentation of an Anthropic-dialect (`/v1/messages`) thinking contract was found (docs.z.ai's own site index carries no such page as of 2026-09-14) — WS-13 §8.2 forbids assuming one dialect's contract on another, so those rows stay untouched. — DEPRECATED 2026-09-25 refresh: absent from Z.AI's current pricing page and chat model enum (https://docs.z.ai/guides/overview/pricing, checked 2026-09-25); Z.AI names no successor or retirement date — its generation is served by glm-5.1/5.2/5.3. — 2026-09-25 audit fix (0.0.24): https://docs.z.ai/guides/llm/glm-5-turbo — glm-5-turbo context length 200K",
5626
5732
  "endpoints": [
5627
5733
  "chat"
5628
5734
  ],
5735
+ "contextWindow": {
5736
+ "value": 200000,
5737
+ "source": "official-doc",
5738
+ "sourceRef": "https://docs.z.ai/guides/llm/glm-5-turbo — glm-5-turbo model guide/table gives 200K context length; K expanded as 1,000 tokens",
5739
+ "confidence": "inferred",
5740
+ "observedAt": "2026-09-25T12:31:56.381Z"
5741
+ },
5629
5742
  "inputModalities": {
5630
5743
  "value": [
5631
5744
  "text"
@@ -5699,10 +5812,17 @@
5699
5812
  "upstreamId": "glm-5.1",
5700
5813
  "displayName": "GLM-5.1",
5701
5814
  "aliases": [],
5702
- "$comment": "R6b-5: duplicated from the `zai-anthropic` sibling's upstream facts with its own key. `toolCalling` fails CLOSED because upstream states no flag for these rows — see WS-13 §8.1. Reasoning evidence added for SDK 0.0.11 (R-10b-8/W18-16, router review findings): GLM exposes readable `reasoning_content` (full-exposed, like DeepSeek), so this row must never be treated as a model with nothing to lose, and its own carriage (as a SOURCE, cross-family) must render `kind=\"exposed\"`, not `kind=\"summary\"`. `continuation` is `plaintext`: Z.ai's own Preserved Thinking / Interleaved Thinking docs (docs.z.ai/guides/capabilities/thinking-mode, and `ChatThinking.clear_thinking` on docs.z.ai/api-reference/llm/chat-completion) document forwarding the full historical `reasoning_content` back in `messages` verbatim and in order — the textual passback contract the schema's `plaintext` value exists for, exactly as `deepseek/deepseek-reasoner` already carries. This ALSO fixes same-domain carriage: `capabilitiesFrom`'s `readableState` gates the OpenAI chat-completions adapter's own `captureExposedReasoning` flag (`chat-completions.ts`), which — with `reasoning: null` — was silently disabling GLM's own tool-loop reasoning replay, contrary to Z.ai's default-on Interleaved Thinking. Passback is opt-in (`clear_thinking` defaults `true` = cleared on the standard API endpoint; the Coding Plan endpoint defaults it `false`) — this row states what the documented contract IS once engaged, not that Winter enables it by default. `toolLoopRequirement` is `silent-degradation`, not DeepSeek's `hard-error`: Z.ai's own wording is “may degrade performance ... or prevent the feature from taking effect”, never a stated error response; nothing in this codebase reads `toolLoopRequirement` for behavior today, so this is evidentiary only. No `zai-anthropic/*` sibling gets this: no official Z.ai documentation of an Anthropic-dialect (`/v1/messages`) thinking contract was found (docs.z.ai's own site index carries no such page as of 2026-09-14) — WS-13 §8.2 forbids assuming one dialect's contract on another, so those rows stay untouched. — 2026-09-25 refresh: OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: yes; thinking.type=disabled. Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs.",
5815
+ "$comment": "R6b-5: duplicated from the `zai-anthropic` sibling's upstream facts with its own key. `toolCalling` fails CLOSED because upstream states no flag for these rows — see WS-13 §8.1. Reasoning evidence added for SDK 0.0.11 (R-10b-8/W18-16, router review findings): GLM exposes readable `reasoning_content` (full-exposed, like DeepSeek), so this row must never be treated as a model with nothing to lose, and its own carriage (as a SOURCE, cross-family) must render `kind=\"exposed\"`, not `kind=\"summary\"`. `continuation` is `plaintext`: Z.ai's own Preserved Thinking / Interleaved Thinking docs (docs.z.ai/guides/capabilities/thinking-mode, and `ChatThinking.clear_thinking` on docs.z.ai/api-reference/llm/chat-completion) document forwarding the full historical `reasoning_content` back in `messages` verbatim and in order — the textual passback contract the schema's `plaintext` value exists for, exactly as `deepseek/deepseek-reasoner` already carries. This ALSO fixes same-domain carriage: `capabilitiesFrom`'s `readableState` gates the OpenAI chat-completions adapter's own `captureExposedReasoning` flag (`chat-completions.ts`), which — with `reasoning: null` — was silently disabling GLM's own tool-loop reasoning replay, contrary to Z.ai's default-on Interleaved Thinking. Passback is opt-in (`clear_thinking` defaults `true` = cleared on the standard API endpoint; the Coding Plan endpoint defaults it `false`) — this row states what the documented contract IS once engaged, not that Winter enables it by default. `toolLoopRequirement` is `silent-degradation`, not DeepSeek's `hard-error`: Z.ai's own wording is “may degrade performance ... or prevent the feature from taking effect”, never a stated error response; nothing in this codebase reads `toolLoopRequirement` for behavior today, so this is evidentiary only. No `zai-anthropic/*` sibling gets this: no official Z.ai documentation of an Anthropic-dialect (`/v1/messages`) thinking contract was found (docs.z.ai's own site index carries no such page as of 2026-09-14) — WS-13 §8.2 forbids assuming one dialect's contract on another, so those rows stay untouched. — 2026-09-25 refresh: OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: yes; thinking.type=disabled. Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs. — 2026-09-25 audit fix (0.0.24): https://docs.z.ai/guides/llm/glm-5.1 — glm-5.1 context length 200K",
5703
5816
  "endpoints": [
5704
5817
  "chat"
5705
5818
  ],
5819
+ "contextWindow": {
5820
+ "value": 200000,
5821
+ "source": "official-doc",
5822
+ "sourceRef": "https://docs.z.ai/guides/llm/glm-5.1 — glm-5.1 model guide/table gives 200K context length; K expanded as 1,000 tokens",
5823
+ "confidence": "inferred",
5824
+ "observedAt": "2026-09-25T12:31:56.381Z"
5825
+ },
5706
5826
  "maxOutputTokens": {
5707
5827
  "value": 131072,
5708
5828
  "source": "official-doc",
@@ -6268,7 +6388,7 @@
6268
6388
  "max"
6269
6389
  ],
6270
6390
  "continuation": "opaque-provider-state",
6271
- "defaultEffort": "low",
6391
+ "defaultEffort": "medium",
6272
6392
  "readableState": {
6273
6393
  "value": "summary",
6274
6394
  "source": "official-doc",
@@ -6560,6 +6680,24 @@
6560
6680
  "sourceRef": "continuity report §4.4, BOTH halves, applied per anthropic/claude-opus-5's own evidence. Own-state acceptance: a Claude model's own thinking blocks are replayed to it unchanged, in order, with signatures intact. Why the domain is NARROW: prior thinking/redacted_thinking blocks are tied to the model that produced them, so \"same provider\" is not automatically \"same continuation domain\" — the domain is this model alone.",
6561
6681
  "confidence": "declared",
6562
6682
  "observedAt": "2026-09-07T00:00:00Z"
6683
+ },
6684
+ "effortRequest": {
6685
+ "value": {
6686
+ "field": "output_config.effort"
6687
+ },
6688
+ "source": "official-doc",
6689
+ "confidence": "declared",
6690
+ "observedAt": "2026-09-25T12:30:00Z",
6691
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
6692
+ },
6693
+ "blockBinding": {
6694
+ "value": {
6695
+ "beta": "thinking-binding-controls-2026-08-01"
6696
+ },
6697
+ "source": "official-doc",
6698
+ "confidence": "declared",
6699
+ "observedAt": "2026-09-25T13:00:00Z",
6700
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting — \"A 400 error says a thinking block signature is invalid\": on Claude Fable 5.1 and Opus 5.5 a replayed thinking block is bound to the conversation (system, tools and earlier messages) and is rejected once that prefix changes (enforced for accounts created on or after 2026-08-31); the documented escape is the `thinking-binding-controls-2026-08-01` beta with `thinking.block_binding.prefix_mismatch_behavior: \"drop_block\"`. See also https://platform.claude.com/docs/en/models/opus-5-5/whats-new-opus-5-5 (\"Thinking blocks are tied to the model and the conversation\")."
6563
6701
  }
6564
6702
  },
6565
6703
  "pricing": {
@@ -6574,7 +6712,12 @@
6574
6712
  "confidence": "declared",
6575
6713
  "observedAt": "2026-09-08T00:00:00Z"
6576
6714
  },
6577
- "unsupportedParameters": [],
6715
+ "unsupportedParameters": [
6716
+ "thinking.type.enabled",
6717
+ "thinking.type.disabled",
6718
+ "tool_choice.any",
6719
+ "tool_choice.tool"
6720
+ ],
6578
6721
  "status": "candidate"
6579
6722
  },
6580
6723
  {
@@ -7334,7 +7477,16 @@
7334
7477
  "max"
7335
7478
  ],
7336
7479
  "continuation": "opaque-provider-state",
7337
- "defaultEffort": "high"
7480
+ "defaultEffort": "high",
7481
+ "effortRequest": {
7482
+ "value": {
7483
+ "field": "output_config.effort"
7484
+ },
7485
+ "source": "official-doc",
7486
+ "confidence": "declared",
7487
+ "observedAt": "2026-09-25T12:30:00Z",
7488
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
7489
+ }
7338
7490
  },
7339
7491
  "pricing": {
7340
7492
  "value": {
@@ -7351,7 +7503,9 @@
7351
7503
  "unsupportedParameters": [
7352
7504
  "temperature",
7353
7505
  "top_p",
7354
- "top_k"
7506
+ "top_k",
7507
+ "thinking.type.enabled",
7508
+ "thinking.type.disabled"
7355
7509
  ],
7356
7510
  "status": "candidate"
7357
7511
  },
@@ -7485,6 +7639,24 @@
7485
7639
  "sourceRef": "continuity report §4.4, BOTH halves, applied per anthropic/claude-opus-5's own evidence. Own-state acceptance: a Claude model's own thinking blocks are replayed to it unchanged, in order, with signatures intact. Why the domain is NARROW: prior thinking/redacted_thinking blocks are tied to the model that produced them, so \"same provider\" is not automatically \"same continuation domain\" — the domain is this model alone.",
7486
7640
  "confidence": "declared",
7487
7641
  "observedAt": "2026-09-16T00:00:00Z"
7642
+ },
7643
+ "effortRequest": {
7644
+ "value": {
7645
+ "field": "output_config.effort"
7646
+ },
7647
+ "source": "official-doc",
7648
+ "confidence": "declared",
7649
+ "observedAt": "2026-09-25T12:30:00Z",
7650
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
7651
+ },
7652
+ "blockBinding": {
7653
+ "value": {
7654
+ "beta": "thinking-binding-controls-2026-08-01"
7655
+ },
7656
+ "source": "official-doc",
7657
+ "confidence": "declared",
7658
+ "observedAt": "2026-09-25T13:00:00Z",
7659
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting — \"A 400 error says a thinking block signature is invalid\": on Claude Fable 5.1 and Opus 5.5 a replayed thinking block is bound to the conversation (system, tools and earlier messages) and is rejected once that prefix changes (enforced for accounts created on or after 2026-08-31); the documented escape is the `thinking-binding-controls-2026-08-01` beta with `thinking.block_binding.prefix_mismatch_behavior: \"drop_block\"`. See also https://platform.claude.com/docs/en/models/opus-5-5/whats-new-opus-5-5 (\"Thinking blocks are tied to the model and the conversation\")."
7488
7660
  }
7489
7661
  },
7490
7662
  "pricing": {
@@ -7499,7 +7671,12 @@
7499
7671
  "confidence": "declared",
7500
7672
  "observedAt": "2026-09-16T00:00:00Z"
7501
7673
  },
7502
- "unsupportedParameters": [],
7674
+ "unsupportedParameters": [
7675
+ "thinking.type.enabled",
7676
+ "thinking.type.disabled",
7677
+ "tool_choice.any",
7678
+ "tool_choice.tool"
7679
+ ],
7503
7680
  "status": "candidate"
7504
7681
  },
7505
7682
  {
@@ -7594,7 +7771,9 @@
7594
7771
  "confidence": "declared",
7595
7772
  "observedAt": "2026-09-16T00:00:00Z"
7596
7773
  },
7597
- "unsupportedParameters": [],
7774
+ "unsupportedParameters": [
7775
+ "thinking.type.adaptive"
7776
+ ],
7598
7777
  "status": "candidate"
7599
7778
  },
7600
7779
  {
@@ -7689,7 +7868,9 @@
7689
7868
  "confidence": "declared",
7690
7869
  "observedAt": "2026-09-19T00:00:00Z"
7691
7870
  },
7692
- "unsupportedParameters": [],
7871
+ "unsupportedParameters": [
7872
+ "thinking.type.adaptive"
7873
+ ],
7693
7874
  "status": "candidate"
7694
7875
  },
7695
7876
  {
@@ -7787,7 +7968,16 @@
7787
7968
  "high"
7788
7969
  ],
7789
7970
  "continuation": "opaque-provider-state",
7790
- "defaultEffort": "high"
7971
+ "defaultEffort": "high",
7972
+ "effortRequest": {
7973
+ "value": {
7974
+ "field": "output_config.effort"
7975
+ },
7976
+ "source": "official-doc",
7977
+ "confidence": "declared",
7978
+ "observedAt": "2026-09-25T12:30:00Z",
7979
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
7980
+ }
7791
7981
  },
7792
7982
  "pricing": {
7793
7983
  "value": {
@@ -7801,7 +7991,9 @@
7801
7991
  "confidence": "declared",
7802
7992
  "observedAt": "2026-09-19T00:00:00Z"
7803
7993
  },
7804
- "unsupportedParameters": [],
7994
+ "unsupportedParameters": [
7995
+ "thinking.type.adaptive"
7996
+ ],
7805
7997
  "status": "candidate"
7806
7998
  },
7807
7999
  {
@@ -7897,7 +8089,16 @@
7897
8089
  "max"
7898
8090
  ],
7899
8091
  "continuation": "opaque-provider-state",
7900
- "defaultEffort": "high"
8092
+ "defaultEffort": "high",
8093
+ "effortRequest": {
8094
+ "value": {
8095
+ "field": "output_config.effort"
8096
+ },
8097
+ "source": "official-doc",
8098
+ "confidence": "declared",
8099
+ "observedAt": "2026-09-25T12:30:00Z",
8100
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
8101
+ }
7901
8102
  },
7902
8103
  "pricing": {
7903
8104
  "value": {
@@ -8008,7 +8209,16 @@
8008
8209
  "max"
8009
8210
  ],
8010
8211
  "continuation": "opaque-provider-state",
8011
- "defaultEffort": "high"
8212
+ "defaultEffort": "high",
8213
+ "effortRequest": {
8214
+ "value": {
8215
+ "field": "output_config.effort"
8216
+ },
8217
+ "source": "official-doc",
8218
+ "confidence": "declared",
8219
+ "observedAt": "2026-09-25T12:30:00Z",
8220
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
8221
+ }
8012
8222
  },
8013
8223
  "pricing": {
8014
8224
  "value": {
@@ -8025,7 +8235,8 @@
8025
8235
  "unsupportedParameters": [
8026
8236
  "temperature",
8027
8237
  "top_p",
8028
- "top_k"
8238
+ "top_k",
8239
+ "thinking.type.enabled"
8029
8240
  ],
8030
8241
  "status": "candidate"
8031
8242
  },
@@ -8123,7 +8334,16 @@
8123
8334
  "max"
8124
8335
  ],
8125
8336
  "continuation": "opaque-provider-state",
8126
- "defaultEffort": "high"
8337
+ "defaultEffort": "high",
8338
+ "effortRequest": {
8339
+ "value": {
8340
+ "field": "output_config.effort"
8341
+ },
8342
+ "source": "official-doc",
8343
+ "confidence": "declared",
8344
+ "observedAt": "2026-09-25T12:30:00Z",
8345
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
8346
+ }
8127
8347
  },
8128
8348
  "pricing": {
8129
8349
  "value": {
@@ -8140,7 +8360,8 @@
8140
8360
  "unsupportedParameters": [
8141
8361
  "temperature",
8142
8362
  "top_p",
8143
- "top_k"
8363
+ "top_k",
8364
+ "thinking.type.enabled"
8144
8365
  ],
8145
8366
  "status": "candidate"
8146
8367
  },
@@ -8281,6 +8502,15 @@
8281
8502
  "sourceRef": "continuity report §4.4, BOTH halves. Own-state acceptance: the model's own `thinking` blocks are replayed to it unchanged, in order, with their signatures intact (§4.4's replay rules and its worked request). Why the domain is NARROW: \"when changing Claude models, prior `thinking` and `redacted_thinking` blocks should be stripped because they are tied to the model that produced them. Therefore 'same provider' is not automatically 'same continuation domain.'\" The domain is this model alone, never the Anthropic provider.",
8282
8503
  "confidence": "declared",
8283
8504
  "observedAt": "2026-09-16T00:00:00Z"
8505
+ },
8506
+ "effortRequest": {
8507
+ "value": {
8508
+ "field": "output_config.effort"
8509
+ },
8510
+ "source": "official-doc",
8511
+ "confidence": "declared",
8512
+ "observedAt": "2026-09-25T12:30:00Z",
8513
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
8284
8514
  }
8285
8515
  },
8286
8516
  "pricing": {
@@ -8298,7 +8528,8 @@
8298
8528
  "unsupportedParameters": [
8299
8529
  "temperature",
8300
8530
  "top_p",
8301
- "top_k"
8531
+ "top_k",
8532
+ "thinking.type.enabled"
8302
8533
  ],
8303
8534
  "status": "candidate"
8304
8535
  },
@@ -8411,7 +8642,9 @@
8411
8642
  "confidence": "declared",
8412
8643
  "observedAt": "2026-09-19T00:00:00Z"
8413
8644
  },
8414
- "unsupportedParameters": [],
8645
+ "unsupportedParameters": [
8646
+ "thinking.type.adaptive"
8647
+ ],
8415
8648
  "status": "candidate"
8416
8649
  },
8417
8650
  {
@@ -8507,7 +8740,16 @@
8507
8740
  "max"
8508
8741
  ],
8509
8742
  "continuation": "opaque-provider-state",
8510
- "defaultEffort": "high"
8743
+ "defaultEffort": "high",
8744
+ "effortRequest": {
8745
+ "value": {
8746
+ "field": "output_config.effort"
8747
+ },
8748
+ "source": "official-doc",
8749
+ "confidence": "declared",
8750
+ "observedAt": "2026-09-25T12:30:00Z",
8751
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
8752
+ }
8511
8753
  },
8512
8754
  "pricing": {
8513
8755
  "value": {
@@ -8663,6 +8905,15 @@
8663
8905
  "sourceRef": "continuity report §4.4, BOTH halves. Own-state acceptance: §4.4 requires the model's own assistant blocks — `thinking` with its signature, `redacted_thinking` — replayed unchanged and in order across a tool loop. Why the domain is NARROW: Anthropic documents model switching as a boundary at which those blocks are STRIPPED, because they are tied to the producing model; \"same provider\" is therefore not \"same continuation domain\", and this domain is the model alone.",
8664
8906
  "confidence": "declared",
8665
8907
  "observedAt": "2026-09-16T00:00:00Z"
8908
+ },
8909
+ "effortRequest": {
8910
+ "value": {
8911
+ "field": "output_config.effort"
8912
+ },
8913
+ "source": "official-doc",
8914
+ "confidence": "declared",
8915
+ "observedAt": "2026-09-25T12:30:00Z",
8916
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
8666
8917
  }
8667
8918
  },
8668
8919
  "pricing": {
@@ -8680,7 +8931,8 @@
8680
8931
  "unsupportedParameters": [
8681
8932
  "temperature",
8682
8933
  "top_p",
8683
- "top_k"
8934
+ "top_k",
8935
+ "thinking.type.enabled"
8684
8936
  ],
8685
8937
  "status": "candidate"
8686
8938
  },
@@ -12934,7 +13186,7 @@
12934
13186
  "upstreamId": "MiniMax-M2.7",
12935
13187
  "displayName": "MiniMax-M2.7",
12936
13188
  "aliases": [],
12937
- "$comment": "OpenAI Chat base https://api.minimax.io/v1. M2.x thinking cannot be disabled; 'disabled' is accepted but ignored. presence_penalty, frequency_penalty, logit_bias are ignored; n supports only 1; function_call deprecated and unsupported. Complete assistant content/thinking/tool calls must be replayed for tool-loop continuity. — 2026-09-25 refresh: OpenAI Chat base https://api.minimax.io/v1. M2.x thinking cannot be disabled; 'disabled' is accepted but ignored. presence_penalty, frequency_penalty, logit_bias are ignored; n supports only 1; function_call deprecated and unsupported. Complete assistant content/thinking/tool calls must be replayed for tool-loop continuity. ",
13189
+ "$comment": "OpenAI Chat base https://api.minimax.io/v1. M2.x thinking cannot be disabled; 'disabled' is accepted but ignored. presence_penalty, frequency_penalty, logit_bias are ignored; n supports only 1; function_call deprecated and unsupported. Complete assistant content/thinking/tool calls must be replayed for tool-loop continuity. — 2026-09-25 refresh: OpenAI Chat base https://api.minimax.io/v1. M2.x thinking cannot be disabled; 'disabled' is accepted but ignored. presence_penalty, frequency_penalty, logit_bias are ignored; n supports only 1; function_call deprecated and unsupported. Complete assistant content/thinking/tool calls must be replayed for tool-loop continuity. — 2026-09-25 audit fix (0.0.24): https://platform.minimax.io/docs/guides/pricing-paygo — live pay-as-you-go table supersedes JavaScript-only subscription-widget link",
12938
13190
  "endpoints": [
12939
13191
  "chat"
12940
13192
  ],
@@ -13010,9 +13262,9 @@
13010
13262
  "cacheWritePerMTokUsd": 0.375
13011
13263
  },
13012
13264
  "source": "official-doc",
13013
- "sourceRef": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise — current MiniMax API pay-as-you-go Token Plan price table for this M2.7 model, per million tokens (retrieved 2026-09-19).",
13265
+ "sourceRef": "https://platform.minimax.io/docs/guides/pricing-paygo — MiniMax-M2.7 Pay as You Go Standard table: $0.3 input/$1.2 output/$0.06 prompt-cache read/$0.375 prompt-cache write per million tokens",
13014
13266
  "confidence": "declared",
13015
- "observedAt": "2026-09-19T00:00:00Z"
13267
+ "observedAt": "2026-09-25T12:31:56.381Z"
13016
13268
  },
13017
13269
  "unsupportedParameters": [
13018
13270
  "function_call"
@@ -13025,7 +13277,7 @@
13025
13277
  "upstreamId": "MiniMax-M2.7-highspeed",
13026
13278
  "displayName": "MiniMax-M2.7-highspeed",
13027
13279
  "aliases": [],
13028
- "$comment": "OpenAI Chat base https://api.minimax.io/v1. M2.x thinking cannot be disabled; 'disabled' is accepted but ignored. presence_penalty, frequency_penalty, logit_bias are ignored; n supports only 1; function_call deprecated and unsupported. Complete assistant content/thinking/tool calls must be replayed for tool-loop continuity. — 2026-09-25 refresh: OpenAI Chat base https://api.minimax.io/v1. M2.x thinking cannot be disabled; 'disabled' is accepted but ignored. presence_penalty, frequency_penalty, logit_bias are ignored; n supports only 1; function_call deprecated and unsupported. Complete assistant content/thinking/tool calls must be replayed for tool-loop continuity. ",
13280
+ "$comment": "OpenAI Chat base https://api.minimax.io/v1. M2.x thinking cannot be disabled; 'disabled' is accepted but ignored. presence_penalty, frequency_penalty, logit_bias are ignored; n supports only 1; function_call deprecated and unsupported. Complete assistant content/thinking/tool calls must be replayed for tool-loop continuity. — 2026-09-25 refresh: OpenAI Chat base https://api.minimax.io/v1. M2.x thinking cannot be disabled; 'disabled' is accepted but ignored. presence_penalty, frequency_penalty, logit_bias are ignored; n supports only 1; function_call deprecated and unsupported. Complete assistant content/thinking/tool calls must be replayed for tool-loop continuity. — 2026-09-25 audit fix (0.0.24): https://platform.minimax.io/docs/guides/pricing-paygo — live pay-as-you-go table supersedes JavaScript-only subscription-widget link",
13029
13281
  "endpoints": [
13030
13282
  "chat"
13031
13283
  ],
@@ -13101,9 +13353,9 @@
13101
13353
  "cacheWritePerMTokUsd": 0.375
13102
13354
  },
13103
13355
  "source": "official-doc",
13104
- "sourceRef": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise — current MiniMax API pay-as-you-go Token Plan price table for this M2.7 model, per million tokens (retrieved 2026-09-19).",
13356
+ "sourceRef": "https://platform.minimax.io/docs/guides/pricing-paygo — MiniMax-M2.7-highspeed Pay as You Go Standard table: $0.6 input/$2.4 output/$0.06 prompt-cache read/$0.375 prompt-cache write per million tokens",
13105
13357
  "confidence": "declared",
13106
- "observedAt": "2026-09-19T00:00:00Z"
13358
+ "observedAt": "2026-09-25T12:31:56.381Z"
13107
13359
  },
13108
13360
  "unsupportedParameters": [
13109
13361
  "function_call"
@@ -21588,10 +21840,17 @@
21588
21840
  "upstreamId": "glm-4.7-flashx",
21589
21841
  "displayName": "GLM-4.7-FLASHX",
21590
21842
  "aliases": [],
21591
- "$comment": "OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: no documented off switch; core parameters describe GLM-4.7 forced thinking (Flash variants inferred). Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs.",
21843
+ "$comment": "OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: no documented off switch; core parameters describe GLM-4.7 forced thinking (Flash variants inferred). Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs. — 2026-09-25 audit fix (0.0.24): https://docs.z.ai/llms-full.txt — glm-4.7-flashx context length 200K",
21592
21844
  "endpoints": [
21593
21845
  "chat"
21594
21846
  ],
21847
+ "contextWindow": {
21848
+ "value": 200000,
21849
+ "source": "official-doc",
21850
+ "sourceRef": "https://docs.z.ai/llms-full.txt — glm-4.7-flashx model guide/table gives 200K context length; K expanded as 1,000 tokens",
21851
+ "confidence": "inferred",
21852
+ "observedAt": "2026-09-25T12:31:56.381Z"
21853
+ },
21595
21854
  "inputModalities": {
21596
21855
  "value": [
21597
21856
  "text"
@@ -21662,10 +21921,17 @@
21662
21921
  "upstreamId": "glm-4.6",
21663
21922
  "displayName": "GLM-4.6",
21664
21923
  "aliases": [],
21665
- "$comment": "OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: yes; thinking.type=disabled. Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs.",
21924
+ "$comment": "OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: yes; thinking.type=disabled. Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs. — 2026-09-25 audit fix (0.0.24): https://docs.z.ai/guides/llm/glm-4.6 — glm-4.6 context length 200K",
21666
21925
  "endpoints": [
21667
21926
  "chat"
21668
21927
  ],
21928
+ "contextWindow": {
21929
+ "value": 200000,
21930
+ "source": "official-doc",
21931
+ "sourceRef": "https://docs.z.ai/guides/llm/glm-4.6 — glm-4.6 model guide/table gives 200K context length; K expanded as 1,000 tokens",
21932
+ "confidence": "inferred",
21933
+ "observedAt": "2026-09-25T12:31:56.381Z"
21934
+ },
21669
21935
  "maxOutputTokens": {
21670
21936
  "value": 131072,
21671
21937
  "source": "official-doc",
@@ -21743,10 +22009,17 @@
21743
22009
  "upstreamId": "glm-4.5",
21744
22010
  "displayName": "GLM-4.5",
21745
22011
  "aliases": [],
21746
- "$comment": "OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: yes; thinking.type=disabled. Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs.",
22012
+ "$comment": "OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: yes; thinking.type=disabled. Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs. — 2026-09-25 audit fix (0.0.24): https://docs.z.ai/guides/llm/glm-4.5 — glm-4.5 context length 128K",
21747
22013
  "endpoints": [
21748
22014
  "chat"
21749
22015
  ],
22016
+ "contextWindow": {
22017
+ "value": 128000,
22018
+ "source": "official-doc",
22019
+ "sourceRef": "https://docs.z.ai/guides/llm/glm-4.5 — glm-4.5 model guide/table gives 128K context length; K expanded as 1,000 tokens",
22020
+ "confidence": "inferred",
22021
+ "observedAt": "2026-09-25T12:31:56.381Z"
22022
+ },
21750
22023
  "maxOutputTokens": {
21751
22024
  "value": 98304,
21752
22025
  "source": "official-doc",
@@ -21824,10 +22097,17 @@
21824
22097
  "upstreamId": "glm-4.5-air",
21825
22098
  "displayName": "GLM-4.5-AIR",
21826
22099
  "aliases": [],
21827
- "$comment": "OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: yes; thinking.type=disabled. Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs.",
22100
+ "$comment": "OpenAI Chat base https://api.z.ai/api/paas/v4. Thinking off: yes; thinking.type=disabled. Effort control: thinking.type toggle only; no documented graded effort. PAYG key support on /api/v1 Responses is not documented; /api/v1 is presented in Coding Plan docs. — 2026-09-25 audit fix (0.0.24): https://docs.z.ai/llms-full.txt — glm-4.5-air context length 128K",
21828
22101
  "endpoints": [
21829
22102
  "chat"
21830
22103
  ],
22104
+ "contextWindow": {
22105
+ "value": 128000,
22106
+ "source": "official-doc",
22107
+ "sourceRef": "https://docs.z.ai/llms-full.txt — glm-4.5-air model guide/table gives 128K context length; K expanded as 1,000 tokens",
22108
+ "confidence": "inferred",
22109
+ "observedAt": "2026-09-25T12:31:56.381Z"
22110
+ },
21831
22111
  "maxOutputTokens": {
21832
22112
  "value": 98304,
21833
22113
  "source": "official-doc",
@@ -25579,7 +25859,7 @@
25579
25859
  "upstreamId": "qwen3.8-max",
25580
25860
  "displayName": "Qwen3.8 Max",
25581
25861
  "aliases": [],
25582
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: reasoning_effort=none (maps to enable_thinking=false), also enable_thinking=false. Chat effort: reasoning_effort low/medium/xhigh; minimal maps low and high/max map xhigh. Do not combine reasoning_effort with thinking_budget. Responses uses reasoning.effort separately. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
25862
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: reasoning_effort=none (maps to enable_thinking=false), also enable_thinking=false. Chat effort: reasoning_effort low/medium/xhigh; minimal maps low and high/max map xhigh. Do not combine reasoning_effort with thinking_budget. Responses uses reasoning.effort separately. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
25583
25863
  "endpoints": [
25584
25864
  "chat"
25585
25865
  ],
@@ -25657,13 +25937,13 @@
25657
25937
  },
25658
25938
  "pricing": {
25659
25939
  "value": {
25660
- "inputPerMTokUsd": 1.65,
25661
- "outputPerMTokUsd": 4.951
25940
+ "inputPerMTokUsd": 1.788,
25941
+ "outputPerMTokUsd": 5.364
25662
25942
  },
25663
25943
  "source": "official-doc",
25664
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.8-max: input $1.65, output $4.951 per million tokens; 0–1M input tokens",
25665
- "confidence": "declared",
25666
- "observedAt": "2026-09-25T09:40:00Z"
25944
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.8-max, 0–1M input tier: ¥12 input/¥36 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
25945
+ "confidence": "inferred",
25946
+ "observedAt": "2026-09-25T12:31:56.381Z"
25667
25947
  },
25668
25948
  "unsupportedParameters": [],
25669
25949
  "status": "candidate"
@@ -25674,7 +25954,7 @@
25674
25954
  "upstreamId": "qwen3.7-max",
25675
25955
  "displayName": "Qwen3.7 Max",
25676
25956
  "aliases": [],
25677
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Currently equivalent to qwen3.7-max-2026-05-20; this mutable alias is not a fixed snapshot. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
25957
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Currently equivalent to qwen3.7-max-2026-05-20; this mutable alias is not a fixed snapshot. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
25678
25958
  "endpoints": [
25679
25959
  "chat"
25680
25960
  ],
@@ -25744,13 +26024,13 @@
25744
26024
  },
25745
26025
  "pricing": {
25746
26026
  "value": {
25747
- "inputPerMTokUsd": 1.65,
25748
- "outputPerMTokUsd": 4.951
26027
+ "inputPerMTokUsd": 1.788,
26028
+ "outputPerMTokUsd": 5.364
25749
26029
  },
25750
26030
  "source": "official-doc",
25751
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.7-max: input $1.65, output $4.951 per million tokens; 0–1M input tokens",
25752
- "confidence": "declared",
25753
- "observedAt": "2026-09-25T09:40:00Z"
26031
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.7-max, 0–1M input tier: ¥12 input/¥36 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
26032
+ "confidence": "inferred",
26033
+ "observedAt": "2026-09-25T12:31:56.381Z"
25754
26034
  },
25755
26035
  "unsupportedParameters": [],
25756
26036
  "status": "candidate"
@@ -25761,7 +26041,7 @@
25761
26041
  "upstreamId": "qwen3.7-plus",
25762
26042
  "displayName": "Qwen3.7 Plus",
25763
26043
  "aliases": [],
25764
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
26044
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
25765
26045
  "endpoints": [
25766
26046
  "chat"
25767
26047
  ],
@@ -25840,13 +26120,13 @@
25840
26120
  },
25841
26121
  "pricing": {
25842
26122
  "value": {
25843
- "inputPerMTokUsd": 0.276,
25844
- "outputPerMTokUsd": 1.101
26123
+ "inputPerMTokUsd": 0.298,
26124
+ "outputPerMTokUsd": 1.192
25845
26125
  },
25846
26126
  "source": "official-doc",
25847
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.7-plus: input $0.276, output $1.101 per million tokens; ≤256K input; >256K–1M $0.826/$3.301; list price before limited-time 20% discount",
25848
- "confidence": "declared",
25849
- "observedAt": "2026-09-25T09:40:00Z"
26127
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.7-plus, ≤256K; >256K–1M: ¥6/¥24; base list price before temporary 20% promotion input tier: ¥2 input/¥8 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
26128
+ "confidence": "inferred",
26129
+ "observedAt": "2026-09-25T12:31:56.381Z"
25850
26130
  },
25851
26131
  "unsupportedParameters": [],
25852
26132
  "status": "candidate"
@@ -25857,7 +26137,7 @@
25857
26137
  "upstreamId": "qwen3.6-plus",
25858
26138
  "displayName": "Qwen3.6 Plus",
25859
26139
  "aliases": [],
25860
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
26140
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
25861
26141
  "endpoints": [
25862
26142
  "chat"
25863
26143
  ],
@@ -25929,13 +26209,13 @@
25929
26209
  },
25930
26210
  "pricing": {
25931
26211
  "value": {
25932
- "inputPerMTokUsd": 0.276,
25933
- "outputPerMTokUsd": 1.651
26212
+ "inputPerMTokUsd": 0.298,
26213
+ "outputPerMTokUsd": 1.788
25934
26214
  },
25935
26215
  "source": "official-doc",
25936
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.6-plus: input $0.276, output $1.651 per million tokens; ≤256K input; >256K–1M $1.101/$6.602",
25937
- "confidence": "declared",
25938
- "observedAt": "2026-09-25T09:40:00Z"
26216
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.6-plus, ≤256K; >256K–1M: ¥8/¥48 input tier: ¥2 input/¥12 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
26217
+ "confidence": "inferred",
26218
+ "observedAt": "2026-09-25T12:31:56.381Z"
25939
26219
  },
25940
26220
  "unsupportedParameters": [],
25941
26221
  "status": "candidate"
@@ -25946,7 +26226,7 @@
25946
26226
  "upstreamId": "qwen3.5-plus",
25947
26227
  "displayName": "Qwen3.5 Plus",
25948
26228
  "aliases": [],
25949
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
26229
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
25950
26230
  "endpoints": [
25951
26231
  "chat"
25952
26232
  ],
@@ -26018,13 +26298,13 @@
26018
26298
  },
26019
26299
  "pricing": {
26020
26300
  "value": {
26021
- "inputPerMTokUsd": 0.115,
26022
- "outputPerMTokUsd": 0.688
26301
+ "inputPerMTokUsd": 0.1192,
26302
+ "outputPerMTokUsd": 0.7152
26023
26303
  },
26024
26304
  "source": "official-doc",
26025
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.5-plus: input $0.115, output $0.688 per million tokens; ≤128K input; >128K–256K $0.287/$1.72; >256K–1M $0.573/$3.44",
26026
- "confidence": "declared",
26027
- "observedAt": "2026-09-25T09:40:00Z"
26305
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.5-plus, ≤128K; 128K–256K: ¥2/¥12; 256K–1M: ¥4/¥24 input tier: ¥0.8 input/¥4.8 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
26306
+ "confidence": "inferred",
26307
+ "observedAt": "2026-09-25T12:31:56.381Z"
26028
26308
  },
26029
26309
  "unsupportedParameters": [],
26030
26310
  "status": "candidate"
@@ -26035,7 +26315,7 @@
26035
26315
  "upstreamId": "qwen3.8-flash",
26036
26316
  "displayName": "Qwen3.8 Flash",
26037
26317
  "aliases": [],
26038
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: reasoning_effort=none (maps to enable_thinking=false), also enable_thinking=false. Chat effort: reasoning_effort low/medium/xhigh; minimal maps low and high/max map xhigh. Do not combine reasoning_effort with thinking_budget. Responses uses reasoning.effort separately. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
26318
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: reasoning_effort=none (maps to enable_thinking=false), also enable_thinking=false. Chat effort: reasoning_effort low/medium/xhigh; minimal maps low and high/max map xhigh. Do not combine reasoning_effort with thinking_budget. Responses uses reasoning.effort separately. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
26039
26319
  "endpoints": [
26040
26320
  "chat"
26041
26321
  ],
@@ -26113,13 +26393,13 @@
26113
26393
  },
26114
26394
  "pricing": {
26115
26395
  "value": {
26116
- "inputPerMTokUsd": 0.113,
26117
- "outputPerMTokUsd": 0.382
26396
+ "inputPerMTokUsd": 0.1192,
26397
+ "outputPerMTokUsd": 0.4023
26118
26398
  },
26119
26399
  "source": "official-doc",
26120
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.8-flash: input $0.113, output $0.382 per million tokens; 0–1M input tokens",
26121
- "confidence": "declared",
26122
- "observedAt": "2026-09-25T09:40:00Z"
26400
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.8-flash, 0–1M input tier: ¥0.8 input/¥2.7 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
26401
+ "confidence": "inferred",
26402
+ "observedAt": "2026-09-25T12:31:56.381Z"
26123
26403
  },
26124
26404
  "unsupportedParameters": [],
26125
26405
  "status": "candidate"
@@ -26130,7 +26410,7 @@
26130
26410
  "upstreamId": "qwen3.6-flash",
26131
26411
  "displayName": "Qwen3.6 Flash",
26132
26412
  "aliases": [],
26133
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
26413
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
26134
26414
  "endpoints": [
26135
26415
  "chat"
26136
26416
  ],
@@ -26209,13 +26489,13 @@
26209
26489
  },
26210
26490
  "pricing": {
26211
26491
  "value": {
26212
- "inputPerMTokUsd": 0.165,
26213
- "outputPerMTokUsd": 0.99
26492
+ "inputPerMTokUsd": 0.1788,
26493
+ "outputPerMTokUsd": 1.0728
26214
26494
  },
26215
26495
  "source": "official-doc",
26216
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.6-flash: input $0.165, output $0.99 per million tokens; ≤256K input; >256K–1M $0.66/$3.961",
26217
- "confidence": "declared",
26218
- "observedAt": "2026-09-25T09:40:00Z"
26496
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.6-flash, ≤256K; 256K–1M: ¥4.8/¥28.8 input tier: ¥1.2 input/¥7.2 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
26497
+ "confidence": "inferred",
26498
+ "observedAt": "2026-09-25T12:31:56.381Z"
26219
26499
  },
26220
26500
  "unsupportedParameters": [],
26221
26501
  "status": "candidate"
@@ -26226,7 +26506,7 @@
26226
26506
  "upstreamId": "qwen3-coder-plus",
26227
26507
  "displayName": "Qwen3 Coder Plus",
26228
26508
  "aliases": [],
26229
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Responses guide omits this ID from explicit enhanced-compatibility roster; keep Chat only until endpoint support confirmed. Conflict adjudication: Sep 24 overview vs Sep 11 model page; structured output=true from newer overview (inferred). Function calling=native for Beijing per region-specific model page; see both URLs in field sourceRefs.",
26509
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Responses guide omits this ID from explicit enhanced-compatibility roster; keep Chat only until endpoint support confirmed. Conflict adjudication: Sep 24 overview vs Sep 11 model page; structured output=true from newer overview (inferred). Function calling=native for Beijing per region-specific model page; see both URLs in field sourceRefs. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
26230
26510
  "endpoints": [
26231
26511
  "chat"
26232
26512
  ],
@@ -26310,13 +26590,13 @@
26310
26590
  },
26311
26591
  "pricing": {
26312
26592
  "value": {
26313
- "inputPerMTokUsd": 0.574,
26314
- "outputPerMTokUsd": 2.294
26593
+ "inputPerMTokUsd": 0.596,
26594
+ "outputPerMTokUsd": 2.384
26315
26595
  },
26316
26596
  "source": "official-doc",
26317
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-coder-plus — China (Beijing) qwen3-coder-plus: input $0.574, output $2.294 per million tokens; ≤32K input; >32K–128K $0.861/$3.441; >128K–256K $1.434/$5.735; >256K–1M $2.868/$28.671",
26318
- "confidence": "declared",
26319
- "observedAt": "2026-09-25T09:40:00Z"
26597
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3-coder-plus, ≤32K; 32K–128K: ¥6/¥24; 128K–256K: ¥10/¥40; 256K–1M: ¥20/¥200 input tier: ¥4 input/¥16 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
26598
+ "confidence": "inferred",
26599
+ "observedAt": "2026-09-25T12:31:56.381Z"
26320
26600
  },
26321
26601
  "unsupportedParameters": [],
26322
26602
  "status": "candidate"
@@ -26327,7 +26607,7 @@
26327
26607
  "upstreamId": "qwen3-coder-next",
26328
26608
  "displayName": "Qwen3 Coder Next",
26329
26609
  "aliases": [],
26330
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Responses guide omits this ID from explicit enhanced-compatibility roster; keep Chat only until endpoint support confirmed. Conflict adjudication: Sep 24 overview vs Sep 11 model page; structured output=true from newer overview (inferred). Function calling=native for Beijing from newer overview; see both URLs in field sourceRefs.",
26610
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Responses guide omits this ID from explicit enhanced-compatibility roster; keep Chat only until endpoint support confirmed. Conflict adjudication: Sep 24 overview vs Sep 11 model page; structured output=true from newer overview (inferred). Function calling=native for Beijing from newer overview; see both URLs in field sourceRefs. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
26331
26611
  "endpoints": [
26332
26612
  "chat"
26333
26613
  ],
@@ -26411,13 +26691,13 @@
26411
26691
  },
26412
26692
  "pricing": {
26413
26693
  "value": {
26414
- "inputPerMTokUsd": 0.144,
26415
- "outputPerMTokUsd": 0.574
26694
+ "inputPerMTokUsd": 0.149,
26695
+ "outputPerMTokUsd": 0.596
26416
26696
  },
26417
26697
  "source": "official-doc",
26418
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-coder-next — China (Beijing) qwen3-coder-next: input $0.144, output $0.574 per million tokens; ≤32K input; >32K–128K $0.216/$0.861; >128K–256K $0.359/$1.434",
26419
- "confidence": "declared",
26420
- "observedAt": "2026-09-25T09:40:00Z"
26698
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3-coder-next, ≤32K; 32K–128K: ¥1.5/¥6; 128K–256K: ¥2.5/¥10 input tier: ¥1 input/¥4 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
26699
+ "confidence": "inferred",
26700
+ "observedAt": "2026-09-25T12:31:56.381Z"
26421
26701
  },
26422
26702
  "unsupportedParameters": [],
26423
26703
  "status": "candidate"
@@ -26428,7 +26708,7 @@
26428
26708
  "upstreamId": "qwen3.5-122b-a10b",
26429
26709
  "displayName": "Qwen3.5 122B A10B",
26430
26710
  "aliases": [],
26431
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Context conflict: individual model page states 262,144; Sep24 vision overview lists 32k context/8k output for vision usage. Catalog uses model-specific limit; vision mode may have lower cap. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
26711
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Context conflict: individual model page states 262,144; Sep24 vision overview lists 32k context/8k output for vision usage. Catalog uses model-specific limit; vision mode may have lower cap. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
26432
26712
  "endpoints": [
26433
26713
  "chat"
26434
26714
  ],
@@ -26514,13 +26794,13 @@
26514
26794
  },
26515
26795
  "pricing": {
26516
26796
  "value": {
26517
- "inputPerMTokUsd": 0.115,
26518
- "outputPerMTokUsd": 0.917
26797
+ "inputPerMTokUsd": 0.1192,
26798
+ "outputPerMTokUsd": 0.9536
26519
26799
  },
26520
26800
  "source": "official-doc",
26521
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-5-122b-a10b — China (Beijing) qwen3.5-122b-a10b: input $0.115, output $0.917 per million tokens; ≤128K input; >128K–256K $0.287/$2.294",
26522
- "confidence": "declared",
26523
- "observedAt": "2026-09-25T09:40:00Z"
26801
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.5-122b-a10b, ≤128K; >128K–256K: ¥2/¥16 input tier: ¥0.8 input/¥6.4 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
26802
+ "confidence": "inferred",
26803
+ "observedAt": "2026-09-25T12:31:56.381Z"
26524
26804
  },
26525
26805
  "unsupportedParameters": [],
26526
26806
  "status": "candidate"
@@ -26531,7 +26811,7 @@
26531
26811
  "upstreamId": "qwen3.5-397b-a17b",
26532
26812
  "displayName": "Qwen3.5 397B A17B",
26533
26813
  "aliases": [],
26534
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Context conflict: individual model page states 262,144; Sep24 vision overview lists 32k context/8k output for vision usage. Catalog uses model-specific limit; vision mode may have lower cap. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
26814
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Context conflict: individual model page states 262,144; Sep24 vision overview lists 32k context/8k output for vision usage. Catalog uses model-specific limit; vision mode may have lower cap. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
26535
26815
  "endpoints": [
26536
26816
  "chat"
26537
26817
  ],
@@ -26617,13 +26897,13 @@
26617
26897
  },
26618
26898
  "pricing": {
26619
26899
  "value": {
26620
- "inputPerMTokUsd": 0.172,
26621
- "outputPerMTokUsd": 1.032
26900
+ "inputPerMTokUsd": 0.1788,
26901
+ "outputPerMTokUsd": 1.0728
26622
26902
  },
26623
26903
  "source": "official-doc",
26624
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-5-397b-a17b — China (Beijing) qwen3.5-397b-a17b: input $0.172, output $1.032 per million tokens; ≤128K input; >128K–256K $0.43/$2.58",
26625
- "confidence": "declared",
26626
- "observedAt": "2026-09-25T09:40:00Z"
26904
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.5-397b-a17b, ≤128K; >128K–256K: ¥3/¥18 input tier: ¥1.2 input/¥7.2 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
26905
+ "confidence": "inferred",
26906
+ "observedAt": "2026-09-25T12:31:56.381Z"
26627
26907
  },
26628
26908
  "unsupportedParameters": [],
26629
26909
  "status": "candidate"
@@ -26634,7 +26914,7 @@
26634
26914
  "upstreamId": "qwen3.6-27b",
26635
26915
  "displayName": "Qwen3.6 27B",
26636
26916
  "aliases": [],
26637
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Responses guide omits this ID from explicit enhanced-compatibility roster; keep Chat only until endpoint support confirmed.",
26917
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. Responses guide omits this ID from explicit enhanced-compatibility roster; keep Chat only until endpoint support confirmed. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
26638
26918
  "endpoints": [
26639
26919
  "chat"
26640
26920
  ],
@@ -26720,13 +27000,13 @@
26720
27000
  },
26721
27001
  "pricing": {
26722
27002
  "value": {
26723
- "inputPerMTokUsd": 0.412564,
26724
- "outputPerMTokUsd": 2.475384
27003
+ "inputPerMTokUsd": 0.447,
27004
+ "outputPerMTokUsd": 2.682
26725
27005
  },
26726
27006
  "source": "official-doc",
26727
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-6-27b — China (Beijing) qwen3.6-27b: input $0.412564, output $2.475384 per million tokens; all input",
26728
- "confidence": "declared",
26729
- "observedAt": "2026-09-25T09:40:00Z"
27007
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.6-27b, ≤256K input tier: ¥3 input/¥18 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
27008
+ "confidence": "inferred",
27009
+ "observedAt": "2026-09-25T12:31:56.381Z"
26730
27010
  },
26731
27011
  "unsupportedParameters": [],
26732
27012
  "status": "candidate"
@@ -26737,7 +27017,7 @@
26737
27017
  "upstreamId": "qwen3.6-35b-a3b",
26738
27018
  "displayName": "Qwen3.6 35B A3B",
26739
27019
  "aliases": [],
26740
- "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
27020
+ "$comment": "PAYG China (Beijing) OpenAI-compatible Chat base: https://dashscope.aliyuncs.com/compatible-mode/v1. Thinking off: enable_thinking=false. Chat effort is via thinking_budget, not reasoning_effort; Winter OpenAI chat adapter cannot send generic effort here, hence efforts=[]. Responses uses reasoning.effort separately. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
26741
27021
  "endpoints": [
26742
27022
  "chat"
26743
27023
  ],
@@ -26823,13 +27103,13 @@
26823
27103
  },
26824
27104
  "pricing": {
26825
27105
  "value": {
26826
- "inputPerMTokUsd": 0.248,
26827
- "outputPerMTokUsd": 1.485
27106
+ "inputPerMTokUsd": 0.2682,
27107
+ "outputPerMTokUsd": 1.6092
26828
27108
  },
26829
27109
  "source": "official-doc",
26830
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-6-35b-a3b — China (Beijing) qwen3.6-35b-a3b: input $0.248, output $1.485 per million tokens; all input",
26831
- "confidence": "declared",
26832
- "observedAt": "2026-09-25T09:40:00Z"
27110
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.6-35b-a3b, ≤256K input tier: ¥1.8 input/¥10.8 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
27111
+ "confidence": "inferred",
27112
+ "observedAt": "2026-09-25T12:31:56.381Z"
26833
27113
  },
26834
27114
  "unsupportedParameters": [],
26835
27115
  "status": "candidate"
@@ -26840,7 +27120,7 @@
26840
27120
  "upstreamId": "qwen3.8-max",
26841
27121
  "displayName": "Qwen3.8 Max (Anthropic dialect)",
26842
27122
  "aliases": [],
26843
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort.",
27123
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
26844
27124
  "endpoints": [
26845
27125
  "chat"
26846
27126
  ],
@@ -26918,13 +27198,13 @@
26918
27198
  },
26919
27199
  "pricing": {
26920
27200
  "value": {
26921
- "inputPerMTokUsd": 1.65,
26922
- "outputPerMTokUsd": 4.951
27201
+ "inputPerMTokUsd": 1.788,
27202
+ "outputPerMTokUsd": 5.364
26923
27203
  },
26924
27204
  "source": "official-doc",
26925
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.8-max: input $1.65, output $4.951 per million tokens; 0–1M input tokens",
26926
- "confidence": "declared",
26927
- "observedAt": "2026-09-25T09:40:00Z"
27205
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.8-max, 0–1M input tier: ¥12 input/¥36 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
27206
+ "confidence": "inferred",
27207
+ "observedAt": "2026-09-25T12:31:56.381Z"
26928
27208
  },
26929
27209
  "unsupportedParameters": [],
26930
27210
  "status": "candidate"
@@ -26935,7 +27215,7 @@
26935
27215
  "upstreamId": "qwen3.7-max",
26936
27216
  "displayName": "Qwen3.7 Max (Anthropic dialect)",
26937
27217
  "aliases": [],
26938
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Currently equivalent to qwen3.7-max-2026-05-20; this mutable alias is not a fixed snapshot.",
27218
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Currently equivalent to qwen3.7-max-2026-05-20; this mutable alias is not a fixed snapshot. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
26939
27219
  "endpoints": [
26940
27220
  "chat"
26941
27221
  ],
@@ -27011,13 +27291,13 @@
27011
27291
  },
27012
27292
  "pricing": {
27013
27293
  "value": {
27014
- "inputPerMTokUsd": 1.65,
27015
- "outputPerMTokUsd": 4.951
27294
+ "inputPerMTokUsd": 1.788,
27295
+ "outputPerMTokUsd": 5.364
27016
27296
  },
27017
27297
  "source": "official-doc",
27018
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.7-max: input $1.65, output $4.951 per million tokens; 0–1M input tokens",
27019
- "confidence": "declared",
27020
- "observedAt": "2026-09-25T09:40:00Z"
27298
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.7-max, 0–1M input tier: ¥12 input/¥36 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
27299
+ "confidence": "inferred",
27300
+ "observedAt": "2026-09-25T12:31:56.381Z"
27021
27301
  },
27022
27302
  "unsupportedParameters": [],
27023
27303
  "status": "candidate"
@@ -27028,7 +27308,7 @@
27028
27308
  "upstreamId": "qwen3.7-plus",
27029
27309
  "displayName": "Qwen3.7 Plus (Anthropic dialect)",
27030
27310
  "aliases": [],
27031
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort.",
27311
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
27032
27312
  "endpoints": [
27033
27313
  "chat"
27034
27314
  ],
@@ -27113,13 +27393,13 @@
27113
27393
  },
27114
27394
  "pricing": {
27115
27395
  "value": {
27116
- "inputPerMTokUsd": 0.276,
27117
- "outputPerMTokUsd": 1.101
27396
+ "inputPerMTokUsd": 0.298,
27397
+ "outputPerMTokUsd": 1.192
27118
27398
  },
27119
27399
  "source": "official-doc",
27120
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.7-plus: input $0.276, output $1.101 per million tokens; ≤256K input; >256K–1M $0.826/$3.301; list price before limited-time 20% discount",
27121
- "confidence": "declared",
27122
- "observedAt": "2026-09-25T09:40:00Z"
27400
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.7-plus, ≤256K; >256K–1M: ¥6/¥24; base list price before temporary 20% promotion input tier: ¥2 input/¥8 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
27401
+ "confidence": "inferred",
27402
+ "observedAt": "2026-09-25T12:31:56.381Z"
27123
27403
  },
27124
27404
  "unsupportedParameters": [],
27125
27405
  "status": "candidate"
@@ -27130,7 +27410,7 @@
27130
27410
  "upstreamId": "qwen3.6-plus",
27131
27411
  "displayName": "Qwen3.6 Plus (Anthropic dialect)",
27132
27412
  "aliases": [],
27133
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced.",
27413
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
27134
27414
  "endpoints": [
27135
27415
  "chat"
27136
27416
  ],
@@ -27208,13 +27488,13 @@
27208
27488
  },
27209
27489
  "pricing": {
27210
27490
  "value": {
27211
- "inputPerMTokUsd": 0.276,
27212
- "outputPerMTokUsd": 1.651
27491
+ "inputPerMTokUsd": 0.298,
27492
+ "outputPerMTokUsd": 1.788
27213
27493
  },
27214
27494
  "source": "official-doc",
27215
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.6-plus: input $0.276, output $1.651 per million tokens; ≤256K input; >256K–1M $1.101/$6.602",
27216
- "confidence": "declared",
27217
- "observedAt": "2026-09-25T09:40:00Z"
27495
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.6-plus, ≤256K; >256K–1M: ¥8/¥48 input tier: ¥2 input/¥12 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
27496
+ "confidence": "inferred",
27497
+ "observedAt": "2026-09-25T12:31:56.381Z"
27218
27498
  },
27219
27499
  "unsupportedParameters": [],
27220
27500
  "status": "candidate"
@@ -27225,7 +27505,7 @@
27225
27505
  "upstreamId": "qwen3.5-plus",
27226
27506
  "displayName": "Qwen3.5 Plus (Anthropic dialect)",
27227
27507
  "aliases": [],
27228
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced.",
27508
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
27229
27509
  "endpoints": [
27230
27510
  "chat"
27231
27511
  ],
@@ -27303,13 +27583,13 @@
27303
27583
  },
27304
27584
  "pricing": {
27305
27585
  "value": {
27306
- "inputPerMTokUsd": 0.115,
27307
- "outputPerMTokUsd": 0.688
27586
+ "inputPerMTokUsd": 0.1192,
27587
+ "outputPerMTokUsd": 0.7152
27308
27588
  },
27309
27589
  "source": "official-doc",
27310
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.5-plus: input $0.115, output $0.688 per million tokens; ≤128K input; >128K–256K $0.287/$1.72; >256K–1M $0.573/$3.44",
27311
- "confidence": "declared",
27312
- "observedAt": "2026-09-25T09:40:00Z"
27590
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.5-plus, ≤128K; 128K–256K: ¥2/¥12; 256K–1M: ¥4/¥24 input tier: ¥0.8 input/¥4.8 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
27591
+ "confidence": "inferred",
27592
+ "observedAt": "2026-09-25T12:31:56.381Z"
27313
27593
  },
27314
27594
  "unsupportedParameters": [],
27315
27595
  "status": "candidate"
@@ -27320,7 +27600,7 @@
27320
27600
  "upstreamId": "qwen3.8-flash",
27321
27601
  "displayName": "Qwen3.8 Flash (Anthropic dialect)",
27322
27602
  "aliases": [],
27323
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort.",
27603
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
27324
27604
  "endpoints": [
27325
27605
  "chat"
27326
27606
  ],
@@ -27398,13 +27678,13 @@
27398
27678
  },
27399
27679
  "pricing": {
27400
27680
  "value": {
27401
- "inputPerMTokUsd": 0.113,
27402
- "outputPerMTokUsd": 0.382
27681
+ "inputPerMTokUsd": 0.1192,
27682
+ "outputPerMTokUsd": 0.4023
27403
27683
  },
27404
27684
  "source": "official-doc",
27405
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.8-flash: input $0.113, output $0.382 per million tokens; 0–1M input tokens",
27406
- "confidence": "declared",
27407
- "observedAt": "2026-09-25T09:40:00Z"
27685
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.8-flash, 0–1M input tier: ¥0.8 input/¥2.7 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
27686
+ "confidence": "inferred",
27687
+ "observedAt": "2026-09-25T12:31:56.381Z"
27408
27688
  },
27409
27689
  "unsupportedParameters": [],
27410
27690
  "status": "candidate"
@@ -27415,7 +27695,7 @@
27415
27695
  "upstreamId": "qwen3.6-flash",
27416
27696
  "displayName": "Qwen3.6 Flash (Anthropic dialect)",
27417
27697
  "aliases": [],
27418
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced.",
27698
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
27419
27699
  "endpoints": [
27420
27700
  "chat"
27421
27701
  ],
@@ -27500,13 +27780,13 @@
27500
27780
  },
27501
27781
  "pricing": {
27502
27782
  "value": {
27503
- "inputPerMTokUsd": 0.165,
27504
- "outputPerMTokUsd": 0.99
27783
+ "inputPerMTokUsd": 0.1788,
27784
+ "outputPerMTokUsd": 1.0728
27505
27785
  },
27506
27786
  "source": "official-doc",
27507
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/model-pricing — China (Beijing) qwen3.6-flash: input $0.165, output $0.99 per million tokens; ≤256K input; >256K–1M $0.66/$3.961",
27508
- "confidence": "declared",
27509
- "observedAt": "2026-09-25T09:40:00Z"
27787
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.6-flash, ≤256K; 256K–1M: ¥4.8/¥28.8 input tier: ¥1.2 input/¥7.2 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
27788
+ "confidence": "inferred",
27789
+ "observedAt": "2026-09-25T12:31:56.381Z"
27510
27790
  },
27511
27791
  "unsupportedParameters": [],
27512
27792
  "status": "candidate"
@@ -27517,7 +27797,7 @@
27517
27797
  "upstreamId": "qwen3-coder-plus",
27518
27798
  "displayName": "Qwen3 Coder Plus (Anthropic dialect)",
27519
27799
  "aliases": [],
27520
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Conflict adjudication: Sep 24 overview vs Sep 11 model page; structured output=true from newer overview (inferred). Function calling=native for Beijing per region-specific model page; see both URLs in field sourceRefs.",
27800
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Conflict adjudication: Sep 24 overview vs Sep 11 model page; structured output=true from newer overview (inferred). Function calling=native for Beijing per region-specific model page; see both URLs in field sourceRefs. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
27521
27801
  "endpoints": [
27522
27802
  "chat"
27523
27803
  ],
@@ -27607,13 +27887,13 @@
27607
27887
  },
27608
27888
  "pricing": {
27609
27889
  "value": {
27610
- "inputPerMTokUsd": 0.574,
27611
- "outputPerMTokUsd": 2.294
27890
+ "inputPerMTokUsd": 0.596,
27891
+ "outputPerMTokUsd": 2.384
27612
27892
  },
27613
27893
  "source": "official-doc",
27614
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-coder-plus — China (Beijing) qwen3-coder-plus: input $0.574, output $2.294 per million tokens; ≤32K input; >32K–128K $0.861/$3.441; >128K–256K $1.434/$5.735; >256K–1M $2.868/$28.671",
27615
- "confidence": "declared",
27616
- "observedAt": "2026-09-25T09:40:00Z"
27894
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3-coder-plus, ≤32K; 32K–128K: ¥6/¥24; 128K–256K: ¥10/¥40; 256K–1M: ¥20/¥200 input tier: ¥4 input/¥16 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
27895
+ "confidence": "inferred",
27896
+ "observedAt": "2026-09-25T12:31:56.381Z"
27617
27897
  },
27618
27898
  "unsupportedParameters": [],
27619
27899
  "status": "candidate"
@@ -27624,7 +27904,7 @@
27624
27904
  "upstreamId": "qwen3-coder-next",
27625
27905
  "displayName": "Qwen3 Coder Next (Anthropic dialect)",
27626
27906
  "aliases": [],
27627
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Conflict adjudication: Sep 24 overview vs Sep 11 model page; structured output=true from newer overview (inferred). Function calling=native for Beijing from newer overview; see both URLs in field sourceRefs.",
27907
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Conflict adjudication: Sep 24 overview vs Sep 11 model page; structured output=true from newer overview (inferred). Function calling=native for Beijing from newer overview; see both URLs in field sourceRefs. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
27628
27908
  "endpoints": [
27629
27909
  "chat"
27630
27910
  ],
@@ -27714,13 +27994,13 @@
27714
27994
  },
27715
27995
  "pricing": {
27716
27996
  "value": {
27717
- "inputPerMTokUsd": 0.144,
27718
- "outputPerMTokUsd": 0.574
27997
+ "inputPerMTokUsd": 0.149,
27998
+ "outputPerMTokUsd": 0.596
27719
27999
  },
27720
28000
  "source": "official-doc",
27721
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-coder-next — China (Beijing) qwen3-coder-next: input $0.144, output $0.574 per million tokens; ≤32K input; >32K–128K $0.216/$0.861; >128K–256K $0.359/$1.434",
27722
- "confidence": "declared",
27723
- "observedAt": "2026-09-25T09:40:00Z"
28001
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3-coder-next, ≤32K; 32K–128K: ¥1.5/¥6; 128K–256K: ¥2.5/¥10 input tier: ¥1 input/¥4 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
28002
+ "confidence": "inferred",
28003
+ "observedAt": "2026-09-25T12:31:56.381Z"
27724
28004
  },
27725
28005
  "unsupportedParameters": [],
27726
28006
  "status": "candidate"
@@ -27731,7 +28011,7 @@
27731
28011
  "upstreamId": "qwen3.5-122b-a10b",
27732
28012
  "displayName": "Qwen3.5 122B A10B (Anthropic dialect)",
27733
28013
  "aliases": [],
27734
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Context conflict: individual model page states 262,144; Sep24 vision overview lists 32k context/8k output for vision usage. Catalog uses model-specific limit; vision mode may have lower cap.",
28014
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Context conflict: individual model page states 262,144; Sep24 vision overview lists 32k context/8k output for vision usage. Catalog uses model-specific limit; vision mode may have lower cap. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
27735
28015
  "endpoints": [
27736
28016
  "chat"
27737
28017
  ],
@@ -27823,13 +28103,13 @@
27823
28103
  },
27824
28104
  "pricing": {
27825
28105
  "value": {
27826
- "inputPerMTokUsd": 0.115,
27827
- "outputPerMTokUsd": 0.917
28106
+ "inputPerMTokUsd": 0.1192,
28107
+ "outputPerMTokUsd": 0.9536
27828
28108
  },
27829
28109
  "source": "official-doc",
27830
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-5-122b-a10b — China (Beijing) qwen3.5-122b-a10b: input $0.115, output $0.917 per million tokens; ≤128K input; >128K–256K $0.287/$2.294",
27831
- "confidence": "declared",
27832
- "observedAt": "2026-09-25T09:40:00Z"
28110
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.5-122b-a10b, ≤128K; >128K–256K: ¥2/¥16 input tier: ¥0.8 input/¥6.4 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
28111
+ "confidence": "inferred",
28112
+ "observedAt": "2026-09-25T12:31:56.381Z"
27833
28113
  },
27834
28114
  "unsupportedParameters": [],
27835
28115
  "status": "candidate"
@@ -27840,7 +28120,7 @@
27840
28120
  "upstreamId": "qwen3.5-397b-a17b",
27841
28121
  "displayName": "Qwen3.5 397B A17B (Anthropic dialect)",
27842
28122
  "aliases": [],
27843
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Context conflict: individual model page states 262,144; Sep24 vision overview lists 32k context/8k output for vision usage. Catalog uses model-specific limit; vision mode may have lower cap.",
28123
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. Alibaba's Sep24 text-generation page labels this model legacy/no longer recommended; no retirement date announced. Context conflict: individual model page states 262,144; Sep24 vision overview lists 32k context/8k output for vision usage. Catalog uses model-specific limit; vision mode may have lower cap. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
27844
28124
  "endpoints": [
27845
28125
  "chat"
27846
28126
  ],
@@ -27932,13 +28212,13 @@
27932
28212
  },
27933
28213
  "pricing": {
27934
28214
  "value": {
27935
- "inputPerMTokUsd": 0.172,
27936
- "outputPerMTokUsd": 1.032
28215
+ "inputPerMTokUsd": 0.1788,
28216
+ "outputPerMTokUsd": 1.0728
27937
28217
  },
27938
28218
  "source": "official-doc",
27939
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-5-397b-a17b — China (Beijing) qwen3.5-397b-a17b: input $0.172, output $1.032 per million tokens; ≤128K input; >128K–256K $0.43/$2.58",
27940
- "confidence": "declared",
27941
- "observedAt": "2026-09-25T09:40:00Z"
28219
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.5-397b-a17b, ≤128K; >128K–256K: ¥3/¥18 input tier: ¥1.2 input/¥7.2 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
28220
+ "confidence": "inferred",
28221
+ "observedAt": "2026-09-25T12:31:56.381Z"
27942
28222
  },
27943
28223
  "unsupportedParameters": [],
27944
28224
  "status": "candidate"
@@ -27949,7 +28229,7 @@
27949
28229
  "upstreamId": "qwen3.6-27b",
27950
28230
  "displayName": "Qwen3.6 27B (Anthropic dialect)",
27951
28231
  "aliases": [],
27952
- "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort.",
28232
+ "$comment": "PAYG China (Beijing) Anthropic Messages base: https://dashscope.aliyuncs.com/apps/anthropic. Thinking off: thinking.type=disabled. Effort: Winter Anthropic adapter sends its fixed low/medium/high/xhigh/max budget_tokens ladder under thinking.type=enabled; Alibaba says budget_tokens remains accepted but will be deprecated in favor of output_config.effort. — 2026-09-25 audit fix (0.0.24): https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY base-tier price and the catalog-wide 0.1490 USD/CNY rate (2026-09-25)",
27953
28233
  "endpoints": [
27954
28234
  "chat"
27955
28235
  ],
@@ -28041,13 +28321,13 @@
28041
28321
  },
28042
28322
  "pricing": {
28043
28323
  "value": {
28044
- "inputPerMTokUsd": 0.412564,
28045
- "outputPerMTokUsd": 2.475384
28324
+ "inputPerMTokUsd": 0.447,
28325
+ "outputPerMTokUsd": 2.682
28046
28326
  },
28047
28327
  "source": "official-doc",
28048
- "sourceRef": "https://www.alibabacloud.com/help/en/model-studio/qwen3-6-27b — China (Beijing) qwen3.6-27b: input $0.412564, output $2.475384 per million tokens; all input",
28049
- "confidence": "declared",
28050
- "observedAt": "2026-09-25T09:40:00Z"
28328
+ "sourceRef": "https://help.aliyun.com/zh/model-studio/model-pricing — Beijing CNY list, qwen3.6-27b, ≤256K input tier: ¥3 input/¥18 output per million; USD = CNY × 0.1490 (CNY→USD reference rate for 2026-09-25, https://www.investing.com/currencies/cny-usd-historical-data) — the one rate every CNY-derived price in this catalog uses; not an Alibaba FX quote.",
28329
+ "confidence": "inferred",
28330
+ "observedAt": "2026-09-25T12:31:56.381Z"
28051
28331
  },
28052
28332
  "unsupportedParameters": [],
28053
28333
  "status": "candidate"
@@ -33777,7 +34057,7 @@
33777
34057
  "observedAt": "2026-09-25T09:52:37Z"
33778
34058
  },
33779
34059
  "unsupportedParameters": [],
33780
- "status": "deprecated"
34060
+ "status": "candidate"
33781
34061
  },
33782
34062
  {
33783
34063
  "key": "xiaomi/mimo-v2.5",
@@ -33875,7 +34155,7 @@
33875
34155
  "observedAt": "2026-09-25T09:52:37Z"
33876
34156
  },
33877
34157
  "unsupportedParameters": [],
33878
- "status": "deprecated"
34158
+ "status": "candidate"
33879
34159
  },
33880
34160
  {
33881
34161
  "key": "xiaomi-anthropic/mimo-v2.6-pro",
@@ -34264,7 +34544,7 @@
34264
34544
  "observedAt": "2026-09-25T09:52:37Z"
34265
34545
  },
34266
34546
  "unsupportedParameters": [],
34267
- "status": "deprecated"
34547
+ "status": "candidate"
34268
34548
  },
34269
34549
  {
34270
34550
  "key": "xiaomi-anthropic/mimo-v2.5",
@@ -34362,7 +34642,7 @@
34362
34642
  "observedAt": "2026-09-25T09:52:37Z"
34363
34643
  },
34364
34644
  "unsupportedParameters": [],
34365
- "status": "deprecated"
34645
+ "status": "candidate"
34366
34646
  },
34367
34647
  {
34368
34648
  "key": "xiaomi-token-plan/mimo-v2.6-pro",
@@ -34968,7 +35248,7 @@
34968
35248
  "continuation": "plaintext"
34969
35249
  },
34970
35250
  "unsupportedParameters": [],
34971
- "status": "deprecated"
35251
+ "status": "candidate"
34972
35252
  },
34973
35253
  {
34974
35254
  "key": "xiaomi-token-plan-cn/mimo-v2.5-pro",
@@ -35052,7 +35332,7 @@
35052
35332
  "continuation": "plaintext"
35053
35333
  },
35054
35334
  "unsupportedParameters": [],
35055
- "status": "deprecated"
35335
+ "status": "candidate"
35056
35336
  },
35057
35337
  {
35058
35338
  "key": "xiaomi-token-plan-eu/mimo-v2.5-pro",
@@ -35136,7 +35416,7 @@
35136
35416
  "continuation": "plaintext"
35137
35417
  },
35138
35418
  "unsupportedParameters": [],
35139
- "status": "deprecated"
35419
+ "status": "candidate"
35140
35420
  },
35141
35421
  {
35142
35422
  "key": "xiaomi-token-plan/mimo-v2.5",
@@ -35223,7 +35503,7 @@
35223
35503
  "continuation": "plaintext"
35224
35504
  },
35225
35505
  "unsupportedParameters": [],
35226
- "status": "deprecated"
35506
+ "status": "candidate"
35227
35507
  },
35228
35508
  {
35229
35509
  "key": "xiaomi-token-plan-cn/mimo-v2.5",
@@ -35310,7 +35590,7 @@
35310
35590
  "continuation": "plaintext"
35311
35591
  },
35312
35592
  "unsupportedParameters": [],
35313
- "status": "deprecated"
35593
+ "status": "candidate"
35314
35594
  },
35315
35595
  {
35316
35596
  "key": "xiaomi-token-plan-eu/mimo-v2.5",
@@ -35397,7 +35677,7 @@
35397
35677
  "continuation": "plaintext"
35398
35678
  },
35399
35679
  "unsupportedParameters": [],
35400
- "status": "deprecated"
35680
+ "status": "candidate"
35401
35681
  },
35402
35682
  {
35403
35683
  "key": "xiaomi-token-plan-anthropic/mimo-v2.6-pro",
@@ -36003,7 +36283,7 @@
36003
36283
  "continuation": "none"
36004
36284
  },
36005
36285
  "unsupportedParameters": [],
36006
- "status": "deprecated"
36286
+ "status": "candidate"
36007
36287
  },
36008
36288
  {
36009
36289
  "key": "xiaomi-token-plan-cn-anthropic/mimo-v2.5-pro",
@@ -36087,7 +36367,7 @@
36087
36367
  "continuation": "none"
36088
36368
  },
36089
36369
  "unsupportedParameters": [],
36090
- "status": "deprecated"
36370
+ "status": "candidate"
36091
36371
  },
36092
36372
  {
36093
36373
  "key": "xiaomi-token-plan-eu-anthropic/mimo-v2.5-pro",
@@ -36171,7 +36451,7 @@
36171
36451
  "continuation": "none"
36172
36452
  },
36173
36453
  "unsupportedParameters": [],
36174
- "status": "deprecated"
36454
+ "status": "candidate"
36175
36455
  },
36176
36456
  {
36177
36457
  "key": "xiaomi-token-plan-anthropic/mimo-v2.5",
@@ -36258,7 +36538,7 @@
36258
36538
  "continuation": "none"
36259
36539
  },
36260
36540
  "unsupportedParameters": [],
36261
- "status": "deprecated"
36541
+ "status": "candidate"
36262
36542
  },
36263
36543
  {
36264
36544
  "key": "xiaomi-token-plan-cn-anthropic/mimo-v2.5",
@@ -36345,7 +36625,7 @@
36345
36625
  "continuation": "none"
36346
36626
  },
36347
36627
  "unsupportedParameters": [],
36348
- "status": "deprecated"
36628
+ "status": "candidate"
36349
36629
  },
36350
36630
  {
36351
36631
  "key": "xiaomi-token-plan-eu-anthropic/mimo-v2.5",
@@ -36432,7 +36712,7 @@
36432
36712
  "continuation": "none"
36433
36713
  },
36434
36714
  "unsupportedParameters": [],
36435
- "status": "deprecated"
36715
+ "status": "candidate"
36436
36716
  },
36437
36717
  {
36438
36718
  "key": "qianfan/ernie-5.1",
@@ -39890,7 +40170,7 @@
39890
40170
  "upstreamId": "hy4-preview",
39891
40171
  "displayName": "Hy4 Preview",
39892
40172
  "aliases": [],
39893
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
40173
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://proxy-hk.tencentcloud.com/pt/document/product/1300/78937 — international USD price table",
39894
40174
  "endpoints": [
39895
40175
  "chat"
39896
40176
  ],
@@ -39978,14 +40258,14 @@
39978
40258
  },
39979
40259
  "pricing": {
39980
40260
  "value": {
39981
- "inputPerMTokUsd": 0.894,
39982
- "outputPerMTokUsd": 2.682,
39983
- "cacheReadPerMTokUsd": 0.0447
40261
+ "inputPerMTokUsd": 0.834,
40262
+ "outputPerMTokUsd": 2.501,
40263
+ "cacheReadPerMTokUsd": 0.042
39984
40264
  },
39985
40265
  "source": "official-doc",
39986
- "sourceRef": "https://cloud.tencent.com/document/product/1823/130055 — hy4-preview: CNY 6 input, 18 output, 0.3 cache hit per million tokens. Converted at 2026-09-25 1 CNY = $0.1490 (https://www.investing.com/currencies/cny-usd-historical-data).",
39987
- "confidence": "inferred",
39988
- "observedAt": "2026-09-25T10:06:30Z"
40266
+ "sourceRef": "https://proxy-hk.tencentcloud.com/pt/document/product/1300/78937 — Singapore international USD list, hy4-preview: $0.834 input/$2.501 output/$0.042 cache-hit per million tokens",
40267
+ "confidence": "declared",
40268
+ "observedAt": "2026-09-25T12:31:56.381Z"
39989
40269
  },
39990
40270
  "unsupportedParameters": [],
39991
40271
  "status": "candidate"
@@ -39996,7 +40276,7 @@
39996
40276
  "upstreamId": "hy3",
39997
40277
  "displayName": "Hy3",
39998
40278
  "aliases": [],
39999
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25).",
40279
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. — The vendor also serves this model on its OpenAI Responses endpoint (same host and key); `endpoints` lists only what this row's Chat Completions adapter routes, so `responses` is omitted until adapter-level Responses support lands (tracked follow-up, 2026-09-25). — 2026-09-25 audit fix (0.0.24): https://proxy-hk.tencentcloud.com/pt/document/product/1300/78937 — international USD price table",
40000
40280
  "endpoints": [
40001
40281
  "chat"
40002
40282
  ],
@@ -40084,14 +40364,14 @@
40084
40364
  },
40085
40365
  "pricing": {
40086
40366
  "value": {
40087
- "inputPerMTokUsd": 0.149,
40088
- "outputPerMTokUsd": 0.596,
40089
- "cacheReadPerMTokUsd": 0.03725
40367
+ "inputPerMTokUsd": 0.132,
40368
+ "outputPerMTokUsd": 0.528,
40369
+ "cacheReadPerMTokUsd": 0.033
40090
40370
  },
40091
40371
  "source": "official-doc",
40092
- "sourceRef": "https://cloud.tencent.com/document/product/1823/130055 — hy3: CNY 1 input, 4 output, 0.25 cache hit per million tokens. Converted at 2026-09-25 1 CNY = $0.1490 (https://www.investing.com/currencies/cny-usd-historical-data).",
40093
- "confidence": "inferred",
40094
- "observedAt": "2026-09-25T10:06:30Z"
40372
+ "sourceRef": "https://proxy-hk.tencentcloud.com/pt/document/product/1300/78937 — Singapore international USD list, hy3: $0.132 input/$0.528 output/$0.033 cache-hit per million tokens",
40373
+ "confidence": "declared",
40374
+ "observedAt": "2026-09-25T12:31:56.381Z"
40095
40375
  },
40096
40376
  "unsupportedParameters": [],
40097
40377
  "status": "candidate"
@@ -40102,7 +40382,7 @@
40102
40382
  "upstreamId": "hy4-preview",
40103
40383
  "displayName": "Hy4 Preview (Anthropic dialect)",
40104
40384
  "aliases": [],
40105
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters.",
40385
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. — 2026-09-25 audit fix (0.0.24): https://proxy-hk.tencentcloud.com/pt/document/product/1300/78937 — international USD price table",
40106
40386
  "endpoints": [
40107
40387
  "chat"
40108
40388
  ],
@@ -40186,14 +40466,14 @@
40186
40466
  },
40187
40467
  "pricing": {
40188
40468
  "value": {
40189
- "inputPerMTokUsd": 0.894,
40190
- "outputPerMTokUsd": 2.682,
40191
- "cacheReadPerMTokUsd": 0.0447
40469
+ "inputPerMTokUsd": 0.834,
40470
+ "outputPerMTokUsd": 2.501,
40471
+ "cacheReadPerMTokUsd": 0.042
40192
40472
  },
40193
40473
  "source": "official-doc",
40194
- "sourceRef": "https://cloud.tencent.com/document/product/1823/130055 — hy4-preview: CNY 6 input, 18 output, 0.3 cache hit per million tokens. Converted at 2026-09-25 1 CNY = $0.1490 (https://www.investing.com/currencies/cny-usd-historical-data).",
40195
- "confidence": "inferred",
40196
- "observedAt": "2026-09-25T10:06:30Z"
40474
+ "sourceRef": "https://proxy-hk.tencentcloud.com/pt/document/product/1300/78937 — Singapore international USD list, hy4-preview: $0.834 input/$2.501 output/$0.042 cache-hit per million tokens",
40475
+ "confidence": "declared",
40476
+ "observedAt": "2026-09-25T12:31:56.381Z"
40197
40477
  },
40198
40478
  "unsupportedParameters": [],
40199
40479
  "status": "candidate"
@@ -40204,7 +40484,7 @@
40204
40484
  "upstreamId": "hy3",
40205
40485
  "displayName": "Hy3 (Anthropic dialect)",
40206
40486
  "aliases": [],
40207
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters.",
40487
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. — 2026-09-25 audit fix (0.0.24): https://proxy-hk.tencentcloud.com/pt/document/product/1300/78937 — international USD price table",
40208
40488
  "endpoints": [
40209
40489
  "chat"
40210
40490
  ],
@@ -40288,14 +40568,14 @@
40288
40568
  },
40289
40569
  "pricing": {
40290
40570
  "value": {
40291
- "inputPerMTokUsd": 0.149,
40292
- "outputPerMTokUsd": 0.596,
40293
- "cacheReadPerMTokUsd": 0.03725
40571
+ "inputPerMTokUsd": 0.132,
40572
+ "outputPerMTokUsd": 0.528,
40573
+ "cacheReadPerMTokUsd": 0.033
40294
40574
  },
40295
40575
  "source": "official-doc",
40296
- "sourceRef": "https://cloud.tencent.com/document/product/1823/130055 — hy3: CNY 1 input, 4 output, 0.25 cache hit per million tokens. Converted at 2026-09-25 1 CNY = $0.1490 (https://www.investing.com/currencies/cny-usd-historical-data).",
40297
- "confidence": "inferred",
40298
- "observedAt": "2026-09-25T10:06:30Z"
40576
+ "sourceRef": "https://proxy-hk.tencentcloud.com/pt/document/product/1300/78937 — Singapore international USD list, hy3: $0.132 input/$0.528 output/$0.033 cache-hit per million tokens",
40577
+ "confidence": "declared",
40578
+ "observedAt": "2026-09-25T12:31:56.381Z"
40299
40579
  },
40300
40580
  "unsupportedParameters": [],
40301
40581
  "status": "candidate"
@@ -40606,10 +40886,31 @@
40606
40886
  "deepseek/deepseek-v4-flash-0731",
40607
40887
  "deepseek/deepseek-v4-flash"
40608
40888
  ],
40609
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
40889
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-flash-202605: context/input/output 1000000/1000000/384000",
40610
40890
  "endpoints": [
40611
40891
  "chat"
40612
40892
  ],
40893
+ "contextWindow": {
40894
+ "value": 1000000,
40895
+ "source": "official-doc",
40896
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-flash-202605 row lists context window 1M tokens; k/M expanded as decimal",
40897
+ "confidence": "inferred",
40898
+ "observedAt": "2026-09-25T12:31:56.381Z"
40899
+ },
40900
+ "maxInputTokens": {
40901
+ "value": 1000000,
40902
+ "source": "official-doc",
40903
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-flash-202605 row lists maximum input 1M tokens; k/M expanded as decimal",
40904
+ "confidence": "inferred",
40905
+ "observedAt": "2026-09-25T12:31:56.381Z"
40906
+ },
40907
+ "maxOutputTokens": {
40908
+ "value": 384000,
40909
+ "source": "official-doc",
40910
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-flash-202605 row lists maximum output 384k tokens; k/M expanded as decimal",
40911
+ "confidence": "inferred",
40912
+ "observedAt": "2026-09-25T12:31:56.381Z"
40913
+ },
40613
40914
  "inputModalities": {
40614
40915
  "value": [
40615
40916
  "text"
@@ -40669,10 +40970,31 @@
40669
40970
  "deepseek/deepseek-v4-pro-0813",
40670
40971
  "deepseek/deepseek-v4-pro"
40671
40972
  ],
40672
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
40973
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-pro-202606: context/input/output 1000000/1000000/384000",
40673
40974
  "endpoints": [
40674
40975
  "chat"
40675
40976
  ],
40977
+ "contextWindow": {
40978
+ "value": 1000000,
40979
+ "source": "official-doc",
40980
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-pro-202606 row lists context window 1M tokens; k/M expanded as decimal",
40981
+ "confidence": "inferred",
40982
+ "observedAt": "2026-09-25T12:31:56.381Z"
40983
+ },
40984
+ "maxInputTokens": {
40985
+ "value": 1000000,
40986
+ "source": "official-doc",
40987
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-pro-202606 row lists maximum input 1M tokens; k/M expanded as decimal",
40988
+ "confidence": "inferred",
40989
+ "observedAt": "2026-09-25T12:31:56.381Z"
40990
+ },
40991
+ "maxOutputTokens": {
40992
+ "value": 384000,
40993
+ "source": "official-doc",
40994
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-pro-202606 row lists maximum output 384k tokens; k/M expanded as decimal",
40995
+ "confidence": "inferred",
40996
+ "observedAt": "2026-09-25T12:31:56.381Z"
40997
+ },
40676
40998
  "inputModalities": {
40677
40999
  "value": [
40678
41000
  "text"
@@ -40731,10 +41053,31 @@
40731
41053
  "aliases": [
40732
41054
  "minimax-m-2-7"
40733
41055
  ],
40734
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
41056
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — minimax-m2.7: context/input/output 200000/200000/128000",
40735
41057
  "endpoints": [
40736
41058
  "chat"
40737
41059
  ],
41060
+ "contextWindow": {
41061
+ "value": 200000,
41062
+ "source": "official-doc",
41063
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — minimax-m2.7 row lists context window 200k tokens; k/M expanded as decimal",
41064
+ "confidence": "inferred",
41065
+ "observedAt": "2026-09-25T12:31:56.381Z"
41066
+ },
41067
+ "maxInputTokens": {
41068
+ "value": 200000,
41069
+ "source": "official-doc",
41070
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — minimax-m2.7 row lists maximum input 200k tokens; k/M expanded as decimal",
41071
+ "confidence": "inferred",
41072
+ "observedAt": "2026-09-25T12:31:56.381Z"
41073
+ },
41074
+ "maxOutputTokens": {
41075
+ "value": 128000,
41076
+ "source": "official-doc",
41077
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — minimax-m2.7 row lists maximum output 128k tokens; k/M expanded as decimal",
41078
+ "confidence": "inferred",
41079
+ "observedAt": "2026-09-25T12:31:56.381Z"
41080
+ },
40738
41081
  "inputModalities": {
40739
41082
  "value": [
40740
41083
  "text"
@@ -40793,10 +41136,24 @@
40793
41136
  "aliases": [
40794
41137
  "minimax-m-3-0"
40795
41138
  ],
40796
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
41139
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — minimax-m3: context/input/output 1000000/1000000/not stated",
40797
41140
  "endpoints": [
40798
41141
  "chat"
40799
41142
  ],
41143
+ "contextWindow": {
41144
+ "value": 1000000,
41145
+ "source": "official-doc",
41146
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — minimax-m3 row lists context window 1M tokens; k/M expanded as decimal",
41147
+ "confidence": "inferred",
41148
+ "observedAt": "2026-09-25T12:31:56.381Z"
41149
+ },
41150
+ "maxInputTokens": {
41151
+ "value": 1000000,
41152
+ "source": "official-doc",
41153
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — minimax-m3 row lists maximum input 1M tokens; k/M expanded as decimal",
41154
+ "confidence": "inferred",
41155
+ "observedAt": "2026-09-25T12:31:56.381Z"
41156
+ },
40800
41157
  "inputModalities": {
40801
41158
  "value": [
40802
41159
  "text"
@@ -40855,10 +41212,31 @@
40855
41212
  "aliases": [
40856
41213
  "glm-5-0"
40857
41214
  ],
40858
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. https://cloud.tencent.com/document/product/1823/130060 — scheduled offline 2026-10-09.",
41215
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. https://cloud.tencent.com/document/product/1823/130060 — scheduled offline 2026-10-09. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — glm-5: context/input/output 200000/200000/128000",
40859
41216
  "endpoints": [
40860
41217
  "chat"
40861
41218
  ],
41219
+ "contextWindow": {
41220
+ "value": 200000,
41221
+ "source": "official-doc",
41222
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5 row lists context window 200k tokens; k/M expanded as decimal",
41223
+ "confidence": "inferred",
41224
+ "observedAt": "2026-09-25T12:31:56.381Z"
41225
+ },
41226
+ "maxInputTokens": {
41227
+ "value": 200000,
41228
+ "source": "official-doc",
41229
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5 row lists maximum input 200k tokens; k/M expanded as decimal",
41230
+ "confidence": "inferred",
41231
+ "observedAt": "2026-09-25T12:31:56.381Z"
41232
+ },
41233
+ "maxOutputTokens": {
41234
+ "value": 128000,
41235
+ "source": "official-doc",
41236
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5 row lists maximum output 128k tokens; k/M expanded as decimal",
41237
+ "confidence": "inferred",
41238
+ "observedAt": "2026-09-25T12:31:56.381Z"
41239
+ },
40862
41240
  "inputModalities": {
40863
41241
  "value": [
40864
41242
  "text"
@@ -40917,10 +41295,31 @@
40917
41295
  "aliases": [
40918
41296
  "glm-5-1"
40919
41297
  ],
40920
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. https://cloud.tencent.com/document/product/1823/130060 — scheduled offline 2026-10-09.",
41298
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. https://cloud.tencent.com/document/product/1823/130060 — scheduled offline 2026-10-09. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — glm-5.1: context/input/output 200000/200000/128000",
40921
41299
  "endpoints": [
40922
41300
  "chat"
40923
41301
  ],
41302
+ "contextWindow": {
41303
+ "value": 200000,
41304
+ "source": "official-doc",
41305
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.1 row lists context window 200k tokens; k/M expanded as decimal",
41306
+ "confidence": "inferred",
41307
+ "observedAt": "2026-09-25T12:31:56.381Z"
41308
+ },
41309
+ "maxInputTokens": {
41310
+ "value": 200000,
41311
+ "source": "official-doc",
41312
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.1 row lists maximum input 200k tokens; k/M expanded as decimal",
41313
+ "confidence": "inferred",
41314
+ "observedAt": "2026-09-25T12:31:56.381Z"
41315
+ },
41316
+ "maxOutputTokens": {
41317
+ "value": 128000,
41318
+ "source": "official-doc",
41319
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.1 row lists maximum output 128k tokens; k/M expanded as decimal",
41320
+ "confidence": "inferred",
41321
+ "observedAt": "2026-09-25T12:31:56.381Z"
41322
+ },
40924
41323
  "inputModalities": {
40925
41324
  "value": [
40926
41325
  "text"
@@ -40979,10 +41378,31 @@
40979
41378
  "aliases": [
40980
41379
  "glm-5-2"
40981
41380
  ],
40982
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
41381
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — glm-5.2: context/input/output 1000000/1000000/128000",
40983
41382
  "endpoints": [
40984
41383
  "chat"
40985
41384
  ],
41385
+ "contextWindow": {
41386
+ "value": 1000000,
41387
+ "source": "official-doc",
41388
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.2 row lists context window 1M tokens; k/M expanded as decimal",
41389
+ "confidence": "inferred",
41390
+ "observedAt": "2026-09-25T12:31:56.381Z"
41391
+ },
41392
+ "maxInputTokens": {
41393
+ "value": 1000000,
41394
+ "source": "official-doc",
41395
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.2 row lists maximum input 1M tokens; k/M expanded as decimal",
41396
+ "confidence": "inferred",
41397
+ "observedAt": "2026-09-25T12:31:56.381Z"
41398
+ },
41399
+ "maxOutputTokens": {
41400
+ "value": 128000,
41401
+ "source": "official-doc",
41402
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.2 row lists maximum output 128k tokens; k/M expanded as decimal",
41403
+ "confidence": "inferred",
41404
+ "observedAt": "2026-09-25T12:31:56.381Z"
41405
+ },
40986
41406
  "inputModalities": {
40987
41407
  "value": [
40988
41408
  "text"
@@ -41041,10 +41461,31 @@
41041
41461
  "aliases": [
41042
41462
  "glm-5-3"
41043
41463
  ],
41044
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
41464
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — glm-5.3: context/input/output 1000000/1000000/128000",
41045
41465
  "endpoints": [
41046
41466
  "chat"
41047
41467
  ],
41468
+ "contextWindow": {
41469
+ "value": 1000000,
41470
+ "source": "official-doc",
41471
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3 row lists context window 1M tokens; k/M expanded as decimal",
41472
+ "confidence": "inferred",
41473
+ "observedAt": "2026-09-25T12:31:56.381Z"
41474
+ },
41475
+ "maxInputTokens": {
41476
+ "value": 1000000,
41477
+ "source": "official-doc",
41478
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3 row lists maximum input 1M tokens; k/M expanded as decimal",
41479
+ "confidence": "inferred",
41480
+ "observedAt": "2026-09-25T12:31:56.381Z"
41481
+ },
41482
+ "maxOutputTokens": {
41483
+ "value": 128000,
41484
+ "source": "official-doc",
41485
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3 row lists maximum output 128k tokens; k/M expanded as decimal",
41486
+ "confidence": "inferred",
41487
+ "observedAt": "2026-09-25T12:31:56.381Z"
41488
+ },
41048
41489
  "inputModalities": {
41049
41490
  "value": [
41050
41491
  "text"
@@ -41101,10 +41542,31 @@
41101
41542
  "upstreamId": "glm-5.3-flash",
41102
41543
  "displayName": "GLM 5.3 Flash (Token Plan)",
41103
41544
  "aliases": [],
41104
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
41545
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — glm-5.3-flash: context/input/output 1000000/1000000/128000",
41105
41546
  "endpoints": [
41106
41547
  "chat"
41107
41548
  ],
41549
+ "contextWindow": {
41550
+ "value": 1000000,
41551
+ "source": "official-doc",
41552
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3-flash row lists context window 1M tokens; k/M expanded as decimal",
41553
+ "confidence": "inferred",
41554
+ "observedAt": "2026-09-25T12:31:56.381Z"
41555
+ },
41556
+ "maxInputTokens": {
41557
+ "value": 1000000,
41558
+ "source": "official-doc",
41559
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3-flash row lists maximum input 1M tokens; k/M expanded as decimal",
41560
+ "confidence": "inferred",
41561
+ "observedAt": "2026-09-25T12:31:56.381Z"
41562
+ },
41563
+ "maxOutputTokens": {
41564
+ "value": 128000,
41565
+ "source": "official-doc",
41566
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3-flash row lists maximum output 128k tokens; k/M expanded as decimal",
41567
+ "confidence": "inferred",
41568
+ "observedAt": "2026-09-25T12:31:56.381Z"
41569
+ },
41108
41570
  "inputModalities": {
41109
41571
  "value": [
41110
41572
  "text"
@@ -41256,10 +41718,31 @@
41256
41718
  "upstreamId": "kimi-k2.7-code",
41257
41719
  "displayName": "Kimi K2.7 Code (Token Plan)",
41258
41720
  "aliases": [],
41259
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
41721
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — kimi-k2.7-code: context/input/output 256000/256000/256000",
41260
41722
  "endpoints": [
41261
41723
  "chat"
41262
41724
  ],
41725
+ "contextWindow": {
41726
+ "value": 256000,
41727
+ "source": "official-doc",
41728
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k2.7-code row lists context window 256k tokens; k/M expanded as decimal",
41729
+ "confidence": "inferred",
41730
+ "observedAt": "2026-09-25T12:31:56.381Z"
41731
+ },
41732
+ "maxInputTokens": {
41733
+ "value": 256000,
41734
+ "source": "official-doc",
41735
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k2.7-code row lists maximum input 256k tokens; k/M expanded as decimal",
41736
+ "confidence": "inferred",
41737
+ "observedAt": "2026-09-25T12:31:56.381Z"
41738
+ },
41739
+ "maxOutputTokens": {
41740
+ "value": 256000,
41741
+ "source": "official-doc",
41742
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k2.7-code row lists maximum output 256k tokens; k/M expanded as decimal",
41743
+ "confidence": "inferred",
41744
+ "observedAt": "2026-09-25T12:31:56.381Z"
41745
+ },
41263
41746
  "inputModalities": {
41264
41747
  "value": [
41265
41748
  "text"
@@ -41316,10 +41799,31 @@
41316
41799
  "upstreamId": "kimi-k3",
41317
41800
  "displayName": "Kimi K3 (Token Plan)",
41318
41801
  "aliases": [],
41319
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
41802
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; OpenAI Chat uses /v1/chat/completions. https://cloud.tencent.com/document/product/1823/135872 — reasoning_effort accepts low/medium/high for thinking models with internal Hy mapping. Off via thinking.type disabled or enable_thinking:false, but model-specific behavior unprobed. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — kimi-k3: context/input/output 1000000/1000000/1000000",
41320
41803
  "endpoints": [
41321
41804
  "chat"
41322
41805
  ],
41806
+ "contextWindow": {
41807
+ "value": 1000000,
41808
+ "source": "official-doc",
41809
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k3 row lists context window 1M tokens; k/M expanded as decimal",
41810
+ "confidence": "inferred",
41811
+ "observedAt": "2026-09-25T12:31:56.381Z"
41812
+ },
41813
+ "maxInputTokens": {
41814
+ "value": 1000000,
41815
+ "source": "official-doc",
41816
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k3 row lists maximum input 1M tokens; k/M expanded as decimal",
41817
+ "confidence": "inferred",
41818
+ "observedAt": "2026-09-25T12:31:56.381Z"
41819
+ },
41820
+ "maxOutputTokens": {
41821
+ "value": 1000000,
41822
+ "source": "official-doc",
41823
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k3 row lists maximum output 1M tokens; k/M expanded as decimal",
41824
+ "confidence": "inferred",
41825
+ "observedAt": "2026-09-25T12:31:56.381Z"
41826
+ },
41323
41827
  "inputModalities": {
41324
41828
  "value": [
41325
41829
  "text"
@@ -41522,10 +42026,31 @@
41522
42026
  "deepseek/deepseek-v4-flash-0731",
41523
42027
  "deepseek/deepseek-v4-flash"
41524
42028
  ],
41525
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
42029
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-flash-202605: context/input/output 1000000/1000000/384000",
41526
42030
  "endpoints": [
41527
42031
  "chat"
41528
42032
  ],
42033
+ "contextWindow": {
42034
+ "value": 1000000,
42035
+ "source": "official-doc",
42036
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-flash-202605 row lists context window 1M tokens; k/M expanded as decimal",
42037
+ "confidence": "inferred",
42038
+ "observedAt": "2026-09-25T12:31:56.381Z"
42039
+ },
42040
+ "maxInputTokens": {
42041
+ "value": 1000000,
42042
+ "source": "official-doc",
42043
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-flash-202605 row lists maximum input 1M tokens; k/M expanded as decimal",
42044
+ "confidence": "inferred",
42045
+ "observedAt": "2026-09-25T12:31:56.381Z"
42046
+ },
42047
+ "maxOutputTokens": {
42048
+ "value": 384000,
42049
+ "source": "official-doc",
42050
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-flash-202605 row lists maximum output 384k tokens; k/M expanded as decimal",
42051
+ "confidence": "inferred",
42052
+ "observedAt": "2026-09-25T12:31:56.381Z"
42053
+ },
41529
42054
  "inputModalities": {
41530
42055
  "value": [
41531
42056
  "text"
@@ -41581,10 +42106,31 @@
41581
42106
  "deepseek/deepseek-v4-pro-0813",
41582
42107
  "deepseek/deepseek-v4-pro"
41583
42108
  ],
41584
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
42109
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-pro-202606: context/input/output 1000000/1000000/384000",
41585
42110
  "endpoints": [
41586
42111
  "chat"
41587
42112
  ],
42113
+ "contextWindow": {
42114
+ "value": 1000000,
42115
+ "source": "official-doc",
42116
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-pro-202606 row lists context window 1M tokens; k/M expanded as decimal",
42117
+ "confidence": "inferred",
42118
+ "observedAt": "2026-09-25T12:31:56.381Z"
42119
+ },
42120
+ "maxInputTokens": {
42121
+ "value": 1000000,
42122
+ "source": "official-doc",
42123
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-pro-202606 row lists maximum input 1M tokens; k/M expanded as decimal",
42124
+ "confidence": "inferred",
42125
+ "observedAt": "2026-09-25T12:31:56.381Z"
42126
+ },
42127
+ "maxOutputTokens": {
42128
+ "value": 384000,
42129
+ "source": "official-doc",
42130
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — deepseek-v4-pro-202606 row lists maximum output 384k tokens; k/M expanded as decimal",
42131
+ "confidence": "inferred",
42132
+ "observedAt": "2026-09-25T12:31:56.381Z"
42133
+ },
41588
42134
  "inputModalities": {
41589
42135
  "value": [
41590
42136
  "text"
@@ -41639,10 +42185,31 @@
41639
42185
  "aliases": [
41640
42186
  "minimax-m-2-7"
41641
42187
  ],
41642
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
42188
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — minimax-m2.7: context/input/output 200000/200000/128000",
41643
42189
  "endpoints": [
41644
42190
  "chat"
41645
42191
  ],
42192
+ "contextWindow": {
42193
+ "value": 200000,
42194
+ "source": "official-doc",
42195
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — minimax-m2.7 row lists context window 200k tokens; k/M expanded as decimal",
42196
+ "confidence": "inferred",
42197
+ "observedAt": "2026-09-25T12:31:56.381Z"
42198
+ },
42199
+ "maxInputTokens": {
42200
+ "value": 200000,
42201
+ "source": "official-doc",
42202
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — minimax-m2.7 row lists maximum input 200k tokens; k/M expanded as decimal",
42203
+ "confidence": "inferred",
42204
+ "observedAt": "2026-09-25T12:31:56.381Z"
42205
+ },
42206
+ "maxOutputTokens": {
42207
+ "value": 128000,
42208
+ "source": "official-doc",
42209
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — minimax-m2.7 row lists maximum output 128k tokens; k/M expanded as decimal",
42210
+ "confidence": "inferred",
42211
+ "observedAt": "2026-09-25T12:31:56.381Z"
42212
+ },
41646
42213
  "inputModalities": {
41647
42214
  "value": [
41648
42215
  "text"
@@ -41697,10 +42264,24 @@
41697
42264
  "aliases": [
41698
42265
  "minimax-m-3-0"
41699
42266
  ],
41700
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
42267
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — minimax-m3: context/input/output 1000000/1000000/not stated",
41701
42268
  "endpoints": [
41702
42269
  "chat"
41703
42270
  ],
42271
+ "contextWindow": {
42272
+ "value": 1000000,
42273
+ "source": "official-doc",
42274
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — minimax-m3 row lists context window 1M tokens; k/M expanded as decimal",
42275
+ "confidence": "inferred",
42276
+ "observedAt": "2026-09-25T12:31:56.381Z"
42277
+ },
42278
+ "maxInputTokens": {
42279
+ "value": 1000000,
42280
+ "source": "official-doc",
42281
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — minimax-m3 row lists maximum input 1M tokens; k/M expanded as decimal",
42282
+ "confidence": "inferred",
42283
+ "observedAt": "2026-09-25T12:31:56.381Z"
42284
+ },
41704
42285
  "inputModalities": {
41705
42286
  "value": [
41706
42287
  "text"
@@ -41755,10 +42336,31 @@
41755
42336
  "aliases": [
41756
42337
  "glm-5-0"
41757
42338
  ],
41758
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. https://cloud.tencent.com/document/product/1823/130060 — scheduled offline 2026-10-09.",
42339
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. https://cloud.tencent.com/document/product/1823/130060 — scheduled offline 2026-10-09. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — glm-5: context/input/output 200000/200000/128000",
41759
42340
  "endpoints": [
41760
42341
  "chat"
41761
42342
  ],
42343
+ "contextWindow": {
42344
+ "value": 200000,
42345
+ "source": "official-doc",
42346
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5 row lists context window 200k tokens; k/M expanded as decimal",
42347
+ "confidence": "inferred",
42348
+ "observedAt": "2026-09-25T12:31:56.381Z"
42349
+ },
42350
+ "maxInputTokens": {
42351
+ "value": 200000,
42352
+ "source": "official-doc",
42353
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5 row lists maximum input 200k tokens; k/M expanded as decimal",
42354
+ "confidence": "inferred",
42355
+ "observedAt": "2026-09-25T12:31:56.381Z"
42356
+ },
42357
+ "maxOutputTokens": {
42358
+ "value": 128000,
42359
+ "source": "official-doc",
42360
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5 row lists maximum output 128k tokens; k/M expanded as decimal",
42361
+ "confidence": "inferred",
42362
+ "observedAt": "2026-09-25T12:31:56.381Z"
42363
+ },
41762
42364
  "inputModalities": {
41763
42365
  "value": [
41764
42366
  "text"
@@ -41813,10 +42415,31 @@
41813
42415
  "aliases": [
41814
42416
  "glm-5-1"
41815
42417
  ],
41816
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. https://cloud.tencent.com/document/product/1823/130060 — scheduled offline 2026-10-09.",
42418
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. https://cloud.tencent.com/document/product/1823/130060 — scheduled offline 2026-10-09. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — glm-5.1: context/input/output 200000/200000/128000",
41817
42419
  "endpoints": [
41818
42420
  "chat"
41819
42421
  ],
42422
+ "contextWindow": {
42423
+ "value": 200000,
42424
+ "source": "official-doc",
42425
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.1 row lists context window 200k tokens; k/M expanded as decimal",
42426
+ "confidence": "inferred",
42427
+ "observedAt": "2026-09-25T12:31:56.381Z"
42428
+ },
42429
+ "maxInputTokens": {
42430
+ "value": 200000,
42431
+ "source": "official-doc",
42432
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.1 row lists maximum input 200k tokens; k/M expanded as decimal",
42433
+ "confidence": "inferred",
42434
+ "observedAt": "2026-09-25T12:31:56.381Z"
42435
+ },
42436
+ "maxOutputTokens": {
42437
+ "value": 128000,
42438
+ "source": "official-doc",
42439
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.1 row lists maximum output 128k tokens; k/M expanded as decimal",
42440
+ "confidence": "inferred",
42441
+ "observedAt": "2026-09-25T12:31:56.381Z"
42442
+ },
41820
42443
  "inputModalities": {
41821
42444
  "value": [
41822
42445
  "text"
@@ -41871,10 +42494,31 @@
41871
42494
  "aliases": [
41872
42495
  "glm-5-2"
41873
42496
  ],
41874
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
42497
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — glm-5.2: context/input/output 1000000/1000000/128000",
41875
42498
  "endpoints": [
41876
42499
  "chat"
41877
42500
  ],
42501
+ "contextWindow": {
42502
+ "value": 1000000,
42503
+ "source": "official-doc",
42504
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.2 row lists context window 1M tokens; k/M expanded as decimal",
42505
+ "confidence": "inferred",
42506
+ "observedAt": "2026-09-25T12:31:56.381Z"
42507
+ },
42508
+ "maxInputTokens": {
42509
+ "value": 1000000,
42510
+ "source": "official-doc",
42511
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.2 row lists maximum input 1M tokens; k/M expanded as decimal",
42512
+ "confidence": "inferred",
42513
+ "observedAt": "2026-09-25T12:31:56.381Z"
42514
+ },
42515
+ "maxOutputTokens": {
42516
+ "value": 128000,
42517
+ "source": "official-doc",
42518
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.2 row lists maximum output 128k tokens; k/M expanded as decimal",
42519
+ "confidence": "inferred",
42520
+ "observedAt": "2026-09-25T12:31:56.381Z"
42521
+ },
41878
42522
  "inputModalities": {
41879
42523
  "value": [
41880
42524
  "text"
@@ -41929,10 +42573,31 @@
41929
42573
  "aliases": [
41930
42574
  "glm-5-3"
41931
42575
  ],
41932
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
42576
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — glm-5.3: context/input/output 1000000/1000000/128000",
41933
42577
  "endpoints": [
41934
42578
  "chat"
41935
42579
  ],
42580
+ "contextWindow": {
42581
+ "value": 1000000,
42582
+ "source": "official-doc",
42583
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3 row lists context window 1M tokens; k/M expanded as decimal",
42584
+ "confidence": "inferred",
42585
+ "observedAt": "2026-09-25T12:31:56.381Z"
42586
+ },
42587
+ "maxInputTokens": {
42588
+ "value": 1000000,
42589
+ "source": "official-doc",
42590
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3 row lists maximum input 1M tokens; k/M expanded as decimal",
42591
+ "confidence": "inferred",
42592
+ "observedAt": "2026-09-25T12:31:56.381Z"
42593
+ },
42594
+ "maxOutputTokens": {
42595
+ "value": 128000,
42596
+ "source": "official-doc",
42597
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3 row lists maximum output 128k tokens; k/M expanded as decimal",
42598
+ "confidence": "inferred",
42599
+ "observedAt": "2026-09-25T12:31:56.381Z"
42600
+ },
41936
42601
  "inputModalities": {
41937
42602
  "value": [
41938
42603
  "text"
@@ -41985,10 +42650,31 @@
41985
42650
  "upstreamId": "glm-5.3-flash",
41986
42651
  "displayName": "GLM 5.3 Flash (Token Plan)",
41987
42652
  "aliases": [],
41988
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
42653
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — glm-5.3-flash: context/input/output 1000000/1000000/128000",
41989
42654
  "endpoints": [
41990
42655
  "chat"
41991
42656
  ],
42657
+ "contextWindow": {
42658
+ "value": 1000000,
42659
+ "source": "official-doc",
42660
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3-flash row lists context window 1M tokens; k/M expanded as decimal",
42661
+ "confidence": "inferred",
42662
+ "observedAt": "2026-09-25T12:31:56.381Z"
42663
+ },
42664
+ "maxInputTokens": {
42665
+ "value": 1000000,
42666
+ "source": "official-doc",
42667
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3-flash row lists maximum input 1M tokens; k/M expanded as decimal",
42668
+ "confidence": "inferred",
42669
+ "observedAt": "2026-09-25T12:31:56.381Z"
42670
+ },
42671
+ "maxOutputTokens": {
42672
+ "value": 128000,
42673
+ "source": "official-doc",
42674
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — glm-5.3-flash row lists maximum output 128k tokens; k/M expanded as decimal",
42675
+ "confidence": "inferred",
42676
+ "observedAt": "2026-09-25T12:31:56.381Z"
42677
+ },
41992
42678
  "inputModalities": {
41993
42679
  "value": [
41994
42680
  "text"
@@ -42132,10 +42818,31 @@
42132
42818
  "upstreamId": "kimi-k2.7-code",
42133
42819
  "displayName": "Kimi K2.7 Code (Token Plan)",
42134
42820
  "aliases": [],
42135
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
42821
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — kimi-k2.7-code: context/input/output 256000/256000/256000",
42136
42822
  "endpoints": [
42137
42823
  "chat"
42138
42824
  ],
42825
+ "contextWindow": {
42826
+ "value": 256000,
42827
+ "source": "official-doc",
42828
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k2.7-code row lists context window 256k tokens; k/M expanded as decimal",
42829
+ "confidence": "inferred",
42830
+ "observedAt": "2026-09-25T12:31:56.381Z"
42831
+ },
42832
+ "maxInputTokens": {
42833
+ "value": 256000,
42834
+ "source": "official-doc",
42835
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k2.7-code row lists maximum input 256k tokens; k/M expanded as decimal",
42836
+ "confidence": "inferred",
42837
+ "observedAt": "2026-09-25T12:31:56.381Z"
42838
+ },
42839
+ "maxOutputTokens": {
42840
+ "value": 256000,
42841
+ "source": "official-doc",
42842
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k2.7-code row lists maximum output 256k tokens; k/M expanded as decimal",
42843
+ "confidence": "inferred",
42844
+ "observedAt": "2026-09-25T12:31:56.381Z"
42845
+ },
42139
42846
  "inputModalities": {
42140
42847
  "value": [
42141
42848
  "text"
@@ -42188,10 +42895,31 @@
42188
42895
  "upstreamId": "kimi-k3",
42189
42896
  "displayName": "Kimi K3 (Token Plan)",
42190
42897
  "aliases": [],
42191
- "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. ",
42898
+ "$comment": "https://cloud.tencent.com/document/product/1823/130079 — Guangzhou https://tokenhub.tencentmaas.com/v1 and Singapore https://tokenhub-intl.tencentmaas.com/v1 are separate sites with non-interchangeable keys; Anthropic Messages uses /v1/messages (adapter root https://tokenhub.tencentmaas.com), not /anthropic. https://cloud.tencent.com/document/product/1823/135874 — thinking.type enabled/disabled/adaptive and thinking.budget_tokens are documented generically; output_config.effort low/medium/high/xhigh/max is generic, not confirmed per Hy model, so efforts empty. Off via thinking.type disabled; model-specific guarantee not stated. https://cloud.tencent.com/document/product/1823/130060 — personal Token Plan is subscription/points, no token pricing; automated API scripts/backends disallowed. https://cloud.tencent.com/document/product/1823/135872 says top_k, repetition_penalty, modalities, audio are silently ignored on Chat, not rejected; excluded from unsupportedParameters. https://cloud.tencent.com/document/product/1823/130060 — available in general personal Token Plan. — 2026-09-25 audit fix (0.0.24): https://cloud.tencent.com/document/product/1823/130051 — kimi-k3: context/input/output 1000000/1000000/1000000",
42192
42899
  "endpoints": [
42193
42900
  "chat"
42194
42901
  ],
42902
+ "contextWindow": {
42903
+ "value": 1000000,
42904
+ "source": "official-doc",
42905
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k3 row lists context window 1M tokens; k/M expanded as decimal",
42906
+ "confidence": "inferred",
42907
+ "observedAt": "2026-09-25T12:31:56.381Z"
42908
+ },
42909
+ "maxInputTokens": {
42910
+ "value": 1000000,
42911
+ "source": "official-doc",
42912
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k3 row lists maximum input 1M tokens; k/M expanded as decimal",
42913
+ "confidence": "inferred",
42914
+ "observedAt": "2026-09-25T12:31:56.381Z"
42915
+ },
42916
+ "maxOutputTokens": {
42917
+ "value": 1000000,
42918
+ "source": "official-doc",
42919
+ "sourceRef": "https://cloud.tencent.com/document/product/1823/130051 — kimi-k3 row lists maximum output 1M tokens; k/M expanded as decimal",
42920
+ "confidence": "inferred",
42921
+ "observedAt": "2026-09-25T12:31:56.381Z"
42922
+ },
42195
42923
  "inputModalities": {
42196
42924
  "value": [
42197
42925
  "text"
@@ -47195,6 +47923,24 @@
47195
47923
  "sourceRef": "continuity report §3 (wire probe of the pinned Claude Agent SDK, 0.3.250→0.3.258) — every uncompacted historical thinking/redacted block is replayed on every later request",
47196
47924
  "confidence": "declared",
47197
47925
  "observedAt": "2026-09-05T00:00:00Z"
47926
+ },
47927
+ "effortRequest": {
47928
+ "value": {
47929
+ "field": "output_config.effort"
47930
+ },
47931
+ "source": "official-doc",
47932
+ "confidence": "declared",
47933
+ "observedAt": "2026-09-25T12:30:00Z",
47934
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
47935
+ },
47936
+ "blockBinding": {
47937
+ "value": {
47938
+ "beta": "thinking-binding-controls-2026-08-01"
47939
+ },
47940
+ "source": "official-doc",
47941
+ "confidence": "declared",
47942
+ "observedAt": "2026-09-25T13:00:00Z",
47943
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting — \"A 400 error says a thinking block signature is invalid\": on Claude Fable 5.1 and Opus 5.5 a replayed thinking block is bound to the conversation (system, tools and earlier messages) and is rejected once that prefix changes (enforced for accounts created on or after 2026-08-31); the documented escape is the `thinking-binding-controls-2026-08-01` beta with `thinking.block_binding.prefix_mismatch_behavior: \"drop_block\"`. See also https://platform.claude.com/docs/en/models/opus-5-5/whats-new-opus-5-5 (\"Thinking blocks are tied to the model and the conversation\")."
47198
47944
  }
47199
47945
  },
47200
47946
  "pricing": {
@@ -47209,7 +47955,12 @@
47209
47955
  "confidence": "declared",
47210
47956
  "observedAt": "2026-09-25T10:25:20Z"
47211
47957
  },
47212
- "unsupportedParameters": [],
47958
+ "unsupportedParameters": [
47959
+ "thinking.type.enabled",
47960
+ "thinking.type.disabled",
47961
+ "tool_choice.any",
47962
+ "tool_choice.tool"
47963
+ ],
47213
47964
  "status": "candidate"
47214
47965
  },
47215
47966
  {
@@ -47351,6 +48102,24 @@
47351
48102
  "sourceRef": "continuity report §3 (wire probe of the pinned Claude Agent SDK, 0.3.250→0.3.258) — every uncompacted historical thinking/redacted block is replayed on every later request",
47352
48103
  "confidence": "declared",
47353
48104
  "observedAt": "2026-09-05T00:00:00Z"
48105
+ },
48106
+ "effortRequest": {
48107
+ "value": {
48108
+ "field": "output_config.effort"
48109
+ },
48110
+ "source": "official-doc",
48111
+ "confidence": "declared",
48112
+ "observedAt": "2026-09-25T12:30:00Z",
48113
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/effort — the effort page's supportedModels list names claude-fable-5-1, claude-fable-5, claude-opus-5-5, claude-opus-5, claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-opus-4-5-20251101, claude-sonnet-5 and claude-sonnet-4-6: effort is requested through `output_config.effort` (Claude 4.7 and later reject a manual `thinking.budget_tokens`, https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting#rejected-configurations)."
48114
+ },
48115
+ "blockBinding": {
48116
+ "value": {
48117
+ "beta": "thinking-binding-controls-2026-08-01"
48118
+ },
48119
+ "source": "official-doc",
48120
+ "confidence": "declared",
48121
+ "observedAt": "2026-09-25T13:00:00Z",
48122
+ "sourceRef": "https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting — \"A 400 error says a thinking block signature is invalid\": on Claude Fable 5.1 and Opus 5.5 a replayed thinking block is bound to the conversation (system, tools and earlier messages) and is rejected once that prefix changes (enforced for accounts created on or after 2026-08-31); the documented escape is the `thinking-binding-controls-2026-08-01` beta with `thinking.block_binding.prefix_mismatch_behavior: \"drop_block\"`. See also https://platform.claude.com/docs/en/models/opus-5-5/whats-new-opus-5-5 (\"Thinking blocks are tied to the model and the conversation\")."
47354
48123
  }
47355
48124
  },
47356
48125
  "pricing": {
@@ -47365,7 +48134,12 @@
47365
48134
  "confidence": "declared",
47366
48135
  "observedAt": "2026-09-25T10:25:20Z"
47367
48136
  },
47368
- "unsupportedParameters": [],
48137
+ "unsupportedParameters": [
48138
+ "thinking.type.enabled",
48139
+ "thinking.type.disabled",
48140
+ "tool_choice.any",
48141
+ "tool_choice.tool"
48142
+ ],
47369
48143
  "status": "candidate"
47370
48144
  },
47371
48145
  {
@@ -49367,7 +50141,7 @@
49367
50141
  "upstreamId": "amazon.nova-premier-v1:0",
49368
50142
  "displayName": "Amazon Nova Premier",
49369
50143
  "aliases": [],
49370
- "$comment": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-premier.html — Bedrock Converse supported at https://bedrock-runtime.us-east-1.amazonaws.com; use modelId us.amazon.nova-premier-v1:0 for US geo cross-region inference. Card says lifecycle Legacy and EOL date September 14, 2026, despite EOL-no-sooner-than October 31, 2026 and still listing a US geo inference profile: conflicting AWS statements; verify serving before shipping. Per-model card max output 25000 conflicts with https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html, whose family table says 10K for all four Nova v1 models. AWS Bedrock pricing page https://aws.amazon.com/bedrock/pricing/ is dynamic and does not expose its Nova rates in static text; the pricing evidence uses an AWS-authored page, not an independent rendered us-east-1 table.",
50144
+ "$comment": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-premier.html — Bedrock Converse supported at https://bedrock-runtime.us-east-1.amazonaws.com; use modelId us.amazon.nova-premier-v1:0 for US geo cross-region inference. Card says lifecycle Legacy and EOL date September 14, 2026, despite EOL-no-sooner-than October 31, 2026 and still listing a US geo inference profile: conflicting AWS statements; verify serving before shipping. Per-model card max output 25000 conflicts with https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html, whose family table says 10K for all four Nova v1 models. AWS Bedrock pricing page https://aws.amazon.com/bedrock/pricing/ is dynamic and does not expose its Nova rates in static text; the pricing evidence uses an AWS-authored page, not an independent rendered us-east-1 table. — 2026-09-25 audit fix (0.0.24): https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-premier.html — per-model modality table lists text, image, video, not pdf",
49371
50145
  "endpoints": [
49372
50146
  "chat"
49373
50147
  ],
@@ -49389,13 +50163,12 @@
49389
50163
  "value": [
49390
50164
  "text",
49391
50165
  "image",
49392
- "video",
49393
- "pdf"
50166
+ "video"
49394
50167
  ],
49395
50168
  "source": "official-doc",
49396
- "sourceRef": "https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html — Amazon Nova Premier supports text, image, video, pdf inputs; PDF is mapped from document support",
49397
- "confidence": "inferred",
49398
- "observedAt": "2026-09-25T11:00:00Z"
50169
+ "sourceRef": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-premier.html — per-model Input Modalities table marks Text, Image, Video supported; PDF is not listed as a separate modality",
50170
+ "confidence": "declared",
50171
+ "observedAt": "2026-09-25T12:31:56.381Z"
49399
50172
  },
49400
50173
  "outputModalities": {
49401
50174
  "value": [
@@ -49454,7 +50227,7 @@
49454
50227
  "upstreamId": "amazon.nova-pro-v1:0",
49455
50228
  "displayName": "Amazon Nova Pro",
49456
50229
  "aliases": [],
49457
- "$comment": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-pro.html — Bedrock Converse supported at https://bedrock-runtime.us-east-1.amazonaws.com; use modelId us.amazon.nova-pro-v1:0 for US geo cross-region inference. Per-model card max output 5000 conflicts with https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html, whose family table says 10K for all four Nova v1 models. AWS Bedrock pricing page https://aws.amazon.com/bedrock/pricing/ is dynamic and does not expose its Nova rates in static text; the pricing evidence uses an AWS-authored page, not an independent rendered us-east-1 table.",
50230
+ "$comment": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-pro.html — Bedrock Converse supported at https://bedrock-runtime.us-east-1.amazonaws.com; use modelId us.amazon.nova-pro-v1:0 for US geo cross-region inference. Per-model card max output 5000 conflicts with https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html, whose family table says 10K for all four Nova v1 models. AWS Bedrock pricing page https://aws.amazon.com/bedrock/pricing/ is dynamic and does not expose its Nova rates in static text; the pricing evidence uses an AWS-authored page, not an independent rendered us-east-1 table. — 2026-09-25 audit fix (0.0.24): https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-pro.html — per-model modality table lists text, image, video, not pdf",
49458
50231
  "endpoints": [
49459
50232
  "chat"
49460
50233
  ],
@@ -49476,13 +50249,12 @@
49476
50249
  "value": [
49477
50250
  "text",
49478
50251
  "image",
49479
- "video",
49480
- "pdf"
50252
+ "video"
49481
50253
  ],
49482
50254
  "source": "official-doc",
49483
- "sourceRef": "https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html — Amazon Nova Pro supports text, image, video, pdf inputs; PDF is mapped from document support",
49484
- "confidence": "inferred",
49485
- "observedAt": "2026-09-25T11:00:00Z"
50255
+ "sourceRef": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-pro.html — per-model Input Modalities table marks Text, Image, Video supported; PDF is not listed as a separate modality",
50256
+ "confidence": "declared",
50257
+ "observedAt": "2026-09-25T12:31:56.381Z"
49486
50258
  },
49487
50259
  "outputModalities": {
49488
50260
  "value": [
@@ -49530,7 +50302,7 @@
49530
50302
  "upstreamId": "amazon.nova-lite-v1:0",
49531
50303
  "displayName": "Amazon Nova Lite",
49532
50304
  "aliases": [],
49533
- "$comment": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-lite.html — Bedrock Converse supported at https://bedrock-runtime.us-east-1.amazonaws.com; use modelId us.amazon.nova-lite-v1:0 for US geo cross-region inference. Per-model card max output 5000 conflicts with https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html, whose family table says 10K for all four Nova v1 models. AWS Bedrock pricing page https://aws.amazon.com/bedrock/pricing/ is dynamic and does not expose its Nova rates in static text; the pricing evidence uses an AWS-authored page, not an independent rendered us-east-1 table.",
50305
+ "$comment": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-lite.html — Bedrock Converse supported at https://bedrock-runtime.us-east-1.amazonaws.com; use modelId us.amazon.nova-lite-v1:0 for US geo cross-region inference. Per-model card max output 5000 conflicts with https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html, whose family table says 10K for all four Nova v1 models. AWS Bedrock pricing page https://aws.amazon.com/bedrock/pricing/ is dynamic and does not expose its Nova rates in static text; the pricing evidence uses an AWS-authored page, not an independent rendered us-east-1 table. — 2026-09-25 audit fix (0.0.24): https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-lite.html — per-model modality table lists text, image, video, not pdf",
49534
50306
  "endpoints": [
49535
50307
  "chat"
49536
50308
  ],
@@ -49552,13 +50324,12 @@
49552
50324
  "value": [
49553
50325
  "text",
49554
50326
  "image",
49555
- "video",
49556
- "pdf"
50327
+ "video"
49557
50328
  ],
49558
50329
  "source": "official-doc",
49559
- "sourceRef": "https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html — Amazon Nova Lite supports text, image, video, pdf inputs; PDF is mapped from document support",
49560
- "confidence": "inferred",
49561
- "observedAt": "2026-09-25T11:00:00Z"
50330
+ "sourceRef": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-lite.html — per-model Input Modalities table marks Text, Image, Video supported; PDF is not listed as a separate modality",
50331
+ "confidence": "declared",
50332
+ "observedAt": "2026-09-25T12:31:56.381Z"
49562
50333
  },
49563
50334
  "outputModalities": {
49564
50335
  "value": [
@@ -49679,7 +50450,7 @@
49679
50450
  "upstreamId": "us.amazon.nova-2-lite-v1:0",
49680
50451
  "displayName": "Amazon Nova 2 Lite (US Geo)",
49681
50452
  "aliases": [],
49682
- "$comment": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-2-lite.html — Bedrock Converse supported at https://bedrock-runtime.us-east-1.amazonaws.com; use modelId us.amazon.nova-2-lite-v1:0 for US geo cross-region inference. Same card also lists in-region amazon.nova-2-lite-v1:0 and global.amazon.nova-2-lite-v1:0; examples and reasoning guide explicitly use us.amazon.nova-2-lite-v1:0. https://docs.aws.amazon.com/nova/latest/nova2-userguide/extended-thinking.html — reasoningConfig goes in additionalModelRequestFields, type enabled/disabled (disabled default), maxReasoningEffort low/medium/high; this is NOT top-level reasoning_effort. High conflicts with temperature/topP/topK; AWS also says maxTokens must be unset at high in a later note. Reasoning content is [REDACTED], billed as output, no recoverable reasoning state. Nova 2 docs list only Lite among generative Nova 2 Converse models; no Nova 2 Pro or Omni on Bedrock model roster as of observation. AWS Bedrock pricing page https://aws.amazon.com/bedrock/pricing/ is dynamic and does not expose its Nova rates in static text; the pricing evidence uses an AWS-authored page, not an independent rendered us-east-1 table.",
50453
+ "$comment": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-2-lite.html — Bedrock Converse supported at https://bedrock-runtime.us-east-1.amazonaws.com; use modelId us.amazon.nova-2-lite-v1:0 for US geo cross-region inference. Same card also lists in-region amazon.nova-2-lite-v1:0 and global.amazon.nova-2-lite-v1:0; examples and reasoning guide explicitly use us.amazon.nova-2-lite-v1:0. https://docs.aws.amazon.com/nova/latest/nova2-userguide/extended-thinking.html — reasoningConfig goes in additionalModelRequestFields, type enabled/disabled (disabled default), maxReasoningEffort low/medium/high; this is NOT top-level reasoning_effort. High conflicts with temperature/topP/topK; AWS also says maxTokens must be unset at high in a later note. Reasoning content is [REDACTED], billed as output, no recoverable reasoning state. Nova 2 docs list only Lite among generative Nova 2 Converse models; no Nova 2 Pro or Omni on Bedrock model roster as of observation. AWS Bedrock pricing page https://aws.amazon.com/bedrock/pricing/ is dynamic and does not expose its Nova rates in static text; the pricing evidence uses an AWS-authored page, not an independent rendered us-east-1 table. — 2026-09-25 audit fix (0.0.24): https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-2-lite.html — per-model modality table lists text, image, video, not pdf",
49683
50454
  "endpoints": [
49684
50455
  "chat"
49685
50456
  ],
@@ -49701,13 +50472,12 @@
49701
50472
  "value": [
49702
50473
  "text",
49703
50474
  "image",
49704
- "video",
49705
- "pdf"
50475
+ "video"
49706
50476
  ],
49707
50477
  "source": "official-doc",
49708
- "sourceRef": "https://docs.aws.amazon.com/nova/latest/nova2-userguide/what-is-nova-2.html — Amazon Nova 2 Lite (US Geo) supports text, image, video, pdf inputs; PDF is mapped from document support",
49709
- "confidence": "inferred",
49710
- "observedAt": "2026-09-25T11:00:00Z"
50478
+ "sourceRef": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-2-lite.html — per-model Input Modalities table marks Text, Image, Video supported; PDF is not listed as a separate modality",
50479
+ "confidence": "declared",
50480
+ "observedAt": "2026-09-25T12:31:56.381Z"
49711
50481
  },
49712
50482
  "outputModalities": {
49713
50483
  "value": [