@opencode/ai 2.0.21 → 2.0.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/cache-policy.js +9 -2
  2. package/dist/llm.d.ts +4 -0
  3. package/dist/promise.d.ts +8 -0
  4. package/dist/protocols/alibaba-chat.d.ts +4 -0
  5. package/dist/protocols/alibaba-chat.js +2 -1
  6. package/dist/protocols/anthropic-messages.js +14 -12
  7. package/dist/protocols/bedrock-converse.js +30 -1
  8. package/dist/protocols/open-responses.d.ts +4 -0
  9. package/dist/protocols/openai-chat.d.ts +8 -0
  10. package/dist/protocols/openai-chat.js +9 -5
  11. package/dist/protocols/openai-responses.js +4 -3
  12. package/dist/protocols/utils/cache.d.ts +5 -0
  13. package/dist/protocols/utils/cache.js +9 -1
  14. package/dist/protocols/utils/tool-stream.d.ts +12 -0
  15. package/dist/protocols/zai-chat.d.ts +4 -0
  16. package/dist/protocols/zai-chat.js +6 -3
  17. package/dist/provider-error.js +20 -6
  18. package/dist/providers/digitalocean.d.ts +469 -0
  19. package/dist/providers/digitalocean.js +53 -0
  20. package/dist/providers/google-vertex-messages.js +1 -1
  21. package/dist/providers/groq.d.ts +4 -0
  22. package/dist/providers/groq.js +1 -1
  23. package/dist/providers/index.d.ts +1 -0
  24. package/dist/providers/index.js +1 -0
  25. package/dist/providers/openrouter.d.ts +4 -0
  26. package/dist/providers/openrouter.js +1 -13
  27. package/dist/route/client.d.ts +4 -0
  28. package/dist/route/executor.js +13 -6
  29. package/dist/route/transport/http.d.ts +2 -1
  30. package/dist/route/transport/http.js +28 -6
  31. package/dist/schema/events.d.ts +176 -0
  32. package/dist/schema/events.js +4 -0
  33. package/dist/schema/options.d.ts +11 -0
  34. package/dist/schema/options.js +15 -2
  35. package/dist/testing.d.ts +32 -0
  36. package/dist/tool-history.js +17 -3
  37. package/package.json +3 -3
@@ -36,6 +36,7 @@ const resolve = (policy) => {
36
36
  // prefix caching, Gemini's implicit + out-of-band CachedContent). Skip the
37
37
  // whole policy pass for these — emitting hints would be harmless but pointless.
38
38
  const RESPECTS_INLINE_HINTS = new Set([
39
+ "alibaba-chat",
39
40
  "alibaba-messages",
40
41
  "anthropic-messages",
41
42
  "anthropic-compatible-messages",
@@ -47,17 +48,19 @@ const RESPECTS_INLINE_HINTS = new Set([
47
48
  "zai-coding-messages",
48
49
  "bedrock-converse",
49
50
  "openrouter",
51
+ "digitalocean",
50
52
  ]);
51
53
  // OpenRouter upstreams other than Anthropic and Alibaba Qwen cache without breakpoints. Gemini uses only the last
52
54
  // breakpoint, so a conversation-tail breakpoint writes a new cache every step and costs more than none. Qwen ignores
53
55
  // breakpoints on tool definitions and caches tools with the system prompt.
56
+ const QWEN = { system: true, messages: { tail: 1 } };
54
57
  const openRouterPolicy = (modelID) => {
55
58
  // `~anthropic/claude-sonnet-latest` style IDs are OpenRouter aliases for the latest model in a family.
56
59
  const id = modelID.replace(/^~/, "");
57
60
  if (id.startsWith("anthropic/"))
58
61
  return AUTO;
59
62
  if (id.startsWith("qwen/"))
60
- return { system: true, messages: { tail: 1 } };
63
+ return QWEN;
61
64
  return NONE;
62
65
  };
63
66
  const makeHint = (ttlSeconds) => ttlSeconds !== undefined ? new CacheHint({ type: "ephemeral", ttlSeconds }) : new CacheHint({ type: "ephemeral" });
@@ -142,7 +145,11 @@ export const applyCachePolicy = (request) => {
142
145
  return request;
143
146
  const policy = request.model.route.id === "openrouter" && (request.cache === undefined || request.cache === "auto")
144
147
  ? openRouterPolicy(request.model.id)
145
- : resolve(request.cache);
148
+ : request.model.route.id === "alibaba-chat" && (request.cache === undefined || request.cache === "auto")
149
+ ? request.model.id.toLowerCase().startsWith("qwen")
150
+ ? QWEN
151
+ : NONE
152
+ : resolve(request.cache);
146
153
  if (!policy.tools && !policy.system && !policy.messages)
147
154
  return request;
148
155
  const hint = makeHint(policy.ttlSeconds);
package/dist/llm.d.ts CHANGED
@@ -206,6 +206,8 @@ export declare class GenerateObjectResponse<T> {
206
206
  readonly reason: {
207
207
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
208
208
  readonly raw?: string | undefined;
209
+ readonly category?: string | undefined;
210
+ readonly explanation?: string | undefined;
209
211
  };
210
212
  readonly index: number;
211
213
  readonly providerMetadata?: {
@@ -219,6 +221,8 @@ export declare class GenerateObjectResponse<T> {
219
221
  readonly reason: {
220
222
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
221
223
  readonly raw?: string | undefined;
224
+ readonly category?: string | undefined;
225
+ readonly explanation?: string | undefined;
222
226
  };
223
227
  readonly providerMetadata?: {
224
228
  readonly [x: string]: {
package/dist/promise.d.ts CHANGED
@@ -233,6 +233,8 @@ export declare const make: (options?: Options) => {
233
233
  readonly reason: {
234
234
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
235
235
  readonly raw?: string | undefined;
236
+ readonly category?: string | undefined;
237
+ readonly explanation?: string | undefined;
236
238
  };
237
239
  readonly index: number;
238
240
  readonly providerMetadata?: {
@@ -246,6 +248,8 @@ export declare const make: (options?: Options) => {
246
248
  readonly reason: {
247
249
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
248
250
  readonly raw?: string | undefined;
251
+ readonly category?: string | undefined;
252
+ readonly explanation?: string | undefined;
249
253
  };
250
254
  readonly providerMetadata?: {
251
255
  readonly [x: string]: {
@@ -720,6 +724,8 @@ export declare const ai: {
720
724
  readonly reason: {
721
725
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
722
726
  readonly raw?: string | undefined;
727
+ readonly category?: string | undefined;
728
+ readonly explanation?: string | undefined;
723
729
  };
724
730
  readonly index: number;
725
731
  readonly providerMetadata?: {
@@ -733,6 +739,8 @@ export declare const ai: {
733
739
  readonly reason: {
734
740
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
735
741
  readonly raw?: string | undefined;
742
+ readonly category?: string | undefined;
743
+ readonly explanation?: string | undefined;
736
744
  };
737
745
  readonly providerMetadata?: {
738
746
  readonly [x: string]: {
@@ -186,6 +186,7 @@ export declare const protocol: Protocol<{
186
186
  } | null | undefined;
187
187
  readonly usage?: {
188
188
  readonly [x: string]: unknown;
189
+ readonly cache_read_input_tokens?: number | null | undefined;
189
190
  readonly cached_tokens?: number | null | undefined;
190
191
  readonly prompt_tokens?: number | null | undefined;
191
192
  readonly completion_tokens?: number | null | undefined;
@@ -196,6 +197,7 @@ export declare const protocol: Protocol<{
196
197
  readonly cache_write_tokens?: number | null | undefined;
197
198
  } | null | undefined;
198
199
  readonly prompt_cache_hit_tokens?: number | null | undefined;
200
+ readonly cache_created_input_tokens?: number | null | undefined;
199
201
  readonly completion_tokens_details?: {
200
202
  readonly [x: string]: unknown;
201
203
  readonly reasoning_tokens?: number | null | undefined;
@@ -225,6 +227,7 @@ export declare const protocol: Protocol<{
225
227
  } | null | undefined;
226
228
  readonly usage?: {
227
229
  readonly [x: string]: unknown;
230
+ readonly cache_read_input_tokens?: number | null | undefined;
228
231
  readonly cached_tokens?: number | null | undefined;
229
232
  readonly prompt_tokens?: number | null | undefined;
230
233
  readonly completion_tokens?: number | null | undefined;
@@ -235,6 +238,7 @@ export declare const protocol: Protocol<{
235
238
  readonly cache_write_tokens?: number | null | undefined;
236
239
  } | null | undefined;
237
240
  readonly prompt_cache_hit_tokens?: number | null | undefined;
241
+ readonly cache_created_input_tokens?: number | null | undefined;
238
242
  readonly completion_tokens_details?: {
239
243
  readonly [x: string]: unknown;
240
244
  readonly reasoning_tokens?: number | null | undefined;
@@ -2,6 +2,7 @@ import { Effect, Schema } from "effect";
2
2
  import { Protocol } from "../route/protocol.js";
3
3
  import { OpenAIChat } from "./openai-chat.js";
4
4
  import { JsonObject, ProviderShared } from "./shared.js";
5
+ import { cacheControl } from "./utils/cache.js";
5
6
  import { OpenResponsesOptions } from "./utils/open-responses-options.js";
6
7
  const Options = Schema.Struct({
7
8
  reasoningEffort: Schema.optional(OpenResponsesOptions.ReasoningEffort),
@@ -53,7 +54,7 @@ export const protocol = Protocol.make({
53
54
  from: Effect.fn("AlibabaChat.fromRequest")(function* (req) {
54
55
  const opts = yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(Options))(req.providerOptions ?? {});
55
56
  return {
56
- ...(yield* OpenAIChat.protocol.body.from(req)),
57
+ ...(yield* OpenAIChat.fromRequest(req, { cacheControl: cacheControl() })),
57
58
  enable_thinking: opts.enableThinking,
58
59
  // Alibaba also rejects an explicit budget that is not below `max_completion_tokens`.
59
60
  thinking_budget: opts.thinkingBudget === undefined
@@ -359,6 +359,7 @@ const AnthropicStreamDelta = Schema.Struct({
359
359
  signature: Schema.optional(Schema.String),
360
360
  stop_reason: optionalNull(Schema.String),
361
361
  stop_sequence: optionalNull(Schema.String),
362
+ stop_details: optionalNull(Schema.Struct({ category: optionalNull(Schema.String), explanation: optionalNull(Schema.String) })),
362
363
  });
363
364
  const decodeAnthropicStreamDelta = Schema.decodeUnknownOption(AnthropicStreamDelta);
364
365
  const AnthropicEvent = Schema.Struct({
@@ -677,16 +678,14 @@ const requireThinkingSignature = (request) => {
677
678
  // Mid-conversation system messages became available with Opus 4.8 and version
678
679
  // 5 of the other supported Claude families. Treat later family versions as
679
680
  // compatible without assuming that every Anthropic Messages model is Claude.
681
+ // Opus 4.8 and every Claude 5 model accept mid-conversation system messages; later versions inherit support.
680
682
  const supportsNativeSystemUpdates = (request) => {
681
- const match = /(?:^|[./])claude-(fable|haiku|mythos|opus|sonnet)-(\d+)(?:[.-](\d+))?/.exec(String(request.model.id).toLowerCase());
682
- if (!match)
683
+ const version = claudeVersion(String(request.model.id));
684
+ if (version === undefined)
683
685
  return false;
684
- const major = Number(match[2]);
685
- if (match[1] !== "opus")
686
- return major >= 5;
687
- if (major !== 4)
688
- return major >= 5;
689
- return match[3] !== undefined && match[3].length <= 2 && Number(match[3]) >= 8;
686
+ if (version.family === "opus" && version.major === 4)
687
+ return version.minor >= 8;
688
+ return version.major >= 5;
690
689
  };
691
690
  const endsInServerToolUse = (message) => {
692
691
  const last = message.content.at(-1);
@@ -851,6 +850,7 @@ const lowerMessages = Effect.fn("AnthropicMessages.lowerMessages")(function* (re
851
850
  }
852
851
  return messages;
853
852
  });
853
+ // Per-turn effort started with Claude Opus 5 and every Claude 5.1 model; later versions of any family inherit it.
854
854
  const supportsEffortUpdates = (model) => {
855
855
  const override = model.compatibility?.supportsEffortUpdates;
856
856
  if (override !== undefined)
@@ -858,10 +858,8 @@ const supportsEffortUpdates = (model) => {
858
858
  const version = claudeVersion(model.id);
859
859
  if (version === undefined)
860
860
  return false;
861
- if (version.family === "opus")
862
- return version.major >= 5;
863
- if (version.family !== "fable" && version.family !== "mythos")
864
- return false;
861
+ if (version.family === "opus" && version.major >= 5)
862
+ return true;
865
863
  return version.major > 5 || (version.major === 5 && version.minor >= 1);
866
864
  };
867
865
  const applyThinkingBindingDefault = (model, thinking) => {
@@ -1227,10 +1225,14 @@ const onMessageDelta = (state, event) => {
1227
1225
  const finishMetadata = stopSequence === null || stopSequence === undefined
1228
1226
  ? state.pendingFinish?.providerMetadata
1229
1227
  : providerMetadata(state.providerMetadataKey, { stopSequence });
1228
+ const category = event.delta?.stop_details?.category;
1229
+ const explanation = event.delta?.stop_details?.explanation;
1230
1230
  return {
1231
1231
  reason: {
1232
1232
  normalized: mapFinishReason(stopReason),
1233
1233
  raw: stopReason,
1234
+ ...(category ? { category } : {}),
1235
+ ...(explanation ? { explanation } : {}),
1234
1236
  },
1235
1237
  providerMetadata: finishMetadata,
1236
1238
  };
@@ -236,13 +236,28 @@ const lowerToolResult = Effect.fn("BedrockConverse.lowerToolResult")(function* (
236
236
  },
237
237
  };
238
238
  });
239
+ // Keep Claude and Nova tool-result images inline; put other models' images beside the result.
240
+ const keepToolImagesInline = (id) => id.includes("anthropic.claude-") || id.includes("amazon.nova-");
239
241
  const lowerMessages = Effect.fn("BedrockConverse.lowerMessages")(function* (request, breakpoints) {
240
242
  const messages = [];
241
243
  const documentNames = new Set();
242
244
  // Mistral can reject replay IDs even when they satisfy Converse's broader ID syntax.
243
245
  const normalizeID = request.model.id.includes("mistral.") ? MistralToolID.normalizer(request) : (id) => id;
244
246
  const providerMetadataKey = request.model.route.providerMetadataKey ?? String(request.model.provider);
247
+ const hoistImages = !keepToolImagesInline(request.model.id);
248
+ // Bedrock expects parallel tool results before any images hoisted beside them.
249
+ const pendingImages = [];
250
+ const flushImages = () => {
251
+ if (pendingImages.length === 0)
252
+ return;
253
+ const previous = messages.at(-1);
254
+ if (previous?.role === "user")
255
+ messages[messages.length - 1] = { role: "user", content: [...previous.content, ...pendingImages] };
256
+ pendingImages.length = 0;
257
+ };
245
258
  for (const message of request.messages) {
259
+ if (message.role !== "tool")
260
+ flushImages();
246
261
  if (message.role === "system") {
247
262
  const part = yield* ProviderShared.wrappedSystemUpdate("Bedrock Converse", message);
248
263
  const content = textWithCache(breakpoints, part.text, part.cache);
@@ -317,7 +332,20 @@ const lowerMessages = Effect.fn("BedrockConverse.lowerMessages")(function* (requ
317
332
  for (const part of message.content) {
318
333
  if (!ProviderShared.supportsContent(part, ["tool-result"]))
319
334
  return yield* ProviderShared.unsupportedContent("Bedrock Converse", "tool", ["tool-result"]);
320
- content.push(yield* lowerToolResult(part, documentNames, normalizeID));
335
+ const result = yield* lowerToolResult(part, documentNames, normalizeID);
336
+ const images = hoistImages
337
+ ? result.toolResult.content.filter((item) => "image" in item)
338
+ : [];
339
+ const nonImageContent = result.toolResult.content.filter((item) => !("image" in item));
340
+ content.push(images.length === 0
341
+ ? result
342
+ : {
343
+ toolResult: {
344
+ ...result.toolResult,
345
+ content: nonImageContent.length > 0 ? nonImageContent : [{ text: "See attached image." }],
346
+ },
347
+ });
348
+ pendingImages.push(...images);
321
349
  const cachePoint = BedrockCache.block(breakpoints, part.cache);
322
350
  if (cachePoint)
323
351
  content.push(cachePoint);
@@ -328,6 +356,7 @@ const lowerMessages = Effect.fn("BedrockConverse.lowerMessages")(function* (requ
328
356
  else
329
357
  messages.push({ role: "user", content });
330
358
  }
359
+ flushImages();
331
360
  return messages;
332
361
  });
333
362
  // System prompts share the cache-point convention: emit the text block, then
@@ -1191,6 +1191,8 @@ export declare const step: (state: ParserState, event: NormalizedEvent) => AIErr
1191
1191
  readonly reason: {
1192
1192
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
1193
1193
  readonly raw?: string | undefined;
1194
+ readonly category?: string | undefined;
1195
+ readonly explanation?: string | undefined;
1194
1196
  };
1195
1197
  readonly index: number;
1196
1198
  readonly providerMetadata?: {
@@ -1204,6 +1206,8 @@ export declare const step: (state: ParserState, event: NormalizedEvent) => AIErr
1204
1206
  readonly reason: {
1205
1207
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
1206
1208
  readonly raw?: string | undefined;
1209
+ readonly category?: string | undefined;
1210
+ readonly explanation?: string | undefined;
1207
1211
  };
1208
1212
  readonly providerMetadata?: {
1209
1213
  readonly [x: string]: {
@@ -315,6 +315,8 @@ export declare const OpenAIChatEvent: Schema.StructWithRest<Schema.Struct<{
315
315
  readonly total_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
316
316
  readonly cached_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
317
317
  readonly prompt_cache_hit_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
318
+ readonly cache_read_input_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
319
+ readonly cache_created_input_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
318
320
  readonly prompt_tokens_details: Schema.optional<Schema.NullOr<Schema.StructWithRest<Schema.Struct<{
319
321
  readonly cached_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
320
322
  readonly cache_write_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
@@ -332,6 +334,8 @@ export declare const OpenAIChatEvent: Schema.StructWithRest<Schema.Struct<{
332
334
  readonly total_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
333
335
  readonly cached_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
334
336
  readonly prompt_cache_hit_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
337
+ readonly cache_read_input_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
338
+ readonly cache_created_input_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
335
339
  readonly prompt_tokens_details: Schema.optional<Schema.NullOr<Schema.StructWithRest<Schema.Struct<{
336
340
  readonly cached_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
337
341
  readonly cache_write_tokens: Schema.optional<Schema.NullOr<Schema.Number>>;
@@ -755,6 +759,7 @@ export declare const protocol: Protocol<{
755
759
  } | null | undefined;
756
760
  readonly usage?: {
757
761
  readonly [x: string]: unknown;
762
+ readonly cache_read_input_tokens?: number | null | undefined;
758
763
  readonly cached_tokens?: number | null | undefined;
759
764
  readonly prompt_tokens?: number | null | undefined;
760
765
  readonly completion_tokens?: number | null | undefined;
@@ -765,6 +770,7 @@ export declare const protocol: Protocol<{
765
770
  readonly cache_write_tokens?: number | null | undefined;
766
771
  } | null | undefined;
767
772
  readonly prompt_cache_hit_tokens?: number | null | undefined;
773
+ readonly cache_created_input_tokens?: number | null | undefined;
768
774
  readonly completion_tokens_details?: {
769
775
  readonly [x: string]: unknown;
770
776
  readonly reasoning_tokens?: number | null | undefined;
@@ -794,6 +800,7 @@ export declare const protocol: Protocol<{
794
800
  } | null | undefined;
795
801
  readonly usage?: {
796
802
  readonly [x: string]: unknown;
803
+ readonly cache_read_input_tokens?: number | null | undefined;
797
804
  readonly cached_tokens?: number | null | undefined;
798
805
  readonly prompt_tokens?: number | null | undefined;
799
806
  readonly completion_tokens?: number | null | undefined;
@@ -804,6 +811,7 @@ export declare const protocol: Protocol<{
804
811
  readonly cache_write_tokens?: number | null | undefined;
805
812
  } | null | undefined;
806
813
  readonly prompt_cache_hit_tokens?: number | null | undefined;
814
+ readonly cache_created_input_tokens?: number | null | undefined;
807
815
  readonly completion_tokens_details?: {
808
816
  readonly [x: string]: unknown;
809
817
  readonly reasoning_tokens?: number | null | undefined;
@@ -161,9 +161,11 @@ const OpenAIChatUsage = Schema.StructWithRest(Schema.Struct({
161
161
  prompt_tokens: optionalNull(Schema.Number),
162
162
  completion_tokens: optionalNull(Schema.Number),
163
163
  total_tokens: optionalNull(Schema.Number),
164
- // Zai reports cache hits as top-level `cached_tokens`; DeepSeek uses `prompt_cache_hit_tokens`.
164
+ // Provider-specific cache accounting fields.
165
165
  cached_tokens: optionalNull(Schema.Number),
166
166
  prompt_cache_hit_tokens: optionalNull(Schema.Number),
167
+ cache_read_input_tokens: optionalNull(Schema.Number),
168
+ cache_created_input_tokens: optionalNull(Schema.Number),
167
169
  prompt_tokens_details: optionalNull(Schema.StructWithRest(Schema.Struct({
168
170
  cached_tokens: optionalNull(Schema.Number),
169
171
  cache_write_tokens: optionalNull(Schema.Number),
@@ -719,17 +721,19 @@ const mapFinishReason = Effect.fn("OpenAIChat.mapFinishReason")(function* (event
719
721
  // satisfied on both sides.
720
722
  // Providers differ on cache-hit location: OpenAI uses
721
723
  // `prompt_tokens_details.cached_tokens`, DeepSeek uses
722
- // `prompt_cache_hit_tokens`, and Zai uses top-level `cached_tokens`.
724
+ // `prompt_cache_hit_tokens`, Zai uses top-level `cached_tokens`, and
725
+ // DigitalOcean uses top-level `cache_read_input_tokens` / `cache_created_input_tokens`.
723
726
  const mapUsage = (usage, providerMetadataKey) => {
724
727
  if (!usage)
725
728
  return undefined;
726
729
  const input = usage.prompt_tokens ?? undefined;
727
730
  const output = usage.completion_tokens ?? undefined;
728
- const cached = (usage.prompt_tokens_details?.cached_tokens ??
731
+ const cached = usage.prompt_tokens_details?.cached_tokens ??
729
732
  usage.prompt_cache_hit_tokens ??
730
733
  usage.cached_tokens ??
731
- undefined);
732
- const cacheWrite = usage.prompt_tokens_details?.cache_write_tokens ?? undefined;
734
+ usage.cache_read_input_tokens ??
735
+ undefined;
736
+ const cacheWrite = usage.prompt_tokens_details?.cache_write_tokens ?? usage.cache_created_input_tokens ?? undefined;
733
737
  const reasoning = usage.completion_tokens_details?.reasoning_tokens ?? undefined;
734
738
  const nonCached = ProviderShared.subtractTokens(input, ProviderShared.sumTokens(cached, cacheWrite));
735
739
  return new Usage({
@@ -111,8 +111,8 @@ const adapter = {
111
111
  name: NAME,
112
112
  restoreHostedToolItem: (item) => (Schema.is(OpenAIResponsesHostedToolItem)(item) ? item : undefined),
113
113
  };
114
- // GPT-6 Astra, Sol, and Luna accept `configuration_update` only in standard mode (not `reasoning.mode: "pro"` or
115
- // `-pro` slugs), and never alongside automatic `context_management` compaction.
114
+ // GPT-6 and later default to `configuration_update` support, except in `reasoning.mode: "pro"`
115
+ // or alongside automatic `context_management` compaction.
116
116
  const supportsEffortUpdates = (request) => {
117
117
  if (request.providerOptions?.contextManagement !== undefined)
118
118
  return false;
@@ -121,7 +121,8 @@ const supportsEffortUpdates = (request) => {
121
121
  const override = request.model.compatibility?.supportsEffortUpdates;
122
122
  if (override !== undefined)
123
123
  return override;
124
- return /(?:^|\/)gpt-6-(?:astra|sol|luna)$/i.test(request.model.id);
124
+ const match = /(?:^|\/)gpt-(\d+)(?:\.\d+)?(?:-|$)/i.exec(request.model.id);
125
+ return match !== null && Number(match[1]) >= 6;
125
126
  };
126
127
  const nativeImageToolInput = (tool) => {
127
128
  const native = tool.native?.openai;
@@ -1,6 +1,11 @@
1
+ import type { CacheHint } from "../../schema/index.js";
1
2
  export interface Breakpoints {
2
3
  remaining: number;
3
4
  dropped: number;
4
5
  }
5
6
  export declare const newBreakpoints: (cap: number) => Breakpoints;
6
7
  export declare const ttlBucket: (ttlSeconds: number | undefined) => "1h" | undefined;
8
+ export declare const cacheControl: () => (cache: CacheHint | undefined) => {
9
+ type: "ephemeral";
10
+ ttl: "1h" | undefined;
11
+ } | undefined;
@@ -1,5 +1,13 @@
1
- // Shared counter and TTL mapping for provider cache-marker lowering.
2
1
  export const newBreakpoints = (cap) => ({ remaining: cap, dropped: 0 });
3
2
  // Requests of at least one hour use the explicit `"1h"` bucket; shorter
4
3
  // requests omit the wire TTL and use the provider default.
5
4
  export const ttlBucket = (ttlSeconds) => ttlSeconds !== undefined && ttlSeconds >= 3600 ? "1h" : undefined;
5
+ export const cacheControl = () => {
6
+ const breakpoints = newBreakpoints(4);
7
+ return (cache) => {
8
+ if (cache === undefined || breakpoints.remaining === 0)
9
+ return undefined;
10
+ breakpoints.remaining -= 1;
11
+ return { type: "ephemeral", ttl: ttlBucket(cache.ttlSeconds) };
12
+ };
13
+ };
@@ -258,6 +258,8 @@ export declare const finish: <K extends StreamKey>(route: string, tools: State<K
258
258
  readonly reason: {
259
259
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
260
260
  readonly raw?: string | undefined;
261
+ readonly category?: string | undefined;
262
+ readonly explanation?: string | undefined;
261
263
  };
262
264
  readonly index: number;
263
265
  readonly providerMetadata?: {
@@ -271,6 +273,8 @@ export declare const finish: <K extends StreamKey>(route: string, tools: State<K
271
273
  readonly reason: {
272
274
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
273
275
  readonly raw?: string | undefined;
276
+ readonly category?: string | undefined;
277
+ readonly explanation?: string | undefined;
274
278
  };
275
279
  readonly providerMetadata?: {
276
280
  readonly [x: string]: {
@@ -481,6 +485,8 @@ export declare const finishWithInput: <K extends StreamKey>(route: string, tools
481
485
  readonly reason: {
482
486
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
483
487
  readonly raw?: string | undefined;
488
+ readonly category?: string | undefined;
489
+ readonly explanation?: string | undefined;
484
490
  };
485
491
  readonly index: number;
486
492
  readonly providerMetadata?: {
@@ -494,6 +500,8 @@ export declare const finishWithInput: <K extends StreamKey>(route: string, tools
494
500
  readonly reason: {
495
501
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
496
502
  readonly raw?: string | undefined;
503
+ readonly category?: string | undefined;
504
+ readonly explanation?: string | undefined;
497
505
  };
498
506
  readonly providerMetadata?: {
499
507
  readonly [x: string]: {
@@ -701,6 +709,8 @@ export declare const finishAll: <K extends StreamKey>(route: string, tools: Stat
701
709
  readonly reason: {
702
710
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
703
711
  readonly raw?: string | undefined;
712
+ readonly category?: string | undefined;
713
+ readonly explanation?: string | undefined;
704
714
  };
705
715
  readonly index: number;
706
716
  readonly providerMetadata?: {
@@ -714,6 +724,8 @@ export declare const finishAll: <K extends StreamKey>(route: string, tools: Stat
714
724
  readonly reason: {
715
725
  readonly normalized: "length" | "stop" | "tool-calls" | "content-filter" | "error" | "unknown";
716
726
  readonly raw?: string | undefined;
727
+ readonly category?: string | undefined;
728
+ readonly explanation?: string | undefined;
717
729
  };
718
730
  readonly providerMetadata?: {
719
731
  readonly [x: string]: {
@@ -163,6 +163,7 @@ export declare const protocol: Protocol<{
163
163
  } | null | undefined;
164
164
  readonly usage?: {
165
165
  readonly [x: string]: unknown;
166
+ readonly cache_read_input_tokens?: number | null | undefined;
166
167
  readonly cached_tokens?: number | null | undefined;
167
168
  readonly prompt_tokens?: number | null | undefined;
168
169
  readonly completion_tokens?: number | null | undefined;
@@ -173,6 +174,7 @@ export declare const protocol: Protocol<{
173
174
  readonly cache_write_tokens?: number | null | undefined;
174
175
  } | null | undefined;
175
176
  readonly prompt_cache_hit_tokens?: number | null | undefined;
177
+ readonly cache_created_input_tokens?: number | null | undefined;
176
178
  readonly completion_tokens_details?: {
177
179
  readonly [x: string]: unknown;
178
180
  readonly reasoning_tokens?: number | null | undefined;
@@ -202,6 +204,7 @@ export declare const protocol: Protocol<{
202
204
  } | null | undefined;
203
205
  readonly usage?: {
204
206
  readonly [x: string]: unknown;
207
+ readonly cache_read_input_tokens?: number | null | undefined;
205
208
  readonly cached_tokens?: number | null | undefined;
206
209
  readonly prompt_tokens?: number | null | undefined;
207
210
  readonly completion_tokens?: number | null | undefined;
@@ -212,6 +215,7 @@ export declare const protocol: Protocol<{
212
215
  readonly cache_write_tokens?: number | null | undefined;
213
216
  } | null | undefined;
214
217
  readonly prompt_cache_hit_tokens?: number | null | undefined;
218
+ readonly cache_created_input_tokens?: number | null | undefined;
215
219
  readonly completion_tokens_details?: {
216
220
  readonly [x: string]: unknown;
217
221
  readonly reasoning_tokens?: number | null | undefined;
@@ -19,15 +19,18 @@ const Body = Schema.Struct({
19
19
  request_id: Options.fields.requestID,
20
20
  user_id: Options.fields.userID,
21
21
  });
22
+ // Tool streaming was introduced in GLM-4.6; later versions inherit support.
23
+ const supportsToolStreaming = (modelID) => {
24
+ const match = /(?:^|\/)glm-(\d+)(?:\.(\d+))?(?:-|$)/i.exec(modelID);
25
+ return match !== null && (Number(match[1]) > 4 || (Number(match[1]) === 4 && Number(match[2] ?? 0) >= 6));
26
+ };
22
27
  const fromRequest = Effect.fn("ZAIChat.fromRequest")(function* (request) {
23
28
  const options = yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(Options))(request.providerOptions ?? {});
24
29
  const body = yield* OpenAIChat.protocol.body.from(request);
25
30
  return {
26
31
  ...body,
27
32
  thinking: options.thinking,
28
- // Tool streaming was introduced in GLM-4.6; older models must not receive the opt-in.
29
- tool_stream: options.toolStream ??
30
- (body.tools?.length && /^glm-(?:4\.[67]|5(?:[.-]|$))/i.test(request.model.id) ? true : undefined),
33
+ tool_stream: options.toolStream ?? (body.tools?.length && supportsToolStreaming(request.model.id) ? true : undefined),
31
34
  do_sample: options.doSample,
32
35
  response_format: options.responseFormat,
33
36
  request_id: options.requestID,
@@ -13,18 +13,24 @@ const patterns = [
13
13
  /tokens in request more than max tokens allowed/i,
14
14
  /maximum prompt length is \d+/i,
15
15
  /reduce the length of the messages/i,
16
+ // DeepInfra
17
+ /requested input length \d+ exceeds maximum input length/i,
16
18
  /maximum context length is \d+ tokens/i,
17
19
  /exceeds (?:the )?maximum allowed input length of [\d,]+ tokens?/i,
18
- /input \(\d+ tokens\) is longer than the model'?s context length \(\d+ tokens\)/i,
20
+ // Novita omits the token counts.
21
+ /input(?: \(\d+ tokens\))? is longer than the model'?s context length/i,
19
22
  /exceeds the limit of \d+/i,
20
23
  /exceeds the available context size/i,
21
24
  /greater than the context length/i,
25
+ // Hugging Face Text Generation Inference, e.g. Together
26
+ /`inputs` tokens \+ `max_new_tokens` must be <= \d+/i,
22
27
  /context window exceeds limit/i,
23
28
  /exceeded model token limit/i,
24
29
  /context[_ ]length[_ ]exceeded/i,
25
30
  /context length is only \d+ tokens/i,
26
31
  /input length.*exceeds.*context length/i,
27
- /prompt too long; exceeded (?:max )?context length/i,
32
+ // Z.ai code 1261 arrives as `Prompt too long` or `Prompt 超长`.
33
+ /prompt (?:too long|超长)/i,
28
34
  /too large for model with \d+ maximum context length/i,
29
35
  /prompt has [\d,]+ tokens?, but the configured context size is [\d,]+ tokens?/i,
30
36
  /model_context_window_exceeded/i,
@@ -92,7 +98,8 @@ const QUOTA_CODES = new Set([
92
98
  "creditlimitexceeded",
93
99
  ]);
94
100
  // Google reports an invalid API key as HTTP 400 INVALID_ARGUMENT with this `details[].reason`.
95
- const AUTH_CODES = new Set(["authentication_error", "permission_error", "api_key_invalid"]);
101
+ // Z.ai's Responses API reports account and plan rejections mid-stream as `permission_denied`.
102
+ const AUTH_CODES = new Set(["authentication_error", "permission_error", "permission_denied", "api_key_invalid"]);
96
103
  const SERVER_CODES = new Set([
97
104
  "api_error",
98
105
  "internal_error",
@@ -106,6 +113,7 @@ const SERVER_CODES = new Set([
106
113
  ]);
107
114
  // `invalid_request` is the Vercel AI Gateway's code for an upstream request rejection.
108
115
  const INVALID_REQUEST_CODES = new Set([
116
+ "model_not_found",
109
117
  "invalid_prompt",
110
118
  "invalid_request",
111
119
  "invalid_request_error",
@@ -125,14 +133,17 @@ const CONTENT_POLICY_CODES = new Set([
125
133
  // OpenCode Zen replaces upstream codes outside its allow-list but keeps the original
126
134
  // as a `[code]` label at the start of the rewritten message.
127
135
  const GATEWAY_CODE_LABEL = /^[^:\n]+: \[([A-Za-z0-9_.-]+)\]/;
136
+ // xAI reports an invalid API key as HTTP 400 with the generic `invalid-argument` code.
137
+ const AUTH_TEXT = /incorrect api key provided/i;
128
138
  const RATE_LIMIT_TEXT = /rate increased too quickly|rate[-_\s]?limit|too[_\s]?many[_\s]?requests/i;
129
139
  // Only consulted on 429, where throttles and account caps share a status.
130
- const QUOTA_TEXT = /insufficient[-_\s]?quota|quota[-_\s]?exceeded|budget exceeded|usage limit/i;
140
+ // Z.ai reports balance, plan expiry, plan limits, and plan model access on 429.
141
+ const QUOTA_TEXT = /insufficient[-_\s]?(?:quota|balance)|quota[-_\s]?exceeded|budget exceeded|usage limit|limit exhausted|package has expired|plan does not yet include/i;
131
142
  // Policy rejections without a dedicated code, matched against the provider's own
132
143
  // explanation only. OpenAI reuses `invalid_prompt` for usage-policy rejections while
133
144
  // Bedrock Mantle reuses it for schema validation; Anthropic reports blocked output
134
145
  // under `invalid_request_error`.
135
- const CONTENT_POLICY_TEXT = /violating our usage policy|blocked by content filtering policy|content[-_\s]?policy|rejected as a result of our safety system/i;
146
+ const CONTENT_POLICY_TEXT = /violating our usage policy|blocked by content filtering policy|content[-_\s]?policy|rejected as a result of our safety system|detected potentially unsafe or sensitive content/i;
136
147
  const SERVER_ERROR_TEXT = /\b(?:try again|(?:please |you can )?retry (?:the |this |your )?request|try (?:the |this |your )?request again|(?:currently |temporarily )?at capacity|overloaded|temporarily unavailable|service[-_\s]?unavailable|(?:server|internal)[-_\s]?error|server (?:is )?busy|provider returned (?:an )?error|resource[-_\s]?exhausted|upstream (?:connect|connection|request)|request buffer limit while retrying upstream)\b/i;
137
148
  const Message = Schema.String.check(Schema.isPattern(/\S/));
138
149
  const messageAt = (fields, message) => Schema.Struct(fields).pipe(Schema.decodeTo(Schema.String, {
@@ -185,7 +196,10 @@ export function classifyProviderFailure(input) {
185
196
  codes.some((code) => QUOTA_CODES.has(code)) ||
186
197
  (input.status === 429 && QUOTA_TEXT.test(text)))
187
198
  return new QuotaExceededError(details);
188
- if (input.status === 401 || input.status === 403 || codes.some((code) => AUTH_CODES.has(code)))
199
+ if (input.status === 401 ||
200
+ input.status === 403 ||
201
+ codes.some((code) => AUTH_CODES.has(code)) ||
202
+ (input.status === 400 && AUTH_TEXT.test(text)))
189
203
  return new AuthenticationError(details);
190
204
  if (input.status === 429 ||
191
205
  codes.some((code) => code.includes("rate_limit") || code === "too_many_requests" || code === "throttlingexception") ||