@gajae-code/ai 0.12.13 → 0.12.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,11 +2,19 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.12.15] - 2026-08-06
6
+
7
+ ### Fixed
8
+
9
+ - Anthropic requests rejected with `A maximum of 4 blocks with cache_control may be provided. Found N.` now retry once with generated caching suppressed, instead of dying on the first attempt (#3934). An Anthropic-compatible gateway may attach its own block-level cache markers before forwarding, and those never appear in the params we serialize, so the total is unpredictable locally and the rejection itself is the only usable signal. The retry keeps generated caching off for the rest of the provider session so later turns do not re-trigger the same 400. Only a genuine breakpoint-overflow `invalid_request_error` is claimed — other `cache_control` complaints, unrelated 400s, non-400 statuses, and our own pre-flight validation failure still surface immediately. The classifier is exported as `isAnthropicCacheBreakpointOverflowError`.
10
+
11
+ ## [0.12.14] - 2026-08-06
12
+
5
13
  ## [0.12.13] - 2026-08-06
6
14
 
7
15
  ### Changed
8
16
 
9
- - Anthropic prompt caching now defaults to top-level automatic caching (`cache_control: { type: "ephemeral" }`) for every Claude-family model, including through non-canonical Anthropic-compatible gateways (Cloudflare AI Gateway, GitHub Copilot, GitLab Duo, Vercel AI Gateway, zenmux, etc.), instead of only `api.anthropic.com`. Non-Claude models on unknown compatible endpoints keep the previous no-cache default; `compat.promptCacheMode: "none"`, `compat.promptCacheMode: "explicit"`, and configured or per-request `cacheRetention: "none"` still opt out. Non-canonical Claude models get the default ~5m cache lifetime unless the endpoint sets `compat.supportsLongCacheRetention: true`.
17
+ - Anthropic prompt caching now defaults to top-level automatic caching (`cache_control: { type: "ephemeral" }`) on the canonical Anthropic API and explicit block-level caching for Claude-family models on non-canonical Anthropic-compatible gateways (Cloudflare AI Gateway, GitHub Copilot, GitLab Duo, Vercel AI Gateway, zenmux, CLIProxyAPI, etc.). Explicit mode is the safer compatible default because gateways commonly inject, rewrite, or reject the top-level field; verified gateways can opt into it with `compat.promptCacheMode: "automatic"`. Non-Claude models on unknown compatible endpoints keep the no-cache default; `promptCacheMode: "none"` and configured or per-request `cacheRetention: "none"` still opt out. Non-canonical Claude models get the default ~5m lifetime unless the endpoint sets `compat.supportsLongCacheRetention: true`.
10
18
 
11
19
  ### Fixed
12
20
 
@@ -63,6 +63,19 @@ export declare function isAnthropicThinkingSignatureInvalidError(error: unknown)
63
63
  * treating it as a thinking-replay rejection.
64
64
  */
65
65
  export declare function isAnthropicMaskedProxyRejection(error: unknown): boolean;
66
+ /**
67
+ * Anthropic rejects a request carrying more than four `cache_control`
68
+ * breakpoints. An Anthropic-compatible gateway may attach its own block-level
69
+ * markers before forwarding, and those never appear in the params we serialize,
70
+ * so no amount of local counting can predict the total. The rejection is the
71
+ * only evidence that our generated marker is one too many, and it is worth
72
+ * exactly one retry with generated caching suppressed.
73
+ *
74
+ * Our own pre-flight `validateCacheControls` failure is deliberately not
75
+ * matched: it carries no `invalid_request_error` wording, so a local bug stays
76
+ * loud instead of being silently retried.
77
+ */
78
+ export declare function isAnthropicCacheBreakpointOverflowError(error: unknown): boolean;
66
79
  export declare const claudeCodeVersion = "2.1.219";
67
80
  export declare const claudeCodeEntrypoint = "sdk-cli";
68
81
  export declare const claudeToolPrefix: string;
@@ -790,10 +790,10 @@ export interface AnthropicCompat extends ToolChoiceCompat {
790
790
  supportsLongCacheRetention?: boolean;
791
791
  /**
792
792
  * Prompt-cache transport accepted by this Anthropic-compatible endpoint.
793
- * Canonical Anthropic and Claude-family models default to `"automatic"`;
794
- * noncanonical non-Claude endpoints default to `"none"`. Set `"automatic"` to
795
- * opt an otherwise unknown compatible endpoint into top-level caching, `"none"`
796
- * to opt out, or `"explicit"` for endpoints that require block-level markers.
793
+ * Canonical Anthropic defaults to `"automatic"`; Claude-family models on
794
+ * noncanonical compatible endpoints default to `"explicit"`; non-Claude
795
+ * compatible endpoints default to `"none"`. Set `"automatic"` to opt into
796
+ * top-level caching, `"none"` to opt out, or `"explicit"` for block markers.
797
797
  */
798
798
  promptCacheMode?: "none" | "explicit" | "automatic";
799
799
  }
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.12.13",
4
+ "version": "0.12.15",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -40,7 +40,7 @@
40
40
  "dependencies": {
41
41
  "@anthropic-ai/sdk": "^0.94.0",
42
42
  "@bufbuild/protobuf": "^2.12.0",
43
- "@gajae-code/utils": "0.12.13",
43
+ "@gajae-code/utils": "0.12.15",
44
44
  "openai": "^6.36.0",
45
45
  "partial-json": "^0.1.7",
46
46
  "zod": "4.4.3"
@@ -319,15 +319,18 @@ const ANTHROPIC_PROVIDER_SESSION_STATE_KEY = "anthropic-messages";
319
319
  type AnthropicProviderSessionState = ProviderSessionState & {
320
320
  strictToolsDisabled: boolean;
321
321
  fastModeDisabled: boolean;
322
+ generatedCachingDisabled: boolean;
322
323
  };
323
324
 
324
325
  function createAnthropicProviderSessionState(): AnthropicProviderSessionState {
325
326
  const state: AnthropicProviderSessionState = {
326
327
  strictToolsDisabled: false,
327
328
  fastModeDisabled: false,
329
+ generatedCachingDisabled: false,
328
330
  close: () => {
329
331
  state.strictToolsDisabled = false;
330
332
  state.fastModeDisabled = false;
333
+ state.generatedCachingDisabled = false;
331
334
  },
332
335
  };
333
336
  return state;
@@ -456,6 +459,28 @@ export function isAnthropicMaskedProxyRejection(error: unknown): boolean {
456
459
  return /"type"\s*:\s*"api_error"/.test(message) && /an error occurred while processing/i.test(message);
457
460
  }
458
461
 
462
+ /**
463
+ * Anthropic rejects a request carrying more than four `cache_control`
464
+ * breakpoints. An Anthropic-compatible gateway may attach its own block-level
465
+ * markers before forwarding, and those never appear in the params we serialize,
466
+ * so no amount of local counting can predict the total. The rejection is the
467
+ * only evidence that our generated marker is one too many, and it is worth
468
+ * exactly one retry with generated caching suppressed.
469
+ *
470
+ * Our own pre-flight `validateCacheControls` failure is deliberately not
471
+ * matched: it carries no `invalid_request_error` wording, so a local bug stays
472
+ * loud instead of being silently retried.
473
+ */
474
+ export function isAnthropicCacheBreakpointOverflowError(error: unknown): boolean {
475
+ if (!isAnthropicInvalidRequestStatus(error)) return false;
476
+ const message = error instanceof Error ? error.message : String(error);
477
+ if (!/invalid_request_error/i.test(message)) return false;
478
+ if (!/cache_control/i.test(message)) return false;
479
+ // Observed: "A maximum of 4 blocks with cache_control may be provided. Found 5."
480
+ // Stay tolerant of phrasing drift around the limit and the reported total.
481
+ return /maximum of \d+ blocks/i.test(message) || /at most \d+ blocks/i.test(message);
482
+ }
483
+
459
484
  function hasStrictAnthropicTools(params: MessageCreateParamsStreaming): boolean {
460
485
  const tools = params.tools as Array<{ strict?: unknown }> | undefined;
461
486
  return tools?.some(tool => tool.strict === true) ?? false;
@@ -493,7 +518,12 @@ function getCacheControl(
493
518
  model: Model<"anthropic-messages">,
494
519
  baseUrl: string,
495
520
  cacheRetention?: CacheRetention,
521
+ suppressGeneratedCaching = false,
496
522
  ): { mode: AnthropicCacheMode; cacheControl?: AnthropicCacheControl } {
523
+ // A gateway already at Anthropic's four-breakpoint limit rejected our
524
+ // generated marker on a previous attempt. The extra markers are invisible
525
+ // here, so the only safe retry is to add none of our own.
526
+ if (suppressGeneratedCaching) return { mode: "none" };
497
527
  const retention = resolveCacheRetention(cacheRetention ?? model.cacheRetention, "long");
498
528
  if (retention === "none") return { mode: "none" };
499
529
 
@@ -504,9 +534,13 @@ function getCacheControl(
504
534
  ? "none"
505
535
  : promptCacheMode === "explicit"
506
536
  ? "explicit"
507
- : promptCacheMode === "automatic" || isCanonicalApi || isClaudeFamilyModel(model)
537
+ : promptCacheMode === "automatic"
508
538
  ? "automatic"
509
- : "none";
539
+ : isCanonicalApi
540
+ ? "automatic"
541
+ : isClaudeFamilyModel(model)
542
+ ? "explicit"
543
+ : "none";
510
544
  if (mode === "none") return { mode };
511
545
 
512
546
  const supportsLongCacheRetention = isCanonicalApi
@@ -1411,16 +1445,23 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1411
1445
  let droppedForcedToolChoice = false;
1412
1446
  let repairLatestAssistantThinking = false;
1413
1447
  let repairAllAssistantThinking = false;
1448
+ let suppressGeneratedCaching = providerSessionState?.generatedCachingDisabled ?? false;
1414
1449
  const prepareParams = async (): Promise<MessageCreateParamsStreaming> => {
1415
1450
  // Degradation state is cumulative: every fallback rebuild must merge all
1416
1451
  // repairs activated so far. Rebuilding from only the immediate call lets
1417
1452
  // a later strict/forced-tool/fast-mode fallback reintroduce the rejected
1418
1453
  // shape (e.g. invalid thinking signatures or forced tool_choice), and
1419
1454
  // the one-shot thinking-repair guard then blocks recovery.
1420
- let nextParams = buildParams(model, baseUrl, context, isOAuthToken, options, disableStrictTools, {
1421
- repairLatestAssistantThinking,
1422
- repairAllAssistantThinking,
1423
- });
1455
+ let nextParams = buildParams(
1456
+ model,
1457
+ baseUrl,
1458
+ context,
1459
+ isOAuthToken,
1460
+ options,
1461
+ disableStrictTools,
1462
+ { repairLatestAssistantThinking, repairAllAssistantThinking },
1463
+ suppressGeneratedCaching,
1464
+ );
1424
1465
  if (droppedForcedToolChoice) {
1425
1466
  delete nextParams.tool_choice;
1426
1467
  }
@@ -1933,6 +1974,29 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1933
1974
  resetOutputForRetry();
1934
1975
  continue;
1935
1976
  }
1977
+ if (
1978
+ !options?.fallbackManaged &&
1979
+ !suppressGeneratedCaching &&
1980
+ firstTokenTime === undefined &&
1981
+ isAnthropicCacheBreakpointOverflowError(streamFailure)
1982
+ ) {
1983
+ // The gateway's own markers already fill Anthropic's four slots, so
1984
+ // our generated breakpoint is the fifth. We cannot see the others,
1985
+ // which makes the rejection itself the only usable signal; retry
1986
+ // once with generated caching off and keep it off for the session.
1987
+ logger.debug("anthropic: cache breakpoint limit exceeded, retrying without generated caching", {
1988
+ model: model.id,
1989
+ error: streamFailure instanceof Error ? streamFailure.message : String(streamFailure),
1990
+ });
1991
+ if (providerSessionState) {
1992
+ providerSessionState.generatedCachingDisabled = true;
1993
+ }
1994
+ suppressGeneratedCaching = true;
1995
+ params = await prepareParams();
1996
+ providerRetryAttempt = 0;
1997
+ resetOutputForRetry();
1998
+ continue;
1999
+ }
1936
2000
  const isTransientEnvelopeFailure =
1937
2001
  isTransientStreamParseError(streamFailure) || isTransientStreamEnvelopeError(streamFailure);
1938
2002
  const canRetryTransientEnvelopeFailure = isTransientEnvelopeFailure && !streamedReplayUnsafeContent;
@@ -2396,6 +2460,7 @@ function enforceCacheControlLimit(params: MessageCreateParamsStreaming, maxBreak
2396
2460
  if (maxBreakpoints !== 4) throw new Error("Anthropic supports exactly four cache breakpoints");
2397
2461
  validateCacheControls(params as AnthropicCacheParams);
2398
2462
  }
2463
+
2399
2464
  function buildParams(
2400
2465
  model: Model<"anthropic-messages">,
2401
2466
  baseUrl: string,
@@ -2404,8 +2469,14 @@ function buildParams(
2404
2469
  options?: AnthropicOptions,
2405
2470
  disableStrictTools = false,
2406
2471
  thinkingRepair?: { repairLatestAssistantThinking?: boolean; repairAllAssistantThinking?: boolean },
2472
+ suppressGeneratedCaching = false,
2407
2473
  ): MessageCreateParamsStreaming {
2408
- const { mode: cacheMode, cacheControl } = getCacheControl(model, baseUrl, options?.cacheRetention);
2474
+ const { mode: cacheMode, cacheControl } = getCacheControl(
2475
+ model,
2476
+ baseUrl,
2477
+ options?.cacheRetention,
2478
+ suppressGeneratedCaching,
2479
+ );
2409
2480
 
2410
2481
  const params: AnthropicSamplingParams = {
2411
2482
  model: model.id,
package/src/types.ts CHANGED
@@ -951,10 +951,10 @@ export interface AnthropicCompat extends ToolChoiceCompat {
951
951
  supportsLongCacheRetention?: boolean;
952
952
  /**
953
953
  * Prompt-cache transport accepted by this Anthropic-compatible endpoint.
954
- * Canonical Anthropic and Claude-family models default to `"automatic"`;
955
- * noncanonical non-Claude endpoints default to `"none"`. Set `"automatic"` to
956
- * opt an otherwise unknown compatible endpoint into top-level caching, `"none"`
957
- * to opt out, or `"explicit"` for endpoints that require block-level markers.
954
+ * Canonical Anthropic defaults to `"automatic"`; Claude-family models on
955
+ * noncanonical compatible endpoints default to `"explicit"`; non-Claude
956
+ * compatible endpoints default to `"none"`. Set `"automatic"` to opt into
957
+ * top-level caching, `"none"` to opt out, or `"explicit"` for block markers.
958
958
  */
959
959
  promptCacheMode?: "none" | "explicit" | "automatic";
960
960
  }