gitlab-ai-provider 6.12.0 → 6.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,11 @@
2
2
 
3
3
  All notable changes to this project will be documented in this file. See [Conventional Commits](https://conventionalcommits.org) for commit guidelines.
4
4
 
5
+ ## <small>6.12.1 (2026-07-29)</small>
6
+
7
+ - Merge branch 'ghavenga-cache-breakpoint-placement' into 'main' ([e8d0fe5](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/e8d0fe5))
8
+ - perf(anthropic): place cache breakpoints on final two messages ([5f2e13e](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/5f2e13e))
9
+
5
10
  ## 6.12.0 (2026-07-27)
6
11
 
7
12
  - Merge branch 'feature-add-opus-5' into 'main' ([9a5f447](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/9a5f447))
package/dist/index.d.mts CHANGED
@@ -88,11 +88,17 @@ declare class GitLabAnthropicLanguageModel implements LanguageModelV3 {
88
88
  *
89
89
  * Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
90
90
  * 1. The system prompt content block — static across all turns.
91
- * 2. The last content block of the second-to-last message — the boundary
92
- * between conversation history and the current turn.
91
+ * 2. The last content block of each of the final two messages.
93
92
  *
94
- * This lets Anthropic cache the system prompt and the accumulated
95
- * conversation prefix, so each new turn only pays for the new content.
93
+ * Two trailing breakpoints (rather than a single one on the penultimate
94
+ * message) keep a cache write within Anthropic's 20-block lookback window
95
+ * as an agentic conversation grows several messages per turn (assistant
96
+ * tool-call → tool-result → …). With a single breakpoint the most recent
97
+ * write can drift more than 20 blocks behind the current position, so the
98
+ * next request fails to prefix-match and pays for a fresh cache write
99
+ * instead of a cheap read. The extra breakpoint costs nothing (breakpoints
100
+ * themselves are free; you only pay for tokens actually written/read) and
101
+ * materially raises the cache hit rate in multi-turn tool-using sessions.
96
102
  *
97
103
  * @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
98
104
  */
package/dist/index.d.ts CHANGED
@@ -88,11 +88,17 @@ declare class GitLabAnthropicLanguageModel implements LanguageModelV3 {
88
88
  *
89
89
  * Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
90
90
  * 1. The system prompt content block — static across all turns.
91
- * 2. The last content block of the second-to-last message — the boundary
92
- * between conversation history and the current turn.
91
+ * 2. The last content block of each of the final two messages.
93
92
  *
94
- * This lets Anthropic cache the system prompt and the accumulated
95
- * conversation prefix, so each new turn only pays for the new content.
93
+ * Two trailing breakpoints (rather than a single one on the penultimate
94
+ * message) keep a cache write within Anthropic's 20-block lookback window
95
+ * as an agentic conversation grows several messages per turn (assistant
96
+ * tool-call → tool-result → …). With a single breakpoint the most recent
97
+ * write can drift more than 20 blocks behind the current position, so the
98
+ * next request fails to prefix-match and pays for a fresh cache write
99
+ * instead of a cheap read. The extra breakpoint costs nothing (breakpoints
100
+ * themselves are free; you only pay for tokens actually written/read) and
101
+ * materially raises the cache hit rate in multi-turn tool-using sessions.
96
102
  *
97
103
  * @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
98
104
  */
package/dist/index.js CHANGED
@@ -929,11 +929,17 @@ var GitLabAnthropicLanguageModel = class {
929
929
  *
930
930
  * Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
931
931
  * 1. The system prompt content block — static across all turns.
932
- * 2. The last content block of the second-to-last message — the boundary
933
- * between conversation history and the current turn.
932
+ * 2. The last content block of each of the final two messages.
934
933
  *
935
- * This lets Anthropic cache the system prompt and the accumulated
936
- * conversation prefix, so each new turn only pays for the new content.
934
+ * Two trailing breakpoints (rather than a single one on the penultimate
935
+ * message) keep a cache write within Anthropic's 20-block lookback window
936
+ * as an agentic conversation grows several messages per turn (assistant
937
+ * tool-call → tool-result → …). With a single breakpoint the most recent
938
+ * write can drift more than 20 blocks behind the current position, so the
939
+ * next request fails to prefix-match and pays for a fresh cache write
940
+ * instead of a cheap read. The extra breakpoint costs nothing (breakpoints
941
+ * themselves are free; you only pay for tokens actually written/read) and
942
+ * materially raises the cache hit rate in multi-turn tool-using sessions.
937
943
  *
938
944
  * @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
939
945
  */
@@ -1022,10 +1028,11 @@ ${message.content}` : message.content;
1022
1028
  cache_control: { type: "ephemeral" }
1023
1029
  }
1024
1030
  ] : void 0;
1025
- if (messages.length >= 2) {
1026
- const penultimate = messages[messages.length - 2];
1027
- if (Array.isArray(penultimate.content)) {
1028
- const lastBlock = penultimate.content[penultimate.content.length - 1];
1031
+ const breakpointCount = Math.min(2, messages.length);
1032
+ for (let i = messages.length - breakpointCount; i < messages.length; i++) {
1033
+ const message = messages[i];
1034
+ if (Array.isArray(message.content) && message.content.length > 0) {
1035
+ const lastBlock = message.content[message.content.length - 1];
1029
1036
  lastBlock.cache_control = {
1030
1037
  type: "ephemeral"
1031
1038
  };
@@ -2519,7 +2526,7 @@ var import_node_async_hooks = require("async_hooks");
2519
2526
  var import_isomorphic_ws = __toESM(require("isomorphic-ws"));
2520
2527
 
2521
2528
  // src/version.ts
2522
- var VERSION = true ? "6.11.1" : "0.0.0-dev";
2529
+ var VERSION = true ? "6.12.0" : "0.0.0-dev";
2523
2530
 
2524
2531
  // src/gitlab-workflow-client.ts
2525
2532
  var WS_CONNECT_TIMEOUT_MS = 3e4;