gitlab-ai-provider 6.11.1 → 6.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -850,11 +850,17 @@ var GitLabAnthropicLanguageModel = class {
850
850
  *
851
851
  * Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
852
852
  * 1. The system prompt content block — static across all turns.
853
- * 2. The last content block of the second-to-last message — the boundary
854
- * between conversation history and the current turn.
853
+ * 2. The last content block of each of the final two messages.
855
854
  *
856
- * This lets Anthropic cache the system prompt and the accumulated
857
- * conversation prefix, so each new turn only pays for the new content.
855
+ * Two trailing breakpoints (rather than a single one on the penultimate
856
+ * message) keep a cache write within Anthropic's 20-block lookback window
857
+ * as an agentic conversation grows several messages per turn (assistant
858
+ * tool-call → tool-result → …). With a single breakpoint the most recent
859
+ * write can drift more than 20 blocks behind the current position, so the
860
+ * next request fails to prefix-match and pays for a fresh cache write
861
+ * instead of a cheap read. The extra breakpoint costs nothing (breakpoints
862
+ * themselves are free; you only pay for tokens actually written/read) and
863
+ * materially raises the cache hit rate in multi-turn tool-using sessions.
858
864
  *
859
865
  * @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
860
866
  */
@@ -943,10 +949,11 @@ ${message.content}` : message.content;
943
949
  cache_control: { type: "ephemeral" }
944
950
  }
945
951
  ] : void 0;
946
- if (messages.length >= 2) {
947
- const penultimate = messages[messages.length - 2];
948
- if (Array.isArray(penultimate.content)) {
949
- const lastBlock = penultimate.content[penultimate.content.length - 1];
952
+ const breakpointCount = Math.min(2, messages.length);
953
+ for (let i = messages.length - breakpointCount; i < messages.length; i++) {
954
+ const message = messages[i];
955
+ if (Array.isArray(message.content) && message.content.length > 0) {
956
+ const lastBlock = message.content[message.content.length - 1];
950
957
  lastBlock.cache_control = {
951
958
  type: "ephemeral"
952
959
  };
@@ -1379,6 +1386,7 @@ import OpenAI from "openai";
1379
1386
  var MODEL_MAPPINGS = {
1380
1387
  // Anthropic models
1381
1388
  "duo-chat-fable-5": { provider: "anthropic", model: "claude-fable-5" },
1389
+ "duo-chat-opus-5": { provider: "anthropic", model: "claude-opus-5" },
1382
1390
  "duo-chat-opus-4-8": { provider: "anthropic", model: "claude-opus-4-8" },
1383
1391
  "duo-chat-opus-4-7": { provider: "anthropic", model: "claude-opus-4-7" },
1384
1392
  "duo-chat-opus-4-6": { provider: "anthropic", model: "claude-opus-4-6" },
@@ -1437,6 +1445,7 @@ var MODEL_MAPPINGS = {
1437
1445
  model: "anthropic/claude-sonnet-4-5-20250929"
1438
1446
  },
1439
1447
  "duo-workflow-sonnet-5": { provider: "workflow", model: "claude_sonnet_5" },
1448
+ "duo-workflow-opus-5": { provider: "workflow", model: "claude_opus_5" },
1440
1449
  "duo-workflow-sonnet-4-6": { provider: "workflow", model: "claude_sonnet_4_6" },
1441
1450
  "duo-workflow-opus-4-5": {
1442
1451
  provider: "workflow",
@@ -2438,7 +2447,7 @@ import { AsyncResource } from "async_hooks";
2438
2447
  import WebSocket from "isomorphic-ws";
2439
2448
 
2440
2449
  // src/version.ts
2441
- var VERSION = true ? "6.11.0" : "0.0.0-dev";
2450
+ var VERSION = true ? "6.12.0" : "0.0.0-dev";
2442
2451
 
2443
2452
  // src/gitlab-workflow-client.ts
2444
2453
  var WS_CONNECT_TIMEOUT_MS = 3e4;