gitlab-ai-provider 6.12.0 → 6.12.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,16 @@
2
2
 
3
3
  All notable changes to this project will be documented in this file. See [Conventional Commits](https://conventionalcommits.org) for commit guidelines.
4
4
 
5
+ ## <small>6.12.2 (2026-08-24)</small>
6
+
7
+ - Merge branch 'fix/workflow-lock-reconnect-race' into 'main' ([51a9277](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/51a9277))
8
+ - fix(workflow): wait out the workflow lock before reconnecting after approval ([35ca694](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/35ca694))
9
+
10
+ ## <small>6.12.1 (2026-07-29)</small>
11
+
12
+ - Merge branch 'ghavenga-cache-breakpoint-placement' into 'main' ([e8d0fe5](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/e8d0fe5))
13
+ - perf(anthropic): place cache breakpoints on final two messages ([5f2e13e](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/5f2e13e))
14
+
5
15
  ## 6.12.0 (2026-07-27)
6
16
 
7
17
  - Merge branch 'feature-add-opus-5' into 'main' ([9a5f447](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/9a5f447))
package/dist/index.d.mts CHANGED
@@ -88,11 +88,17 @@ declare class GitLabAnthropicLanguageModel implements LanguageModelV3 {
88
88
  *
89
89
  * Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
90
90
  * 1. The system prompt content block — static across all turns.
91
- * 2. The last content block of the second-to-last message — the boundary
92
- * between conversation history and the current turn.
91
+ * 2. The last content block of each of the final two messages.
93
92
  *
94
- * This lets Anthropic cache the system prompt and the accumulated
95
- * conversation prefix, so each new turn only pays for the new content.
93
+ * Two trailing breakpoints (rather than a single one on the penultimate
94
+ * message) keep a cache write within Anthropic's 20-block lookback window
95
+ * as an agentic conversation grows several messages per turn (assistant
96
+ * tool-call → tool-result → …). With a single breakpoint the most recent
97
+ * write can drift more than 20 blocks behind the current position, so the
98
+ * next request fails to prefix-match and pays for a fresh cache write
99
+ * instead of a cheap read. The extra breakpoint costs nothing (breakpoints
100
+ * themselves are free; you only pay for tokens actually written/read) and
101
+ * materially raises the cache hit rate in multi-turn tool-using sessions.
96
102
  *
97
103
  * @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
98
104
  */
@@ -888,6 +894,15 @@ declare class GitLabWorkflowLanguageModel implements LanguageModelV3 {
888
894
  private executeToolAndRespond;
889
895
  private cleanupClient;
890
896
  private approveAndResume;
897
+ /**
898
+ * Close the active client and wait for its close handshake, so a following
899
+ * reconnect to the same workflow does not race the Workhorse lock.
900
+ */
901
+ private closeClientAndWait;
902
+ /**
903
+ * Surface a terminal error on the stream and drop the workflow.
904
+ */
905
+ private failStream;
891
906
  private buildWorkflowMetadata;
892
907
  private getGitInfo;
893
908
  /**
@@ -1560,6 +1575,22 @@ declare class GitLabWorkflowClient {
1560
1575
  * Close the WebSocket connection.
1561
1576
  */
1562
1577
  close(): void;
1578
+ /**
1579
+ * Close the socket and wait for the close handshake to actually land.
1580
+ *
1581
+ * `close()` returns as soon as the close frame is queued. That is too early
1582
+ * when the caller is about to reconnect to the same workflow, because
1583
+ * Workhorse releases the workflow lock only once its own stop handshake with
1584
+ * DWS completes. Awaiting `onclose` (or the timeout) keeps the reconnect from
1585
+ * racing the lock and getting rejected with close code 1013.
1586
+ *
1587
+ * Best effort: never rejects, and resolves after `timeoutMs` if the server
1588
+ * never answers.
1589
+ *
1590
+ * @param timeoutMs - Maximum time to wait for the close event
1591
+ * @default WS_CLOSE_WAIT_TIMEOUT_MS
1592
+ */
1593
+ closeAndWait(timeoutMs?: number): Promise<void>;
1563
1594
  /**
1564
1595
  * Check if the WebSocket is currently connected.
1565
1596
  */
package/dist/index.d.ts CHANGED
@@ -88,11 +88,17 @@ declare class GitLabAnthropicLanguageModel implements LanguageModelV3 {
88
88
  *
89
89
  * Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
90
90
  * 1. The system prompt content block — static across all turns.
91
- * 2. The last content block of the second-to-last message — the boundary
92
- * between conversation history and the current turn.
91
+ * 2. The last content block of each of the final two messages.
93
92
  *
94
- * This lets Anthropic cache the system prompt and the accumulated
95
- * conversation prefix, so each new turn only pays for the new content.
93
+ * Two trailing breakpoints (rather than a single one on the penultimate
94
+ * message) keep a cache write within Anthropic's 20-block lookback window
95
+ * as an agentic conversation grows several messages per turn (assistant
96
+ * tool-call → tool-result → …). With a single breakpoint the most recent
97
+ * write can drift more than 20 blocks behind the current position, so the
98
+ * next request fails to prefix-match and pays for a fresh cache write
99
+ * instead of a cheap read. The extra breakpoint costs nothing (breakpoints
100
+ * themselves are free; you only pay for tokens actually written/read) and
101
+ * materially raises the cache hit rate in multi-turn tool-using sessions.
96
102
  *
97
103
  * @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
98
104
  */
@@ -888,6 +894,15 @@ declare class GitLabWorkflowLanguageModel implements LanguageModelV3 {
888
894
  private executeToolAndRespond;
889
895
  private cleanupClient;
890
896
  private approveAndResume;
897
+ /**
898
+ * Close the active client and wait for its close handshake, so a following
899
+ * reconnect to the same workflow does not race the Workhorse lock.
900
+ */
901
+ private closeClientAndWait;
902
+ /**
903
+ * Surface a terminal error on the stream and drop the workflow.
904
+ */
905
+ private failStream;
891
906
  private buildWorkflowMetadata;
892
907
  private getGitInfo;
893
908
  /**
@@ -1560,6 +1575,22 @@ declare class GitLabWorkflowClient {
1560
1575
  * Close the WebSocket connection.
1561
1576
  */
1562
1577
  close(): void;
1578
+ /**
1579
+ * Close the socket and wait for the close handshake to actually land.
1580
+ *
1581
+ * `close()` returns as soon as the close frame is queued. That is too early
1582
+ * when the caller is about to reconnect to the same workflow, because
1583
+ * Workhorse releases the workflow lock only once its own stop handshake with
1584
+ * DWS completes. Awaiting `onclose` (or the timeout) keeps the reconnect from
1585
+ * racing the lock and getting rejected with close code 1013.
1586
+ *
1587
+ * Best effort: never rejects, and resolves after `timeoutMs` if the server
1588
+ * never answers.
1589
+ *
1590
+ * @param timeoutMs - Maximum time to wait for the close event
1591
+ * @default WS_CLOSE_WAIT_TIMEOUT_MS
1592
+ */
1593
+ closeAndWait(timeoutMs?: number): Promise<void>;
1563
1594
  /**
1564
1595
  * Check if the WebSocket is currently connected.
1565
1596
  */
package/dist/index.js CHANGED
@@ -482,6 +482,10 @@ var WorkflowType = /* @__PURE__ */ ((WorkflowType2) => {
482
482
  })(WorkflowType || {});
483
483
  var WS_KEEPALIVE_PING_INTERVAL_MS = 45e3;
484
484
  var WS_HEARTBEAT_INTERVAL_MS = 6e4;
485
+ var WS_CLOSE_WORKFLOW_LOCKED = 1013;
486
+ var WS_CLOSE_WAIT_TIMEOUT_MS = 3e3;
487
+ var WS_WORKFLOW_LOCK_RETRY_DELAY_MS = 1500;
488
+ var WS_WORKFLOW_LOCK_MAX_RETRIES = 8;
485
489
  var DEFAULT_WORKFLOW_DEFINITION = "chat" /* CHAT */;
486
490
  var DEFAULT_CLIENT_CAPABILITIES = ["shell_command"];
487
491
  var CLIENT_VERSION = "8.51.0";
@@ -929,11 +933,17 @@ var GitLabAnthropicLanguageModel = class {
929
933
  *
930
934
  * Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
931
935
  * 1. The system prompt content block — static across all turns.
932
- * 2. The last content block of the second-to-last message — the boundary
933
- * between conversation history and the current turn.
936
+ * 2. The last content block of each of the final two messages.
934
937
  *
935
- * This lets Anthropic cache the system prompt and the accumulated
936
- * conversation prefix, so each new turn only pays for the new content.
938
+ * Two trailing breakpoints (rather than a single one on the penultimate
939
+ * message) keep a cache write within Anthropic's 20-block lookback window
940
+ * as an agentic conversation grows several messages per turn (assistant
941
+ * tool-call → tool-result → …). With a single breakpoint the most recent
942
+ * write can drift more than 20 blocks behind the current position, so the
943
+ * next request fails to prefix-match and pays for a fresh cache write
944
+ * instead of a cheap read. The extra breakpoint costs nothing (breakpoints
945
+ * themselves are free; you only pay for tokens actually written/read) and
946
+ * materially raises the cache hit rate in multi-turn tool-using sessions.
937
947
  *
938
948
  * @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
939
949
  */
@@ -1022,10 +1032,11 @@ ${message.content}` : message.content;
1022
1032
  cache_control: { type: "ephemeral" }
1023
1033
  }
1024
1034
  ] : void 0;
1025
- if (messages.length >= 2) {
1026
- const penultimate = messages[messages.length - 2];
1027
- if (Array.isArray(penultimate.content)) {
1028
- const lastBlock = penultimate.content[penultimate.content.length - 1];
1035
+ const breakpointCount = Math.min(2, messages.length);
1036
+ for (let i = messages.length - breakpointCount; i < messages.length; i++) {
1037
+ const message = messages[i];
1038
+ if (Array.isArray(message.content) && message.content.length > 0) {
1039
+ const lastBlock = message.content[message.content.length - 1];
1029
1040
  lastBlock.cache_control = {
1030
1041
  type: "ephemeral"
1031
1042
  };
@@ -2519,7 +2530,7 @@ var import_node_async_hooks = require("async_hooks");
2519
2530
  var import_isomorphic_ws = __toESM(require("isomorphic-ws"));
2520
2531
 
2521
2532
  // src/version.ts
2522
- var VERSION = true ? "6.11.1" : "0.0.0-dev";
2533
+ var VERSION = true ? "6.12.1" : "0.0.0-dev";
2523
2534
 
2524
2535
  // src/gitlab-workflow-client.ts
2525
2536
  var WS_CONNECT_TIMEOUT_MS = 3e4;
@@ -2669,6 +2680,48 @@ var GitLabWorkflowClient = class {
2669
2680
  }
2670
2681
  }
2671
2682
  }
2683
+ /**
2684
+ * Close the socket and wait for the close handshake to actually land.
2685
+ *
2686
+ * `close()` returns as soon as the close frame is queued. That is too early
2687
+ * when the caller is about to reconnect to the same workflow, because
2688
+ * Workhorse releases the workflow lock only once its own stop handshake with
2689
+ * DWS completes. Awaiting `onclose` (or the timeout) keeps the reconnect from
2690
+ * racing the lock and getting rejected with close code 1013.
2691
+ *
2692
+ * Best effort: never rejects, and resolves after `timeoutMs` if the server
2693
+ * never answers.
2694
+ *
2695
+ * @param timeoutMs - Maximum time to wait for the close event
2696
+ * @default WS_CLOSE_WAIT_TIMEOUT_MS
2697
+ */
2698
+ closeAndWait(timeoutMs = WS_CLOSE_WAIT_TIMEOUT_MS) {
2699
+ const sock = this.socket;
2700
+ const pending = sock && (sock.readyState === import_isomorphic_ws.default.OPEN || sock.readyState === import_isomorphic_ws.default.CONNECTING);
2701
+ if (this.closed || !pending) {
2702
+ this.close();
2703
+ return Promise.resolve();
2704
+ }
2705
+ return new Promise((resolve3) => {
2706
+ let settled = false;
2707
+ const finish = () => {
2708
+ if (settled) return;
2709
+ settled = true;
2710
+ clearTimeout(timer);
2711
+ resolve3();
2712
+ };
2713
+ const timer = setTimeout(finish, timeoutMs);
2714
+ const previous = sock.onclose;
2715
+ sock.onclose = (event) => {
2716
+ try {
2717
+ previous?.call(sock, event);
2718
+ } finally {
2719
+ finish();
2720
+ }
2721
+ };
2722
+ this.close();
2723
+ });
2724
+ }
2672
2725
  /**
2673
2726
  * Check if the WebSocket is currently connected.
2674
2727
  */
@@ -4628,19 +4681,20 @@ var GitLabWorkflowLanguageModel = class _GitLabWorkflowLanguageModel {
4628
4681
  }
4629
4682
  ss.approvalPending = false;
4630
4683
  ss.deferredClose = null;
4631
- this.cleanupClient(ss, false);
4632
4684
  const approval = decision.approved ? { approval: { tool_name: tools[0]?.name, tool_args_json: tools[0]?.args } } : { rejection: { message: decision.message ?? "User rejected" } };
4633
4685
  const newStartReq = {
4634
4686
  ...startReq,
4635
4687
  approval,
4636
4688
  preapproved_tools: decision.approved ? [...startReq.preapproved_tools ?? [], ...tools.map((t) => t.name)] : startReq.preapproved_tools ?? []
4637
4689
  };
4638
- const newClient = new GitLabWorkflowClient();
4639
- this.activeClients.add(newClient);
4640
- ss.activeClient = newClient;
4641
- const modelRef = await this.resolveModelRef();
4642
- try {
4643
- await newClient.connect(
4690
+ const reconnect = async () => {
4691
+ if (ss.streamClosed) return;
4692
+ await this.closeClientAndWait(ss);
4693
+ const client = new GitLabWorkflowClient();
4694
+ this.activeClients.add(client);
4695
+ ss.activeClient = client;
4696
+ const modelRef = await this.resolveModelRef();
4697
+ await client.connect(
4644
4698
  {
4645
4699
  instanceUrl: this.config.instanceUrl,
4646
4700
  modelRef,
@@ -4651,23 +4705,63 @@ var GitLabWorkflowLanguageModel = class _GitLabWorkflowLanguageModel {
4651
4705
  aiCatalogItemVersionId: wsExtras?.aiCatalogItemVersionId,
4652
4706
  workflowDefinition: wsExtras?.workflowDefinition
4653
4707
  },
4654
- (event) => this.handleWorkflowEvent(
4708
+ (event) => {
4709
+ if (event.type === "closed" && event.code === WS_CLOSE_WORKFLOW_LOCKED) {
4710
+ void retryWhileLocked();
4711
+ return;
4712
+ }
4713
+ this.handleWorkflowEvent(
4714
+ ss,
4715
+ event,
4716
+ controller,
4717
+ client,
4718
+ toolExecutor,
4719
+ nextTextId,
4720
+ availableToolNames,
4721
+ newStartReq,
4722
+ wsExtras
4723
+ );
4724
+ }
4725
+ );
4726
+ client.sendStartRequest(newStartReq);
4727
+ };
4728
+ let lockRetries = 0;
4729
+ const retryWhileLocked = async () => {
4730
+ if (ss.streamClosed) return;
4731
+ if (++lockRetries > WS_WORKFLOW_LOCK_MAX_RETRIES) {
4732
+ this.failStream(
4655
4733
  ss,
4656
- event,
4657
4734
  controller,
4658
- newClient,
4659
- toolExecutor,
4660
- nextTextId,
4661
- availableToolNames,
4662
- newStartReq,
4663
- wsExtras
4664
- )
4665
- );
4666
- newClient.sendStartRequest(newStartReq);
4667
- } catch (err) {
4668
- this.cleanupClient(ss, true);
4669
- if (!ss.streamClosed) controller.error(err);
4670
- }
4735
+ new GitLabError({
4736
+ message: "Failed to acquire lock on workflow after reconnect retries",
4737
+ statusCode: WS_CLOSE_WORKFLOW_LOCKED
4738
+ })
4739
+ );
4740
+ return;
4741
+ }
4742
+ await new Promise((r) => setTimeout(r, WS_WORKFLOW_LOCK_RETRY_DELAY_MS * lockRetries));
4743
+ await reconnect().catch((err) => this.failStream(ss, controller, err));
4744
+ };
4745
+ await reconnect().catch((err) => this.failStream(ss, controller, err));
4746
+ }
4747
+ /**
4748
+ * Close the active client and wait for its close handshake, so a following
4749
+ * reconnect to the same workflow does not race the Workhorse lock.
4750
+ */
4751
+ async closeClientAndWait(ss) {
4752
+ const client = ss.activeClient;
4753
+ if (!client) return;
4754
+ this.activeClients.delete(client);
4755
+ ss.activeClient = null;
4756
+ await client.closeAndWait();
4757
+ }
4758
+ /**
4759
+ * Surface a terminal error on the stream and drop the workflow.
4760
+ */
4761
+ failStream(ss, controller, error) {
4762
+ this.cleanupClient(ss, true);
4763
+ if (ss.streamClosed) return;
4764
+ controller.error(error);
4671
4765
  }
4672
4766
  // ---------------------------------------------------------------------------
4673
4767
  // Workflow metadata