gitlab-ai-provider 6.12.0 → 6.12.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -403,6 +403,10 @@ var WorkflowType = /* @__PURE__ */ ((WorkflowType2) => {
403
403
  })(WorkflowType || {});
404
404
  var WS_KEEPALIVE_PING_INTERVAL_MS = 45e3;
405
405
  var WS_HEARTBEAT_INTERVAL_MS = 6e4;
406
+ var WS_CLOSE_WORKFLOW_LOCKED = 1013;
407
+ var WS_CLOSE_WAIT_TIMEOUT_MS = 3e3;
408
+ var WS_WORKFLOW_LOCK_RETRY_DELAY_MS = 1500;
409
+ var WS_WORKFLOW_LOCK_MAX_RETRIES = 8;
406
410
  var DEFAULT_WORKFLOW_DEFINITION = "chat" /* CHAT */;
407
411
  var DEFAULT_CLIENT_CAPABILITIES = ["shell_command"];
408
412
  var CLIENT_VERSION = "8.51.0";
@@ -850,11 +854,17 @@ var GitLabAnthropicLanguageModel = class {
850
854
  *
851
855
  * Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
852
856
  * 1. The system prompt content block — static across all turns.
853
- * 2. The last content block of the second-to-last message — the boundary
854
- * between conversation history and the current turn.
857
+ * 2. The last content block of each of the final two messages.
855
858
  *
856
- * This lets Anthropic cache the system prompt and the accumulated
857
- * conversation prefix, so each new turn only pays for the new content.
859
+ * Two trailing breakpoints (rather than a single one on the penultimate
860
+ * message) keep a cache write within Anthropic's 20-block lookback window
861
+ * as an agentic conversation grows several messages per turn (assistant
862
+ * tool-call → tool-result → …). With a single breakpoint the most recent
863
+ * write can drift more than 20 blocks behind the current position, so the
864
+ * next request fails to prefix-match and pays for a fresh cache write
865
+ * instead of a cheap read. The extra breakpoint costs nothing (breakpoints
866
+ * themselves are free; you only pay for tokens actually written/read) and
867
+ * materially raises the cache hit rate in multi-turn tool-using sessions.
858
868
  *
859
869
  * @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
860
870
  */
@@ -943,10 +953,11 @@ ${message.content}` : message.content;
943
953
  cache_control: { type: "ephemeral" }
944
954
  }
945
955
  ] : void 0;
946
- if (messages.length >= 2) {
947
- const penultimate = messages[messages.length - 2];
948
- if (Array.isArray(penultimate.content)) {
949
- const lastBlock = penultimate.content[penultimate.content.length - 1];
956
+ const breakpointCount = Math.min(2, messages.length);
957
+ for (let i = messages.length - breakpointCount; i < messages.length; i++) {
958
+ const message = messages[i];
959
+ if (Array.isArray(message.content) && message.content.length > 0) {
960
+ const lastBlock = message.content[message.content.length - 1];
950
961
  lastBlock.cache_control = {
951
962
  type: "ephemeral"
952
963
  };
@@ -2440,7 +2451,7 @@ import { AsyncResource } from "async_hooks";
2440
2451
  import WebSocket from "isomorphic-ws";
2441
2452
 
2442
2453
  // src/version.ts
2443
- var VERSION = true ? "6.11.1" : "0.0.0-dev";
2454
+ var VERSION = true ? "6.12.1" : "0.0.0-dev";
2444
2455
 
2445
2456
  // src/gitlab-workflow-client.ts
2446
2457
  var WS_CONNECT_TIMEOUT_MS = 3e4;
@@ -2590,6 +2601,48 @@ var GitLabWorkflowClient = class {
2590
2601
  }
2591
2602
  }
2592
2603
  }
2604
+ /**
2605
+ * Close the socket and wait for the close handshake to actually land.
2606
+ *
2607
+ * `close()` returns as soon as the close frame is queued. That is too early
2608
+ * when the caller is about to reconnect to the same workflow, because
2609
+ * Workhorse releases the workflow lock only once its own stop handshake with
2610
+ * DWS completes. Awaiting `onclose` (or the timeout) keeps the reconnect from
2611
+ * racing the lock and getting rejected with close code 1013.
2612
+ *
2613
+ * Best effort: never rejects, and resolves after `timeoutMs` if the server
2614
+ * never answers.
2615
+ *
2616
+ * @param timeoutMs - Maximum time to wait for the close event
2617
+ * @default WS_CLOSE_WAIT_TIMEOUT_MS
2618
+ */
2619
+ closeAndWait(timeoutMs = WS_CLOSE_WAIT_TIMEOUT_MS) {
2620
+ const sock = this.socket;
2621
+ const pending = sock && (sock.readyState === WebSocket.OPEN || sock.readyState === WebSocket.CONNECTING);
2622
+ if (this.closed || !pending) {
2623
+ this.close();
2624
+ return Promise.resolve();
2625
+ }
2626
+ return new Promise((resolve3) => {
2627
+ let settled = false;
2628
+ const finish = () => {
2629
+ if (settled) return;
2630
+ settled = true;
2631
+ clearTimeout(timer);
2632
+ resolve3();
2633
+ };
2634
+ const timer = setTimeout(finish, timeoutMs);
2635
+ const previous = sock.onclose;
2636
+ sock.onclose = (event) => {
2637
+ try {
2638
+ previous?.call(sock, event);
2639
+ } finally {
2640
+ finish();
2641
+ }
2642
+ };
2643
+ this.close();
2644
+ });
2645
+ }
2593
2646
  /**
2594
2647
  * Check if the WebSocket is currently connected.
2595
2648
  */
@@ -4549,19 +4602,20 @@ var GitLabWorkflowLanguageModel = class _GitLabWorkflowLanguageModel {
4549
4602
  }
4550
4603
  ss.approvalPending = false;
4551
4604
  ss.deferredClose = null;
4552
- this.cleanupClient(ss, false);
4553
4605
  const approval = decision.approved ? { approval: { tool_name: tools[0]?.name, tool_args_json: tools[0]?.args } } : { rejection: { message: decision.message ?? "User rejected" } };
4554
4606
  const newStartReq = {
4555
4607
  ...startReq,
4556
4608
  approval,
4557
4609
  preapproved_tools: decision.approved ? [...startReq.preapproved_tools ?? [], ...tools.map((t) => t.name)] : startReq.preapproved_tools ?? []
4558
4610
  };
4559
- const newClient = new GitLabWorkflowClient();
4560
- this.activeClients.add(newClient);
4561
- ss.activeClient = newClient;
4562
- const modelRef = await this.resolveModelRef();
4563
- try {
4564
- await newClient.connect(
4611
+ const reconnect = async () => {
4612
+ if (ss.streamClosed) return;
4613
+ await this.closeClientAndWait(ss);
4614
+ const client = new GitLabWorkflowClient();
4615
+ this.activeClients.add(client);
4616
+ ss.activeClient = client;
4617
+ const modelRef = await this.resolveModelRef();
4618
+ await client.connect(
4565
4619
  {
4566
4620
  instanceUrl: this.config.instanceUrl,
4567
4621
  modelRef,
@@ -4572,23 +4626,63 @@ var GitLabWorkflowLanguageModel = class _GitLabWorkflowLanguageModel {
4572
4626
  aiCatalogItemVersionId: wsExtras?.aiCatalogItemVersionId,
4573
4627
  workflowDefinition: wsExtras?.workflowDefinition
4574
4628
  },
4575
- (event) => this.handleWorkflowEvent(
4629
+ (event) => {
4630
+ if (event.type === "closed" && event.code === WS_CLOSE_WORKFLOW_LOCKED) {
4631
+ void retryWhileLocked();
4632
+ return;
4633
+ }
4634
+ this.handleWorkflowEvent(
4635
+ ss,
4636
+ event,
4637
+ controller,
4638
+ client,
4639
+ toolExecutor,
4640
+ nextTextId,
4641
+ availableToolNames,
4642
+ newStartReq,
4643
+ wsExtras
4644
+ );
4645
+ }
4646
+ );
4647
+ client.sendStartRequest(newStartReq);
4648
+ };
4649
+ let lockRetries = 0;
4650
+ const retryWhileLocked = async () => {
4651
+ if (ss.streamClosed) return;
4652
+ if (++lockRetries > WS_WORKFLOW_LOCK_MAX_RETRIES) {
4653
+ this.failStream(
4576
4654
  ss,
4577
- event,
4578
4655
  controller,
4579
- newClient,
4580
- toolExecutor,
4581
- nextTextId,
4582
- availableToolNames,
4583
- newStartReq,
4584
- wsExtras
4585
- )
4586
- );
4587
- newClient.sendStartRequest(newStartReq);
4588
- } catch (err) {
4589
- this.cleanupClient(ss, true);
4590
- if (!ss.streamClosed) controller.error(err);
4591
- }
4656
+ new GitLabError({
4657
+ message: "Failed to acquire lock on workflow after reconnect retries",
4658
+ statusCode: WS_CLOSE_WORKFLOW_LOCKED
4659
+ })
4660
+ );
4661
+ return;
4662
+ }
4663
+ await new Promise((r) => setTimeout(r, WS_WORKFLOW_LOCK_RETRY_DELAY_MS * lockRetries));
4664
+ await reconnect().catch((err) => this.failStream(ss, controller, err));
4665
+ };
4666
+ await reconnect().catch((err) => this.failStream(ss, controller, err));
4667
+ }
4668
+ /**
4669
+ * Close the active client and wait for its close handshake, so a following
4670
+ * reconnect to the same workflow does not race the Workhorse lock.
4671
+ */
4672
+ async closeClientAndWait(ss) {
4673
+ const client = ss.activeClient;
4674
+ if (!client) return;
4675
+ this.activeClients.delete(client);
4676
+ ss.activeClient = null;
4677
+ await client.closeAndWait();
4678
+ }
4679
+ /**
4680
+ * Surface a terminal error on the stream and drop the workflow.
4681
+ */
4682
+ failStream(ss, controller, error) {
4683
+ this.cleanupClient(ss, true);
4684
+ if (ss.streamClosed) return;
4685
+ controller.error(error);
4592
4686
  }
4593
4687
  // ---------------------------------------------------------------------------
4594
4688
  // Workflow metadata