gitlab-ai-provider 6.12.0 → 6.12.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/dist/gitlab-ai-provider-6.12.2.tgz +0 -0
- package/dist/index.d.mts +35 -4
- package/dist/index.d.ts +35 -4
- package/dist/index.js +125 -31
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +125 -31
- package/dist/index.mjs.map +1 -1
- package/package.json +1 -1
- package/dist/gitlab-ai-provider-6.12.0.tgz +0 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,16 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file. See [Conventional Commits](https://conventionalcommits.org) for commit guidelines.
|
|
4
4
|
|
|
5
|
+
## <small>6.12.2 (2026-08-24)</small>
|
|
6
|
+
|
|
7
|
+
- Merge branch 'fix/workflow-lock-reconnect-race' into 'main' ([51a9277](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/51a9277))
|
|
8
|
+
- fix(workflow): wait out the workflow lock before reconnecting after approval ([35ca694](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/35ca694))
|
|
9
|
+
|
|
10
|
+
## <small>6.12.1 (2026-07-29)</small>
|
|
11
|
+
|
|
12
|
+
- Merge branch 'ghavenga-cache-breakpoint-placement' into 'main' ([e8d0fe5](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/e8d0fe5))
|
|
13
|
+
- perf(anthropic): place cache breakpoints on final two messages ([5f2e13e](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/5f2e13e))
|
|
14
|
+
|
|
5
15
|
## 6.12.0 (2026-07-27)
|
|
6
16
|
|
|
7
17
|
- Merge branch 'feature-add-opus-5' into 'main' ([9a5f447](https://gitlab.com/vglafirov/gitlab-ai-provider/commit/9a5f447))
|
|
Binary file
|
package/dist/index.d.mts
CHANGED
|
@@ -88,11 +88,17 @@ declare class GitLabAnthropicLanguageModel implements LanguageModelV3 {
|
|
|
88
88
|
*
|
|
89
89
|
* Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
|
|
90
90
|
* 1. The system prompt content block — static across all turns.
|
|
91
|
-
* 2. The last content block of
|
|
92
|
-
* between conversation history and the current turn.
|
|
91
|
+
* 2. The last content block of each of the final two messages.
|
|
93
92
|
*
|
|
94
|
-
*
|
|
95
|
-
*
|
|
93
|
+
* Two trailing breakpoints (rather than a single one on the penultimate
|
|
94
|
+
* message) keep a cache write within Anthropic's 20-block lookback window
|
|
95
|
+
* as an agentic conversation grows several messages per turn (assistant
|
|
96
|
+
* tool-call → tool-result → …). With a single breakpoint the most recent
|
|
97
|
+
* write can drift more than 20 blocks behind the current position, so the
|
|
98
|
+
* next request fails to prefix-match and pays for a fresh cache write
|
|
99
|
+
* instead of a cheap read. The extra breakpoint costs nothing (breakpoints
|
|
100
|
+
* themselves are free; you only pay for tokens actually written/read) and
|
|
101
|
+
* materially raises the cache hit rate in multi-turn tool-using sessions.
|
|
96
102
|
*
|
|
97
103
|
* @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
|
|
98
104
|
*/
|
|
@@ -888,6 +894,15 @@ declare class GitLabWorkflowLanguageModel implements LanguageModelV3 {
|
|
|
888
894
|
private executeToolAndRespond;
|
|
889
895
|
private cleanupClient;
|
|
890
896
|
private approveAndResume;
|
|
897
|
+
/**
|
|
898
|
+
* Close the active client and wait for its close handshake, so a following
|
|
899
|
+
* reconnect to the same workflow does not race the Workhorse lock.
|
|
900
|
+
*/
|
|
901
|
+
private closeClientAndWait;
|
|
902
|
+
/**
|
|
903
|
+
* Surface a terminal error on the stream and drop the workflow.
|
|
904
|
+
*/
|
|
905
|
+
private failStream;
|
|
891
906
|
private buildWorkflowMetadata;
|
|
892
907
|
private getGitInfo;
|
|
893
908
|
/**
|
|
@@ -1560,6 +1575,22 @@ declare class GitLabWorkflowClient {
|
|
|
1560
1575
|
* Close the WebSocket connection.
|
|
1561
1576
|
*/
|
|
1562
1577
|
close(): void;
|
|
1578
|
+
/**
|
|
1579
|
+
* Close the socket and wait for the close handshake to actually land.
|
|
1580
|
+
*
|
|
1581
|
+
* `close()` returns as soon as the close frame is queued. That is too early
|
|
1582
|
+
* when the caller is about to reconnect to the same workflow, because
|
|
1583
|
+
* Workhorse releases the workflow lock only once its own stop handshake with
|
|
1584
|
+
* DWS completes. Awaiting `onclose` (or the timeout) keeps the reconnect from
|
|
1585
|
+
* racing the lock and getting rejected with close code 1013.
|
|
1586
|
+
*
|
|
1587
|
+
* Best effort: never rejects, and resolves after `timeoutMs` if the server
|
|
1588
|
+
* never answers.
|
|
1589
|
+
*
|
|
1590
|
+
* @param timeoutMs - Maximum time to wait for the close event
|
|
1591
|
+
* @default WS_CLOSE_WAIT_TIMEOUT_MS
|
|
1592
|
+
*/
|
|
1593
|
+
closeAndWait(timeoutMs?: number): Promise<void>;
|
|
1563
1594
|
/**
|
|
1564
1595
|
* Check if the WebSocket is currently connected.
|
|
1565
1596
|
*/
|
package/dist/index.d.ts
CHANGED
|
@@ -88,11 +88,17 @@ declare class GitLabAnthropicLanguageModel implements LanguageModelV3 {
|
|
|
88
88
|
*
|
|
89
89
|
* Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
|
|
90
90
|
* 1. The system prompt content block — static across all turns.
|
|
91
|
-
* 2. The last content block of
|
|
92
|
-
* between conversation history and the current turn.
|
|
91
|
+
* 2. The last content block of each of the final two messages.
|
|
93
92
|
*
|
|
94
|
-
*
|
|
95
|
-
*
|
|
93
|
+
* Two trailing breakpoints (rather than a single one on the penultimate
|
|
94
|
+
* message) keep a cache write within Anthropic's 20-block lookback window
|
|
95
|
+
* as an agentic conversation grows several messages per turn (assistant
|
|
96
|
+
* tool-call → tool-result → …). With a single breakpoint the most recent
|
|
97
|
+
* write can drift more than 20 blocks behind the current position, so the
|
|
98
|
+
* next request fails to prefix-match and pays for a fresh cache write
|
|
99
|
+
* instead of a cheap read. The extra breakpoint costs nothing (breakpoints
|
|
100
|
+
* themselves are free; you only pay for tokens actually written/read) and
|
|
101
|
+
* materially raises the cache hit rate in multi-turn tool-using sessions.
|
|
96
102
|
*
|
|
97
103
|
* @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
|
|
98
104
|
*/
|
|
@@ -888,6 +894,15 @@ declare class GitLabWorkflowLanguageModel implements LanguageModelV3 {
|
|
|
888
894
|
private executeToolAndRespond;
|
|
889
895
|
private cleanupClient;
|
|
890
896
|
private approveAndResume;
|
|
897
|
+
/**
|
|
898
|
+
* Close the active client and wait for its close handshake, so a following
|
|
899
|
+
* reconnect to the same workflow does not race the Workhorse lock.
|
|
900
|
+
*/
|
|
901
|
+
private closeClientAndWait;
|
|
902
|
+
/**
|
|
903
|
+
* Surface a terminal error on the stream and drop the workflow.
|
|
904
|
+
*/
|
|
905
|
+
private failStream;
|
|
891
906
|
private buildWorkflowMetadata;
|
|
892
907
|
private getGitInfo;
|
|
893
908
|
/**
|
|
@@ -1560,6 +1575,22 @@ declare class GitLabWorkflowClient {
|
|
|
1560
1575
|
* Close the WebSocket connection.
|
|
1561
1576
|
*/
|
|
1562
1577
|
close(): void;
|
|
1578
|
+
/**
|
|
1579
|
+
* Close the socket and wait for the close handshake to actually land.
|
|
1580
|
+
*
|
|
1581
|
+
* `close()` returns as soon as the close frame is queued. That is too early
|
|
1582
|
+
* when the caller is about to reconnect to the same workflow, because
|
|
1583
|
+
* Workhorse releases the workflow lock only once its own stop handshake with
|
|
1584
|
+
* DWS completes. Awaiting `onclose` (or the timeout) keeps the reconnect from
|
|
1585
|
+
* racing the lock and getting rejected with close code 1013.
|
|
1586
|
+
*
|
|
1587
|
+
* Best effort: never rejects, and resolves after `timeoutMs` if the server
|
|
1588
|
+
* never answers.
|
|
1589
|
+
*
|
|
1590
|
+
* @param timeoutMs - Maximum time to wait for the close event
|
|
1591
|
+
* @default WS_CLOSE_WAIT_TIMEOUT_MS
|
|
1592
|
+
*/
|
|
1593
|
+
closeAndWait(timeoutMs?: number): Promise<void>;
|
|
1563
1594
|
/**
|
|
1564
1595
|
* Check if the WebSocket is currently connected.
|
|
1565
1596
|
*/
|
package/dist/index.js
CHANGED
|
@@ -482,6 +482,10 @@ var WorkflowType = /* @__PURE__ */ ((WorkflowType2) => {
|
|
|
482
482
|
})(WorkflowType || {});
|
|
483
483
|
var WS_KEEPALIVE_PING_INTERVAL_MS = 45e3;
|
|
484
484
|
var WS_HEARTBEAT_INTERVAL_MS = 6e4;
|
|
485
|
+
var WS_CLOSE_WORKFLOW_LOCKED = 1013;
|
|
486
|
+
var WS_CLOSE_WAIT_TIMEOUT_MS = 3e3;
|
|
487
|
+
var WS_WORKFLOW_LOCK_RETRY_DELAY_MS = 1500;
|
|
488
|
+
var WS_WORKFLOW_LOCK_MAX_RETRIES = 8;
|
|
485
489
|
var DEFAULT_WORKFLOW_DEFINITION = "chat" /* CHAT */;
|
|
486
490
|
var DEFAULT_CLIENT_CAPABILITIES = ["shell_command"];
|
|
487
491
|
var CLIENT_VERSION = "8.51.0";
|
|
@@ -929,11 +933,17 @@ var GitLabAnthropicLanguageModel = class {
|
|
|
929
933
|
*
|
|
930
934
|
* Cache breakpoints (`cache_control: { type: "ephemeral" }`) are placed on:
|
|
931
935
|
* 1. The system prompt content block — static across all turns.
|
|
932
|
-
* 2. The last content block of
|
|
933
|
-
* between conversation history and the current turn.
|
|
936
|
+
* 2. The last content block of each of the final two messages.
|
|
934
937
|
*
|
|
935
|
-
*
|
|
936
|
-
*
|
|
938
|
+
* Two trailing breakpoints (rather than a single one on the penultimate
|
|
939
|
+
* message) keep a cache write within Anthropic's 20-block lookback window
|
|
940
|
+
* as an agentic conversation grows several messages per turn (assistant
|
|
941
|
+
* tool-call → tool-result → …). With a single breakpoint the most recent
|
|
942
|
+
* write can drift more than 20 blocks behind the current position, so the
|
|
943
|
+
* next request fails to prefix-match and pays for a fresh cache write
|
|
944
|
+
* instead of a cheap read. The extra breakpoint costs nothing (breakpoints
|
|
945
|
+
* themselves are free; you only pay for tokens actually written/read) and
|
|
946
|
+
* materially raises the cache hit rate in multi-turn tool-using sessions.
|
|
937
947
|
*
|
|
938
948
|
* @see https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
|
|
939
949
|
*/
|
|
@@ -1022,10 +1032,11 @@ ${message.content}` : message.content;
|
|
|
1022
1032
|
cache_control: { type: "ephemeral" }
|
|
1023
1033
|
}
|
|
1024
1034
|
] : void 0;
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1035
|
+
const breakpointCount = Math.min(2, messages.length);
|
|
1036
|
+
for (let i = messages.length - breakpointCount; i < messages.length; i++) {
|
|
1037
|
+
const message = messages[i];
|
|
1038
|
+
if (Array.isArray(message.content) && message.content.length > 0) {
|
|
1039
|
+
const lastBlock = message.content[message.content.length - 1];
|
|
1029
1040
|
lastBlock.cache_control = {
|
|
1030
1041
|
type: "ephemeral"
|
|
1031
1042
|
};
|
|
@@ -2519,7 +2530,7 @@ var import_node_async_hooks = require("async_hooks");
|
|
|
2519
2530
|
var import_isomorphic_ws = __toESM(require("isomorphic-ws"));
|
|
2520
2531
|
|
|
2521
2532
|
// src/version.ts
|
|
2522
|
-
var VERSION = true ? "6.
|
|
2533
|
+
var VERSION = true ? "6.12.1" : "0.0.0-dev";
|
|
2523
2534
|
|
|
2524
2535
|
// src/gitlab-workflow-client.ts
|
|
2525
2536
|
var WS_CONNECT_TIMEOUT_MS = 3e4;
|
|
@@ -2669,6 +2680,48 @@ var GitLabWorkflowClient = class {
|
|
|
2669
2680
|
}
|
|
2670
2681
|
}
|
|
2671
2682
|
}
|
|
2683
|
+
/**
|
|
2684
|
+
* Close the socket and wait for the close handshake to actually land.
|
|
2685
|
+
*
|
|
2686
|
+
* `close()` returns as soon as the close frame is queued. That is too early
|
|
2687
|
+
* when the caller is about to reconnect to the same workflow, because
|
|
2688
|
+
* Workhorse releases the workflow lock only once its own stop handshake with
|
|
2689
|
+
* DWS completes. Awaiting `onclose` (or the timeout) keeps the reconnect from
|
|
2690
|
+
* racing the lock and getting rejected with close code 1013.
|
|
2691
|
+
*
|
|
2692
|
+
* Best effort: never rejects, and resolves after `timeoutMs` if the server
|
|
2693
|
+
* never answers.
|
|
2694
|
+
*
|
|
2695
|
+
* @param timeoutMs - Maximum time to wait for the close event
|
|
2696
|
+
* @default WS_CLOSE_WAIT_TIMEOUT_MS
|
|
2697
|
+
*/
|
|
2698
|
+
closeAndWait(timeoutMs = WS_CLOSE_WAIT_TIMEOUT_MS) {
|
|
2699
|
+
const sock = this.socket;
|
|
2700
|
+
const pending = sock && (sock.readyState === import_isomorphic_ws.default.OPEN || sock.readyState === import_isomorphic_ws.default.CONNECTING);
|
|
2701
|
+
if (this.closed || !pending) {
|
|
2702
|
+
this.close();
|
|
2703
|
+
return Promise.resolve();
|
|
2704
|
+
}
|
|
2705
|
+
return new Promise((resolve3) => {
|
|
2706
|
+
let settled = false;
|
|
2707
|
+
const finish = () => {
|
|
2708
|
+
if (settled) return;
|
|
2709
|
+
settled = true;
|
|
2710
|
+
clearTimeout(timer);
|
|
2711
|
+
resolve3();
|
|
2712
|
+
};
|
|
2713
|
+
const timer = setTimeout(finish, timeoutMs);
|
|
2714
|
+
const previous = sock.onclose;
|
|
2715
|
+
sock.onclose = (event) => {
|
|
2716
|
+
try {
|
|
2717
|
+
previous?.call(sock, event);
|
|
2718
|
+
} finally {
|
|
2719
|
+
finish();
|
|
2720
|
+
}
|
|
2721
|
+
};
|
|
2722
|
+
this.close();
|
|
2723
|
+
});
|
|
2724
|
+
}
|
|
2672
2725
|
/**
|
|
2673
2726
|
* Check if the WebSocket is currently connected.
|
|
2674
2727
|
*/
|
|
@@ -4628,19 +4681,20 @@ var GitLabWorkflowLanguageModel = class _GitLabWorkflowLanguageModel {
|
|
|
4628
4681
|
}
|
|
4629
4682
|
ss.approvalPending = false;
|
|
4630
4683
|
ss.deferredClose = null;
|
|
4631
|
-
this.cleanupClient(ss, false);
|
|
4632
4684
|
const approval = decision.approved ? { approval: { tool_name: tools[0]?.name, tool_args_json: tools[0]?.args } } : { rejection: { message: decision.message ?? "User rejected" } };
|
|
4633
4685
|
const newStartReq = {
|
|
4634
4686
|
...startReq,
|
|
4635
4687
|
approval,
|
|
4636
4688
|
preapproved_tools: decision.approved ? [...startReq.preapproved_tools ?? [], ...tools.map((t) => t.name)] : startReq.preapproved_tools ?? []
|
|
4637
4689
|
};
|
|
4638
|
-
const
|
|
4639
|
-
|
|
4640
|
-
|
|
4641
|
-
|
|
4642
|
-
|
|
4643
|
-
|
|
4690
|
+
const reconnect = async () => {
|
|
4691
|
+
if (ss.streamClosed) return;
|
|
4692
|
+
await this.closeClientAndWait(ss);
|
|
4693
|
+
const client = new GitLabWorkflowClient();
|
|
4694
|
+
this.activeClients.add(client);
|
|
4695
|
+
ss.activeClient = client;
|
|
4696
|
+
const modelRef = await this.resolveModelRef();
|
|
4697
|
+
await client.connect(
|
|
4644
4698
|
{
|
|
4645
4699
|
instanceUrl: this.config.instanceUrl,
|
|
4646
4700
|
modelRef,
|
|
@@ -4651,23 +4705,63 @@ var GitLabWorkflowLanguageModel = class _GitLabWorkflowLanguageModel {
|
|
|
4651
4705
|
aiCatalogItemVersionId: wsExtras?.aiCatalogItemVersionId,
|
|
4652
4706
|
workflowDefinition: wsExtras?.workflowDefinition
|
|
4653
4707
|
},
|
|
4654
|
-
(event) =>
|
|
4708
|
+
(event) => {
|
|
4709
|
+
if (event.type === "closed" && event.code === WS_CLOSE_WORKFLOW_LOCKED) {
|
|
4710
|
+
void retryWhileLocked();
|
|
4711
|
+
return;
|
|
4712
|
+
}
|
|
4713
|
+
this.handleWorkflowEvent(
|
|
4714
|
+
ss,
|
|
4715
|
+
event,
|
|
4716
|
+
controller,
|
|
4717
|
+
client,
|
|
4718
|
+
toolExecutor,
|
|
4719
|
+
nextTextId,
|
|
4720
|
+
availableToolNames,
|
|
4721
|
+
newStartReq,
|
|
4722
|
+
wsExtras
|
|
4723
|
+
);
|
|
4724
|
+
}
|
|
4725
|
+
);
|
|
4726
|
+
client.sendStartRequest(newStartReq);
|
|
4727
|
+
};
|
|
4728
|
+
let lockRetries = 0;
|
|
4729
|
+
const retryWhileLocked = async () => {
|
|
4730
|
+
if (ss.streamClosed) return;
|
|
4731
|
+
if (++lockRetries > WS_WORKFLOW_LOCK_MAX_RETRIES) {
|
|
4732
|
+
this.failStream(
|
|
4655
4733
|
ss,
|
|
4656
|
-
event,
|
|
4657
4734
|
controller,
|
|
4658
|
-
|
|
4659
|
-
|
|
4660
|
-
|
|
4661
|
-
|
|
4662
|
-
|
|
4663
|
-
|
|
4664
|
-
|
|
4665
|
-
);
|
|
4666
|
-
|
|
4667
|
-
}
|
|
4668
|
-
|
|
4669
|
-
|
|
4670
|
-
|
|
4735
|
+
new GitLabError({
|
|
4736
|
+
message: "Failed to acquire lock on workflow after reconnect retries",
|
|
4737
|
+
statusCode: WS_CLOSE_WORKFLOW_LOCKED
|
|
4738
|
+
})
|
|
4739
|
+
);
|
|
4740
|
+
return;
|
|
4741
|
+
}
|
|
4742
|
+
await new Promise((r) => setTimeout(r, WS_WORKFLOW_LOCK_RETRY_DELAY_MS * lockRetries));
|
|
4743
|
+
await reconnect().catch((err) => this.failStream(ss, controller, err));
|
|
4744
|
+
};
|
|
4745
|
+
await reconnect().catch((err) => this.failStream(ss, controller, err));
|
|
4746
|
+
}
|
|
4747
|
+
/**
|
|
4748
|
+
* Close the active client and wait for its close handshake, so a following
|
|
4749
|
+
* reconnect to the same workflow does not race the Workhorse lock.
|
|
4750
|
+
*/
|
|
4751
|
+
async closeClientAndWait(ss) {
|
|
4752
|
+
const client = ss.activeClient;
|
|
4753
|
+
if (!client) return;
|
|
4754
|
+
this.activeClients.delete(client);
|
|
4755
|
+
ss.activeClient = null;
|
|
4756
|
+
await client.closeAndWait();
|
|
4757
|
+
}
|
|
4758
|
+
/**
|
|
4759
|
+
* Surface a terminal error on the stream and drop the workflow.
|
|
4760
|
+
*/
|
|
4761
|
+
failStream(ss, controller, error) {
|
|
4762
|
+
this.cleanupClient(ss, true);
|
|
4763
|
+
if (ss.streamClosed) return;
|
|
4764
|
+
controller.error(error);
|
|
4671
4765
|
}
|
|
4672
4766
|
// ---------------------------------------------------------------------------
|
|
4673
4767
|
// Workflow metadata
|