gitlab-ai-provider 6.13.0 → 6.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -970,7 +970,12 @@ ${message.content}` : message.content;
970
970
  } else if (message.role === "assistant") {
971
971
  const content = [];
972
972
  for (const part of message.content) {
973
- if (part.type === "text") {
973
+ if (part.type === "reasoning") {
974
+ const thinkingBlock = this.convertReasoningPart(part);
975
+ if (thinkingBlock) {
976
+ content.push(thinkingBlock);
977
+ }
978
+ } else if (part.type === "text") {
974
979
  content.push({ type: "text", text: part.text });
975
980
  } else if (part.type === "tool-call") {
976
981
  let toolInput = part.input;
@@ -1085,17 +1090,97 @@ ${message.content}` : message.content;
1085
1090
  raw: params?.raw
1086
1091
  };
1087
1092
  }
1093
+ /**
1094
+ * Translates the camelCase GitLab thinking config into the Anthropic Messages
1095
+ * request fields. `adaptive` is required by effort-based models such as Claude
1096
+ * Opus 4.8 and pairs `thinking: { type: 'adaptive' }` with a sibling
1097
+ * `output_config: { effort }`; `enabled`/`disabled` map through directly.
1098
+ */
1099
+ buildThinkingParams(config) {
1100
+ if (!config) return {};
1101
+ if (config.type === "adaptive") {
1102
+ return {
1103
+ // Newer adaptive-only models (Claude Sonnet 5+, Opus 4.7+) default
1104
+ // `thinking.display` to `'omitted'`, which returns *empty* thinking
1105
+ // blocks — the model still reasons and effort still affects
1106
+ // behavior/token usage, but no visible chain-of-thought text comes
1107
+ // back. Older models already default to `'summarized'`, so it's
1108
+ // safe to always request it explicitly rather than gating on model
1109
+ // version.
1110
+ thinking: { type: "adaptive", display: "summarized" },
1111
+ output_config: { effort: config.effort }
1112
+ };
1113
+ }
1114
+ if (config.type === "enabled") {
1115
+ return { thinking: { type: "enabled", budget_tokens: config.budgetTokens } };
1116
+ }
1117
+ return { thinking: { type: "disabled" } };
1118
+ }
1119
+ /** Per-call thinking configuration takes precedence over the model-level default. */
1120
+ resolveThinkingConfig(options) {
1121
+ const gitlabOptions = options.providerOptions?.["gitlab"];
1122
+ return gitlabOptions?.thinking ?? this.config.thinking;
1123
+ }
1124
+ validateThinkingConfig(config, maxTokens) {
1125
+ if (config?.type === "enabled" && config.budgetTokens >= maxTokens) {
1126
+ throw new GitLabError({
1127
+ message: `Anthropic thinking budgetTokens (${config.budgetTokens}) must be less than maxTokens (${maxTokens}).`,
1128
+ statusCode: 400
1129
+ });
1130
+ }
1131
+ }
1132
+ /**
1133
+ * Rebuild an Anthropic thinking block from a prior-turn reasoning part.
1134
+ * Anthropic requires thinking blocks to be replayed unmodified (with their
1135
+ * signature) during tool-use loops; the signature / redacted payload are
1136
+ * carried in the reasoning part's provider metadata under `gitlab`.
1137
+ */
1138
+ convertReasoningPart(part) {
1139
+ const meta = part.providerOptions?.["gitlab"];
1140
+ const redactedData = meta?.["redactedData"];
1141
+ if (typeof redactedData === "string") {
1142
+ return { type: "redacted_thinking", data: redactedData };
1143
+ }
1144
+ const signature = meta?.["signature"];
1145
+ if (typeof signature !== "string") {
1146
+ return void 0;
1147
+ }
1148
+ return { type: "thinking", thinking: part.text, signature };
1149
+ }
1150
+ isThinkingActive(config) {
1151
+ return config?.type === "enabled" || config?.type === "adaptive";
1152
+ }
1153
+ /**
1154
+ * Anthropic rejects `temperature` changes while thinking is active and requires
1155
+ * `top_p` within [0.95, 1]; drop those params when thinking is engaged so the
1156
+ * gateway does not 400. Undefined values are omitted either way.
1157
+ */
1158
+ buildSamplingParams(options, thinkingConfig) {
1159
+ const params = {};
1160
+ const thinkingActive = this.isThinkingActive(thinkingConfig);
1161
+ if (options.temperature != null && !thinkingActive) {
1162
+ params.temperature = options.temperature;
1163
+ }
1164
+ if (options.topP != null) {
1165
+ if (!thinkingActive || options.topP >= 0.95 && options.topP <= 1) {
1166
+ params.top_p = options.topP;
1167
+ }
1168
+ }
1169
+ return params;
1170
+ }
1088
1171
  async doGenerate(options) {
1089
1172
  return this.doGenerateWithRetry(options, false);
1090
1173
  }
1091
1174
  async doGenerateWithRetry(options, isRetry) {
1092
- const client = await this.getAnthropicClient(isRetry);
1093
1175
  const { system, messages } = this.convertPrompt(options.prompt);
1094
1176
  const toolsDisabled = options.toolChoice?.type === "none";
1095
1177
  const tools = toolsDisabled ? void 0 : this.convertTools(options.tools);
1096
1178
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1097
1179
  const anthropicModel = this.config.anthropicModel || "claude-sonnet-4-5-20250929";
1098
1180
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
1181
+ const thinkingConfig = this.resolveThinkingConfig(options);
1182
+ this.validateThinkingConfig(thinkingConfig, maxTokens);
1183
+ const client = await this.getAnthropicClient(isRetry);
1099
1184
  const generateParams = {
1100
1185
  model: anthropicModel,
1101
1186
  max_tokens: maxTokens,
@@ -1103,15 +1188,27 @@ ${message.content}` : message.content;
1103
1188
  messages,
1104
1189
  tools,
1105
1190
  tool_choice: tools ? toolChoice : void 0,
1106
- temperature: options.temperature,
1107
- top_p: options.topP,
1108
- stop_sequences: options.stopSequences
1191
+ ...this.buildSamplingParams(options, thinkingConfig),
1192
+ stop_sequences: options.stopSequences,
1193
+ ...this.buildThinkingParams(thinkingConfig)
1109
1194
  };
1110
1195
  try {
1111
1196
  const response = options.abortSignal ? await client.messages.create(generateParams, { signal: options.abortSignal }) : await client.messages.create(generateParams);
1112
1197
  const content = [];
1113
1198
  for (const block of response.content) {
1114
- if (block.type === "text") {
1199
+ if (block.type === "thinking") {
1200
+ content.push({
1201
+ type: "reasoning",
1202
+ text: block.thinking,
1203
+ providerMetadata: { gitlab: { signature: block.signature } }
1204
+ });
1205
+ } else if (block.type === "redacted_thinking") {
1206
+ content.push({
1207
+ type: "reasoning",
1208
+ text: "",
1209
+ providerMetadata: { gitlab: { redactedData: block.data } }
1210
+ });
1211
+ } else if (block.type === "text") {
1115
1212
  content.push({
1116
1213
  type: "text",
1117
1214
  text: block.text
@@ -1165,13 +1262,15 @@ ${message.content}` : message.content;
1165
1262
  return this.doStreamWithRetry(options, false);
1166
1263
  }
1167
1264
  async doStreamWithRetry(options, isRetry) {
1168
- const client = await this.getAnthropicClient(isRetry);
1169
1265
  const { system, messages } = this.convertPrompt(options.prompt);
1170
1266
  const toolsDisabled = options.toolChoice?.type === "none";
1171
1267
  const tools = toolsDisabled ? void 0 : this.convertTools(options.tools);
1172
1268
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1173
1269
  const anthropicModel = this.config.anthropicModel || "claude-sonnet-4-5-20250929";
1174
1270
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
1271
+ const thinkingConfig = this.resolveThinkingConfig(options);
1272
+ this.validateThinkingConfig(thinkingConfig, maxTokens);
1273
+ const client = await this.getAnthropicClient(isRetry);
1175
1274
  const requestBody = {
1176
1275
  model: anthropicModel,
1177
1276
  max_tokens: maxTokens,
@@ -1179,9 +1278,9 @@ ${message.content}` : message.content;
1179
1278
  messages,
1180
1279
  tools,
1181
1280
  tool_choice: tools ? toolChoice : void 0,
1182
- temperature: options.temperature,
1183
- top_p: options.topP,
1281
+ ...this.buildSamplingParams(options, thinkingConfig),
1184
1282
  stop_sequences: options.stopSequences,
1283
+ ...this.buildThinkingParams(thinkingConfig),
1185
1284
  stream: true
1186
1285
  };
1187
1286
  const self = this;
@@ -1280,6 +1379,24 @@ ${message.content}` : message.content;
1280
1379
  type: "text-start",
1281
1380
  id: textId
1282
1381
  });
1382
+ } else if (event.content_block.type === "thinking") {
1383
+ const reasoningId = `reasoning-${event.index}`;
1384
+ contentBlocks[event.index] = { type: "reasoning", id: reasoningId };
1385
+ safeEnqueue({
1386
+ type: "reasoning-start",
1387
+ id: reasoningId
1388
+ });
1389
+ } else if (event.content_block.type === "redacted_thinking") {
1390
+ const reasoningId = `reasoning-${event.index}`;
1391
+ contentBlocks[event.index] = {
1392
+ type: "reasoning",
1393
+ id: reasoningId,
1394
+ redactedData: event.content_block.data
1395
+ };
1396
+ safeEnqueue({
1397
+ type: "reasoning-start",
1398
+ id: reasoningId
1399
+ });
1283
1400
  } else if (event.content_block.type === "tool_use") {
1284
1401
  contentBlocks[event.index] = {
1285
1402
  type: "tool-call",
@@ -1302,6 +1419,14 @@ ${message.content}` : message.content;
1302
1419
  id: block.id,
1303
1420
  delta: event.delta.text
1304
1421
  });
1422
+ } else if (event.delta.type === "thinking_delta" && block?.type === "reasoning") {
1423
+ safeEnqueue({
1424
+ type: "reasoning-delta",
1425
+ id: block.id,
1426
+ delta: event.delta.thinking
1427
+ });
1428
+ } else if (event.delta.type === "signature_delta" && block?.type === "reasoning") {
1429
+ block.signature = event.delta.signature;
1305
1430
  } else if (event.delta.type === "input_json_delta" && block?.type === "tool-call") {
1306
1431
  block.input += event.delta.partial_json;
1307
1432
  safeEnqueue({
@@ -1319,6 +1444,13 @@ ${message.content}` : message.content;
1319
1444
  type: "text-end",
1320
1445
  id: block.id
1321
1446
  });
1447
+ } else if (block?.type === "reasoning") {
1448
+ const gitlabMeta = block.redactedData != null ? { redactedData: block.redactedData } : block.signature != null ? { signature: block.signature } : void 0;
1449
+ safeEnqueue({
1450
+ type: "reasoning-end",
1451
+ id: block.id,
1452
+ ...gitlabMeta && { providerMetadata: { gitlab: gitlabMeta } }
1453
+ });
1322
1454
  } else if (block?.type === "tool-call") {
1323
1455
  safeEnqueue({
1324
1456
  type: "tool-input-end",
@@ -1671,6 +1803,17 @@ var GitLabOpenAILanguageModel = class {
1671
1803
  }
1672
1804
  return parts.join(" | ");
1673
1805
  }
1806
+ /**
1807
+ * Per-call `providerOptions.gitlab.reasoningEffort` takes precedence over the
1808
+ * model-level default.
1809
+ */
1810
+ resolveReasoningEffort(options) {
1811
+ const gitlabOptions = options.providerOptions?.["gitlab"];
1812
+ return gitlabOptions?.reasoningEffort ?? this.config.reasoningEffort;
1813
+ }
1814
+ asOpenAIReasoningEffort(effort) {
1815
+ return effort;
1816
+ }
1674
1817
  convertTools(tools) {
1675
1818
  if (!tools || tools.length === 0) {
1676
1819
  return void 0;
@@ -1945,6 +2088,7 @@ var GitLabOpenAILanguageModel = class {
1945
2088
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1946
2089
  const openaiModel = this.config.openaiModel || "gpt-4o";
1947
2090
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2091
+ const reasoningEffort = this.resolveReasoningEffort(options);
1948
2092
  const generateParams = {
1949
2093
  model: openaiModel,
1950
2094
  max_completion_tokens: maxTokens,
@@ -1953,7 +2097,10 @@ var GitLabOpenAILanguageModel = class {
1953
2097
  tool_choice: tools ? toolChoice : void 0,
1954
2098
  temperature: options.temperature,
1955
2099
  top_p: options.topP,
1956
- stop: options.stopSequences
2100
+ stop: options.stopSequences,
2101
+ ...reasoningEffort && {
2102
+ reasoning_effort: this.asOpenAIReasoningEffort(reasoningEffort)
2103
+ }
1957
2104
  };
1958
2105
  try {
1959
2106
  const response = options.abortSignal ? await client.chat.completions.create(generateParams, { signal: options.abortSignal }) : await client.chat.completions.create(generateParams);
@@ -2016,6 +2163,7 @@ var GitLabOpenAILanguageModel = class {
2016
2163
  const instructions = this.extractSystemInstructions(options.prompt);
2017
2164
  const openaiModel = this.config.openaiModel || "gpt-5-codex";
2018
2165
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2166
+ const reasoningEffort = this.resolveReasoningEffort(options);
2019
2167
  const generateParams = {
2020
2168
  model: openaiModel,
2021
2169
  input,
@@ -2024,7 +2172,18 @@ var GitLabOpenAILanguageModel = class {
2024
2172
  max_output_tokens: maxTokens,
2025
2173
  temperature: options.temperature,
2026
2174
  top_p: options.topP,
2027
- store: false
2175
+ store: false,
2176
+ ...reasoningEffort && {
2177
+ // `summary: 'auto'` opts into visible reasoning summary text.
2178
+ // Without it, the Responses API never returns summary content —
2179
+ // the model still reasons and effort still affects behavior, but
2180
+ // no chain-of-thought text comes back (mirrors the Anthropic
2181
+ // `thinking.display` gap fixed in buildThinkingParams).
2182
+ reasoning: {
2183
+ effort: this.asOpenAIReasoningEffort(reasoningEffort),
2184
+ summary: "auto"
2185
+ }
2186
+ }
2028
2187
  };
2029
2188
  try {
2030
2189
  const response = options.abortSignal ? await client.responses.create(generateParams, { signal: options.abortSignal }) : await client.responses.create(generateParams);
@@ -2045,6 +2204,12 @@ var GitLabOpenAILanguageModel = class {
2045
2204
  toolName: item.name,
2046
2205
  input: item.arguments
2047
2206
  });
2207
+ } else if (item.type === "reasoning") {
2208
+ for (const summary of item.summary || []) {
2209
+ if (summary.type === "summary_text" && summary.text) {
2210
+ content.push({ type: "reasoning", text: summary.text });
2211
+ }
2212
+ }
2048
2213
  }
2049
2214
  }
2050
2215
  const usage = this.createUsage({
@@ -2097,6 +2262,7 @@ var GitLabOpenAILanguageModel = class {
2097
2262
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
2098
2263
  const openaiModel = this.config.openaiModel || "gpt-4o";
2099
2264
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2265
+ const reasoningEffort = this.resolveReasoningEffort(options);
2100
2266
  const requestBody = {
2101
2267
  model: openaiModel,
2102
2268
  max_completion_tokens: maxTokens,
@@ -2106,6 +2272,9 @@ var GitLabOpenAILanguageModel = class {
2106
2272
  temperature: options.temperature,
2107
2273
  top_p: options.topP,
2108
2274
  stop: options.stopSequences,
2275
+ ...reasoningEffort && {
2276
+ reasoning_effort: this.asOpenAIReasoningEffort(reasoningEffort)
2277
+ },
2109
2278
  stream: true,
2110
2279
  stream_options: { include_usage: true }
2111
2280
  };
@@ -2313,6 +2482,7 @@ var GitLabOpenAILanguageModel = class {
2313
2482
  const instructions = this.extractSystemInstructions(options.prompt);
2314
2483
  const openaiModel = this.config.openaiModel || "gpt-5-codex";
2315
2484
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2485
+ const reasoningEffort = this.resolveReasoningEffort(options);
2316
2486
  const requestBody = {
2317
2487
  model: openaiModel,
2318
2488
  input,
@@ -2322,6 +2492,17 @@ var GitLabOpenAILanguageModel = class {
2322
2492
  temperature: options.temperature,
2323
2493
  top_p: options.topP,
2324
2494
  store: false,
2495
+ ...reasoningEffort && {
2496
+ // `summary: 'auto'` opts into visible reasoning summary text.
2497
+ // Without it, the Responses API never returns summary content —
2498
+ // the model still reasons and effort still affects behavior, but
2499
+ // no chain-of-thought text comes back (mirrors the Anthropic
2500
+ // `thinking.display` gap fixed in buildThinkingParams).
2501
+ reasoning: {
2502
+ effort: this.asOpenAIReasoningEffort(reasoningEffort),
2503
+ summary: "auto"
2504
+ }
2505
+ },
2325
2506
  stream: true
2326
2507
  };
2327
2508
  const self = this;
@@ -2360,11 +2541,38 @@ var GitLabOpenAILanguageModel = class {
2360
2541
  }
2361
2542
  };
2362
2543
  const toolCalls = {};
2544
+ const activeReasoning = /* @__PURE__ */ new Set();
2545
+ const knownReasoning = /* @__PURE__ */ new Set();
2546
+ const reasoningWithText = /* @__PURE__ */ new Set();
2363
2547
  let usage = self.createUsage();
2364
2548
  let finishReason = { unified: "other", raw: void 0 };
2365
2549
  let textStarted = false;
2366
2550
  let contentEmitted = false;
2367
2551
  const textId = "text-0";
2552
+ const reasoningId = (outputIndex, summaryIndex) => `reasoning-${outputIndex}-${summaryIndex}`;
2553
+ const startReasoning = (outputIndex, summaryIndex) => {
2554
+ const id = reasoningId(outputIndex, summaryIndex);
2555
+ if (!knownReasoning.has(id)) {
2556
+ knownReasoning.add(id);
2557
+ activeReasoning.add(id);
2558
+ contentEmitted = true;
2559
+ safeEnqueue({ type: "reasoning-start", id });
2560
+ }
2561
+ return id;
2562
+ };
2563
+ const emitReasoningDelta = (outputIndex, summaryIndex, delta) => {
2564
+ const id = startReasoning(outputIndex, summaryIndex);
2565
+ if (activeReasoning.has(id) && delta) {
2566
+ reasoningWithText.add(id);
2567
+ safeEnqueue({ type: "reasoning-delta", id, delta });
2568
+ }
2569
+ };
2570
+ const endReasoning = (outputIndex, summaryIndex) => {
2571
+ const id = reasoningId(outputIndex, summaryIndex);
2572
+ if (activeReasoning.delete(id)) {
2573
+ safeEnqueue({ type: "reasoning-end", id });
2574
+ }
2575
+ };
2368
2576
  try {
2369
2577
  const openaiStream = await client.responses.create(
2370
2578
  {
@@ -2397,6 +2605,30 @@ var GitLabOpenAILanguageModel = class {
2397
2605
  toolName: event.item.name
2398
2606
  });
2399
2607
  }
2608
+ } else if (event.type === "response.reasoning_summary_part.added") {
2609
+ startReasoning(event.output_index, event.summary_index);
2610
+ } else if (event.type === "response.reasoning_summary_text.delta") {
2611
+ emitReasoningDelta(event.output_index, event.summary_index, event.delta);
2612
+ } else if (event.type === "response.reasoning_summary_text.done") {
2613
+ const id = startReasoning(event.output_index, event.summary_index);
2614
+ if (!reasoningWithText.has(id)) {
2615
+ emitReasoningDelta(event.output_index, event.summary_index, event.text);
2616
+ }
2617
+ endReasoning(event.output_index, event.summary_index);
2618
+ } else if (event.type === "response.reasoning_summary_part.done") {
2619
+ const id = startReasoning(event.output_index, event.summary_index);
2620
+ if (!reasoningWithText.has(id)) {
2621
+ emitReasoningDelta(event.output_index, event.summary_index, event.part.text);
2622
+ }
2623
+ endReasoning(event.output_index, event.summary_index);
2624
+ } else if (event.type === "response.output_item.done" && event.item.type === "reasoning") {
2625
+ const prefix = `reasoning-${event.output_index}-`;
2626
+ for (const id of [...activeReasoning]) {
2627
+ if (id.startsWith(prefix)) {
2628
+ activeReasoning.delete(id);
2629
+ safeEnqueue({ type: "reasoning-end", id });
2630
+ }
2631
+ }
2400
2632
  } else if (event.type === "response.output_text.delta") {
2401
2633
  contentEmitted = true;
2402
2634
  if (!textStarted) {
@@ -2439,6 +2671,10 @@ var GitLabOpenAILanguageModel = class {
2439
2671
  }
2440
2672
  }
2441
2673
  }
2674
+ for (const id of activeReasoning) {
2675
+ safeEnqueue({ type: "reasoning-end", id });
2676
+ }
2677
+ activeReasoning.clear();
2442
2678
  if (textStarted) {
2443
2679
  safeEnqueue({ type: "text-end", id: textId });
2444
2680
  }
@@ -2531,7 +2767,7 @@ var import_node_async_hooks = require("async_hooks");
2531
2767
  var import_isomorphic_ws = __toESM(require("isomorphic-ws"));
2532
2768
 
2533
2769
  // src/version.ts
2534
- var VERSION = true ? "6.12.2" : "0.0.0-dev";
2770
+ var VERSION = true ? "6.13.0" : "0.0.0-dev";
2535
2771
 
2536
2772
  // src/gitlab-workflow-client.ts
2537
2773
  var WS_CONNECT_TIMEOUT_MS = 3e4;
@@ -5378,12 +5614,14 @@ function createGitLab(options = {}) {
5378
5614
  if (mapping.provider === "openai") {
5379
5615
  return new GitLabOpenAILanguageModel(modelId, {
5380
5616
  ...baseConfig,
5381
- openaiModel: agenticOptions?.providerModel ?? mapping.model
5617
+ openaiModel: agenticOptions?.providerModel ?? mapping.model,
5618
+ reasoningEffort: agenticOptions?.reasoningEffort
5382
5619
  });
5383
5620
  }
5384
5621
  return new GitLabAnthropicLanguageModel(modelId, {
5385
5622
  ...baseConfig,
5386
- anthropicModel: agenticOptions?.providerModel ?? mapping.model
5623
+ anthropicModel: agenticOptions?.providerModel ?? mapping.model,
5624
+ thinking: agenticOptions?.thinking
5387
5625
  });
5388
5626
  };
5389
5627
  const createWorkflowChatModel = (modelId, workflowOptions) => {