gitlab-ai-provider 6.12.2 → 6.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -970,7 +970,12 @@ ${message.content}` : message.content;
970
970
  } else if (message.role === "assistant") {
971
971
  const content = [];
972
972
  for (const part of message.content) {
973
- if (part.type === "text") {
973
+ if (part.type === "reasoning") {
974
+ const thinkingBlock = this.convertReasoningPart(part);
975
+ if (thinkingBlock) {
976
+ content.push(thinkingBlock);
977
+ }
978
+ } else if (part.type === "text") {
974
979
  content.push({ type: "text", text: part.text });
975
980
  } else if (part.type === "tool-call") {
976
981
  let toolInput = part.input;
@@ -1085,17 +1090,97 @@ ${message.content}` : message.content;
1085
1090
  raw: params?.raw
1086
1091
  };
1087
1092
  }
1093
+ /**
1094
+ * Translates the camelCase GitLab thinking config into the Anthropic Messages
1095
+ * request fields. `adaptive` is required by effort-based models such as Claude
1096
+ * Opus 4.8 and pairs `thinking: { type: 'adaptive' }` with a sibling
1097
+ * `output_config: { effort }`; `enabled`/`disabled` map through directly.
1098
+ */
1099
+ buildThinkingParams(config) {
1100
+ if (!config) return {};
1101
+ if (config.type === "adaptive") {
1102
+ return {
1103
+ // Newer adaptive-only models (Claude Sonnet 5+, Opus 4.7+) default
1104
+ // `thinking.display` to `'omitted'`, which returns *empty* thinking
1105
+ // blocks — the model still reasons and effort still affects
1106
+ // behavior/token usage, but no visible chain-of-thought text comes
1107
+ // back. Older models already default to `'summarized'`, so it's
1108
+ // safe to always request it explicitly rather than gating on model
1109
+ // version.
1110
+ thinking: { type: "adaptive", display: "summarized" },
1111
+ output_config: { effort: config.effort }
1112
+ };
1113
+ }
1114
+ if (config.type === "enabled") {
1115
+ return { thinking: { type: "enabled", budget_tokens: config.budgetTokens } };
1116
+ }
1117
+ return { thinking: { type: "disabled" } };
1118
+ }
1119
+ /** Per-call thinking configuration takes precedence over the model-level default. */
1120
+ resolveThinkingConfig(options) {
1121
+ const gitlabOptions = options.providerOptions?.["gitlab"];
1122
+ return gitlabOptions?.thinking ?? this.config.thinking;
1123
+ }
1124
+ validateThinkingConfig(config, maxTokens) {
1125
+ if (config?.type === "enabled" && config.budgetTokens >= maxTokens) {
1126
+ throw new GitLabError({
1127
+ message: `Anthropic thinking budgetTokens (${config.budgetTokens}) must be less than maxTokens (${maxTokens}).`,
1128
+ statusCode: 400
1129
+ });
1130
+ }
1131
+ }
1132
+ /**
1133
+ * Rebuild an Anthropic thinking block from a prior-turn reasoning part.
1134
+ * Anthropic requires thinking blocks to be replayed unmodified (with their
1135
+ * signature) during tool-use loops; the signature / redacted payload are
1136
+ * carried in the reasoning part's provider metadata under `gitlab`.
1137
+ */
1138
+ convertReasoningPart(part) {
1139
+ const meta = part.providerOptions?.["gitlab"];
1140
+ const redactedData = meta?.["redactedData"];
1141
+ if (typeof redactedData === "string") {
1142
+ return { type: "redacted_thinking", data: redactedData };
1143
+ }
1144
+ const signature = meta?.["signature"];
1145
+ if (typeof signature !== "string") {
1146
+ return void 0;
1147
+ }
1148
+ return { type: "thinking", thinking: part.text, signature };
1149
+ }
1150
+ isThinkingActive(config) {
1151
+ return config?.type === "enabled" || config?.type === "adaptive";
1152
+ }
1153
+ /**
1154
+ * Anthropic rejects `temperature` changes while thinking is active and requires
1155
+ * `top_p` within [0.95, 1]; drop those params when thinking is engaged so the
1156
+ * gateway does not 400. Undefined values are omitted either way.
1157
+ */
1158
+ buildSamplingParams(options, thinkingConfig) {
1159
+ const params = {};
1160
+ const thinkingActive = this.isThinkingActive(thinkingConfig);
1161
+ if (options.temperature != null && !thinkingActive) {
1162
+ params.temperature = options.temperature;
1163
+ }
1164
+ if (options.topP != null) {
1165
+ if (!thinkingActive || options.topP >= 0.95 && options.topP <= 1) {
1166
+ params.top_p = options.topP;
1167
+ }
1168
+ }
1169
+ return params;
1170
+ }
1088
1171
  async doGenerate(options) {
1089
1172
  return this.doGenerateWithRetry(options, false);
1090
1173
  }
1091
1174
  async doGenerateWithRetry(options, isRetry) {
1092
- const client = await this.getAnthropicClient(isRetry);
1093
1175
  const { system, messages } = this.convertPrompt(options.prompt);
1094
1176
  const toolsDisabled = options.toolChoice?.type === "none";
1095
1177
  const tools = toolsDisabled ? void 0 : this.convertTools(options.tools);
1096
1178
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1097
1179
  const anthropicModel = this.config.anthropicModel || "claude-sonnet-4-5-20250929";
1098
1180
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
1181
+ const thinkingConfig = this.resolveThinkingConfig(options);
1182
+ this.validateThinkingConfig(thinkingConfig, maxTokens);
1183
+ const client = await this.getAnthropicClient(isRetry);
1099
1184
  const generateParams = {
1100
1185
  model: anthropicModel,
1101
1186
  max_tokens: maxTokens,
@@ -1103,15 +1188,27 @@ ${message.content}` : message.content;
1103
1188
  messages,
1104
1189
  tools,
1105
1190
  tool_choice: tools ? toolChoice : void 0,
1106
- temperature: options.temperature,
1107
- top_p: options.topP,
1108
- stop_sequences: options.stopSequences
1191
+ ...this.buildSamplingParams(options, thinkingConfig),
1192
+ stop_sequences: options.stopSequences,
1193
+ ...this.buildThinkingParams(thinkingConfig)
1109
1194
  };
1110
1195
  try {
1111
1196
  const response = options.abortSignal ? await client.messages.create(generateParams, { signal: options.abortSignal }) : await client.messages.create(generateParams);
1112
1197
  const content = [];
1113
1198
  for (const block of response.content) {
1114
- if (block.type === "text") {
1199
+ if (block.type === "thinking") {
1200
+ content.push({
1201
+ type: "reasoning",
1202
+ text: block.thinking,
1203
+ providerMetadata: { gitlab: { signature: block.signature } }
1204
+ });
1205
+ } else if (block.type === "redacted_thinking") {
1206
+ content.push({
1207
+ type: "reasoning",
1208
+ text: "",
1209
+ providerMetadata: { gitlab: { redactedData: block.data } }
1210
+ });
1211
+ } else if (block.type === "text") {
1115
1212
  content.push({
1116
1213
  type: "text",
1117
1214
  text: block.text
@@ -1165,13 +1262,15 @@ ${message.content}` : message.content;
1165
1262
  return this.doStreamWithRetry(options, false);
1166
1263
  }
1167
1264
  async doStreamWithRetry(options, isRetry) {
1168
- const client = await this.getAnthropicClient(isRetry);
1169
1265
  const { system, messages } = this.convertPrompt(options.prompt);
1170
1266
  const toolsDisabled = options.toolChoice?.type === "none";
1171
1267
  const tools = toolsDisabled ? void 0 : this.convertTools(options.tools);
1172
1268
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1173
1269
  const anthropicModel = this.config.anthropicModel || "claude-sonnet-4-5-20250929";
1174
1270
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
1271
+ const thinkingConfig = this.resolveThinkingConfig(options);
1272
+ this.validateThinkingConfig(thinkingConfig, maxTokens);
1273
+ const client = await this.getAnthropicClient(isRetry);
1175
1274
  const requestBody = {
1176
1275
  model: anthropicModel,
1177
1276
  max_tokens: maxTokens,
@@ -1179,9 +1278,9 @@ ${message.content}` : message.content;
1179
1278
  messages,
1180
1279
  tools,
1181
1280
  tool_choice: tools ? toolChoice : void 0,
1182
- temperature: options.temperature,
1183
- top_p: options.topP,
1281
+ ...this.buildSamplingParams(options, thinkingConfig),
1184
1282
  stop_sequences: options.stopSequences,
1283
+ ...this.buildThinkingParams(thinkingConfig),
1185
1284
  stream: true
1186
1285
  };
1187
1286
  const self = this;
@@ -1280,6 +1379,24 @@ ${message.content}` : message.content;
1280
1379
  type: "text-start",
1281
1380
  id: textId
1282
1381
  });
1382
+ } else if (event.content_block.type === "thinking") {
1383
+ const reasoningId = `reasoning-${event.index}`;
1384
+ contentBlocks[event.index] = { type: "reasoning", id: reasoningId };
1385
+ safeEnqueue({
1386
+ type: "reasoning-start",
1387
+ id: reasoningId
1388
+ });
1389
+ } else if (event.content_block.type === "redacted_thinking") {
1390
+ const reasoningId = `reasoning-${event.index}`;
1391
+ contentBlocks[event.index] = {
1392
+ type: "reasoning",
1393
+ id: reasoningId,
1394
+ redactedData: event.content_block.data
1395
+ };
1396
+ safeEnqueue({
1397
+ type: "reasoning-start",
1398
+ id: reasoningId
1399
+ });
1283
1400
  } else if (event.content_block.type === "tool_use") {
1284
1401
  contentBlocks[event.index] = {
1285
1402
  type: "tool-call",
@@ -1302,6 +1419,14 @@ ${message.content}` : message.content;
1302
1419
  id: block.id,
1303
1420
  delta: event.delta.text
1304
1421
  });
1422
+ } else if (event.delta.type === "thinking_delta" && block?.type === "reasoning") {
1423
+ safeEnqueue({
1424
+ type: "reasoning-delta",
1425
+ id: block.id,
1426
+ delta: event.delta.thinking
1427
+ });
1428
+ } else if (event.delta.type === "signature_delta" && block?.type === "reasoning") {
1429
+ block.signature = event.delta.signature;
1305
1430
  } else if (event.delta.type === "input_json_delta" && block?.type === "tool-call") {
1306
1431
  block.input += event.delta.partial_json;
1307
1432
  safeEnqueue({
@@ -1319,6 +1444,13 @@ ${message.content}` : message.content;
1319
1444
  type: "text-end",
1320
1445
  id: block.id
1321
1446
  });
1447
+ } else if (block?.type === "reasoning") {
1448
+ const gitlabMeta = block.redactedData != null ? { redactedData: block.redactedData } : block.signature != null ? { signature: block.signature } : void 0;
1449
+ safeEnqueue({
1450
+ type: "reasoning-end",
1451
+ id: block.id,
1452
+ ...gitlabMeta && { providerMetadata: { gitlab: gitlabMeta } }
1453
+ });
1322
1454
  } else if (block?.type === "tool-call") {
1323
1455
  safeEnqueue({
1324
1456
  type: "tool-input-end",
@@ -1468,6 +1600,7 @@ var import_openai = __toESM(require("openai"));
1468
1600
  // src/model-mappings.ts
1469
1601
  var MODEL_MAPPINGS = {
1470
1602
  // Anthropic models
1603
+ "duo-chat-fable-5-1": { provider: "anthropic", model: "claude-fable-5-1" },
1471
1604
  "duo-chat-fable-5": { provider: "anthropic", model: "claude-fable-5" },
1472
1605
  "duo-chat-opus-5": { provider: "anthropic", model: "claude-opus-5" },
1473
1606
  "duo-chat-opus-4-8": { provider: "anthropic", model: "claude-opus-4-8" },
@@ -1670,6 +1803,17 @@ var GitLabOpenAILanguageModel = class {
1670
1803
  }
1671
1804
  return parts.join(" | ");
1672
1805
  }
1806
+ /**
1807
+ * Per-call `providerOptions.gitlab.reasoningEffort` takes precedence over the
1808
+ * model-level default.
1809
+ */
1810
+ resolveReasoningEffort(options) {
1811
+ const gitlabOptions = options.providerOptions?.["gitlab"];
1812
+ return gitlabOptions?.reasoningEffort ?? this.config.reasoningEffort;
1813
+ }
1814
+ asOpenAIReasoningEffort(effort) {
1815
+ return effort;
1816
+ }
1673
1817
  convertTools(tools) {
1674
1818
  if (!tools || tools.length === 0) {
1675
1819
  return void 0;
@@ -1944,6 +2088,7 @@ var GitLabOpenAILanguageModel = class {
1944
2088
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1945
2089
  const openaiModel = this.config.openaiModel || "gpt-4o";
1946
2090
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2091
+ const reasoningEffort = this.resolveReasoningEffort(options);
1947
2092
  const generateParams = {
1948
2093
  model: openaiModel,
1949
2094
  max_completion_tokens: maxTokens,
@@ -1952,7 +2097,10 @@ var GitLabOpenAILanguageModel = class {
1952
2097
  tool_choice: tools ? toolChoice : void 0,
1953
2098
  temperature: options.temperature,
1954
2099
  top_p: options.topP,
1955
- stop: options.stopSequences
2100
+ stop: options.stopSequences,
2101
+ ...reasoningEffort && {
2102
+ reasoning_effort: this.asOpenAIReasoningEffort(reasoningEffort)
2103
+ }
1956
2104
  };
1957
2105
  try {
1958
2106
  const response = options.abortSignal ? await client.chat.completions.create(generateParams, { signal: options.abortSignal }) : await client.chat.completions.create(generateParams);
@@ -2015,6 +2163,7 @@ var GitLabOpenAILanguageModel = class {
2015
2163
  const instructions = this.extractSystemInstructions(options.prompt);
2016
2164
  const openaiModel = this.config.openaiModel || "gpt-5-codex";
2017
2165
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2166
+ const reasoningEffort = this.resolveReasoningEffort(options);
2018
2167
  const generateParams = {
2019
2168
  model: openaiModel,
2020
2169
  input,
@@ -2023,7 +2172,18 @@ var GitLabOpenAILanguageModel = class {
2023
2172
  max_output_tokens: maxTokens,
2024
2173
  temperature: options.temperature,
2025
2174
  top_p: options.topP,
2026
- store: false
2175
+ store: false,
2176
+ ...reasoningEffort && {
2177
+ // `summary: 'auto'` opts into visible reasoning summary text.
2178
+ // Without it, the Responses API never returns summary content —
2179
+ // the model still reasons and effort still affects behavior, but
2180
+ // no chain-of-thought text comes back (mirrors the Anthropic
2181
+ // `thinking.display` gap fixed in buildThinkingParams).
2182
+ reasoning: {
2183
+ effort: this.asOpenAIReasoningEffort(reasoningEffort),
2184
+ summary: "auto"
2185
+ }
2186
+ }
2027
2187
  };
2028
2188
  try {
2029
2189
  const response = options.abortSignal ? await client.responses.create(generateParams, { signal: options.abortSignal }) : await client.responses.create(generateParams);
@@ -2044,6 +2204,12 @@ var GitLabOpenAILanguageModel = class {
2044
2204
  toolName: item.name,
2045
2205
  input: item.arguments
2046
2206
  });
2207
+ } else if (item.type === "reasoning") {
2208
+ for (const summary of item.summary || []) {
2209
+ if (summary.type === "summary_text" && summary.text) {
2210
+ content.push({ type: "reasoning", text: summary.text });
2211
+ }
2212
+ }
2047
2213
  }
2048
2214
  }
2049
2215
  const usage = this.createUsage({
@@ -2096,6 +2262,7 @@ var GitLabOpenAILanguageModel = class {
2096
2262
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
2097
2263
  const openaiModel = this.config.openaiModel || "gpt-4o";
2098
2264
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2265
+ const reasoningEffort = this.resolveReasoningEffort(options);
2099
2266
  const requestBody = {
2100
2267
  model: openaiModel,
2101
2268
  max_completion_tokens: maxTokens,
@@ -2105,6 +2272,9 @@ var GitLabOpenAILanguageModel = class {
2105
2272
  temperature: options.temperature,
2106
2273
  top_p: options.topP,
2107
2274
  stop: options.stopSequences,
2275
+ ...reasoningEffort && {
2276
+ reasoning_effort: this.asOpenAIReasoningEffort(reasoningEffort)
2277
+ },
2108
2278
  stream: true,
2109
2279
  stream_options: { include_usage: true }
2110
2280
  };
@@ -2312,6 +2482,7 @@ var GitLabOpenAILanguageModel = class {
2312
2482
  const instructions = this.extractSystemInstructions(options.prompt);
2313
2483
  const openaiModel = this.config.openaiModel || "gpt-5-codex";
2314
2484
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2485
+ const reasoningEffort = this.resolveReasoningEffort(options);
2315
2486
  const requestBody = {
2316
2487
  model: openaiModel,
2317
2488
  input,
@@ -2321,6 +2492,17 @@ var GitLabOpenAILanguageModel = class {
2321
2492
  temperature: options.temperature,
2322
2493
  top_p: options.topP,
2323
2494
  store: false,
2495
+ ...reasoningEffort && {
2496
+ // `summary: 'auto'` opts into visible reasoning summary text.
2497
+ // Without it, the Responses API never returns summary content —
2498
+ // the model still reasons and effort still affects behavior, but
2499
+ // no chain-of-thought text comes back (mirrors the Anthropic
2500
+ // `thinking.display` gap fixed in buildThinkingParams).
2501
+ reasoning: {
2502
+ effort: this.asOpenAIReasoningEffort(reasoningEffort),
2503
+ summary: "auto"
2504
+ }
2505
+ },
2324
2506
  stream: true
2325
2507
  };
2326
2508
  const self = this;
@@ -2359,11 +2541,38 @@ var GitLabOpenAILanguageModel = class {
2359
2541
  }
2360
2542
  };
2361
2543
  const toolCalls = {};
2544
+ const activeReasoning = /* @__PURE__ */ new Set();
2545
+ const knownReasoning = /* @__PURE__ */ new Set();
2546
+ const reasoningWithText = /* @__PURE__ */ new Set();
2362
2547
  let usage = self.createUsage();
2363
2548
  let finishReason = { unified: "other", raw: void 0 };
2364
2549
  let textStarted = false;
2365
2550
  let contentEmitted = false;
2366
2551
  const textId = "text-0";
2552
+ const reasoningId = (outputIndex, summaryIndex) => `reasoning-${outputIndex}-${summaryIndex}`;
2553
+ const startReasoning = (outputIndex, summaryIndex) => {
2554
+ const id = reasoningId(outputIndex, summaryIndex);
2555
+ if (!knownReasoning.has(id)) {
2556
+ knownReasoning.add(id);
2557
+ activeReasoning.add(id);
2558
+ contentEmitted = true;
2559
+ safeEnqueue({ type: "reasoning-start", id });
2560
+ }
2561
+ return id;
2562
+ };
2563
+ const emitReasoningDelta = (outputIndex, summaryIndex, delta) => {
2564
+ const id = startReasoning(outputIndex, summaryIndex);
2565
+ if (activeReasoning.has(id) && delta) {
2566
+ reasoningWithText.add(id);
2567
+ safeEnqueue({ type: "reasoning-delta", id, delta });
2568
+ }
2569
+ };
2570
+ const endReasoning = (outputIndex, summaryIndex) => {
2571
+ const id = reasoningId(outputIndex, summaryIndex);
2572
+ if (activeReasoning.delete(id)) {
2573
+ safeEnqueue({ type: "reasoning-end", id });
2574
+ }
2575
+ };
2367
2576
  try {
2368
2577
  const openaiStream = await client.responses.create(
2369
2578
  {
@@ -2396,6 +2605,30 @@ var GitLabOpenAILanguageModel = class {
2396
2605
  toolName: event.item.name
2397
2606
  });
2398
2607
  }
2608
+ } else if (event.type === "response.reasoning_summary_part.added") {
2609
+ startReasoning(event.output_index, event.summary_index);
2610
+ } else if (event.type === "response.reasoning_summary_text.delta") {
2611
+ emitReasoningDelta(event.output_index, event.summary_index, event.delta);
2612
+ } else if (event.type === "response.reasoning_summary_text.done") {
2613
+ const id = startReasoning(event.output_index, event.summary_index);
2614
+ if (!reasoningWithText.has(id)) {
2615
+ emitReasoningDelta(event.output_index, event.summary_index, event.text);
2616
+ }
2617
+ endReasoning(event.output_index, event.summary_index);
2618
+ } else if (event.type === "response.reasoning_summary_part.done") {
2619
+ const id = startReasoning(event.output_index, event.summary_index);
2620
+ if (!reasoningWithText.has(id)) {
2621
+ emitReasoningDelta(event.output_index, event.summary_index, event.part.text);
2622
+ }
2623
+ endReasoning(event.output_index, event.summary_index);
2624
+ } else if (event.type === "response.output_item.done" && event.item.type === "reasoning") {
2625
+ const prefix = `reasoning-${event.output_index}-`;
2626
+ for (const id of [...activeReasoning]) {
2627
+ if (id.startsWith(prefix)) {
2628
+ activeReasoning.delete(id);
2629
+ safeEnqueue({ type: "reasoning-end", id });
2630
+ }
2631
+ }
2399
2632
  } else if (event.type === "response.output_text.delta") {
2400
2633
  contentEmitted = true;
2401
2634
  if (!textStarted) {
@@ -2438,6 +2671,10 @@ var GitLabOpenAILanguageModel = class {
2438
2671
  }
2439
2672
  }
2440
2673
  }
2674
+ for (const id of activeReasoning) {
2675
+ safeEnqueue({ type: "reasoning-end", id });
2676
+ }
2677
+ activeReasoning.clear();
2441
2678
  if (textStarted) {
2442
2679
  safeEnqueue({ type: "text-end", id: textId });
2443
2680
  }
@@ -2530,7 +2767,7 @@ var import_node_async_hooks = require("async_hooks");
2530
2767
  var import_isomorphic_ws = __toESM(require("isomorphic-ws"));
2531
2768
 
2532
2769
  // src/version.ts
2533
- var VERSION = true ? "6.12.1" : "0.0.0-dev";
2770
+ var VERSION = true ? "6.13.0" : "0.0.0-dev";
2534
2771
 
2535
2772
  // src/gitlab-workflow-client.ts
2536
2773
  var WS_CONNECT_TIMEOUT_MS = 3e4;
@@ -5377,12 +5614,14 @@ function createGitLab(options = {}) {
5377
5614
  if (mapping.provider === "openai") {
5378
5615
  return new GitLabOpenAILanguageModel(modelId, {
5379
5616
  ...baseConfig,
5380
- openaiModel: agenticOptions?.providerModel ?? mapping.model
5617
+ openaiModel: agenticOptions?.providerModel ?? mapping.model,
5618
+ reasoningEffort: agenticOptions?.reasoningEffort
5381
5619
  });
5382
5620
  }
5383
5621
  return new GitLabAnthropicLanguageModel(modelId, {
5384
5622
  ...baseConfig,
5385
- anthropicModel: agenticOptions?.providerModel ?? mapping.model
5623
+ anthropicModel: agenticOptions?.providerModel ?? mapping.model,
5624
+ thinking: agenticOptions?.thinking
5386
5625
  });
5387
5626
  };
5388
5627
  const createWorkflowChatModel = (modelId, workflowOptions) => {