gitlab-ai-provider 6.12.2 → 6.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -891,7 +891,12 @@ ${message.content}` : message.content;
891
891
  } else if (message.role === "assistant") {
892
892
  const content = [];
893
893
  for (const part of message.content) {
894
- if (part.type === "text") {
894
+ if (part.type === "reasoning") {
895
+ const thinkingBlock = this.convertReasoningPart(part);
896
+ if (thinkingBlock) {
897
+ content.push(thinkingBlock);
898
+ }
899
+ } else if (part.type === "text") {
895
900
  content.push({ type: "text", text: part.text });
896
901
  } else if (part.type === "tool-call") {
897
902
  let toolInput = part.input;
@@ -1006,17 +1011,97 @@ ${message.content}` : message.content;
1006
1011
  raw: params?.raw
1007
1012
  };
1008
1013
  }
1014
+ /**
1015
+ * Translates the camelCase GitLab thinking config into the Anthropic Messages
1016
+ * request fields. `adaptive` is required by effort-based models such as Claude
1017
+ * Opus 4.8 and pairs `thinking: { type: 'adaptive' }` with a sibling
1018
+ * `output_config: { effort }`; `enabled`/`disabled` map through directly.
1019
+ */
1020
+ buildThinkingParams(config) {
1021
+ if (!config) return {};
1022
+ if (config.type === "adaptive") {
1023
+ return {
1024
+ // Newer adaptive-only models (Claude Sonnet 5+, Opus 4.7+) default
1025
+ // `thinking.display` to `'omitted'`, which returns *empty* thinking
1026
+ // blocks — the model still reasons and effort still affects
1027
+ // behavior/token usage, but no visible chain-of-thought text comes
1028
+ // back. Older models already default to `'summarized'`, so it's
1029
+ // safe to always request it explicitly rather than gating on model
1030
+ // version.
1031
+ thinking: { type: "adaptive", display: "summarized" },
1032
+ output_config: { effort: config.effort }
1033
+ };
1034
+ }
1035
+ if (config.type === "enabled") {
1036
+ return { thinking: { type: "enabled", budget_tokens: config.budgetTokens } };
1037
+ }
1038
+ return { thinking: { type: "disabled" } };
1039
+ }
1040
+ /** Per-call thinking configuration takes precedence over the model-level default. */
1041
+ resolveThinkingConfig(options) {
1042
+ const gitlabOptions = options.providerOptions?.["gitlab"];
1043
+ return gitlabOptions?.thinking ?? this.config.thinking;
1044
+ }
1045
+ validateThinkingConfig(config, maxTokens) {
1046
+ if (config?.type === "enabled" && config.budgetTokens >= maxTokens) {
1047
+ throw new GitLabError({
1048
+ message: `Anthropic thinking budgetTokens (${config.budgetTokens}) must be less than maxTokens (${maxTokens}).`,
1049
+ statusCode: 400
1050
+ });
1051
+ }
1052
+ }
1053
+ /**
1054
+ * Rebuild an Anthropic thinking block from a prior-turn reasoning part.
1055
+ * Anthropic requires thinking blocks to be replayed unmodified (with their
1056
+ * signature) during tool-use loops; the signature / redacted payload are
1057
+ * carried in the reasoning part's provider metadata under `gitlab`.
1058
+ */
1059
+ convertReasoningPart(part) {
1060
+ const meta = part.providerOptions?.["gitlab"];
1061
+ const redactedData = meta?.["redactedData"];
1062
+ if (typeof redactedData === "string") {
1063
+ return { type: "redacted_thinking", data: redactedData };
1064
+ }
1065
+ const signature = meta?.["signature"];
1066
+ if (typeof signature !== "string") {
1067
+ return void 0;
1068
+ }
1069
+ return { type: "thinking", thinking: part.text, signature };
1070
+ }
1071
+ isThinkingActive(config) {
1072
+ return config?.type === "enabled" || config?.type === "adaptive";
1073
+ }
1074
+ /**
1075
+ * Anthropic rejects `temperature` changes while thinking is active and requires
1076
+ * `top_p` within [0.95, 1]; drop those params when thinking is engaged so the
1077
+ * gateway does not 400. Undefined values are omitted either way.
1078
+ */
1079
+ buildSamplingParams(options, thinkingConfig) {
1080
+ const params = {};
1081
+ const thinkingActive = this.isThinkingActive(thinkingConfig);
1082
+ if (options.temperature != null && !thinkingActive) {
1083
+ params.temperature = options.temperature;
1084
+ }
1085
+ if (options.topP != null) {
1086
+ if (!thinkingActive || options.topP >= 0.95 && options.topP <= 1) {
1087
+ params.top_p = options.topP;
1088
+ }
1089
+ }
1090
+ return params;
1091
+ }
1009
1092
  async doGenerate(options) {
1010
1093
  return this.doGenerateWithRetry(options, false);
1011
1094
  }
1012
1095
  async doGenerateWithRetry(options, isRetry) {
1013
- const client = await this.getAnthropicClient(isRetry);
1014
1096
  const { system, messages } = this.convertPrompt(options.prompt);
1015
1097
  const toolsDisabled = options.toolChoice?.type === "none";
1016
1098
  const tools = toolsDisabled ? void 0 : this.convertTools(options.tools);
1017
1099
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1018
1100
  const anthropicModel = this.config.anthropicModel || "claude-sonnet-4-5-20250929";
1019
1101
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
1102
+ const thinkingConfig = this.resolveThinkingConfig(options);
1103
+ this.validateThinkingConfig(thinkingConfig, maxTokens);
1104
+ const client = await this.getAnthropicClient(isRetry);
1020
1105
  const generateParams = {
1021
1106
  model: anthropicModel,
1022
1107
  max_tokens: maxTokens,
@@ -1024,15 +1109,27 @@ ${message.content}` : message.content;
1024
1109
  messages,
1025
1110
  tools,
1026
1111
  tool_choice: tools ? toolChoice : void 0,
1027
- temperature: options.temperature,
1028
- top_p: options.topP,
1029
- stop_sequences: options.stopSequences
1112
+ ...this.buildSamplingParams(options, thinkingConfig),
1113
+ stop_sequences: options.stopSequences,
1114
+ ...this.buildThinkingParams(thinkingConfig)
1030
1115
  };
1031
1116
  try {
1032
1117
  const response = options.abortSignal ? await client.messages.create(generateParams, { signal: options.abortSignal }) : await client.messages.create(generateParams);
1033
1118
  const content = [];
1034
1119
  for (const block of response.content) {
1035
- if (block.type === "text") {
1120
+ if (block.type === "thinking") {
1121
+ content.push({
1122
+ type: "reasoning",
1123
+ text: block.thinking,
1124
+ providerMetadata: { gitlab: { signature: block.signature } }
1125
+ });
1126
+ } else if (block.type === "redacted_thinking") {
1127
+ content.push({
1128
+ type: "reasoning",
1129
+ text: "",
1130
+ providerMetadata: { gitlab: { redactedData: block.data } }
1131
+ });
1132
+ } else if (block.type === "text") {
1036
1133
  content.push({
1037
1134
  type: "text",
1038
1135
  text: block.text
@@ -1086,13 +1183,15 @@ ${message.content}` : message.content;
1086
1183
  return this.doStreamWithRetry(options, false);
1087
1184
  }
1088
1185
  async doStreamWithRetry(options, isRetry) {
1089
- const client = await this.getAnthropicClient(isRetry);
1090
1186
  const { system, messages } = this.convertPrompt(options.prompt);
1091
1187
  const toolsDisabled = options.toolChoice?.type === "none";
1092
1188
  const tools = toolsDisabled ? void 0 : this.convertTools(options.tools);
1093
1189
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1094
1190
  const anthropicModel = this.config.anthropicModel || "claude-sonnet-4-5-20250929";
1095
1191
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
1192
+ const thinkingConfig = this.resolveThinkingConfig(options);
1193
+ this.validateThinkingConfig(thinkingConfig, maxTokens);
1194
+ const client = await this.getAnthropicClient(isRetry);
1096
1195
  const requestBody = {
1097
1196
  model: anthropicModel,
1098
1197
  max_tokens: maxTokens,
@@ -1100,9 +1199,9 @@ ${message.content}` : message.content;
1100
1199
  messages,
1101
1200
  tools,
1102
1201
  tool_choice: tools ? toolChoice : void 0,
1103
- temperature: options.temperature,
1104
- top_p: options.topP,
1202
+ ...this.buildSamplingParams(options, thinkingConfig),
1105
1203
  stop_sequences: options.stopSequences,
1204
+ ...this.buildThinkingParams(thinkingConfig),
1106
1205
  stream: true
1107
1206
  };
1108
1207
  const self = this;
@@ -1201,6 +1300,24 @@ ${message.content}` : message.content;
1201
1300
  type: "text-start",
1202
1301
  id: textId
1203
1302
  });
1303
+ } else if (event.content_block.type === "thinking") {
1304
+ const reasoningId = `reasoning-${event.index}`;
1305
+ contentBlocks[event.index] = { type: "reasoning", id: reasoningId };
1306
+ safeEnqueue({
1307
+ type: "reasoning-start",
1308
+ id: reasoningId
1309
+ });
1310
+ } else if (event.content_block.type === "redacted_thinking") {
1311
+ const reasoningId = `reasoning-${event.index}`;
1312
+ contentBlocks[event.index] = {
1313
+ type: "reasoning",
1314
+ id: reasoningId,
1315
+ redactedData: event.content_block.data
1316
+ };
1317
+ safeEnqueue({
1318
+ type: "reasoning-start",
1319
+ id: reasoningId
1320
+ });
1204
1321
  } else if (event.content_block.type === "tool_use") {
1205
1322
  contentBlocks[event.index] = {
1206
1323
  type: "tool-call",
@@ -1223,6 +1340,14 @@ ${message.content}` : message.content;
1223
1340
  id: block.id,
1224
1341
  delta: event.delta.text
1225
1342
  });
1343
+ } else if (event.delta.type === "thinking_delta" && block?.type === "reasoning") {
1344
+ safeEnqueue({
1345
+ type: "reasoning-delta",
1346
+ id: block.id,
1347
+ delta: event.delta.thinking
1348
+ });
1349
+ } else if (event.delta.type === "signature_delta" && block?.type === "reasoning") {
1350
+ block.signature = event.delta.signature;
1226
1351
  } else if (event.delta.type === "input_json_delta" && block?.type === "tool-call") {
1227
1352
  block.input += event.delta.partial_json;
1228
1353
  safeEnqueue({
@@ -1240,6 +1365,13 @@ ${message.content}` : message.content;
1240
1365
  type: "text-end",
1241
1366
  id: block.id
1242
1367
  });
1368
+ } else if (block?.type === "reasoning") {
1369
+ const gitlabMeta = block.redactedData != null ? { redactedData: block.redactedData } : block.signature != null ? { signature: block.signature } : void 0;
1370
+ safeEnqueue({
1371
+ type: "reasoning-end",
1372
+ id: block.id,
1373
+ ...gitlabMeta && { providerMetadata: { gitlab: gitlabMeta } }
1374
+ });
1243
1375
  } else if (block?.type === "tool-call") {
1244
1376
  safeEnqueue({
1245
1377
  type: "tool-input-end",
@@ -1389,6 +1521,7 @@ import OpenAI from "openai";
1389
1521
  // src/model-mappings.ts
1390
1522
  var MODEL_MAPPINGS = {
1391
1523
  // Anthropic models
1524
+ "duo-chat-fable-5-1": { provider: "anthropic", model: "claude-fable-5-1" },
1392
1525
  "duo-chat-fable-5": { provider: "anthropic", model: "claude-fable-5" },
1393
1526
  "duo-chat-opus-5": { provider: "anthropic", model: "claude-opus-5" },
1394
1527
  "duo-chat-opus-4-8": { provider: "anthropic", model: "claude-opus-4-8" },
@@ -1591,6 +1724,17 @@ var GitLabOpenAILanguageModel = class {
1591
1724
  }
1592
1725
  return parts.join(" | ");
1593
1726
  }
1727
+ /**
1728
+ * Per-call `providerOptions.gitlab.reasoningEffort` takes precedence over the
1729
+ * model-level default.
1730
+ */
1731
+ resolveReasoningEffort(options) {
1732
+ const gitlabOptions = options.providerOptions?.["gitlab"];
1733
+ return gitlabOptions?.reasoningEffort ?? this.config.reasoningEffort;
1734
+ }
1735
+ asOpenAIReasoningEffort(effort) {
1736
+ return effort;
1737
+ }
1594
1738
  convertTools(tools) {
1595
1739
  if (!tools || tools.length === 0) {
1596
1740
  return void 0;
@@ -1865,6 +2009,7 @@ var GitLabOpenAILanguageModel = class {
1865
2009
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1866
2010
  const openaiModel = this.config.openaiModel || "gpt-4o";
1867
2011
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2012
+ const reasoningEffort = this.resolveReasoningEffort(options);
1868
2013
  const generateParams = {
1869
2014
  model: openaiModel,
1870
2015
  max_completion_tokens: maxTokens,
@@ -1873,7 +2018,10 @@ var GitLabOpenAILanguageModel = class {
1873
2018
  tool_choice: tools ? toolChoice : void 0,
1874
2019
  temperature: options.temperature,
1875
2020
  top_p: options.topP,
1876
- stop: options.stopSequences
2021
+ stop: options.stopSequences,
2022
+ ...reasoningEffort && {
2023
+ reasoning_effort: this.asOpenAIReasoningEffort(reasoningEffort)
2024
+ }
1877
2025
  };
1878
2026
  try {
1879
2027
  const response = options.abortSignal ? await client.chat.completions.create(generateParams, { signal: options.abortSignal }) : await client.chat.completions.create(generateParams);
@@ -1936,6 +2084,7 @@ var GitLabOpenAILanguageModel = class {
1936
2084
  const instructions = this.extractSystemInstructions(options.prompt);
1937
2085
  const openaiModel = this.config.openaiModel || "gpt-5-codex";
1938
2086
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2087
+ const reasoningEffort = this.resolveReasoningEffort(options);
1939
2088
  const generateParams = {
1940
2089
  model: openaiModel,
1941
2090
  input,
@@ -1944,7 +2093,18 @@ var GitLabOpenAILanguageModel = class {
1944
2093
  max_output_tokens: maxTokens,
1945
2094
  temperature: options.temperature,
1946
2095
  top_p: options.topP,
1947
- store: false
2096
+ store: false,
2097
+ ...reasoningEffort && {
2098
+ // `summary: 'auto'` opts into visible reasoning summary text.
2099
+ // Without it, the Responses API never returns summary content —
2100
+ // the model still reasons and effort still affects behavior, but
2101
+ // no chain-of-thought text comes back (mirrors the Anthropic
2102
+ // `thinking.display` gap fixed in buildThinkingParams).
2103
+ reasoning: {
2104
+ effort: this.asOpenAIReasoningEffort(reasoningEffort),
2105
+ summary: "auto"
2106
+ }
2107
+ }
1948
2108
  };
1949
2109
  try {
1950
2110
  const response = options.abortSignal ? await client.responses.create(generateParams, { signal: options.abortSignal }) : await client.responses.create(generateParams);
@@ -1965,6 +2125,12 @@ var GitLabOpenAILanguageModel = class {
1965
2125
  toolName: item.name,
1966
2126
  input: item.arguments
1967
2127
  });
2128
+ } else if (item.type === "reasoning") {
2129
+ for (const summary of item.summary || []) {
2130
+ if (summary.type === "summary_text" && summary.text) {
2131
+ content.push({ type: "reasoning", text: summary.text });
2132
+ }
2133
+ }
1968
2134
  }
1969
2135
  }
1970
2136
  const usage = this.createUsage({
@@ -2017,6 +2183,7 @@ var GitLabOpenAILanguageModel = class {
2017
2183
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
2018
2184
  const openaiModel = this.config.openaiModel || "gpt-4o";
2019
2185
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2186
+ const reasoningEffort = this.resolveReasoningEffort(options);
2020
2187
  const requestBody = {
2021
2188
  model: openaiModel,
2022
2189
  max_completion_tokens: maxTokens,
@@ -2026,6 +2193,9 @@ var GitLabOpenAILanguageModel = class {
2026
2193
  temperature: options.temperature,
2027
2194
  top_p: options.topP,
2028
2195
  stop: options.stopSequences,
2196
+ ...reasoningEffort && {
2197
+ reasoning_effort: this.asOpenAIReasoningEffort(reasoningEffort)
2198
+ },
2029
2199
  stream: true,
2030
2200
  stream_options: { include_usage: true }
2031
2201
  };
@@ -2233,6 +2403,7 @@ var GitLabOpenAILanguageModel = class {
2233
2403
  const instructions = this.extractSystemInstructions(options.prompt);
2234
2404
  const openaiModel = this.config.openaiModel || "gpt-5-codex";
2235
2405
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2406
+ const reasoningEffort = this.resolveReasoningEffort(options);
2236
2407
  const requestBody = {
2237
2408
  model: openaiModel,
2238
2409
  input,
@@ -2242,6 +2413,17 @@ var GitLabOpenAILanguageModel = class {
2242
2413
  temperature: options.temperature,
2243
2414
  top_p: options.topP,
2244
2415
  store: false,
2416
+ ...reasoningEffort && {
2417
+ // `summary: 'auto'` opts into visible reasoning summary text.
2418
+ // Without it, the Responses API never returns summary content —
2419
+ // the model still reasons and effort still affects behavior, but
2420
+ // no chain-of-thought text comes back (mirrors the Anthropic
2421
+ // `thinking.display` gap fixed in buildThinkingParams).
2422
+ reasoning: {
2423
+ effort: this.asOpenAIReasoningEffort(reasoningEffort),
2424
+ summary: "auto"
2425
+ }
2426
+ },
2245
2427
  stream: true
2246
2428
  };
2247
2429
  const self = this;
@@ -2280,11 +2462,38 @@ var GitLabOpenAILanguageModel = class {
2280
2462
  }
2281
2463
  };
2282
2464
  const toolCalls = {};
2465
+ const activeReasoning = /* @__PURE__ */ new Set();
2466
+ const knownReasoning = /* @__PURE__ */ new Set();
2467
+ const reasoningWithText = /* @__PURE__ */ new Set();
2283
2468
  let usage = self.createUsage();
2284
2469
  let finishReason = { unified: "other", raw: void 0 };
2285
2470
  let textStarted = false;
2286
2471
  let contentEmitted = false;
2287
2472
  const textId = "text-0";
2473
+ const reasoningId = (outputIndex, summaryIndex) => `reasoning-${outputIndex}-${summaryIndex}`;
2474
+ const startReasoning = (outputIndex, summaryIndex) => {
2475
+ const id = reasoningId(outputIndex, summaryIndex);
2476
+ if (!knownReasoning.has(id)) {
2477
+ knownReasoning.add(id);
2478
+ activeReasoning.add(id);
2479
+ contentEmitted = true;
2480
+ safeEnqueue({ type: "reasoning-start", id });
2481
+ }
2482
+ return id;
2483
+ };
2484
+ const emitReasoningDelta = (outputIndex, summaryIndex, delta) => {
2485
+ const id = startReasoning(outputIndex, summaryIndex);
2486
+ if (activeReasoning.has(id) && delta) {
2487
+ reasoningWithText.add(id);
2488
+ safeEnqueue({ type: "reasoning-delta", id, delta });
2489
+ }
2490
+ };
2491
+ const endReasoning = (outputIndex, summaryIndex) => {
2492
+ const id = reasoningId(outputIndex, summaryIndex);
2493
+ if (activeReasoning.delete(id)) {
2494
+ safeEnqueue({ type: "reasoning-end", id });
2495
+ }
2496
+ };
2288
2497
  try {
2289
2498
  const openaiStream = await client.responses.create(
2290
2499
  {
@@ -2317,6 +2526,30 @@ var GitLabOpenAILanguageModel = class {
2317
2526
  toolName: event.item.name
2318
2527
  });
2319
2528
  }
2529
+ } else if (event.type === "response.reasoning_summary_part.added") {
2530
+ startReasoning(event.output_index, event.summary_index);
2531
+ } else if (event.type === "response.reasoning_summary_text.delta") {
2532
+ emitReasoningDelta(event.output_index, event.summary_index, event.delta);
2533
+ } else if (event.type === "response.reasoning_summary_text.done") {
2534
+ const id = startReasoning(event.output_index, event.summary_index);
2535
+ if (!reasoningWithText.has(id)) {
2536
+ emitReasoningDelta(event.output_index, event.summary_index, event.text);
2537
+ }
2538
+ endReasoning(event.output_index, event.summary_index);
2539
+ } else if (event.type === "response.reasoning_summary_part.done") {
2540
+ const id = startReasoning(event.output_index, event.summary_index);
2541
+ if (!reasoningWithText.has(id)) {
2542
+ emitReasoningDelta(event.output_index, event.summary_index, event.part.text);
2543
+ }
2544
+ endReasoning(event.output_index, event.summary_index);
2545
+ } else if (event.type === "response.output_item.done" && event.item.type === "reasoning") {
2546
+ const prefix = `reasoning-${event.output_index}-`;
2547
+ for (const id of [...activeReasoning]) {
2548
+ if (id.startsWith(prefix)) {
2549
+ activeReasoning.delete(id);
2550
+ safeEnqueue({ type: "reasoning-end", id });
2551
+ }
2552
+ }
2320
2553
  } else if (event.type === "response.output_text.delta") {
2321
2554
  contentEmitted = true;
2322
2555
  if (!textStarted) {
@@ -2359,6 +2592,10 @@ var GitLabOpenAILanguageModel = class {
2359
2592
  }
2360
2593
  }
2361
2594
  }
2595
+ for (const id of activeReasoning) {
2596
+ safeEnqueue({ type: "reasoning-end", id });
2597
+ }
2598
+ activeReasoning.clear();
2362
2599
  if (textStarted) {
2363
2600
  safeEnqueue({ type: "text-end", id: textId });
2364
2601
  }
@@ -2451,7 +2688,7 @@ import { AsyncResource } from "async_hooks";
2451
2688
  import WebSocket from "isomorphic-ws";
2452
2689
 
2453
2690
  // src/version.ts
2454
- var VERSION = true ? "6.12.1" : "0.0.0-dev";
2691
+ var VERSION = true ? "6.13.0" : "0.0.0-dev";
2455
2692
 
2456
2693
  // src/gitlab-workflow-client.ts
2457
2694
  var WS_CONNECT_TIMEOUT_MS = 3e4;
@@ -5298,12 +5535,14 @@ function createGitLab(options = {}) {
5298
5535
  if (mapping.provider === "openai") {
5299
5536
  return new GitLabOpenAILanguageModel(modelId, {
5300
5537
  ...baseConfig,
5301
- openaiModel: agenticOptions?.providerModel ?? mapping.model
5538
+ openaiModel: agenticOptions?.providerModel ?? mapping.model,
5539
+ reasoningEffort: agenticOptions?.reasoningEffort
5302
5540
  });
5303
5541
  }
5304
5542
  return new GitLabAnthropicLanguageModel(modelId, {
5305
5543
  ...baseConfig,
5306
- anthropicModel: agenticOptions?.providerModel ?? mapping.model
5544
+ anthropicModel: agenticOptions?.providerModel ?? mapping.model,
5545
+ thinking: agenticOptions?.thinking
5307
5546
  });
5308
5547
  };
5309
5548
  const createWorkflowChatModel = (modelId, workflowOptions) => {