gitlab-ai-provider 6.13.0 → 6.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -891,7 +891,12 @@ ${message.content}` : message.content;
891
891
  } else if (message.role === "assistant") {
892
892
  const content = [];
893
893
  for (const part of message.content) {
894
- if (part.type === "text") {
894
+ if (part.type === "reasoning") {
895
+ const thinkingBlock = this.convertReasoningPart(part);
896
+ if (thinkingBlock) {
897
+ content.push(thinkingBlock);
898
+ }
899
+ } else if (part.type === "text") {
895
900
  content.push({ type: "text", text: part.text });
896
901
  } else if (part.type === "tool-call") {
897
902
  let toolInput = part.input;
@@ -1006,17 +1011,97 @@ ${message.content}` : message.content;
1006
1011
  raw: params?.raw
1007
1012
  };
1008
1013
  }
1014
+ /**
1015
+ * Translates the camelCase GitLab thinking config into the Anthropic Messages
1016
+ * request fields. `adaptive` is required by effort-based models such as Claude
1017
+ * Opus 4.8 and pairs `thinking: { type: 'adaptive' }` with a sibling
1018
+ * `output_config: { effort }`; `enabled`/`disabled` map through directly.
1019
+ */
1020
+ buildThinkingParams(config) {
1021
+ if (!config) return {};
1022
+ if (config.type === "adaptive") {
1023
+ return {
1024
+ // Newer adaptive-only models (Claude Sonnet 5+, Opus 4.7+) default
1025
+ // `thinking.display` to `'omitted'`, which returns *empty* thinking
1026
+ // blocks — the model still reasons and effort still affects
1027
+ // behavior/token usage, but no visible chain-of-thought text comes
1028
+ // back. Older models already default to `'summarized'`, so it's
1029
+ // safe to always request it explicitly rather than gating on model
1030
+ // version.
1031
+ thinking: { type: "adaptive", display: "summarized" },
1032
+ output_config: { effort: config.effort }
1033
+ };
1034
+ }
1035
+ if (config.type === "enabled") {
1036
+ return { thinking: { type: "enabled", budget_tokens: config.budgetTokens } };
1037
+ }
1038
+ return { thinking: { type: "disabled" } };
1039
+ }
1040
+ /** Per-call thinking configuration takes precedence over the model-level default. */
1041
+ resolveThinkingConfig(options) {
1042
+ const gitlabOptions = options.providerOptions?.["gitlab"];
1043
+ return gitlabOptions?.thinking ?? this.config.thinking;
1044
+ }
1045
+ validateThinkingConfig(config, maxTokens) {
1046
+ if (config?.type === "enabled" && config.budgetTokens >= maxTokens) {
1047
+ throw new GitLabError({
1048
+ message: `Anthropic thinking budgetTokens (${config.budgetTokens}) must be less than maxTokens (${maxTokens}).`,
1049
+ statusCode: 400
1050
+ });
1051
+ }
1052
+ }
1053
+ /**
1054
+ * Rebuild an Anthropic thinking block from a prior-turn reasoning part.
1055
+ * Anthropic requires thinking blocks to be replayed unmodified (with their
1056
+ * signature) during tool-use loops; the signature / redacted payload are
1057
+ * carried in the reasoning part's provider metadata under `gitlab`.
1058
+ */
1059
+ convertReasoningPart(part) {
1060
+ const meta = part.providerOptions?.["gitlab"];
1061
+ const redactedData = meta?.["redactedData"];
1062
+ if (typeof redactedData === "string") {
1063
+ return { type: "redacted_thinking", data: redactedData };
1064
+ }
1065
+ const signature = meta?.["signature"];
1066
+ if (typeof signature !== "string") {
1067
+ return void 0;
1068
+ }
1069
+ return { type: "thinking", thinking: part.text, signature };
1070
+ }
1071
+ isThinkingActive(config) {
1072
+ return config?.type === "enabled" || config?.type === "adaptive";
1073
+ }
1074
+ /**
1075
+ * Anthropic rejects `temperature` changes while thinking is active and requires
1076
+ * `top_p` within [0.95, 1]; drop those params when thinking is engaged so the
1077
+ * gateway does not 400. Undefined values are omitted either way.
1078
+ */
1079
+ buildSamplingParams(options, thinkingConfig) {
1080
+ const params = {};
1081
+ const thinkingActive = this.isThinkingActive(thinkingConfig);
1082
+ if (options.temperature != null && !thinkingActive) {
1083
+ params.temperature = options.temperature;
1084
+ }
1085
+ if (options.topP != null) {
1086
+ if (!thinkingActive || options.topP >= 0.95 && options.topP <= 1) {
1087
+ params.top_p = options.topP;
1088
+ }
1089
+ }
1090
+ return params;
1091
+ }
1009
1092
  async doGenerate(options) {
1010
1093
  return this.doGenerateWithRetry(options, false);
1011
1094
  }
1012
1095
  async doGenerateWithRetry(options, isRetry) {
1013
- const client = await this.getAnthropicClient(isRetry);
1014
1096
  const { system, messages } = this.convertPrompt(options.prompt);
1015
1097
  const toolsDisabled = options.toolChoice?.type === "none";
1016
1098
  const tools = toolsDisabled ? void 0 : this.convertTools(options.tools);
1017
1099
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1018
1100
  const anthropicModel = this.config.anthropicModel || "claude-sonnet-4-5-20250929";
1019
1101
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
1102
+ const thinkingConfig = this.resolveThinkingConfig(options);
1103
+ this.validateThinkingConfig(thinkingConfig, maxTokens);
1104
+ const client = await this.getAnthropicClient(isRetry);
1020
1105
  const generateParams = {
1021
1106
  model: anthropicModel,
1022
1107
  max_tokens: maxTokens,
@@ -1024,15 +1109,27 @@ ${message.content}` : message.content;
1024
1109
  messages,
1025
1110
  tools,
1026
1111
  tool_choice: tools ? toolChoice : void 0,
1027
- temperature: options.temperature,
1028
- top_p: options.topP,
1029
- stop_sequences: options.stopSequences
1112
+ ...this.buildSamplingParams(options, thinkingConfig),
1113
+ stop_sequences: options.stopSequences,
1114
+ ...this.buildThinkingParams(thinkingConfig)
1030
1115
  };
1031
1116
  try {
1032
1117
  const response = options.abortSignal ? await client.messages.create(generateParams, { signal: options.abortSignal }) : await client.messages.create(generateParams);
1033
1118
  const content = [];
1034
1119
  for (const block of response.content) {
1035
- if (block.type === "text") {
1120
+ if (block.type === "thinking") {
1121
+ content.push({
1122
+ type: "reasoning",
1123
+ text: block.thinking,
1124
+ providerMetadata: { gitlab: { signature: block.signature } }
1125
+ });
1126
+ } else if (block.type === "redacted_thinking") {
1127
+ content.push({
1128
+ type: "reasoning",
1129
+ text: "",
1130
+ providerMetadata: { gitlab: { redactedData: block.data } }
1131
+ });
1132
+ } else if (block.type === "text") {
1036
1133
  content.push({
1037
1134
  type: "text",
1038
1135
  text: block.text
@@ -1086,13 +1183,15 @@ ${message.content}` : message.content;
1086
1183
  return this.doStreamWithRetry(options, false);
1087
1184
  }
1088
1185
  async doStreamWithRetry(options, isRetry) {
1089
- const client = await this.getAnthropicClient(isRetry);
1090
1186
  const { system, messages } = this.convertPrompt(options.prompt);
1091
1187
  const toolsDisabled = options.toolChoice?.type === "none";
1092
1188
  const tools = toolsDisabled ? void 0 : this.convertTools(options.tools);
1093
1189
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1094
1190
  const anthropicModel = this.config.anthropicModel || "claude-sonnet-4-5-20250929";
1095
1191
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
1192
+ const thinkingConfig = this.resolveThinkingConfig(options);
1193
+ this.validateThinkingConfig(thinkingConfig, maxTokens);
1194
+ const client = await this.getAnthropicClient(isRetry);
1096
1195
  const requestBody = {
1097
1196
  model: anthropicModel,
1098
1197
  max_tokens: maxTokens,
@@ -1100,9 +1199,9 @@ ${message.content}` : message.content;
1100
1199
  messages,
1101
1200
  tools,
1102
1201
  tool_choice: tools ? toolChoice : void 0,
1103
- temperature: options.temperature,
1104
- top_p: options.topP,
1202
+ ...this.buildSamplingParams(options, thinkingConfig),
1105
1203
  stop_sequences: options.stopSequences,
1204
+ ...this.buildThinkingParams(thinkingConfig),
1106
1205
  stream: true
1107
1206
  };
1108
1207
  const self = this;
@@ -1201,6 +1300,24 @@ ${message.content}` : message.content;
1201
1300
  type: "text-start",
1202
1301
  id: textId
1203
1302
  });
1303
+ } else if (event.content_block.type === "thinking") {
1304
+ const reasoningId = `reasoning-${event.index}`;
1305
+ contentBlocks[event.index] = { type: "reasoning", id: reasoningId };
1306
+ safeEnqueue({
1307
+ type: "reasoning-start",
1308
+ id: reasoningId
1309
+ });
1310
+ } else if (event.content_block.type === "redacted_thinking") {
1311
+ const reasoningId = `reasoning-${event.index}`;
1312
+ contentBlocks[event.index] = {
1313
+ type: "reasoning",
1314
+ id: reasoningId,
1315
+ redactedData: event.content_block.data
1316
+ };
1317
+ safeEnqueue({
1318
+ type: "reasoning-start",
1319
+ id: reasoningId
1320
+ });
1204
1321
  } else if (event.content_block.type === "tool_use") {
1205
1322
  contentBlocks[event.index] = {
1206
1323
  type: "tool-call",
@@ -1223,6 +1340,14 @@ ${message.content}` : message.content;
1223
1340
  id: block.id,
1224
1341
  delta: event.delta.text
1225
1342
  });
1343
+ } else if (event.delta.type === "thinking_delta" && block?.type === "reasoning") {
1344
+ safeEnqueue({
1345
+ type: "reasoning-delta",
1346
+ id: block.id,
1347
+ delta: event.delta.thinking
1348
+ });
1349
+ } else if (event.delta.type === "signature_delta" && block?.type === "reasoning") {
1350
+ block.signature = event.delta.signature;
1226
1351
  } else if (event.delta.type === "input_json_delta" && block?.type === "tool-call") {
1227
1352
  block.input += event.delta.partial_json;
1228
1353
  safeEnqueue({
@@ -1240,6 +1365,13 @@ ${message.content}` : message.content;
1240
1365
  type: "text-end",
1241
1366
  id: block.id
1242
1367
  });
1368
+ } else if (block?.type === "reasoning") {
1369
+ const gitlabMeta = block.redactedData != null ? { redactedData: block.redactedData } : block.signature != null ? { signature: block.signature } : void 0;
1370
+ safeEnqueue({
1371
+ type: "reasoning-end",
1372
+ id: block.id,
1373
+ ...gitlabMeta && { providerMetadata: { gitlab: gitlabMeta } }
1374
+ });
1243
1375
  } else if (block?.type === "tool-call") {
1244
1376
  safeEnqueue({
1245
1377
  type: "tool-input-end",
@@ -1592,6 +1724,17 @@ var GitLabOpenAILanguageModel = class {
1592
1724
  }
1593
1725
  return parts.join(" | ");
1594
1726
  }
1727
+ /**
1728
+ * Per-call `providerOptions.gitlab.reasoningEffort` takes precedence over the
1729
+ * model-level default.
1730
+ */
1731
+ resolveReasoningEffort(options) {
1732
+ const gitlabOptions = options.providerOptions?.["gitlab"];
1733
+ return gitlabOptions?.reasoningEffort ?? this.config.reasoningEffort;
1734
+ }
1735
+ asOpenAIReasoningEffort(effort) {
1736
+ return effort;
1737
+ }
1595
1738
  convertTools(tools) {
1596
1739
  if (!tools || tools.length === 0) {
1597
1740
  return void 0;
@@ -1866,6 +2009,7 @@ var GitLabOpenAILanguageModel = class {
1866
2009
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
1867
2010
  const openaiModel = this.config.openaiModel || "gpt-4o";
1868
2011
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2012
+ const reasoningEffort = this.resolveReasoningEffort(options);
1869
2013
  const generateParams = {
1870
2014
  model: openaiModel,
1871
2015
  max_completion_tokens: maxTokens,
@@ -1874,7 +2018,10 @@ var GitLabOpenAILanguageModel = class {
1874
2018
  tool_choice: tools ? toolChoice : void 0,
1875
2019
  temperature: options.temperature,
1876
2020
  top_p: options.topP,
1877
- stop: options.stopSequences
2021
+ stop: options.stopSequences,
2022
+ ...reasoningEffort && {
2023
+ reasoning_effort: this.asOpenAIReasoningEffort(reasoningEffort)
2024
+ }
1878
2025
  };
1879
2026
  try {
1880
2027
  const response = options.abortSignal ? await client.chat.completions.create(generateParams, { signal: options.abortSignal }) : await client.chat.completions.create(generateParams);
@@ -1937,6 +2084,7 @@ var GitLabOpenAILanguageModel = class {
1937
2084
  const instructions = this.extractSystemInstructions(options.prompt);
1938
2085
  const openaiModel = this.config.openaiModel || "gpt-5-codex";
1939
2086
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2087
+ const reasoningEffort = this.resolveReasoningEffort(options);
1940
2088
  const generateParams = {
1941
2089
  model: openaiModel,
1942
2090
  input,
@@ -1945,7 +2093,18 @@ var GitLabOpenAILanguageModel = class {
1945
2093
  max_output_tokens: maxTokens,
1946
2094
  temperature: options.temperature,
1947
2095
  top_p: options.topP,
1948
- store: false
2096
+ store: false,
2097
+ ...reasoningEffort && {
2098
+ // `summary: 'auto'` opts into visible reasoning summary text.
2099
+ // Without it, the Responses API never returns summary content —
2100
+ // the model still reasons and effort still affects behavior, but
2101
+ // no chain-of-thought text comes back (mirrors the Anthropic
2102
+ // `thinking.display` gap fixed in buildThinkingParams).
2103
+ reasoning: {
2104
+ effort: this.asOpenAIReasoningEffort(reasoningEffort),
2105
+ summary: "auto"
2106
+ }
2107
+ }
1949
2108
  };
1950
2109
  try {
1951
2110
  const response = options.abortSignal ? await client.responses.create(generateParams, { signal: options.abortSignal }) : await client.responses.create(generateParams);
@@ -1966,6 +2125,12 @@ var GitLabOpenAILanguageModel = class {
1966
2125
  toolName: item.name,
1967
2126
  input: item.arguments
1968
2127
  });
2128
+ } else if (item.type === "reasoning") {
2129
+ for (const summary of item.summary || []) {
2130
+ if (summary.type === "summary_text" && summary.text) {
2131
+ content.push({ type: "reasoning", text: summary.text });
2132
+ }
2133
+ }
1969
2134
  }
1970
2135
  }
1971
2136
  const usage = this.createUsage({
@@ -2018,6 +2183,7 @@ var GitLabOpenAILanguageModel = class {
2018
2183
  const toolChoice = toolsDisabled ? void 0 : this.convertToolChoice(options.toolChoice);
2019
2184
  const openaiModel = this.config.openaiModel || "gpt-4o";
2020
2185
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2186
+ const reasoningEffort = this.resolveReasoningEffort(options);
2021
2187
  const requestBody = {
2022
2188
  model: openaiModel,
2023
2189
  max_completion_tokens: maxTokens,
@@ -2027,6 +2193,9 @@ var GitLabOpenAILanguageModel = class {
2027
2193
  temperature: options.temperature,
2028
2194
  top_p: options.topP,
2029
2195
  stop: options.stopSequences,
2196
+ ...reasoningEffort && {
2197
+ reasoning_effort: this.asOpenAIReasoningEffort(reasoningEffort)
2198
+ },
2030
2199
  stream: true,
2031
2200
  stream_options: { include_usage: true }
2032
2201
  };
@@ -2234,6 +2403,7 @@ var GitLabOpenAILanguageModel = class {
2234
2403
  const instructions = this.extractSystemInstructions(options.prompt);
2235
2404
  const openaiModel = this.config.openaiModel || "gpt-5-codex";
2236
2405
  const maxTokens = options.maxOutputTokens || this.config.maxTokens || 8192;
2406
+ const reasoningEffort = this.resolveReasoningEffort(options);
2237
2407
  const requestBody = {
2238
2408
  model: openaiModel,
2239
2409
  input,
@@ -2243,6 +2413,17 @@ var GitLabOpenAILanguageModel = class {
2243
2413
  temperature: options.temperature,
2244
2414
  top_p: options.topP,
2245
2415
  store: false,
2416
+ ...reasoningEffort && {
2417
+ // `summary: 'auto'` opts into visible reasoning summary text.
2418
+ // Without it, the Responses API never returns summary content —
2419
+ // the model still reasons and effort still affects behavior, but
2420
+ // no chain-of-thought text comes back (mirrors the Anthropic
2421
+ // `thinking.display` gap fixed in buildThinkingParams).
2422
+ reasoning: {
2423
+ effort: this.asOpenAIReasoningEffort(reasoningEffort),
2424
+ summary: "auto"
2425
+ }
2426
+ },
2246
2427
  stream: true
2247
2428
  };
2248
2429
  const self = this;
@@ -2281,11 +2462,38 @@ var GitLabOpenAILanguageModel = class {
2281
2462
  }
2282
2463
  };
2283
2464
  const toolCalls = {};
2465
+ const activeReasoning = /* @__PURE__ */ new Set();
2466
+ const knownReasoning = /* @__PURE__ */ new Set();
2467
+ const reasoningWithText = /* @__PURE__ */ new Set();
2284
2468
  let usage = self.createUsage();
2285
2469
  let finishReason = { unified: "other", raw: void 0 };
2286
2470
  let textStarted = false;
2287
2471
  let contentEmitted = false;
2288
2472
  const textId = "text-0";
2473
+ const reasoningId = (outputIndex, summaryIndex) => `reasoning-${outputIndex}-${summaryIndex}`;
2474
+ const startReasoning = (outputIndex, summaryIndex) => {
2475
+ const id = reasoningId(outputIndex, summaryIndex);
2476
+ if (!knownReasoning.has(id)) {
2477
+ knownReasoning.add(id);
2478
+ activeReasoning.add(id);
2479
+ contentEmitted = true;
2480
+ safeEnqueue({ type: "reasoning-start", id });
2481
+ }
2482
+ return id;
2483
+ };
2484
+ const emitReasoningDelta = (outputIndex, summaryIndex, delta) => {
2485
+ const id = startReasoning(outputIndex, summaryIndex);
2486
+ if (activeReasoning.has(id) && delta) {
2487
+ reasoningWithText.add(id);
2488
+ safeEnqueue({ type: "reasoning-delta", id, delta });
2489
+ }
2490
+ };
2491
+ const endReasoning = (outputIndex, summaryIndex) => {
2492
+ const id = reasoningId(outputIndex, summaryIndex);
2493
+ if (activeReasoning.delete(id)) {
2494
+ safeEnqueue({ type: "reasoning-end", id });
2495
+ }
2496
+ };
2289
2497
  try {
2290
2498
  const openaiStream = await client.responses.create(
2291
2499
  {
@@ -2318,6 +2526,30 @@ var GitLabOpenAILanguageModel = class {
2318
2526
  toolName: event.item.name
2319
2527
  });
2320
2528
  }
2529
+ } else if (event.type === "response.reasoning_summary_part.added") {
2530
+ startReasoning(event.output_index, event.summary_index);
2531
+ } else if (event.type === "response.reasoning_summary_text.delta") {
2532
+ emitReasoningDelta(event.output_index, event.summary_index, event.delta);
2533
+ } else if (event.type === "response.reasoning_summary_text.done") {
2534
+ const id = startReasoning(event.output_index, event.summary_index);
2535
+ if (!reasoningWithText.has(id)) {
2536
+ emitReasoningDelta(event.output_index, event.summary_index, event.text);
2537
+ }
2538
+ endReasoning(event.output_index, event.summary_index);
2539
+ } else if (event.type === "response.reasoning_summary_part.done") {
2540
+ const id = startReasoning(event.output_index, event.summary_index);
2541
+ if (!reasoningWithText.has(id)) {
2542
+ emitReasoningDelta(event.output_index, event.summary_index, event.part.text);
2543
+ }
2544
+ endReasoning(event.output_index, event.summary_index);
2545
+ } else if (event.type === "response.output_item.done" && event.item.type === "reasoning") {
2546
+ const prefix = `reasoning-${event.output_index}-`;
2547
+ for (const id of [...activeReasoning]) {
2548
+ if (id.startsWith(prefix)) {
2549
+ activeReasoning.delete(id);
2550
+ safeEnqueue({ type: "reasoning-end", id });
2551
+ }
2552
+ }
2321
2553
  } else if (event.type === "response.output_text.delta") {
2322
2554
  contentEmitted = true;
2323
2555
  if (!textStarted) {
@@ -2360,6 +2592,10 @@ var GitLabOpenAILanguageModel = class {
2360
2592
  }
2361
2593
  }
2362
2594
  }
2595
+ for (const id of activeReasoning) {
2596
+ safeEnqueue({ type: "reasoning-end", id });
2597
+ }
2598
+ activeReasoning.clear();
2363
2599
  if (textStarted) {
2364
2600
  safeEnqueue({ type: "text-end", id: textId });
2365
2601
  }
@@ -2452,7 +2688,7 @@ import { AsyncResource } from "async_hooks";
2452
2688
  import WebSocket from "isomorphic-ws";
2453
2689
 
2454
2690
  // src/version.ts
2455
- var VERSION = true ? "6.12.2" : "0.0.0-dev";
2691
+ var VERSION = true ? "6.13.0" : "0.0.0-dev";
2456
2692
 
2457
2693
  // src/gitlab-workflow-client.ts
2458
2694
  var WS_CONNECT_TIMEOUT_MS = 3e4;
@@ -5299,12 +5535,14 @@ function createGitLab(options = {}) {
5299
5535
  if (mapping.provider === "openai") {
5300
5536
  return new GitLabOpenAILanguageModel(modelId, {
5301
5537
  ...baseConfig,
5302
- openaiModel: agenticOptions?.providerModel ?? mapping.model
5538
+ openaiModel: agenticOptions?.providerModel ?? mapping.model,
5539
+ reasoningEffort: agenticOptions?.reasoningEffort
5303
5540
  });
5304
5541
  }
5305
5542
  return new GitLabAnthropicLanguageModel(modelId, {
5306
5543
  ...baseConfig,
5307
- anthropicModel: agenticOptions?.providerModel ?? mapping.model
5544
+ anthropicModel: agenticOptions?.providerModel ?? mapping.model,
5545
+ thinking: agenticOptions?.thinking
5308
5546
  });
5309
5547
  };
5310
5548
  const createWorkflowChatModel = (modelId, workflowOptions) => {