flavor-code 1.3.11 → 1.3.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -209,6 +209,13 @@ var ProviderConfigSchema = z2.object({
209
209
  cheapModel: z2.string().min(1).optional(),
210
210
  models: z2.array(z2.string().min(1)).max(100).optional(),
211
211
  maxOutputTokens: z2.number().int().positive().optional(),
212
+ // Extended-thinking budget in output tokens for the Anthropic protocol.
213
+ // Defaults to DEFAULT_THINKING_BUDGET; set to 0 to disable the request
214
+ // parameter. Provider thinking deltas are forwarded whenever they arrive.
215
+ thinkingBudget: z2.number().int().min(0).optional(),
216
+ // Reasoning effort requested from the OpenAI Responses protocol. Omitted by
217
+ // default so strict compatible gateways never see an unknown parameter.
218
+ thinkingEffort: z2.enum(["low", "medium", "high"]).optional(),
212
219
  // Send the Claude Code client fingerprint (User-Agent/x-app) with Anthropic requests,
213
220
  // for gateways that restrict the Anthropic protocol to Claude clients.
214
221
  claudeClient: z2.boolean().optional(),
@@ -5874,7 +5881,12 @@ function providerValidMessages(input) {
5874
5881
  const calls = original.toolCalls.filter((call) => call.id && call.name && availableResults.has(call.id));
5875
5882
  calls.forEach((call) => announced.add(call.id));
5876
5883
  if (!original.content && calls.length === 0) continue;
5877
- output.push({ role: "assistant", content: original.content, ...calls.length === 0 ? {} : { toolCalls: calls.map((call) => ({ ...call })) } });
5884
+ output.push({
5885
+ role: "assistant",
5886
+ content: original.content,
5887
+ ...calls.length === 0 ? {} : { toolCalls: calls.map((call) => ({ ...call })) },
5888
+ ...original.thinkingBlocks === void 0 ? {} : { thinkingBlocks: original.thinkingBlocks.map((block) => ({ ...block })) }
5889
+ });
5878
5890
  continue;
5879
5891
  }
5880
5892
  if (original.role === "tool") {
@@ -6597,6 +6609,7 @@ var AgentLoop = class {
6597
6609
  }
6598
6610
  let assistantText = "";
6599
6611
  const collectedToolCalls = [];
6612
+ const collectedThinking = [];
6600
6613
  let completed = false;
6601
6614
  let reactiveRetried = false;
6602
6615
  let attempt = 1;
@@ -6634,6 +6647,7 @@ var AgentLoop = class {
6634
6647
  }
6635
6648
  assistantText = "";
6636
6649
  collectedToolCalls.length = 0;
6650
+ collectedThinking.length = 0;
6637
6651
  completed = false;
6638
6652
  let terminalError;
6639
6653
  let providerError = false;
@@ -6654,6 +6668,10 @@ var AgentLoop = class {
6654
6668
  assistantText += event.text;
6655
6669
  accumulatedText += event.text;
6656
6670
  yield event;
6671
+ } else if (event.type === "thinking") {
6672
+ yield event;
6673
+ } else if (event.type === "thinking-block") {
6674
+ collectedThinking.push({ text: event.text, ...event.signature === void 0 ? {} : { signature: event.signature } });
6657
6675
  } else if (event.type === "tool-call") {
6658
6676
  collectedToolCalls.push({ kind: "valid", id: event.id, name: event.name, input: event.input });
6659
6677
  } else if (event.type === "invalid-tool-call") {
@@ -7001,7 +7019,8 @@ var AgentLoop = class {
7001
7019
  const assistantMessage = {
7002
7020
  role: "assistant",
7003
7021
  content: assistantText,
7004
- toolCalls: toolCalls.map(({ id, name, input }) => ({ id, name, input }))
7022
+ toolCalls: toolCalls.map(({ id, name, input }) => ({ id, name, input })),
7023
+ ...collectedThinking.length === 0 ? {} : { thinkingBlocks: collectedThinking.map((block) => ({ ...block })) }
7005
7024
  };
7006
7025
  this.#options.context.appendMany([assistantMessage, ...stagedMessages]);
7007
7026
  for (const { call, result } of stagedResults) {
@@ -13287,6 +13306,7 @@ var CLAUDE_CLIENT_HEADERS = {
13287
13306
  "x-app": "cli"
13288
13307
  };
13289
13308
  var DEFAULT_MAX_OUTPUT_TOKENS = 32768;
13309
+ var DEFAULT_THINKING_BUDGET = 8192;
13290
13310
  var MAX_CACHE_MARKERS = 4;
13291
13311
  function collectCacheMarkers(system, messages) {
13292
13312
  const markers = [];
@@ -13345,9 +13365,26 @@ function formatCacheUsage(model, snapshot4, shape) {
13345
13365
  ...shape === void 0 ? {} : { requestMessages: shape.messages, requestMarkers: shape.markers }
13346
13366
  });
13347
13367
  }
13368
+ function stripThinkingBlocks(messages) {
13369
+ for (const message2 of messages) {
13370
+ if (message2.role !== "assistant" || !Array.isArray(message2.content)) continue;
13371
+ const blocks = message2.content;
13372
+ if (!blocks.some((block) => block?.type === "thinking")) continue;
13373
+ message2.content = blocks.filter((block) => block?.type !== "thinking");
13374
+ }
13375
+ }
13376
+ function isThinkingParamRejected(error) {
13377
+ const status = error?.status;
13378
+ if (status !== void 0 && status !== 400) return false;
13379
+ const message2 = error instanceof Error ? error.message : String(error ?? "");
13380
+ return /\bthinking\b/iu.test(message2) && /(budget_tokens|expected|not supported|unsupported|unknown|invalid|extra inputs|unrecognized)/iu.test(message2);
13381
+ }
13348
13382
  var AnthropicModelAdapter = class {
13349
13383
  client;
13350
13384
  maxOutputTokens;
13385
+ thinkingBudget;
13386
+ /** Set once an endpoint rejects the `thinking` parameter; later requests stop sending it. */
13387
+ #thinkingRejected = false;
13351
13388
  debugUsage;
13352
13389
  constructor(options) {
13353
13390
  const maxOutputTokens = options.maxOutputTokens ?? DEFAULT_MAX_OUTPUT_TOKENS;
@@ -13355,6 +13392,7 @@ var AnthropicModelAdapter = class {
13355
13392
  throw new Error("maxOutputTokens must be a positive integer");
13356
13393
  }
13357
13394
  this.maxOutputTokens = maxOutputTokens;
13395
+ this.thinkingBudget = options.thinkingBudget ?? DEFAULT_THINKING_BUDGET;
13358
13396
  this.debugUsage = options.debugUsage ?? isEnvTruthy(process.env.FLAVOR_DEBUG_USAGE);
13359
13397
  this.client = options.client ?? new Anthropic({
13360
13398
  ...options.apiKey === void 0 ? {} : { apiKey: options.apiKey },
@@ -13362,6 +13400,16 @@ var AnthropicModelAdapter = class {
13362
13400
  ...options.headers === void 0 ? {} : { defaultHeaders: options.headers }
13363
13401
  });
13364
13402
  }
13403
+ /**
13404
+ * Extended-thinking request parameter when enabled, otherwise undefined.
13405
+ * Anthropic requires the budget to fit inside max_tokens with a 1024 floor.
13406
+ */
13407
+ #thinkingParam() {
13408
+ if (this.#thinkingRejected || this.thinkingBudget <= 0) return void 0;
13409
+ const budget = Math.min(this.thinkingBudget, this.maxOutputTokens - 1);
13410
+ if (budget < 1024) return void 0;
13411
+ return { type: "enabled", budget_tokens: budget };
13412
+ }
13365
13413
  #logUsage(request, snapshot4, shape) {
13366
13414
  const line = formatCacheUsage(request.model, snapshot4, shape);
13367
13415
  if (this.debugUsage) {
@@ -13417,7 +13465,13 @@ var AnthropicModelAdapter = class {
13417
13465
  }
13418
13466
  messages.push({ role: "user", content: results });
13419
13467
  } else if (message2.role === "assistant" && message2.toolCalls?.length) {
13468
+ const echoThinking = this.#thinkingRejected ? [] : (message2.thinkingBlocks ?? []).filter((block) => block.signature !== void 0);
13420
13469
  const content = [
13470
+ ...echoThinking.map((block) => ({
13471
+ type: "thinking",
13472
+ thinking: block.text,
13473
+ signature: block.signature
13474
+ })),
13421
13475
  ...modelContentText(message2.content) ? [{ type: "text", text: modelContentText(message2.content) }] : [],
13422
13476
  ...message2.toolCalls.map((call) => ({
13423
13477
  type: "tool_use",
@@ -13479,18 +13533,56 @@ var AnthropicModelAdapter = class {
13479
13533
  stream: true,
13480
13534
  messages,
13481
13535
  ...system ? { system } : {},
13536
+ ...(() => {
13537
+ const thinking = this.#thinkingParam();
13538
+ return thinking === void 0 ? {} : { thinking };
13539
+ })(),
13482
13540
  tools: [...request.tools].sort((a, b) => a.name.localeCompare(b.name)).map((tool) => ({
13483
13541
  name: tool.name,
13484
13542
  description: tool.description,
13485
13543
  input_schema: { ...tool.inputSchema, type: "object" }
13486
13544
  }))
13487
13545
  };
13488
- const stream = await this.client.messages.create(body, { signal: request.signal });
13546
+ let stream;
13547
+ try {
13548
+ stream = await this.client.messages.create(body, { signal: request.signal });
13549
+ } catch (error) {
13550
+ if (this.thinkingBudget > 0 && !this.#thinkingRejected && isThinkingParamRejected(error)) {
13551
+ this.#thinkingRejected = true;
13552
+ delete body.thinking;
13553
+ stripThinkingBlocks(body.messages);
13554
+ stream = await this.client.messages.create(body, { signal: request.signal });
13555
+ } else {
13556
+ throw error;
13557
+ }
13558
+ }
13559
+ const thinkingBuffers = /* @__PURE__ */ new Map();
13489
13560
  for await (const event of stream) {
13490
13561
  if (event.type === "message_start") {
13491
13562
  hasUsage = event.message?.usage !== void 0;
13492
13563
  inputTokens = updateInputUsage(inputUsage, event.message?.usage);
13493
13564
  outputTokens = event.message?.usage?.output_tokens ?? outputTokens;
13565
+ } else if (event.type === "content_block_start" && event.content_block?.type === "thinking" && event.index !== void 0) {
13566
+ thinkingBuffers.set(event.index, { text: "" });
13567
+ } else if (event.type === "content_block_delta" && event.delta?.type === "thinking_delta") {
13568
+ if (event.index !== void 0 && event.delta.thinking) {
13569
+ const pending2 = thinkingBuffers.get(event.index);
13570
+ if (pending2) pending2.text += event.delta.thinking;
13571
+ }
13572
+ if (event.delta.thinking) yield { type: "thinking", text: event.delta.thinking };
13573
+ } else if (event.type === "content_block_delta" && event.delta?.type === "signature_delta" && event.index !== void 0) {
13574
+ const pending2 = thinkingBuffers.get(event.index);
13575
+ if (pending2) pending2.signature = (pending2.signature ?? "") + (event.delta.signature ?? "");
13576
+ } else if (event.type === "content_block_stop" && event.index !== void 0 && thinkingBuffers.has(event.index)) {
13577
+ const pending2 = thinkingBuffers.get(event.index);
13578
+ if (pending2.text.length > 0 || pending2.signature !== void 0) {
13579
+ yield {
13580
+ type: "thinking-block",
13581
+ text: pending2.text,
13582
+ ...pending2.signature === void 0 ? {} : { signature: pending2.signature }
13583
+ };
13584
+ }
13585
+ thinkingBuffers.delete(event.index);
13494
13586
  } else if (event.type === "content_block_start" && event.content_block?.type === "tool_use" && event.index !== void 0 && event.content_block.id && event.content_block.name) {
13495
13587
  pendingTools.set(event.index, {
13496
13588
  id: event.content_block.id,
@@ -13693,12 +13785,14 @@ function formatOpenAIUsage(model, breakdown) {
13693
13785
  var OpenAIModelAdapter = class {
13694
13786
  client;
13695
13787
  debugUsage;
13788
+ thinkingEffort;
13696
13789
  constructor(options) {
13697
13790
  this.client = options.client ?? new OpenAI({
13698
13791
  ...options.apiKey === void 0 ? {} : { apiKey: options.apiKey },
13699
13792
  ...options.baseURL === void 0 ? {} : { baseURL: options.baseURL }
13700
13793
  });
13701
13794
  this.debugUsage = options.debugUsage ?? isEnvTruthy(process.env.FLAVOR_DEBUG_USAGE);
13795
+ this.thinkingEffort = options.thinkingEffort;
13702
13796
  }
13703
13797
  #logUsage(model, breakdown) {
13704
13798
  if (breakdown === void 0) return;
@@ -13720,6 +13814,7 @@ var OpenAIModelAdapter = class {
13720
13814
  const body = {
13721
13815
  model: request.model,
13722
13816
  input: (await Promise.all(request.messages.map(toInput))).flat(),
13817
+ ...this.thinkingEffort === void 0 ? {} : { reasoning: { effort: this.thinkingEffort } },
13723
13818
  tools: [...request.tools].sort((a, b) => a.name.localeCompare(b.name)).map((tool) => ({
13724
13819
  type: "function",
13725
13820
  name: tool.name,
@@ -13752,6 +13847,8 @@ var OpenAIModelAdapter = class {
13752
13847
  }
13753
13848
  } else if (event.type === "response.output_text.delta" && event.delta) {
13754
13849
  yield { type: "text", text: event.delta };
13850
+ } else if ((event.type === "response.reasoning_summary_text.delta" || event.type === "response.reasoning_text.delta") && event.delta) {
13851
+ yield { type: "thinking", text: event.delta };
13755
13852
  } else if (event.type === "response.function_call_arguments.done" && event.output_index !== void 0 && event.name) {
13756
13853
  const callId = callIds.get(event.output_index);
13757
13854
  if (callId && !emittedCalls.has(event.output_index)) {
@@ -15301,6 +15398,13 @@ import { basename as basename5, dirname as dirname15, join as join23, relative a
15301
15398
  import { randomUUID as randomUUID13 } from "crypto";
15302
15399
  import { z as z27 } from "zod";
15303
15400
 
15401
+ // src/ui/thinking-line.ts
15402
+ var THINKING_MAX_STORED_CHARS = 4e3;
15403
+ function appendThinkingText(previous, delta) {
15404
+ const next = (previous ?? "") + delta;
15405
+ return next.length > THINKING_MAX_STORED_CHARS ? next.slice(-THINKING_MAX_STORED_CHARS) : next;
15406
+ }
15407
+
15304
15408
  // src/ui/transcript.ts
15305
15409
  function createTranscriptState() {
15306
15410
  return { completed: [], nextId: 1 };
@@ -15397,6 +15501,9 @@ function transcriptReducer(state, action) {
15397
15501
  if (event.type === "text") {
15398
15502
  return { ...state, active: addText(withoutModelActivity(state.active), event.text) };
15399
15503
  }
15504
+ if (event.type === "thinking") {
15505
+ return { ...state, active: attachThinking(state.active, event.text) };
15506
+ }
15400
15507
  if (event.type === "tool-start") return upsertStatus({ ...state, active: withoutModelActivity(state.active) }, {
15401
15508
  kind: "status",
15402
15509
  id: `tool:${event.id}`,
@@ -15697,6 +15804,18 @@ function finishActive(state) {
15697
15804
  ...state.taskSnapshot === void 0 ? {} : { taskSnapshot: state.taskSnapshot }
15698
15805
  };
15699
15806
  }
15807
+ function attachThinking(turn, text) {
15808
+ const blocks = [...turn.blocks];
15809
+ const index = blocks.findIndex((block2) => block2.kind === "status" && block2.activity === "model" && block2.state === "running");
15810
+ if (index < 0) return turn;
15811
+ const block = blocks[index];
15812
+ if (block === void 0 || block.kind !== "status") return turn;
15813
+ blocks[index] = {
15814
+ ...block,
15815
+ thinkingText: appendThinkingText(block.thinkingText, text)
15816
+ };
15817
+ return { ...turn, blocks };
15818
+ }
15700
15819
  function withoutModelActivity(turn, id) {
15701
15820
  const blocks = turn.blocks.filter((block) => block.kind !== "status" || block.activity !== "model" || id !== void 0 && block.id !== id);
15702
15821
  if (blocks.length === turn.blocks.length) return turn;
@@ -26151,8 +26270,12 @@ ${renderMemoryDocument(entries)}`;
26151
26270
  };
26152
26271
  registry.register(next.providerId, next.apiType === "anthropic" ? new AnthropicModelAdapter({
26153
26272
  ...adapterOptions,
26273
+ ...providerConfig?.thinkingBudget === void 0 ? {} : { thinkingBudget: providerConfig.thinkingBudget },
26154
26274
  ...providerConfig?.claudeClient === true ? { headers: CLAUDE_CLIENT_HEADERS } : {}
26155
- }) : new OpenAIModelAdapter(adapterOptions));
26275
+ }) : new OpenAIModelAdapter({
26276
+ ...adapterOptions,
26277
+ ...providerConfig?.thinkingEffort === void 0 ? {} : { thinkingEffort: providerConfig.thinkingEffort }
26278
+ }));
26156
26279
  effectiveLlm = next;
26157
26280
  mainModel = `${next.providerId}:${next.defaultModel}`;
26158
26281
  childModel = `${next.providerId}:${next.cheapModel}`;
@@ -26738,8 +26861,12 @@ async function registerConfiguredAdapters(providers, registry, environment, diag
26738
26861
  };
26739
26862
  const adapter = apiProtocol === "anthropic" ? new AnthropicModelAdapter({
26740
26863
  ...adapterOptions,
26864
+ ...runtimeProvider.thinkingBudget === void 0 ? {} : { thinkingBudget: runtimeProvider.thinkingBudget },
26741
26865
  ...runtimeProvider.claudeClient === true ? { headers: CLAUDE_CLIENT_HEADERS } : {}
26742
- }) : new OpenAIModelAdapter(adapterOptions);
26866
+ }) : new OpenAIModelAdapter({
26867
+ ...adapterOptions,
26868
+ ...runtimeProvider.thinkingEffort === void 0 ? {} : { thinkingEffort: runtimeProvider.thinkingEffort }
26869
+ });
26743
26870
  const cacheProfile = resolveCacheProfile({ apiType: apiProtocol, baseURL: runtimeProvider.baseURL });
26744
26871
  if (apiProtocol === "openai" && isDashScopeBaseURL(runtimeProvider.baseURL)) {
26745
26872
  diagnostics.push(