@prestyj/ai 5.28.1 → 5.29.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -35,14 +35,17 @@ __export(index_exports, {
35
35
  ProviderError: () => ProviderError,
36
36
  REDACTION_MARKER: () => REDACTED,
37
37
  StreamResult: () => StreamResult,
38
+ TOOL_CALL_NAME_RULES: () => TOOL_CALL_NAME_RULES,
38
39
  clampProviderContextImages: () => clampProviderContextImages,
39
40
  classifyProviderError: () => classifyProviderError,
41
+ dropInvalidToolCalls: () => dropInvalidToolCalls,
40
42
  environmentSecrets: () => environmentSecrets,
41
43
  formatError: () => formatError,
42
44
  formatErrorForDisplay: () => formatErrorForDisplay,
43
45
  hasLoneSurrogate: () => hasLoneSurrogate,
44
46
  isHardBillingMessage: () => isHardBillingMessage,
45
47
  isUsageLimitError: () => isUsageLimitError,
48
+ isValidToolCallName: () => isValidToolCallName,
46
49
  localWireModelId: () => localWireModelId,
47
50
  palsuAssistantMessage: () => palsuAssistantMessage,
48
51
  palsuText: () => palsuText,
@@ -61,7 +64,9 @@ __export(index_exports, {
61
64
  stream: () => stream,
62
65
  toAnthropicMessages: () => toAnthropicMessages,
63
66
  toOpenAIMessages: () => toOpenAIMessages,
64
- toWellFormedText: () => toWellFormedText
67
+ toWellFormedText: () => toWellFormedText,
68
+ toolCallNameRuleFor: () => toolCallNameRuleFor,
69
+ usesResponsesLite: () => usesResponsesLite
65
70
  });
66
71
  module.exports = __toCommonJS(index_exports);
67
72
 
@@ -515,7 +520,11 @@ function zodToJsonSchema(schema) {
515
520
  return normalized;
516
521
  }
517
522
  function resolveToolSchema(tool) {
518
- return tool.rawInputSchema ?? zodToJsonSchema(tool.parameters);
523
+ const schema = tool.rawInputSchema ?? zodToJsonSchema(tool.parameters);
524
+ if (schema.type === "object" && schema.properties === void 0) {
525
+ return { ...schema, properties: {} };
526
+ }
527
+ return schema;
519
528
  }
520
529
  function normalizeRootForAnthropic(schema) {
521
530
  const branches = schema.oneOf ?? schema.anyOf;
@@ -722,6 +731,10 @@ var ANTHROPIC_INPUT_BLOCK_TYPES = /* @__PURE__ */ new Set([
722
731
  "connector_text",
723
732
  "container_upload",
724
733
  "document",
734
+ // Server-side refusal fallback marker. Only reaches the wire when the request
735
+ // carries the server-side-fallback beta; otherwise applyServerFallbackReplay
736
+ // strips it first (see toAnthropicMessages' `fallbackBlocks` option).
737
+ "fallback",
725
738
  "image",
726
739
  "mid_conv_system",
727
740
  "redacted_thinking",
@@ -743,6 +756,42 @@ function isPositionSensitiveThinking(part) {
743
756
  if (part.type === "thinking") return hasValidThinkingSignature(part);
744
757
  return isRawThinking(part);
745
758
  }
759
+ function isServerFallbackBlock(part) {
760
+ return part.type === "raw" && part.data.type === "fallback";
761
+ }
762
+ function applyServerFallbackReplay(content, keepMarkers, droppedToolCallIds) {
763
+ let lastFallbackIdx = -1;
764
+ content.forEach((part, idx) => {
765
+ if (isServerFallbackBlock(part)) lastFallbackIdx = idx;
766
+ });
767
+ if (lastFallbackIdx === -1) return content;
768
+ const resultIds = /* @__PURE__ */ new Set();
769
+ for (const part of content) {
770
+ if (part.type === "server_tool_result") resultIds.add(part.toolUseId);
771
+ else if (part.type === "raw" && typeof part.data.tool_use_id === "string")
772
+ resultIds.add(part.data.tool_use_id);
773
+ }
774
+ const out = [];
775
+ content.forEach((part, idx) => {
776
+ if (isServerFallbackBlock(part)) {
777
+ if (keepMarkers) out.push(part);
778
+ return;
779
+ }
780
+ if (idx < lastFallbackIdx) {
781
+ if (part.type === "thinking" || isRawThinking(part)) return;
782
+ if (part.type === "raw" && part.data.type === "connector_text") return;
783
+ if (part.type === "tool_call") {
784
+ droppedToolCallIds?.add(part.id);
785
+ return;
786
+ }
787
+ if (part.type === "server_tool_call" && !resultIds.has(part.id)) return;
788
+ if (part.type === "raw" && part.data.type === "server_tool_use" && !resultIds.has(String(part.data.id)))
789
+ return;
790
+ }
791
+ out.push(part);
792
+ });
793
+ return out.every(isServerFallbackBlock) ? [] : out;
794
+ }
746
795
  function toAnthropicAssistantPart(part, idMap) {
747
796
  if (part.type === "text") return { type: "text", text: part.text };
748
797
  if (part.type === "thinking") {
@@ -810,10 +859,17 @@ function countContextImages(messages) {
810
859
  }
811
860
  return count;
812
861
  }
862
+ var IMAGE_DROP_BATCH = 30;
863
+ function providerImageDropCount(imageCount2, budget) {
864
+ const overflow = imageCount2 - budget;
865
+ if (overflow <= 0) return 0;
866
+ const batch = Math.max(1, Math.min(IMAGE_DROP_BATCH, Math.floor(budget / 3)));
867
+ return Math.min(imageCount2, Math.ceil(overflow / batch) * batch);
868
+ }
813
869
  function clampProviderContextImages(messages, provider, supportsImages) {
814
870
  if (supportsImages === false) return messages;
815
871
  const budget = PROVIDER_IMAGE_BUDGETS[provider] ?? 5;
816
- let remainingToRemove = countContextImages(messages) - budget;
872
+ let remainingToRemove = providerImageDropCount(countContextImages(messages), budget);
817
873
  if (remainingToRemove <= 0) return messages;
818
874
  return messages.map((message) => {
819
875
  if (message.role === "user" && Array.isArray(message.content)) {
@@ -847,6 +903,144 @@ function clampProviderContextImages(messages, provider, supportsImages) {
847
903
  return message;
848
904
  });
849
905
  }
906
+ var TOOL_CALL_NAME_RULES = {
907
+ // Anthropic Messages API: tool / tool_use name `^[a-zA-Z0-9_-]{1,128}$`.
908
+ anthropic: { id: "anthropic", maxLength: 128, pattern: /^[a-zA-Z0-9_-]{1,128}$/ },
909
+ // OpenAI Chat Completions: function name a-z A-Z 0-9 _ -, max length 64.
910
+ "openai-chat": { id: "openai-chat", maxLength: 64, pattern: /^[a-zA-Z0-9_-]{1,64}$/ },
911
+ // OpenAI Responses (Codex): same charset; replayed `input[N].name` over 128
912
+ // chars is rejected with `string_above_max_length`.
913
+ "openai-responses": {
914
+ id: "openai-responses",
915
+ maxLength: 128,
916
+ pattern: /^[a-zA-Z0-9_-]{1,128}$/
917
+ },
918
+ // Gemini FunctionDeclaration name: starts with a letter or underscore, then
919
+ // a-z A-Z 0-9 _ . : -, max length 128.
920
+ gemini: { id: "gemini", maxLength: 128, pattern: /^[a-zA-Z_][a-zA-Z0-9_.:-]{0,127}$/ },
921
+ // OpenAI-compatible third parties (GLM, Kimi, DeepSeek, OpenRouter, xAI,
922
+ // local servers, MiniMax, …) don't share one documented charset, so only
923
+ // reject what no declared tool name can contain: blank, >128 chars, or any
924
+ // whitespace / control character (the signature of invocation text).
925
+ generic: { id: "generic", maxLength: 128 }
926
+ };
927
+ var TOOL_NAME_FORBIDDEN_CHARS = /[\s\p{Cc}]/u;
928
+ function isValidToolCallName(name, rule) {
929
+ if (typeof name !== "string" || name.trim().length === 0) return false;
930
+ if (name.length > rule.maxLength) return false;
931
+ if (TOOL_NAME_FORBIDDEN_CHARS.test(name)) return false;
932
+ return rule.pattern ? rule.pattern.test(name) : true;
933
+ }
934
+ function isDefaultOrHost(baseUrl, host) {
935
+ if (!baseUrl) return true;
936
+ try {
937
+ return new URL(baseUrl).hostname === host;
938
+ } catch {
939
+ return false;
940
+ }
941
+ }
942
+ function toolCallNameRuleFor(provider, options) {
943
+ switch (provider) {
944
+ case "anthropic":
945
+ return TOOL_CALL_NAME_RULES.anthropic;
946
+ case "openai":
947
+ if (options?.accountId) return TOOL_CALL_NAME_RULES["openai-responses"];
948
+ return isDefaultOrHost(options?.baseUrl, "api.openai.com") ? TOOL_CALL_NAME_RULES["openai-chat"] : TOOL_CALL_NAME_RULES.generic;
949
+ case "gemini":
950
+ return TOOL_CALL_NAME_RULES.gemini;
951
+ default:
952
+ return TOOL_CALL_NAME_RULES.generic;
953
+ }
954
+ }
955
+ function isValidToolCallId(id) {
956
+ return typeof id === "string" && id.trim().length > 0;
957
+ }
958
+ function isRawReasoning(part) {
959
+ return part.type === "raw" && part.data.type === "reasoning";
960
+ }
961
+ function isDanglingReasoning(parts, idx) {
962
+ for (let i = idx + 1; i < parts.length; i++) {
963
+ const next = parts[i];
964
+ if (isRawReasoning(next)) return true;
965
+ if (next.type === "text" || next.type === "tool_call") return false;
966
+ }
967
+ return true;
968
+ }
969
+ function hasReplayableAssistantContent(parts) {
970
+ return parts.some((part) => {
971
+ if (part.type === "text") return part.text.length > 0;
972
+ if (part.type === "thinking") return false;
973
+ if (part.type === "raw") {
974
+ const t = part.data.type;
975
+ return !(t === "thinking" || t === "redacted_thinking" || t === "reasoning" || t === "fallback");
976
+ }
977
+ return true;
978
+ });
979
+ }
980
+ function dropInvalidToolCalls(messages, rule) {
981
+ const isInvalid = (part) => part.type === "tool_call" && (!isValidToolCallName(part.name, rule) || !isValidToolCallId(part.id));
982
+ const hasInvalid = messages.some(
983
+ (m) => m.role === "assistant" && Array.isArray(m.content) && m.content.some(isInvalid)
984
+ );
985
+ if (!hasInvalid) return messages;
986
+ const out = [];
987
+ let pending = /* @__PURE__ */ new Map();
988
+ let prunedAssistant = null;
989
+ for (const msg of messages) {
990
+ if (msg.role === "user" || msg.role === "system") {
991
+ if (msg.role === "user") pending = /* @__PURE__ */ new Map();
992
+ out.push(msg);
993
+ continue;
994
+ }
995
+ if (msg.role === "assistant") {
996
+ pending = /* @__PURE__ */ new Map();
997
+ if (typeof msg.content === "string") {
998
+ out.push(msg);
999
+ continue;
1000
+ }
1001
+ const original = msg.content;
1002
+ let dropped = false;
1003
+ const kept = [];
1004
+ for (const part of original) {
1005
+ if (part.type === "tool_call") {
1006
+ const keep = !isInvalid(part);
1007
+ const queue = pending.get(part.id) ?? [];
1008
+ queue.push(keep);
1009
+ pending.set(part.id, queue);
1010
+ if (!keep) {
1011
+ dropped = true;
1012
+ continue;
1013
+ }
1014
+ }
1015
+ kept.push(part);
1016
+ }
1017
+ if (!dropped) {
1018
+ out.push(msg);
1019
+ continue;
1020
+ }
1021
+ const content = kept.filter(
1022
+ (part, idx) => !isRawReasoning(part) || !isDanglingReasoning(kept, idx) || isDanglingReasoning(original, original.indexOf(part))
1023
+ );
1024
+ if (hasReplayableAssistantContent(content)) {
1025
+ const pruned = { ...msg, content };
1026
+ out.push(pruned);
1027
+ prunedAssistant = pruned;
1028
+ }
1029
+ continue;
1030
+ }
1031
+ let changed = false;
1032
+ const results = msg.content.filter((result) => {
1033
+ const queue = pending.get(result.toolCallId);
1034
+ const keep = queue && queue.length > 0 ? queue.shift() : true;
1035
+ if (!keep) changed = true;
1036
+ return keep;
1037
+ });
1038
+ if (!changed) out.push(msg);
1039
+ else if (results.length > 0) out.push({ ...msg, content: results });
1040
+ }
1041
+ if (prunedAssistant && out[out.length - 1] === prunedAssistant) out.pop();
1042
+ return out;
1043
+ }
850
1044
  var NON_VISION_USER_IMAGE_PLACEHOLDER = "(image omitted: model does not support images)";
851
1045
  var NON_VISION_TOOL_IMAGE_PLACEHOLDER = "(tool image omitted: model does not support images)";
852
1046
  var NON_VIDEO_USER_PLACEHOLDER = "(video omitted: model does not support video)";
@@ -953,10 +1147,12 @@ function remapAnthropicToolCallId(id, idMap) {
953
1147
  idMap.set(id, mapped);
954
1148
  return mapped;
955
1149
  }
956
- function toAnthropicMessages(messages, cacheControl) {
1150
+ function toAnthropicMessages(messages, cacheControl, options) {
957
1151
  let systemText;
958
1152
  const out = [];
959
1153
  const idMap = /* @__PURE__ */ new Map();
1154
+ const keepFallbackBlocks = options?.fallbackBlocks === true;
1155
+ const droppedToolCallIds = /* @__PURE__ */ new Set();
960
1156
  const trajectoryStartIdx = messages.reduce(
961
1157
  (last, m, i) => m.role === "user" ? i : last,
962
1158
  -1
@@ -1002,17 +1198,23 @@ function toAnthropicMessages(messages, cacheControl) {
1002
1198
  }
1003
1199
  if (msg.role === "assistant") {
1004
1200
  if (typeof msg.content === "string" && msg.content === "") continue;
1005
- const content = typeof msg.content === "string" ? msg.content : toAnthropicAssistantContent(msg.content, msgIdx > trajectoryStartIdx, idMap);
1201
+ const content = typeof msg.content === "string" ? msg.content : toAnthropicAssistantContent(
1202
+ applyServerFallbackReplay(msg.content, keepFallbackBlocks, droppedToolCallIds),
1203
+ msgIdx > trajectoryStartIdx,
1204
+ idMap
1205
+ );
1006
1206
  if (Array.isArray(content) && content.length === 0) continue;
1007
1207
  out.push({ role: "assistant", content });
1008
1208
  continue;
1009
1209
  }
1010
1210
  if (msg.role === "tool") {
1211
+ const results = droppedToolCallIds.size ? msg.content.filter((r) => !droppedToolCallIds.has(r.toolCallId)) : msg.content;
1212
+ if (results.length === 0) continue;
1011
1213
  out.push({
1012
1214
  role: "user",
1013
1215
  // Cast covers the video block (used by the Anthropic-compatible MiniMax
1014
1216
  // API), which isn't in the first-party Anthropic tool_result types.
1015
- content: msg.content.map((result) => ({
1217
+ content: results.map((result) => ({
1016
1218
  type: "tool_result",
1017
1219
  tool_use_id: remapAnthropicToolCallId(result.toolCallId, idMap),
1018
1220
  content: toAnthropicToolResultContent(result.content),
@@ -1356,6 +1558,48 @@ function fineGrainedToolStreamingEnabled() {
1356
1558
  const v = raw.trim().toLowerCase();
1357
1559
  return v === "1" || v === "true" || v === "yes" || v === "on";
1358
1560
  }
1561
+ var SERVER_FALLBACK_BETA = "server-side-fallback-2026-07-01";
1562
+ var anthropicServerFallback = {
1563
+ disabled: /* @__PURE__ */ new Set(),
1564
+ reset() {
1565
+ this.disabled.clear();
1566
+ }
1567
+ };
1568
+ function serverFallbackKey(options) {
1569
+ const auth = options.apiKey?.startsWith("sk-ant-oat") ? "oauth" : "key";
1570
+ return `${options.baseUrl ?? process.env.ANTHROPIC_BASE_URL ?? ""}|${auth}`;
1571
+ }
1572
+ function isDirectAnthropicApi(baseUrl) {
1573
+ const effective = baseUrl ?? process.env.ANTHROPIC_BASE_URL;
1574
+ if (!effective) return true;
1575
+ try {
1576
+ const url = new URL(effective);
1577
+ return url.protocol === "https:" && url.hostname === "api.anthropic.com";
1578
+ } catch {
1579
+ return false;
1580
+ }
1581
+ }
1582
+ function isServerFallbackRejection(err) {
1583
+ const status = err?.status;
1584
+ if (status !== 400) return false;
1585
+ const message = err instanceof Error ? err.message : String(err);
1586
+ return /fallbacks|server-side-fallback/i.test(message);
1587
+ }
1588
+ function sumUsageIterations(usage) {
1589
+ const iterations = usage?.iterations;
1590
+ if (!Array.isArray(iterations) || iterations.length === 0) return null;
1591
+ const num = (v) => typeof v === "number" && Number.isFinite(v) ? v : 0;
1592
+ const totals = { inputTokens: 0, outputTokens: 0, cacheRead: 0, cacheWrite: 0 };
1593
+ for (const it of iterations) {
1594
+ if (!it || typeof it !== "object") continue;
1595
+ const rec = it;
1596
+ totals.inputTokens += num(rec.input_tokens);
1597
+ totals.outputTokens += num(rec.output_tokens);
1598
+ totals.cacheRead += num(rec.cache_read_input_tokens);
1599
+ totals.cacheWrite += num(rec.cache_creation_input_tokens);
1600
+ }
1601
+ return totals;
1602
+ }
1359
1603
  function createClient(options) {
1360
1604
  const isOAuth = options.apiKey?.startsWith("sk-ant-oat");
1361
1605
  const userAgent = isOAuth ? options.userAgent ?? "claude-cli/2.1.75 (external, cli)" : "";
@@ -1450,12 +1694,16 @@ function streamAnthropic(options) {
1450
1694
  async function* runStream(options) {
1451
1695
  const client = createClient(options);
1452
1696
  const isOAuth = options.apiKey?.startsWith("sk-ant-oat");
1453
- const useStreaming = options.streaming !== false;
1697
+ const useStreaming = options.streaming !== false && !options.prewarm;
1454
1698
  const cacheControl = toAnthropicCacheControl(options.cacheRetention, options.baseUrl);
1455
1699
  const supportsFirstPartyToolExtras = !options.baseUrl || options.baseUrl.includes("api.anthropic.com");
1456
1700
  const downgradedImages = downgradeUnsupportedImages(options.messages, options.supportsImages);
1457
1701
  const downgradedMessages = downgradeUnsupportedVideos(downgradedImages, options.supportsVideo);
1458
- const { system: rawSystem, messages } = toAnthropicMessages(downgradedMessages, cacheControl);
1702
+ const fallbackKey = serverFallbackKey(options);
1703
+ const useServerFallback = options.provider !== "minimax" && isDirectAnthropicApi(options.baseUrl) && !anthropicServerFallback.disabled.has(fallbackKey);
1704
+ const { system: rawSystem, messages } = toAnthropicMessages(downgradedMessages, cacheControl, {
1705
+ fallbackBlocks: useServerFallback
1706
+ });
1459
1707
  const system = isOAuth ? [
1460
1708
  {
1461
1709
  type: "text",
@@ -1474,6 +1722,17 @@ async function* runStream(options) {
1474
1722
  outputConfig = t.outputConfig;
1475
1723
  }
1476
1724
  }
1725
+ if (options.prewarm) {
1726
+ const budget = thinking?.budget_tokens;
1727
+ if (budget != null && budget >= 1) {
1728
+ return {
1729
+ message: { role: "assistant", content: [] },
1730
+ stopReason: "end_turn",
1731
+ usage: { inputTokens: 0, outputTokens: 0 }
1732
+ };
1733
+ }
1734
+ maxTokens = 1;
1735
+ }
1477
1736
  const params = {
1478
1737
  model: options.model,
1479
1738
  max_tokens: maxTokens,
@@ -1514,6 +1773,7 @@ async function* runStream(options) {
1514
1773
  ];
1515
1774
  return contextEdits.length ? { context_management: { edits: contextEdits } } : {};
1516
1775
  })(),
1776
+ ...useServerFallback ? { fallbacks: "default" } : {},
1517
1777
  stream: useStreaming
1518
1778
  };
1519
1779
  const hasAdaptiveThinking = isAdaptiveThinkingModel(options.model);
@@ -1530,18 +1790,39 @@ async function* runStream(options) {
1530
1790
  // it Anthropic silently ignores ttl:"1h" and falls back to the 5-min default,
1531
1791
  // so a pre-warmed cache expires before the user's first turn. cacheControl.ttl
1532
1792
  // is only "1h" on the first-party endpoint (see toAnthropicCacheControl).
1533
- ...cacheControl?.ttl === "1h" ? ["extended-cache-ttl-2025-04-11"] : []
1793
+ ...cacheControl?.ttl === "1h" ? ["extended-cache-ttl-2025-04-11"] : [],
1794
+ ...useServerFallback ? [SERVER_FALLBACK_BETA] : []
1534
1795
  ];
1535
- const requestOptions = {
1796
+ const toRequestOptions = (betas) => ({
1536
1797
  signal: options.signal ?? void 0,
1537
- ...betaHeaders.length ? { headers: { "anthropic-beta": betaHeaders.join(",") } } : {}
1798
+ ...betas.length ? { headers: { "anthropic-beta": betas.join(",") } } : {}
1799
+ });
1800
+ const requestOptions = toRequestOptions(betaHeaders);
1801
+ const send = async (create) => {
1802
+ try {
1803
+ return await create(params, requestOptions);
1804
+ } catch (err) {
1805
+ if (!useServerFallback || !isServerFallbackRejection(err)) throw err;
1806
+ anthropicServerFallback.disabled.add(fallbackKey);
1807
+ const { fallbacks: _dropped, ...rest } = params;
1808
+ const retryParams = {
1809
+ ...rest,
1810
+ messages: toAnthropicMessages(downgradedMessages, cacheControl, { fallbackBlocks: false }).messages
1811
+ };
1812
+ return create(
1813
+ retryParams,
1814
+ toRequestOptions(betaHeaders.filter((b) => b !== SERVER_FALLBACK_BETA))
1815
+ );
1816
+ }
1538
1817
  };
1539
1818
  if (!useStreaming) {
1540
1819
  const nonStreamingClient = client.withOptions({ timeout: NON_STREAMING_TIMEOUT_MS });
1541
1820
  try {
1542
- const message = await nonStreamingClient.messages.create(
1543
- { ...params, stream: false },
1544
- requestOptions
1821
+ const message = await send(
1822
+ (p, o) => nonStreamingClient.messages.create(
1823
+ { ...p, stream: false },
1824
+ o
1825
+ )
1545
1826
  );
1546
1827
  yield* synthesizeEventsFromMessage(message);
1547
1828
  return messageToResponse(message);
@@ -1559,9 +1840,8 @@ async function* runStream(options) {
1559
1840
  const keepalive = { type: "keepalive" };
1560
1841
  let receivedAnyEvent = false;
1561
1842
  try {
1562
- const stream2 = await client.messages.create(
1563
- params,
1564
- requestOptions
1843
+ const stream2 = await send(
1844
+ (p, o) => client.messages.create(p, o)
1565
1845
  );
1566
1846
  for await (const event of stream2) {
1567
1847
  receivedAnyEvent = true;
@@ -1739,6 +2019,13 @@ async function* runStream(options) {
1739
2019
  if (usage?.output_tokens != null) {
1740
2020
  outputTokens = usage.output_tokens;
1741
2021
  }
2022
+ const totals = sumUsageIterations(usage);
2023
+ if (totals) {
2024
+ inputTokens = totals.inputTokens;
2025
+ outputTokens = totals.outputTokens;
2026
+ cacheRead = totals.cacheRead;
2027
+ cacheWrite = totals.cacheWrite;
2028
+ }
1742
2029
  yield keepalive;
1743
2030
  break;
1744
2031
  }
@@ -1868,10 +2155,11 @@ function messageToResponse(message) {
1868
2155
  }
1869
2156
  }
1870
2157
  const usage = message.usage;
1871
- const inputTokens = usage.input_tokens ?? 0;
1872
- const outputTokens = usage.output_tokens ?? 0;
1873
- const cacheRead = usage.cache_read_input_tokens;
1874
- const cacheWrite = usage.cache_creation_input_tokens;
2158
+ const totals = sumUsageIterations(usage);
2159
+ const inputTokens = totals?.inputTokens ?? usage.input_tokens ?? 0;
2160
+ const outputTokens = totals?.outputTokens ?? usage.output_tokens ?? 0;
2161
+ const cacheRead = totals?.cacheRead ?? usage.cache_read_input_tokens;
2162
+ const cacheWrite = totals?.cacheWrite ?? usage.cache_creation_input_tokens;
1875
2163
  return {
1876
2164
  message: {
1877
2165
  role: "assistant",
@@ -2141,7 +2429,7 @@ async function* runStream2(options) {
2141
2429
  ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 && !isLocal ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
2142
2430
  ...options.tools?.length ? {
2143
2431
  tools: toOpenAITools(options.tools, {
2144
- strict: supportsStrictToolSampling(options.provider)
2432
+ strict: supportsStrictToolSampling(options.provider) && (options.strictTools ?? true)
2145
2433
  })
2146
2434
  } : {},
2147
2435
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
@@ -2659,6 +2947,7 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2659
2947
  const downgraded = downgradeUnsupportedVideos(downgradedImages, options.supportsVideo);
2660
2948
  const { system, input } = toCodexInput(downgraded, { supportsImages: options.supportsImages });
2661
2949
  const responsesLite = usesResponsesLite(options.model);
2950
+ const liteShape = options.responsesLite ?? responsesLite;
2662
2951
  const body = {
2663
2952
  model: options.model,
2664
2953
  store: false,
@@ -2666,11 +2955,11 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2666
2955
  instructions: system,
2667
2956
  input,
2668
2957
  tool_choice: toCodexToolChoice(options.toolChoice, options.tools),
2669
- parallel_tool_calls: !responsesLite,
2958
+ parallel_tool_calls: !liteShape,
2670
2959
  include: ["reasoning.encrypted_content"]
2671
2960
  };
2672
2961
  if (options.tools?.length) {
2673
- body.tools = toCodexTools(options.tools);
2962
+ body.tools = toCodexTools(options.tools, options.strictTools ?? true);
2674
2963
  }
2675
2964
  body.prompt_cache_key = normalizePromptCacheKey(options.promptCacheKey ?? "ezcoder");
2676
2965
  if (options.temperature != null && !options.thinking) {
@@ -2682,7 +2971,7 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2682
2971
  // `ultra` is a client orchestration preset, not a Codex API effort.
2683
2972
  effort: options.thinking === "ultra" ? "max" : options.thinking ?? (responsesLite ? "low" : "none"),
2684
2973
  summary: "auto",
2685
- ...responsesLite ? { context: "all_turns" } : {}
2974
+ ...liteShape ? { context: "all_turns" } : {}
2686
2975
  };
2687
2976
  if (responsesLite) {
2688
2977
  body.text = { verbosity: "low" };
@@ -2694,10 +2983,8 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2694
2983
  "OpenAI-Beta": "responses=experimental",
2695
2984
  originator: responsesLite ? "codex_cli_rs" : "ezcoder",
2696
2985
  "User-Agent": responsesLite ? `codex_cli_rs/${CODEX_CLIENT_VERSION}` : `ezcoder (${import_node_os.default.platform()} ${import_node_os.default.release()}; ${import_node_os.default.arch()})`,
2697
- ...responsesLite ? {
2698
- version: CODEX_CLIENT_VERSION,
2699
- "X-OpenAI-Internal-Codex-Responses-Lite": "true"
2700
- } : {}
2986
+ ...responsesLite ? { version: CODEX_CLIENT_VERSION } : {},
2987
+ ...liteShape ? { "X-OpenAI-Internal-Codex-Responses-Lite": "true" } : {}
2701
2988
  };
2702
2989
  if (options.accountId) {
2703
2990
  headers["chatgpt-account-id"] = options.accountId;
@@ -3161,15 +3448,17 @@ function toCodexInput(messages, options) {
3161
3448
  }
3162
3449
  return { system, input };
3163
3450
  }
3164
- function toCodexTools(tools) {
3451
+ function toCodexTools(tools, strictTools) {
3165
3452
  return tools.map((tool) => {
3166
3453
  let parameters = resolveToolSchema(tool);
3167
3454
  let strict = null;
3168
- try {
3169
- parameters = makeStrictToolSchema(parameters);
3170
- strict = true;
3171
- } catch (error) {
3172
- if (!(error instanceof UnsupportedStrictSchemaError)) throw error;
3455
+ if (strictTools) {
3456
+ try {
3457
+ parameters = makeStrictToolSchema(parameters);
3458
+ strict = true;
3459
+ } catch (error) {
3460
+ if (!(error instanceof UnsupportedStrictSchemaError)) throw error;
3461
+ }
3173
3462
  }
3174
3463
  return {
3175
3464
  type: "function",
@@ -3406,6 +3695,17 @@ function stripUnsupportedSchemaFields(value) {
3406
3695
  }
3407
3696
  delete value.$schema;
3408
3697
  delete value.additionalProperties;
3698
+ for (const [key, target, step] of [
3699
+ ["exclusiveMinimum", "minimum", 1],
3700
+ ["exclusiveMaximum", "maximum", -1]
3701
+ ]) {
3702
+ const bound = value[key];
3703
+ if (bound === void 0) continue;
3704
+ delete value[key];
3705
+ if (typeof bound === "number" && value[target] === void 0) {
3706
+ value[target] = value.type === "integer" ? bound + step : bound;
3707
+ }
3708
+ }
3409
3709
  for (const item of Object.values(value)) {
3410
3710
  if (isJsonObject(item) || Array.isArray(item)) {
3411
3711
  stripUnsupportedSchemaFields(item);
@@ -3905,6 +4205,41 @@ function sanitizeMessagesForWire(messages) {
3905
4205
  return sanitized ?? messages;
3906
4206
  }
3907
4207
 
4208
+ // src/utils/context-observation.ts
4209
+ function imageCount(message) {
4210
+ if (!message || !Array.isArray(message.content)) return 0;
4211
+ if (message.role === "user")
4212
+ return message.content.filter((part) => part.type === "image").length;
4213
+ if (message.role !== "tool") return 0;
4214
+ return message.content.reduce(
4215
+ (count, result) => count + (Array.isArray(result.content) ? result.content.filter((part) => part.type === "image").length : 0),
4216
+ 0
4217
+ );
4218
+ }
4219
+ function observePreparedContext(before, after, tools) {
4220
+ let imagesBefore = 0;
4221
+ let imagesAfter = 0;
4222
+ let firstImageDropMessage = null;
4223
+ for (let index = 0; index < before.length; index++) {
4224
+ const oldCount = imageCount(before[index]);
4225
+ const newCount = imageCount(after[index]);
4226
+ imagesBefore += oldCount;
4227
+ imagesAfter += newCount;
4228
+ if (firstImageDropMessage === null && newCount < oldCount) firstImageDropMessage = index;
4229
+ }
4230
+ return {
4231
+ messages: after,
4232
+ tools: tools.map((tool) => ({
4233
+ name: tool.name,
4234
+ description: tool.description,
4235
+ parameters: resolveToolSchema(tool)
4236
+ })),
4237
+ imagesBefore,
4238
+ imagesAfter,
4239
+ firstImageDropMessage
4240
+ };
4241
+ }
4242
+
3908
4243
  // src/stream.ts
3909
4244
  var GLM_CODING_BASE_URL = "https://api.z.ai/api/coding/paas/v4";
3910
4245
  var KIMI_CODE_USER_AGENT = `kimi-code-cli/${process.env.KIMI_CODE_VERSION ?? "1.0.11"}`;
@@ -4053,11 +4388,27 @@ function stream(options) {
4053
4388
  throw new VideoUnsupportedError();
4054
4389
  }
4055
4390
  const wireMessages = stripMessageProvenance(options.messages);
4056
- const messages = clampProviderContextImages(
4057
- sanitizeMessagesForWire(wireMessages),
4058
- options.provider,
4059
- options.supportsImages
4391
+ const sanitized = sanitizeMessagesForWire(wireMessages);
4392
+ const replayable = dropInvalidToolCalls(
4393
+ sanitized,
4394
+ toolCallNameRuleFor(options.provider, {
4395
+ accountId: options.accountId,
4396
+ baseUrl: options.baseUrl
4397
+ })
4060
4398
  );
4399
+ const messages = clampProviderContextImages(replayable, options.provider, options.supportsImages);
4400
+ if (options.onContextPrepared) {
4401
+ try {
4402
+ options.onContextPrepared(
4403
+ observePreparedContext(
4404
+ replayable === sanitized ? wireMessages : replayable,
4405
+ messages,
4406
+ options.tools ?? []
4407
+ )
4408
+ );
4409
+ } catch {
4410
+ }
4411
+ }
4061
4412
  return entry.stream(messages === options.messages ? options : { ...options, messages });
4062
4413
  }
4063
4414
  function stripMessageProvenance(messages) {
@@ -4469,14 +4820,17 @@ function registerPalsuProvider(config) {
4469
4820
  ProviderError,
4470
4821
  REDACTION_MARKER,
4471
4822
  StreamResult,
4823
+ TOOL_CALL_NAME_RULES,
4472
4824
  clampProviderContextImages,
4473
4825
  classifyProviderError,
4826
+ dropInvalidToolCalls,
4474
4827
  environmentSecrets,
4475
4828
  formatError,
4476
4829
  formatErrorForDisplay,
4477
4830
  hasLoneSurrogate,
4478
4831
  isHardBillingMessage,
4479
4832
  isUsageLimitError,
4833
+ isValidToolCallName,
4480
4834
  localWireModelId,
4481
4835
  palsuAssistantMessage,
4482
4836
  palsuText,
@@ -4495,6 +4849,8 @@ function registerPalsuProvider(config) {
4495
4849
  stream,
4496
4850
  toAnthropicMessages,
4497
4851
  toOpenAIMessages,
4498
- toWellFormedText
4852
+ toWellFormedText,
4853
+ toolCallNameRuleFor,
4854
+ usesResponsesLite
4499
4855
  });
4500
4856
  //# sourceMappingURL=index.cjs.map