@prestyj/ai 5.28.0 → 5.29.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -35,14 +35,17 @@ __export(index_exports, {
35
35
  ProviderError: () => ProviderError,
36
36
  REDACTION_MARKER: () => REDACTED,
37
37
  StreamResult: () => StreamResult,
38
+ TOOL_CALL_NAME_RULES: () => TOOL_CALL_NAME_RULES,
38
39
  clampProviderContextImages: () => clampProviderContextImages,
39
40
  classifyProviderError: () => classifyProviderError,
41
+ dropInvalidToolCalls: () => dropInvalidToolCalls,
40
42
  environmentSecrets: () => environmentSecrets,
41
43
  formatError: () => formatError,
42
44
  formatErrorForDisplay: () => formatErrorForDisplay,
43
45
  hasLoneSurrogate: () => hasLoneSurrogate,
44
46
  isHardBillingMessage: () => isHardBillingMessage,
45
47
  isUsageLimitError: () => isUsageLimitError,
48
+ isValidToolCallName: () => isValidToolCallName,
46
49
  localWireModelId: () => localWireModelId,
47
50
  palsuAssistantMessage: () => palsuAssistantMessage,
48
51
  palsuText: () => palsuText,
@@ -61,7 +64,8 @@ __export(index_exports, {
61
64
  stream: () => stream,
62
65
  toAnthropicMessages: () => toAnthropicMessages,
63
66
  toOpenAIMessages: () => toOpenAIMessages,
64
- toWellFormedText: () => toWellFormedText
67
+ toWellFormedText: () => toWellFormedText,
68
+ toolCallNameRuleFor: () => toolCallNameRuleFor
65
69
  });
66
70
  module.exports = __toCommonJS(index_exports);
67
71
 
@@ -515,7 +519,11 @@ function zodToJsonSchema(schema) {
515
519
  return normalized;
516
520
  }
517
521
  function resolveToolSchema(tool) {
518
- return tool.rawInputSchema ?? zodToJsonSchema(tool.parameters);
522
+ const schema = tool.rawInputSchema ?? zodToJsonSchema(tool.parameters);
523
+ if (schema.type === "object" && schema.properties === void 0) {
524
+ return { ...schema, properties: {} };
525
+ }
526
+ return schema;
519
527
  }
520
528
  function normalizeRootForAnthropic(schema) {
521
529
  const branches = schema.oneOf ?? schema.anyOf;
@@ -722,6 +730,10 @@ var ANTHROPIC_INPUT_BLOCK_TYPES = /* @__PURE__ */ new Set([
722
730
  "connector_text",
723
731
  "container_upload",
724
732
  "document",
733
+ // Server-side refusal fallback marker. Only reaches the wire when the request
734
+ // carries the server-side-fallback beta; otherwise applyServerFallbackReplay
735
+ // strips it first (see toAnthropicMessages' `fallbackBlocks` option).
736
+ "fallback",
725
737
  "image",
726
738
  "mid_conv_system",
727
739
  "redacted_thinking",
@@ -743,6 +755,42 @@ function isPositionSensitiveThinking(part) {
743
755
  if (part.type === "thinking") return hasValidThinkingSignature(part);
744
756
  return isRawThinking(part);
745
757
  }
758
+ function isServerFallbackBlock(part) {
759
+ return part.type === "raw" && part.data.type === "fallback";
760
+ }
761
+ function applyServerFallbackReplay(content, keepMarkers, droppedToolCallIds) {
762
+ let lastFallbackIdx = -1;
763
+ content.forEach((part, idx) => {
764
+ if (isServerFallbackBlock(part)) lastFallbackIdx = idx;
765
+ });
766
+ if (lastFallbackIdx === -1) return content;
767
+ const resultIds = /* @__PURE__ */ new Set();
768
+ for (const part of content) {
769
+ if (part.type === "server_tool_result") resultIds.add(part.toolUseId);
770
+ else if (part.type === "raw" && typeof part.data.tool_use_id === "string")
771
+ resultIds.add(part.data.tool_use_id);
772
+ }
773
+ const out = [];
774
+ content.forEach((part, idx) => {
775
+ if (isServerFallbackBlock(part)) {
776
+ if (keepMarkers) out.push(part);
777
+ return;
778
+ }
779
+ if (idx < lastFallbackIdx) {
780
+ if (part.type === "thinking" || isRawThinking(part)) return;
781
+ if (part.type === "raw" && part.data.type === "connector_text") return;
782
+ if (part.type === "tool_call") {
783
+ droppedToolCallIds?.add(part.id);
784
+ return;
785
+ }
786
+ if (part.type === "server_tool_call" && !resultIds.has(part.id)) return;
787
+ if (part.type === "raw" && part.data.type === "server_tool_use" && !resultIds.has(String(part.data.id)))
788
+ return;
789
+ }
790
+ out.push(part);
791
+ });
792
+ return out.every(isServerFallbackBlock) ? [] : out;
793
+ }
746
794
  function toAnthropicAssistantPart(part, idMap) {
747
795
  if (part.type === "text") return { type: "text", text: part.text };
748
796
  if (part.type === "thinking") {
@@ -810,10 +858,17 @@ function countContextImages(messages) {
810
858
  }
811
859
  return count;
812
860
  }
861
+ var IMAGE_DROP_BATCH = 30;
862
+ function providerImageDropCount(imageCount2, budget) {
863
+ const overflow = imageCount2 - budget;
864
+ if (overflow <= 0) return 0;
865
+ const batch = Math.max(1, Math.min(IMAGE_DROP_BATCH, Math.floor(budget / 3)));
866
+ return Math.min(imageCount2, Math.ceil(overflow / batch) * batch);
867
+ }
813
868
  function clampProviderContextImages(messages, provider, supportsImages) {
814
869
  if (supportsImages === false) return messages;
815
870
  const budget = PROVIDER_IMAGE_BUDGETS[provider] ?? 5;
816
- let remainingToRemove = countContextImages(messages) - budget;
871
+ let remainingToRemove = providerImageDropCount(countContextImages(messages), budget);
817
872
  if (remainingToRemove <= 0) return messages;
818
873
  return messages.map((message) => {
819
874
  if (message.role === "user" && Array.isArray(message.content)) {
@@ -847,6 +902,144 @@ function clampProviderContextImages(messages, provider, supportsImages) {
847
902
  return message;
848
903
  });
849
904
  }
905
+ var TOOL_CALL_NAME_RULES = {
906
+ // Anthropic Messages API: tool / tool_use name `^[a-zA-Z0-9_-]{1,128}$`.
907
+ anthropic: { id: "anthropic", maxLength: 128, pattern: /^[a-zA-Z0-9_-]{1,128}$/ },
908
+ // OpenAI Chat Completions: function name a-z A-Z 0-9 _ -, max length 64.
909
+ "openai-chat": { id: "openai-chat", maxLength: 64, pattern: /^[a-zA-Z0-9_-]{1,64}$/ },
910
+ // OpenAI Responses (Codex): same charset; replayed `input[N].name` over 128
911
+ // chars is rejected with `string_above_max_length`.
912
+ "openai-responses": {
913
+ id: "openai-responses",
914
+ maxLength: 128,
915
+ pattern: /^[a-zA-Z0-9_-]{1,128}$/
916
+ },
917
+ // Gemini FunctionDeclaration name: starts with a letter or underscore, then
918
+ // a-z A-Z 0-9 _ . : -, max length 128.
919
+ gemini: { id: "gemini", maxLength: 128, pattern: /^[a-zA-Z_][a-zA-Z0-9_.:-]{0,127}$/ },
920
+ // OpenAI-compatible third parties (GLM, Kimi, DeepSeek, OpenRouter, xAI,
921
+ // local servers, MiniMax, …) don't share one documented charset, so only
922
+ // reject what no declared tool name can contain: blank, >128 chars, or any
923
+ // whitespace / control character (the signature of invocation text).
924
+ generic: { id: "generic", maxLength: 128 }
925
+ };
926
+ var TOOL_NAME_FORBIDDEN_CHARS = /[\s\p{Cc}]/u;
927
+ function isValidToolCallName(name, rule) {
928
+ if (typeof name !== "string" || name.trim().length === 0) return false;
929
+ if (name.length > rule.maxLength) return false;
930
+ if (TOOL_NAME_FORBIDDEN_CHARS.test(name)) return false;
931
+ return rule.pattern ? rule.pattern.test(name) : true;
932
+ }
933
+ function isDefaultOrHost(baseUrl, host) {
934
+ if (!baseUrl) return true;
935
+ try {
936
+ return new URL(baseUrl).hostname === host;
937
+ } catch {
938
+ return false;
939
+ }
940
+ }
941
+ function toolCallNameRuleFor(provider, options) {
942
+ switch (provider) {
943
+ case "anthropic":
944
+ return TOOL_CALL_NAME_RULES.anthropic;
945
+ case "openai":
946
+ if (options?.accountId) return TOOL_CALL_NAME_RULES["openai-responses"];
947
+ return isDefaultOrHost(options?.baseUrl, "api.openai.com") ? TOOL_CALL_NAME_RULES["openai-chat"] : TOOL_CALL_NAME_RULES.generic;
948
+ case "gemini":
949
+ return TOOL_CALL_NAME_RULES.gemini;
950
+ default:
951
+ return TOOL_CALL_NAME_RULES.generic;
952
+ }
953
+ }
954
+ function isValidToolCallId(id) {
955
+ return typeof id === "string" && id.trim().length > 0;
956
+ }
957
+ function isRawReasoning(part) {
958
+ return part.type === "raw" && part.data.type === "reasoning";
959
+ }
960
+ function isDanglingReasoning(parts, idx) {
961
+ for (let i = idx + 1; i < parts.length; i++) {
962
+ const next = parts[i];
963
+ if (isRawReasoning(next)) return true;
964
+ if (next.type === "text" || next.type === "tool_call") return false;
965
+ }
966
+ return true;
967
+ }
968
+ function hasReplayableAssistantContent(parts) {
969
+ return parts.some((part) => {
970
+ if (part.type === "text") return part.text.length > 0;
971
+ if (part.type === "thinking") return false;
972
+ if (part.type === "raw") {
973
+ const t = part.data.type;
974
+ return !(t === "thinking" || t === "redacted_thinking" || t === "reasoning" || t === "fallback");
975
+ }
976
+ return true;
977
+ });
978
+ }
979
+ function dropInvalidToolCalls(messages, rule) {
980
+ const isInvalid = (part) => part.type === "tool_call" && (!isValidToolCallName(part.name, rule) || !isValidToolCallId(part.id));
981
+ const hasInvalid = messages.some(
982
+ (m) => m.role === "assistant" && Array.isArray(m.content) && m.content.some(isInvalid)
983
+ );
984
+ if (!hasInvalid) return messages;
985
+ const out = [];
986
+ let pending = /* @__PURE__ */ new Map();
987
+ let prunedAssistant = null;
988
+ for (const msg of messages) {
989
+ if (msg.role === "user" || msg.role === "system") {
990
+ if (msg.role === "user") pending = /* @__PURE__ */ new Map();
991
+ out.push(msg);
992
+ continue;
993
+ }
994
+ if (msg.role === "assistant") {
995
+ pending = /* @__PURE__ */ new Map();
996
+ if (typeof msg.content === "string") {
997
+ out.push(msg);
998
+ continue;
999
+ }
1000
+ const original = msg.content;
1001
+ let dropped = false;
1002
+ const kept = [];
1003
+ for (const part of original) {
1004
+ if (part.type === "tool_call") {
1005
+ const keep = !isInvalid(part);
1006
+ const queue = pending.get(part.id) ?? [];
1007
+ queue.push(keep);
1008
+ pending.set(part.id, queue);
1009
+ if (!keep) {
1010
+ dropped = true;
1011
+ continue;
1012
+ }
1013
+ }
1014
+ kept.push(part);
1015
+ }
1016
+ if (!dropped) {
1017
+ out.push(msg);
1018
+ continue;
1019
+ }
1020
+ const content = kept.filter(
1021
+ (part, idx) => !isRawReasoning(part) || !isDanglingReasoning(kept, idx) || isDanglingReasoning(original, original.indexOf(part))
1022
+ );
1023
+ if (hasReplayableAssistantContent(content)) {
1024
+ const pruned = { ...msg, content };
1025
+ out.push(pruned);
1026
+ prunedAssistant = pruned;
1027
+ }
1028
+ continue;
1029
+ }
1030
+ let changed = false;
1031
+ const results = msg.content.filter((result) => {
1032
+ const queue = pending.get(result.toolCallId);
1033
+ const keep = queue && queue.length > 0 ? queue.shift() : true;
1034
+ if (!keep) changed = true;
1035
+ return keep;
1036
+ });
1037
+ if (!changed) out.push(msg);
1038
+ else if (results.length > 0) out.push({ ...msg, content: results });
1039
+ }
1040
+ if (prunedAssistant && out[out.length - 1] === prunedAssistant) out.pop();
1041
+ return out;
1042
+ }
850
1043
  var NON_VISION_USER_IMAGE_PLACEHOLDER = "(image omitted: model does not support images)";
851
1044
  var NON_VISION_TOOL_IMAGE_PLACEHOLDER = "(tool image omitted: model does not support images)";
852
1045
  var NON_VIDEO_USER_PLACEHOLDER = "(video omitted: model does not support video)";
@@ -953,10 +1146,12 @@ function remapAnthropicToolCallId(id, idMap) {
953
1146
  idMap.set(id, mapped);
954
1147
  return mapped;
955
1148
  }
956
- function toAnthropicMessages(messages, cacheControl) {
1149
+ function toAnthropicMessages(messages, cacheControl, options) {
957
1150
  let systemText;
958
1151
  const out = [];
959
1152
  const idMap = /* @__PURE__ */ new Map();
1153
+ const keepFallbackBlocks = options?.fallbackBlocks === true;
1154
+ const droppedToolCallIds = /* @__PURE__ */ new Set();
960
1155
  const trajectoryStartIdx = messages.reduce(
961
1156
  (last, m, i) => m.role === "user" ? i : last,
962
1157
  -1
@@ -1002,17 +1197,23 @@ function toAnthropicMessages(messages, cacheControl) {
1002
1197
  }
1003
1198
  if (msg.role === "assistant") {
1004
1199
  if (typeof msg.content === "string" && msg.content === "") continue;
1005
- const content = typeof msg.content === "string" ? msg.content : toAnthropicAssistantContent(msg.content, msgIdx > trajectoryStartIdx, idMap);
1200
+ const content = typeof msg.content === "string" ? msg.content : toAnthropicAssistantContent(
1201
+ applyServerFallbackReplay(msg.content, keepFallbackBlocks, droppedToolCallIds),
1202
+ msgIdx > trajectoryStartIdx,
1203
+ idMap
1204
+ );
1006
1205
  if (Array.isArray(content) && content.length === 0) continue;
1007
1206
  out.push({ role: "assistant", content });
1008
1207
  continue;
1009
1208
  }
1010
1209
  if (msg.role === "tool") {
1210
+ const results = droppedToolCallIds.size ? msg.content.filter((r) => !droppedToolCallIds.has(r.toolCallId)) : msg.content;
1211
+ if (results.length === 0) continue;
1011
1212
  out.push({
1012
1213
  role: "user",
1013
1214
  // Cast covers the video block (used by the Anthropic-compatible MiniMax
1014
1215
  // API), which isn't in the first-party Anthropic tool_result types.
1015
- content: msg.content.map((result) => ({
1216
+ content: results.map((result) => ({
1016
1217
  type: "tool_result",
1017
1218
  tool_use_id: remapAnthropicToolCallId(result.toolCallId, idMap),
1018
1219
  content: toAnthropicToolResultContent(result.content),
@@ -1356,6 +1557,48 @@ function fineGrainedToolStreamingEnabled() {
1356
1557
  const v = raw.trim().toLowerCase();
1357
1558
  return v === "1" || v === "true" || v === "yes" || v === "on";
1358
1559
  }
1560
+ var SERVER_FALLBACK_BETA = "server-side-fallback-2026-07-01";
1561
+ var anthropicServerFallback = {
1562
+ disabled: /* @__PURE__ */ new Set(),
1563
+ reset() {
1564
+ this.disabled.clear();
1565
+ }
1566
+ };
1567
+ function serverFallbackKey(options) {
1568
+ const auth = options.apiKey?.startsWith("sk-ant-oat") ? "oauth" : "key";
1569
+ return `${options.baseUrl ?? process.env.ANTHROPIC_BASE_URL ?? ""}|${auth}`;
1570
+ }
1571
+ function isDirectAnthropicApi(baseUrl) {
1572
+ const effective = baseUrl ?? process.env.ANTHROPIC_BASE_URL;
1573
+ if (!effective) return true;
1574
+ try {
1575
+ const url = new URL(effective);
1576
+ return url.protocol === "https:" && url.hostname === "api.anthropic.com";
1577
+ } catch {
1578
+ return false;
1579
+ }
1580
+ }
1581
+ function isServerFallbackRejection(err) {
1582
+ const status = err?.status;
1583
+ if (status !== 400) return false;
1584
+ const message = err instanceof Error ? err.message : String(err);
1585
+ return /fallbacks|server-side-fallback/i.test(message);
1586
+ }
1587
+ function sumUsageIterations(usage) {
1588
+ const iterations = usage?.iterations;
1589
+ if (!Array.isArray(iterations) || iterations.length === 0) return null;
1590
+ const num = (v) => typeof v === "number" && Number.isFinite(v) ? v : 0;
1591
+ const totals = { inputTokens: 0, outputTokens: 0, cacheRead: 0, cacheWrite: 0 };
1592
+ for (const it of iterations) {
1593
+ if (!it || typeof it !== "object") continue;
1594
+ const rec = it;
1595
+ totals.inputTokens += num(rec.input_tokens);
1596
+ totals.outputTokens += num(rec.output_tokens);
1597
+ totals.cacheRead += num(rec.cache_read_input_tokens);
1598
+ totals.cacheWrite += num(rec.cache_creation_input_tokens);
1599
+ }
1600
+ return totals;
1601
+ }
1359
1602
  function createClient(options) {
1360
1603
  const isOAuth = options.apiKey?.startsWith("sk-ant-oat");
1361
1604
  const userAgent = isOAuth ? options.userAgent ?? "claude-cli/2.1.75 (external, cli)" : "";
@@ -1450,12 +1693,16 @@ function streamAnthropic(options) {
1450
1693
  async function* runStream(options) {
1451
1694
  const client = createClient(options);
1452
1695
  const isOAuth = options.apiKey?.startsWith("sk-ant-oat");
1453
- const useStreaming = options.streaming !== false;
1696
+ const useStreaming = options.streaming !== false && !options.prewarm;
1454
1697
  const cacheControl = toAnthropicCacheControl(options.cacheRetention, options.baseUrl);
1455
1698
  const supportsFirstPartyToolExtras = !options.baseUrl || options.baseUrl.includes("api.anthropic.com");
1456
1699
  const downgradedImages = downgradeUnsupportedImages(options.messages, options.supportsImages);
1457
1700
  const downgradedMessages = downgradeUnsupportedVideos(downgradedImages, options.supportsVideo);
1458
- const { system: rawSystem, messages } = toAnthropicMessages(downgradedMessages, cacheControl);
1701
+ const fallbackKey = serverFallbackKey(options);
1702
+ const useServerFallback = options.provider !== "minimax" && isDirectAnthropicApi(options.baseUrl) && !anthropicServerFallback.disabled.has(fallbackKey);
1703
+ const { system: rawSystem, messages } = toAnthropicMessages(downgradedMessages, cacheControl, {
1704
+ fallbackBlocks: useServerFallback
1705
+ });
1459
1706
  const system = isOAuth ? [
1460
1707
  {
1461
1708
  type: "text",
@@ -1474,6 +1721,17 @@ async function* runStream(options) {
1474
1721
  outputConfig = t.outputConfig;
1475
1722
  }
1476
1723
  }
1724
+ if (options.prewarm) {
1725
+ const budget = thinking?.budget_tokens;
1726
+ if (budget != null && budget >= 1) {
1727
+ return {
1728
+ message: { role: "assistant", content: [] },
1729
+ stopReason: "end_turn",
1730
+ usage: { inputTokens: 0, outputTokens: 0 }
1731
+ };
1732
+ }
1733
+ maxTokens = 1;
1734
+ }
1477
1735
  const params = {
1478
1736
  model: options.model,
1479
1737
  max_tokens: maxTokens,
@@ -1514,6 +1772,7 @@ async function* runStream(options) {
1514
1772
  ];
1515
1773
  return contextEdits.length ? { context_management: { edits: contextEdits } } : {};
1516
1774
  })(),
1775
+ ...useServerFallback ? { fallbacks: "default" } : {},
1517
1776
  stream: useStreaming
1518
1777
  };
1519
1778
  const hasAdaptiveThinking = isAdaptiveThinkingModel(options.model);
@@ -1530,18 +1789,39 @@ async function* runStream(options) {
1530
1789
  // it Anthropic silently ignores ttl:"1h" and falls back to the 5-min default,
1531
1790
  // so a pre-warmed cache expires before the user's first turn. cacheControl.ttl
1532
1791
  // is only "1h" on the first-party endpoint (see toAnthropicCacheControl).
1533
- ...cacheControl?.ttl === "1h" ? ["extended-cache-ttl-2025-04-11"] : []
1792
+ ...cacheControl?.ttl === "1h" ? ["extended-cache-ttl-2025-04-11"] : [],
1793
+ ...useServerFallback ? [SERVER_FALLBACK_BETA] : []
1534
1794
  ];
1535
- const requestOptions = {
1795
+ const toRequestOptions = (betas) => ({
1536
1796
  signal: options.signal ?? void 0,
1537
- ...betaHeaders.length ? { headers: { "anthropic-beta": betaHeaders.join(",") } } : {}
1797
+ ...betas.length ? { headers: { "anthropic-beta": betas.join(",") } } : {}
1798
+ });
1799
+ const requestOptions = toRequestOptions(betaHeaders);
1800
+ const send = async (create) => {
1801
+ try {
1802
+ return await create(params, requestOptions);
1803
+ } catch (err) {
1804
+ if (!useServerFallback || !isServerFallbackRejection(err)) throw err;
1805
+ anthropicServerFallback.disabled.add(fallbackKey);
1806
+ const { fallbacks: _dropped, ...rest } = params;
1807
+ const retryParams = {
1808
+ ...rest,
1809
+ messages: toAnthropicMessages(downgradedMessages, cacheControl, { fallbackBlocks: false }).messages
1810
+ };
1811
+ return create(
1812
+ retryParams,
1813
+ toRequestOptions(betaHeaders.filter((b) => b !== SERVER_FALLBACK_BETA))
1814
+ );
1815
+ }
1538
1816
  };
1539
1817
  if (!useStreaming) {
1540
1818
  const nonStreamingClient = client.withOptions({ timeout: NON_STREAMING_TIMEOUT_MS });
1541
1819
  try {
1542
- const message = await nonStreamingClient.messages.create(
1543
- { ...params, stream: false },
1544
- requestOptions
1820
+ const message = await send(
1821
+ (p, o) => nonStreamingClient.messages.create(
1822
+ { ...p, stream: false },
1823
+ o
1824
+ )
1545
1825
  );
1546
1826
  yield* synthesizeEventsFromMessage(message);
1547
1827
  return messageToResponse(message);
@@ -1559,9 +1839,8 @@ async function* runStream(options) {
1559
1839
  const keepalive = { type: "keepalive" };
1560
1840
  let receivedAnyEvent = false;
1561
1841
  try {
1562
- const stream2 = await client.messages.create(
1563
- params,
1564
- requestOptions
1842
+ const stream2 = await send(
1843
+ (p, o) => client.messages.create(p, o)
1565
1844
  );
1566
1845
  for await (const event of stream2) {
1567
1846
  receivedAnyEvent = true;
@@ -1739,6 +2018,13 @@ async function* runStream(options) {
1739
2018
  if (usage?.output_tokens != null) {
1740
2019
  outputTokens = usage.output_tokens;
1741
2020
  }
2021
+ const totals = sumUsageIterations(usage);
2022
+ if (totals) {
2023
+ inputTokens = totals.inputTokens;
2024
+ outputTokens = totals.outputTokens;
2025
+ cacheRead = totals.cacheRead;
2026
+ cacheWrite = totals.cacheWrite;
2027
+ }
1742
2028
  yield keepalive;
1743
2029
  break;
1744
2030
  }
@@ -1868,10 +2154,11 @@ function messageToResponse(message) {
1868
2154
  }
1869
2155
  }
1870
2156
  const usage = message.usage;
1871
- const inputTokens = usage.input_tokens ?? 0;
1872
- const outputTokens = usage.output_tokens ?? 0;
1873
- const cacheRead = usage.cache_read_input_tokens;
1874
- const cacheWrite = usage.cache_creation_input_tokens;
2157
+ const totals = sumUsageIterations(usage);
2158
+ const inputTokens = totals?.inputTokens ?? usage.input_tokens ?? 0;
2159
+ const outputTokens = totals?.outputTokens ?? usage.output_tokens ?? 0;
2160
+ const cacheRead = totals?.cacheRead ?? usage.cache_read_input_tokens;
2161
+ const cacheWrite = totals?.cacheWrite ?? usage.cache_creation_input_tokens;
1875
2162
  return {
1876
2163
  message: {
1877
2164
  role: "assistant",
@@ -2577,7 +2864,7 @@ function extractRequestIdFromMessage(message) {
2577
2864
 
2578
2865
  // src/providers/openai-codex.ts
2579
2866
  var DEFAULT_BASE_URL = "https://chatgpt.com/backend-api";
2580
- var CODEX_CLIENT_VERSION = "0.155.1";
2867
+ var CODEX_CLIENT_VERSION = "0.159.1";
2581
2868
  var CODEX_REQUEST_COMPRESSION_MIN_BYTES = 16 * 1024;
2582
2869
  var zstdInitPromise;
2583
2870
  async function encodeCodexRequest(body) {
@@ -2623,7 +2910,7 @@ async function encodeCodexRequest(body) {
2623
2910
  }
2624
2911
  }
2625
2912
  function usesResponsesLite(model) {
2626
- return model.startsWith("gpt-5.6-") || model.startsWith("gpt-6-");
2913
+ return model.startsWith("gpt-5.6-") || model.startsWith("gpt-6-") || model.startsWith("gpt-6.");
2627
2914
  }
2628
2915
  function outputTextKey(itemId, contentIndex) {
2629
2916
  return `${itemId ?? ""}:${contentIndex ?? 0}`;
@@ -2745,7 +3032,7 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2745
3032
  if (response.status === 400 && message === `The '${options.model}' model is not supported when using Codex with a ChatGPT account.`) {
2746
3033
  hint = "This model is not available through your ChatGPT account. Switch to a model listed for OpenAI via the model selector, or check your ChatGPT usage limits.";
2747
3034
  } else if (response.status === 404 && text.includes("does not exist")) {
2748
- hint = "This model is not in OpenAI's current catalog for your ChatGPT account. Switch to GPT-6 Astra, GPT-6 Sol, or GPT-6 Luna via the model selector.";
3035
+ hint = "This model is not in OpenAI's current catalog for your ChatGPT account. Switch to GPT-6 Astra, GPT-6.1 Sol, or GPT-6 Luna via the model selector.";
2749
3036
  }
2750
3037
  throw new ProviderError("openai", message, {
2751
3038
  statusCode: response.status,
@@ -2759,6 +3046,8 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2759
3046
  const contentParts = [];
2760
3047
  let textAccum = "";
2761
3048
  const toolCalls = /* @__PURE__ */ new Map();
3049
+ const finishedToolCalls = /* @__PURE__ */ new Set();
3050
+ let terminal;
2762
3051
  const orderedItems = [];
2763
3052
  const outputItemTypes = /* @__PURE__ */ new Map();
2764
3053
  const outputTextByPart = /* @__PURE__ */ new Map();
@@ -2902,6 +3191,7 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2902
3191
  for (const [key, tc] of toolCalls) {
2903
3192
  if (key.endsWith(`|${itemId}`)) {
2904
3193
  tc.argsJson = argsStr;
3194
+ finishedToolCalls.add(key);
2905
3195
  break;
2906
3196
  }
2907
3197
  }
@@ -2927,6 +3217,7 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2927
3217
  const id = `${callId}|${itemId}`;
2928
3218
  const tc = toolCalls.get(id);
2929
3219
  if (tc) {
3220
+ finishedToolCalls.add(id);
2930
3221
  orderedItems.push({ kind: "tool", id });
2931
3222
  const args = parseToolArguments(tc.argsJson);
2932
3223
  yield {
@@ -2938,8 +3229,17 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2938
3229
  }
2939
3230
  }
2940
3231
  }
2941
- if (type === "response.completed" || type === "response.done") {
3232
+ if (type === "response.completed" || type === "response.done" || type === "response.incomplete") {
2942
3233
  const resp = event.response;
3234
+ if (type === "response.incomplete" || resp?.status === "incomplete") {
3235
+ const details = resp?.incomplete_details;
3236
+ terminal = {
3237
+ status: "incomplete",
3238
+ reason: typeof details?.reason === "string" ? details.reason : void 0
3239
+ };
3240
+ } else {
3241
+ terminal = { status: "completed" };
3242
+ }
2943
3243
  const usage = resp?.usage;
2944
3244
  if (usage) {
2945
3245
  cacheRead = usage.input_tokens_details?.cached_tokens ?? 0;
@@ -2949,6 +3249,25 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2949
3249
  }
2950
3250
  }
2951
3251
  }
3252
+ if (!terminal) {
3253
+ throw new ProviderError("openai", "Stream ended before completion (no response.completed).", {
3254
+ statusCode: 504
3255
+ });
3256
+ }
3257
+ const droppedToolCalls = [...toolCalls.keys()].filter((id) => !finishedToolCalls.has(id)).length;
3258
+ if (terminal.status === "completed") {
3259
+ for (const [id, tc] of toolCalls) {
3260
+ if (!finishedToolCalls.has(id)) {
3261
+ throw new ProviderError(
3262
+ "openai",
3263
+ `Codex reply completed with an unfinished tool call: ${tc.name} (${id}).`,
3264
+ { statusCode: 502 }
3265
+ );
3266
+ }
3267
+ }
3268
+ } else {
3269
+ providerDiag("codex_incomplete", { reason: terminal.reason ?? null, droppedToolCalls });
3270
+ }
2952
3271
  const seenTool = /* @__PURE__ */ new Set();
2953
3272
  let textInserted = false;
2954
3273
  for (const entry of orderedItems) {
@@ -2975,7 +3294,7 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2975
3294
  contentParts.push({ type: "text", text: textAccum });
2976
3295
  }
2977
3296
  for (const [id, tc] of toolCalls) {
2978
- if (seenTool.has(id)) continue;
3297
+ if (seenTool.has(id) || !finishedToolCalls.has(id)) continue;
2979
3298
  seenTool.add(id);
2980
3299
  contentParts.push({
2981
3300
  type: "tool_call",
@@ -2984,8 +3303,15 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2984
3303
  args: parseToolArguments(tc.argsJson)
2985
3304
  });
2986
3305
  }
3306
+ if (droppedToolCalls > 0) {
3307
+ let last = contentParts.at(-1);
3308
+ while (last?.type === "raw" && isEncryptedReasoning(last.data)) {
3309
+ contentParts.pop();
3310
+ last = contentParts.at(-1);
3311
+ }
3312
+ }
2987
3313
  const hasToolCalls = contentParts.some((p) => p.type === "tool_call");
2988
- const stopReason = hasToolCalls ? "tool_use" : "end_turn";
3314
+ const stopReason = terminal.status === "incomplete" ? incompleteStopReason(terminal.reason) : hasToolCalls ? "tool_use" : "end_turn";
2989
3315
  const streamResponse = {
2990
3316
  message: {
2991
3317
  role: "assistant",
@@ -3002,6 +3328,11 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
3002
3328
  yield { type: "done", stopReason };
3003
3329
  return streamResponse;
3004
3330
  }
3331
+ function incompleteStopReason(reason) {
3332
+ if (reason === "max_output_tokens") return "max_tokens";
3333
+ if (reason === "content_filter") return "refusal";
3334
+ return "error";
3335
+ }
3005
3336
  async function* parseSSE(body) {
3006
3337
  for await (const event of readSseStream(body)) {
3007
3338
  const data = event.data.trim();
@@ -3861,6 +4192,41 @@ function sanitizeMessagesForWire(messages) {
3861
4192
  return sanitized ?? messages;
3862
4193
  }
3863
4194
 
4195
+ // src/utils/context-observation.ts
4196
+ function imageCount(message) {
4197
+ if (!message || !Array.isArray(message.content)) return 0;
4198
+ if (message.role === "user")
4199
+ return message.content.filter((part) => part.type === "image").length;
4200
+ if (message.role !== "tool") return 0;
4201
+ return message.content.reduce(
4202
+ (count, result) => count + (Array.isArray(result.content) ? result.content.filter((part) => part.type === "image").length : 0),
4203
+ 0
4204
+ );
4205
+ }
4206
+ function observePreparedContext(before, after, tools) {
4207
+ let imagesBefore = 0;
4208
+ let imagesAfter = 0;
4209
+ let firstImageDropMessage = null;
4210
+ for (let index = 0; index < before.length; index++) {
4211
+ const oldCount = imageCount(before[index]);
4212
+ const newCount = imageCount(after[index]);
4213
+ imagesBefore += oldCount;
4214
+ imagesAfter += newCount;
4215
+ if (firstImageDropMessage === null && newCount < oldCount) firstImageDropMessage = index;
4216
+ }
4217
+ return {
4218
+ messages: after,
4219
+ tools: tools.map((tool) => ({
4220
+ name: tool.name,
4221
+ description: tool.description,
4222
+ parameters: resolveToolSchema(tool)
4223
+ })),
4224
+ imagesBefore,
4225
+ imagesAfter,
4226
+ firstImageDropMessage
4227
+ };
4228
+ }
4229
+
3864
4230
  // src/stream.ts
3865
4231
  var GLM_CODING_BASE_URL = "https://api.z.ai/api/coding/paas/v4";
3866
4232
  var KIMI_CODE_USER_AGENT = `kimi-code-cli/${process.env.KIMI_CODE_VERSION ?? "1.0.11"}`;
@@ -4009,11 +4375,27 @@ function stream(options) {
4009
4375
  throw new VideoUnsupportedError();
4010
4376
  }
4011
4377
  const wireMessages = stripMessageProvenance(options.messages);
4012
- const messages = clampProviderContextImages(
4013
- sanitizeMessagesForWire(wireMessages),
4014
- options.provider,
4015
- options.supportsImages
4378
+ const sanitized = sanitizeMessagesForWire(wireMessages);
4379
+ const replayable = dropInvalidToolCalls(
4380
+ sanitized,
4381
+ toolCallNameRuleFor(options.provider, {
4382
+ accountId: options.accountId,
4383
+ baseUrl: options.baseUrl
4384
+ })
4016
4385
  );
4386
+ const messages = clampProviderContextImages(replayable, options.provider, options.supportsImages);
4387
+ if (options.onContextPrepared) {
4388
+ try {
4389
+ options.onContextPrepared(
4390
+ observePreparedContext(
4391
+ replayable === sanitized ? wireMessages : replayable,
4392
+ messages,
4393
+ options.tools ?? []
4394
+ )
4395
+ );
4396
+ } catch {
4397
+ }
4398
+ }
4017
4399
  return entry.stream(messages === options.messages ? options : { ...options, messages });
4018
4400
  }
4019
4401
  function stripMessageProvenance(messages) {
@@ -4137,6 +4519,8 @@ var CIRCULAR = "[CIRCULAR]";
4137
4519
  var SENSITIVE_NAME = /(?:^|[_-])(?:api[_-]?key|access[_-]?token|refresh[_-]?token|token|key|auth(?:orization)?|bearer|cookie|credential|private[_-]?key|password|passwd|secret)(?:$|[_-])/i;
4138
4520
  var ENV_SECRET_ASSIGNMENT = /\b((?:[A-Z0-9]+_)*(?:API_?KEY|ACCESS_TOKEN|REFRESH_TOKEN|TOKEN|KEY|AUTH|AUTHORIZATION|BEARER|CREDENTIALS?|PASSWORD|PASSWD|SECRET))\b(\s*[=:]\s*)(["']?)(?!\$|process\.env|os\.environ|import\.meta|env\.)(?=[^\s,"';}]*\d)([^\s,"';}=$][^\s,"';}]{7,})\3/g;
4139
4521
  var COMPACT_SECRET_ASSIGNMENT = /\b((?:[a-z0-9]+[_-])*(?:api[_-]?key|access[_-]?token|refresh[_-]?token|token|auth|authorization|credentials?|password|passwd|secret)|(?:[a-z0-9]+[_-])+key)=(["']?)(?!\$)([^\s,"'&;}=][^\s,"'&;}]{7,})\2/gi;
4522
+ var URL_USERINFO = /\b([a-z][a-z0-9+.-]*:\/\/)[^\s/?#@:"'`<>]*:[^\s/?#"'`<>]+@/gi;
4523
+ var URL_USERINFO_TO_FIRST_AT = /\b([a-z][a-z0-9+.-]*:\/\/)[^\s/@:]*:[^\s/@]+@/gi;
4140
4524
  function escaped(value) {
4141
4525
  return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
4142
4526
  }
@@ -4160,7 +4544,8 @@ function redactText(text, options = {}) {
4160
4544
  /-----BEGIN (?:[A-Z0-9 ]+ )?PRIVATE KEY-----[\s\S]*?-----END (?:[A-Z0-9 ]+ )?PRIVATE KEY-----/g,
4161
4545
  REDACTED
4162
4546
  );
4163
- result = result.replace(/\b([a-z][a-z0-9+.-]*:\/\/)[^\s/@:]+:[^\s/@]+@/gi, `$1${REDACTED}@`);
4547
+ result = result.replace(URL_USERINFO, `$1${REDACTED}@`);
4548
+ result = result.replace(URL_USERINFO_TO_FIRST_AT, `$1${REDACTED}@`);
4164
4549
  result = result.replace(
4165
4550
  /\b(authorization\s*[:=]\s*)(?:bearer|basic)\s+[^\s,;]+/gi,
4166
4551
  `$1${REDACTED}`
@@ -4422,14 +4807,17 @@ function registerPalsuProvider(config) {
4422
4807
  ProviderError,
4423
4808
  REDACTION_MARKER,
4424
4809
  StreamResult,
4810
+ TOOL_CALL_NAME_RULES,
4425
4811
  clampProviderContextImages,
4426
4812
  classifyProviderError,
4813
+ dropInvalidToolCalls,
4427
4814
  environmentSecrets,
4428
4815
  formatError,
4429
4816
  formatErrorForDisplay,
4430
4817
  hasLoneSurrogate,
4431
4818
  isHardBillingMessage,
4432
4819
  isUsageLimitError,
4820
+ isValidToolCallName,
4433
4821
  localWireModelId,
4434
4822
  palsuAssistantMessage,
4435
4823
  palsuText,
@@ -4448,6 +4836,7 @@ function registerPalsuProvider(config) {
4448
4836
  stream,
4449
4837
  toAnthropicMessages,
4450
4838
  toOpenAIMessages,
4451
- toWellFormedText
4839
+ toWellFormedText,
4840
+ toolCallNameRuleFor
4452
4841
  });
4453
4842
  //# sourceMappingURL=index.cjs.map