@prestyj/ai 5.28.0 → 5.29.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -448,7 +448,11 @@ function zodToJsonSchema(schema) {
448
448
  return normalized;
449
449
  }
450
450
  function resolveToolSchema(tool) {
451
- return tool.rawInputSchema ?? zodToJsonSchema(tool.parameters);
451
+ const schema = tool.rawInputSchema ?? zodToJsonSchema(tool.parameters);
452
+ if (schema.type === "object" && schema.properties === void 0) {
453
+ return { ...schema, properties: {} };
454
+ }
455
+ return schema;
452
456
  }
453
457
  function normalizeRootForAnthropic(schema) {
454
458
  const branches = schema.oneOf ?? schema.anyOf;
@@ -655,6 +659,10 @@ var ANTHROPIC_INPUT_BLOCK_TYPES = /* @__PURE__ */ new Set([
655
659
  "connector_text",
656
660
  "container_upload",
657
661
  "document",
662
+ // Server-side refusal fallback marker. Only reaches the wire when the request
663
+ // carries the server-side-fallback beta; otherwise applyServerFallbackReplay
664
+ // strips it first (see toAnthropicMessages' `fallbackBlocks` option).
665
+ "fallback",
658
666
  "image",
659
667
  "mid_conv_system",
660
668
  "redacted_thinking",
@@ -676,6 +684,42 @@ function isPositionSensitiveThinking(part) {
676
684
  if (part.type === "thinking") return hasValidThinkingSignature(part);
677
685
  return isRawThinking(part);
678
686
  }
687
+ function isServerFallbackBlock(part) {
688
+ return part.type === "raw" && part.data.type === "fallback";
689
+ }
690
+ function applyServerFallbackReplay(content, keepMarkers, droppedToolCallIds) {
691
+ let lastFallbackIdx = -1;
692
+ content.forEach((part, idx) => {
693
+ if (isServerFallbackBlock(part)) lastFallbackIdx = idx;
694
+ });
695
+ if (lastFallbackIdx === -1) return content;
696
+ const resultIds = /* @__PURE__ */ new Set();
697
+ for (const part of content) {
698
+ if (part.type === "server_tool_result") resultIds.add(part.toolUseId);
699
+ else if (part.type === "raw" && typeof part.data.tool_use_id === "string")
700
+ resultIds.add(part.data.tool_use_id);
701
+ }
702
+ const out = [];
703
+ content.forEach((part, idx) => {
704
+ if (isServerFallbackBlock(part)) {
705
+ if (keepMarkers) out.push(part);
706
+ return;
707
+ }
708
+ if (idx < lastFallbackIdx) {
709
+ if (part.type === "thinking" || isRawThinking(part)) return;
710
+ if (part.type === "raw" && part.data.type === "connector_text") return;
711
+ if (part.type === "tool_call") {
712
+ droppedToolCallIds?.add(part.id);
713
+ return;
714
+ }
715
+ if (part.type === "server_tool_call" && !resultIds.has(part.id)) return;
716
+ if (part.type === "raw" && part.data.type === "server_tool_use" && !resultIds.has(String(part.data.id)))
717
+ return;
718
+ }
719
+ out.push(part);
720
+ });
721
+ return out.every(isServerFallbackBlock) ? [] : out;
722
+ }
679
723
  function toAnthropicAssistantPart(part, idMap) {
680
724
  if (part.type === "text") return { type: "text", text: part.text };
681
725
  if (part.type === "thinking") {
@@ -743,10 +787,17 @@ function countContextImages(messages) {
743
787
  }
744
788
  return count;
745
789
  }
790
+ var IMAGE_DROP_BATCH = 30;
791
+ function providerImageDropCount(imageCount2, budget) {
792
+ const overflow = imageCount2 - budget;
793
+ if (overflow <= 0) return 0;
794
+ const batch = Math.max(1, Math.min(IMAGE_DROP_BATCH, Math.floor(budget / 3)));
795
+ return Math.min(imageCount2, Math.ceil(overflow / batch) * batch);
796
+ }
746
797
  function clampProviderContextImages(messages, provider, supportsImages) {
747
798
  if (supportsImages === false) return messages;
748
799
  const budget = PROVIDER_IMAGE_BUDGETS[provider] ?? 5;
749
- let remainingToRemove = countContextImages(messages) - budget;
800
+ let remainingToRemove = providerImageDropCount(countContextImages(messages), budget);
750
801
  if (remainingToRemove <= 0) return messages;
751
802
  return messages.map((message) => {
752
803
  if (message.role === "user" && Array.isArray(message.content)) {
@@ -780,6 +831,144 @@ function clampProviderContextImages(messages, provider, supportsImages) {
780
831
  return message;
781
832
  });
782
833
  }
834
+ var TOOL_CALL_NAME_RULES = {
835
+ // Anthropic Messages API: tool / tool_use name `^[a-zA-Z0-9_-]{1,128}$`.
836
+ anthropic: { id: "anthropic", maxLength: 128, pattern: /^[a-zA-Z0-9_-]{1,128}$/ },
837
+ // OpenAI Chat Completions: function name a-z A-Z 0-9 _ -, max length 64.
838
+ "openai-chat": { id: "openai-chat", maxLength: 64, pattern: /^[a-zA-Z0-9_-]{1,64}$/ },
839
+ // OpenAI Responses (Codex): same charset; replayed `input[N].name` over 128
840
+ // chars is rejected with `string_above_max_length`.
841
+ "openai-responses": {
842
+ id: "openai-responses",
843
+ maxLength: 128,
844
+ pattern: /^[a-zA-Z0-9_-]{1,128}$/
845
+ },
846
+ // Gemini FunctionDeclaration name: starts with a letter or underscore, then
847
+ // a-z A-Z 0-9 _ . : -, max length 128.
848
+ gemini: { id: "gemini", maxLength: 128, pattern: /^[a-zA-Z_][a-zA-Z0-9_.:-]{0,127}$/ },
849
+ // OpenAI-compatible third parties (GLM, Kimi, DeepSeek, OpenRouter, xAI,
850
+ // local servers, MiniMax, …) don't share one documented charset, so only
851
+ // reject what no declared tool name can contain: blank, >128 chars, or any
852
+ // whitespace / control character (the signature of invocation text).
853
+ generic: { id: "generic", maxLength: 128 }
854
+ };
855
+ var TOOL_NAME_FORBIDDEN_CHARS = /[\s\p{Cc}]/u;
856
+ function isValidToolCallName(name, rule) {
857
+ if (typeof name !== "string" || name.trim().length === 0) return false;
858
+ if (name.length > rule.maxLength) return false;
859
+ if (TOOL_NAME_FORBIDDEN_CHARS.test(name)) return false;
860
+ return rule.pattern ? rule.pattern.test(name) : true;
861
+ }
862
+ function isDefaultOrHost(baseUrl, host) {
863
+ if (!baseUrl) return true;
864
+ try {
865
+ return new URL(baseUrl).hostname === host;
866
+ } catch {
867
+ return false;
868
+ }
869
+ }
870
+ function toolCallNameRuleFor(provider, options) {
871
+ switch (provider) {
872
+ case "anthropic":
873
+ return TOOL_CALL_NAME_RULES.anthropic;
874
+ case "openai":
875
+ if (options?.accountId) return TOOL_CALL_NAME_RULES["openai-responses"];
876
+ return isDefaultOrHost(options?.baseUrl, "api.openai.com") ? TOOL_CALL_NAME_RULES["openai-chat"] : TOOL_CALL_NAME_RULES.generic;
877
+ case "gemini":
878
+ return TOOL_CALL_NAME_RULES.gemini;
879
+ default:
880
+ return TOOL_CALL_NAME_RULES.generic;
881
+ }
882
+ }
883
+ function isValidToolCallId(id) {
884
+ return typeof id === "string" && id.trim().length > 0;
885
+ }
886
+ function isRawReasoning(part) {
887
+ return part.type === "raw" && part.data.type === "reasoning";
888
+ }
889
+ function isDanglingReasoning(parts, idx) {
890
+ for (let i = idx + 1; i < parts.length; i++) {
891
+ const next = parts[i];
892
+ if (isRawReasoning(next)) return true;
893
+ if (next.type === "text" || next.type === "tool_call") return false;
894
+ }
895
+ return true;
896
+ }
897
+ function hasReplayableAssistantContent(parts) {
898
+ return parts.some((part) => {
899
+ if (part.type === "text") return part.text.length > 0;
900
+ if (part.type === "thinking") return false;
901
+ if (part.type === "raw") {
902
+ const t = part.data.type;
903
+ return !(t === "thinking" || t === "redacted_thinking" || t === "reasoning" || t === "fallback");
904
+ }
905
+ return true;
906
+ });
907
+ }
908
+ function dropInvalidToolCalls(messages, rule) {
909
+ const isInvalid = (part) => part.type === "tool_call" && (!isValidToolCallName(part.name, rule) || !isValidToolCallId(part.id));
910
+ const hasInvalid = messages.some(
911
+ (m) => m.role === "assistant" && Array.isArray(m.content) && m.content.some(isInvalid)
912
+ );
913
+ if (!hasInvalid) return messages;
914
+ const out = [];
915
+ let pending = /* @__PURE__ */ new Map();
916
+ let prunedAssistant = null;
917
+ for (const msg of messages) {
918
+ if (msg.role === "user" || msg.role === "system") {
919
+ if (msg.role === "user") pending = /* @__PURE__ */ new Map();
920
+ out.push(msg);
921
+ continue;
922
+ }
923
+ if (msg.role === "assistant") {
924
+ pending = /* @__PURE__ */ new Map();
925
+ if (typeof msg.content === "string") {
926
+ out.push(msg);
927
+ continue;
928
+ }
929
+ const original = msg.content;
930
+ let dropped = false;
931
+ const kept = [];
932
+ for (const part of original) {
933
+ if (part.type === "tool_call") {
934
+ const keep = !isInvalid(part);
935
+ const queue = pending.get(part.id) ?? [];
936
+ queue.push(keep);
937
+ pending.set(part.id, queue);
938
+ if (!keep) {
939
+ dropped = true;
940
+ continue;
941
+ }
942
+ }
943
+ kept.push(part);
944
+ }
945
+ if (!dropped) {
946
+ out.push(msg);
947
+ continue;
948
+ }
949
+ const content = kept.filter(
950
+ (part, idx) => !isRawReasoning(part) || !isDanglingReasoning(kept, idx) || isDanglingReasoning(original, original.indexOf(part))
951
+ );
952
+ if (hasReplayableAssistantContent(content)) {
953
+ const pruned = { ...msg, content };
954
+ out.push(pruned);
955
+ prunedAssistant = pruned;
956
+ }
957
+ continue;
958
+ }
959
+ let changed = false;
960
+ const results = msg.content.filter((result) => {
961
+ const queue = pending.get(result.toolCallId);
962
+ const keep = queue && queue.length > 0 ? queue.shift() : true;
963
+ if (!keep) changed = true;
964
+ return keep;
965
+ });
966
+ if (!changed) out.push(msg);
967
+ else if (results.length > 0) out.push({ ...msg, content: results });
968
+ }
969
+ if (prunedAssistant && out[out.length - 1] === prunedAssistant) out.pop();
970
+ return out;
971
+ }
783
972
  var NON_VISION_USER_IMAGE_PLACEHOLDER = "(image omitted: model does not support images)";
784
973
  var NON_VISION_TOOL_IMAGE_PLACEHOLDER = "(tool image omitted: model does not support images)";
785
974
  var NON_VIDEO_USER_PLACEHOLDER = "(video omitted: model does not support video)";
@@ -886,10 +1075,12 @@ function remapAnthropicToolCallId(id, idMap) {
886
1075
  idMap.set(id, mapped);
887
1076
  return mapped;
888
1077
  }
889
- function toAnthropicMessages(messages, cacheControl) {
1078
+ function toAnthropicMessages(messages, cacheControl, options) {
890
1079
  let systemText;
891
1080
  const out = [];
892
1081
  const idMap = /* @__PURE__ */ new Map();
1082
+ const keepFallbackBlocks = options?.fallbackBlocks === true;
1083
+ const droppedToolCallIds = /* @__PURE__ */ new Set();
893
1084
  const trajectoryStartIdx = messages.reduce(
894
1085
  (last, m, i) => m.role === "user" ? i : last,
895
1086
  -1
@@ -935,17 +1126,23 @@ function toAnthropicMessages(messages, cacheControl) {
935
1126
  }
936
1127
  if (msg.role === "assistant") {
937
1128
  if (typeof msg.content === "string" && msg.content === "") continue;
938
- const content = typeof msg.content === "string" ? msg.content : toAnthropicAssistantContent(msg.content, msgIdx > trajectoryStartIdx, idMap);
1129
+ const content = typeof msg.content === "string" ? msg.content : toAnthropicAssistantContent(
1130
+ applyServerFallbackReplay(msg.content, keepFallbackBlocks, droppedToolCallIds),
1131
+ msgIdx > trajectoryStartIdx,
1132
+ idMap
1133
+ );
939
1134
  if (Array.isArray(content) && content.length === 0) continue;
940
1135
  out.push({ role: "assistant", content });
941
1136
  continue;
942
1137
  }
943
1138
  if (msg.role === "tool") {
1139
+ const results = droppedToolCallIds.size ? msg.content.filter((r) => !droppedToolCallIds.has(r.toolCallId)) : msg.content;
1140
+ if (results.length === 0) continue;
944
1141
  out.push({
945
1142
  role: "user",
946
1143
  // Cast covers the video block (used by the Anthropic-compatible MiniMax
947
1144
  // API), which isn't in the first-party Anthropic tool_result types.
948
- content: msg.content.map((result) => ({
1145
+ content: results.map((result) => ({
949
1146
  type: "tool_result",
950
1147
  tool_use_id: remapAnthropicToolCallId(result.toolCallId, idMap),
951
1148
  content: toAnthropicToolResultContent(result.content),
@@ -1289,6 +1486,48 @@ function fineGrainedToolStreamingEnabled() {
1289
1486
  const v = raw.trim().toLowerCase();
1290
1487
  return v === "1" || v === "true" || v === "yes" || v === "on";
1291
1488
  }
1489
+ var SERVER_FALLBACK_BETA = "server-side-fallback-2026-07-01";
1490
+ var anthropicServerFallback = {
1491
+ disabled: /* @__PURE__ */ new Set(),
1492
+ reset() {
1493
+ this.disabled.clear();
1494
+ }
1495
+ };
1496
+ function serverFallbackKey(options) {
1497
+ const auth = options.apiKey?.startsWith("sk-ant-oat") ? "oauth" : "key";
1498
+ return `${options.baseUrl ?? process.env.ANTHROPIC_BASE_URL ?? ""}|${auth}`;
1499
+ }
1500
+ function isDirectAnthropicApi(baseUrl) {
1501
+ const effective = baseUrl ?? process.env.ANTHROPIC_BASE_URL;
1502
+ if (!effective) return true;
1503
+ try {
1504
+ const url = new URL(effective);
1505
+ return url.protocol === "https:" && url.hostname === "api.anthropic.com";
1506
+ } catch {
1507
+ return false;
1508
+ }
1509
+ }
1510
+ function isServerFallbackRejection(err) {
1511
+ const status = err?.status;
1512
+ if (status !== 400) return false;
1513
+ const message = err instanceof Error ? err.message : String(err);
1514
+ return /fallbacks|server-side-fallback/i.test(message);
1515
+ }
1516
+ function sumUsageIterations(usage) {
1517
+ const iterations = usage?.iterations;
1518
+ if (!Array.isArray(iterations) || iterations.length === 0) return null;
1519
+ const num = (v) => typeof v === "number" && Number.isFinite(v) ? v : 0;
1520
+ const totals = { inputTokens: 0, outputTokens: 0, cacheRead: 0, cacheWrite: 0 };
1521
+ for (const it of iterations) {
1522
+ if (!it || typeof it !== "object") continue;
1523
+ const rec = it;
1524
+ totals.inputTokens += num(rec.input_tokens);
1525
+ totals.outputTokens += num(rec.output_tokens);
1526
+ totals.cacheRead += num(rec.cache_read_input_tokens);
1527
+ totals.cacheWrite += num(rec.cache_creation_input_tokens);
1528
+ }
1529
+ return totals;
1530
+ }
1292
1531
  function createClient(options) {
1293
1532
  const isOAuth = options.apiKey?.startsWith("sk-ant-oat");
1294
1533
  const userAgent = isOAuth ? options.userAgent ?? "claude-cli/2.1.75 (external, cli)" : "";
@@ -1383,12 +1622,16 @@ function streamAnthropic(options) {
1383
1622
  async function* runStream(options) {
1384
1623
  const client = createClient(options);
1385
1624
  const isOAuth = options.apiKey?.startsWith("sk-ant-oat");
1386
- const useStreaming = options.streaming !== false;
1625
+ const useStreaming = options.streaming !== false && !options.prewarm;
1387
1626
  const cacheControl = toAnthropicCacheControl(options.cacheRetention, options.baseUrl);
1388
1627
  const supportsFirstPartyToolExtras = !options.baseUrl || options.baseUrl.includes("api.anthropic.com");
1389
1628
  const downgradedImages = downgradeUnsupportedImages(options.messages, options.supportsImages);
1390
1629
  const downgradedMessages = downgradeUnsupportedVideos(downgradedImages, options.supportsVideo);
1391
- const { system: rawSystem, messages } = toAnthropicMessages(downgradedMessages, cacheControl);
1630
+ const fallbackKey = serverFallbackKey(options);
1631
+ const useServerFallback = options.provider !== "minimax" && isDirectAnthropicApi(options.baseUrl) && !anthropicServerFallback.disabled.has(fallbackKey);
1632
+ const { system: rawSystem, messages } = toAnthropicMessages(downgradedMessages, cacheControl, {
1633
+ fallbackBlocks: useServerFallback
1634
+ });
1392
1635
  const system = isOAuth ? [
1393
1636
  {
1394
1637
  type: "text",
@@ -1407,6 +1650,17 @@ async function* runStream(options) {
1407
1650
  outputConfig = t.outputConfig;
1408
1651
  }
1409
1652
  }
1653
+ if (options.prewarm) {
1654
+ const budget = thinking?.budget_tokens;
1655
+ if (budget != null && budget >= 1) {
1656
+ return {
1657
+ message: { role: "assistant", content: [] },
1658
+ stopReason: "end_turn",
1659
+ usage: { inputTokens: 0, outputTokens: 0 }
1660
+ };
1661
+ }
1662
+ maxTokens = 1;
1663
+ }
1410
1664
  const params = {
1411
1665
  model: options.model,
1412
1666
  max_tokens: maxTokens,
@@ -1447,6 +1701,7 @@ async function* runStream(options) {
1447
1701
  ];
1448
1702
  return contextEdits.length ? { context_management: { edits: contextEdits } } : {};
1449
1703
  })(),
1704
+ ...useServerFallback ? { fallbacks: "default" } : {},
1450
1705
  stream: useStreaming
1451
1706
  };
1452
1707
  const hasAdaptiveThinking = isAdaptiveThinkingModel(options.model);
@@ -1463,18 +1718,39 @@ async function* runStream(options) {
1463
1718
  // it Anthropic silently ignores ttl:"1h" and falls back to the 5-min default,
1464
1719
  // so a pre-warmed cache expires before the user's first turn. cacheControl.ttl
1465
1720
  // is only "1h" on the first-party endpoint (see toAnthropicCacheControl).
1466
- ...cacheControl?.ttl === "1h" ? ["extended-cache-ttl-2025-04-11"] : []
1721
+ ...cacheControl?.ttl === "1h" ? ["extended-cache-ttl-2025-04-11"] : [],
1722
+ ...useServerFallback ? [SERVER_FALLBACK_BETA] : []
1467
1723
  ];
1468
- const requestOptions = {
1724
+ const toRequestOptions = (betas) => ({
1469
1725
  signal: options.signal ?? void 0,
1470
- ...betaHeaders.length ? { headers: { "anthropic-beta": betaHeaders.join(",") } } : {}
1726
+ ...betas.length ? { headers: { "anthropic-beta": betas.join(",") } } : {}
1727
+ });
1728
+ const requestOptions = toRequestOptions(betaHeaders);
1729
+ const send = async (create) => {
1730
+ try {
1731
+ return await create(params, requestOptions);
1732
+ } catch (err) {
1733
+ if (!useServerFallback || !isServerFallbackRejection(err)) throw err;
1734
+ anthropicServerFallback.disabled.add(fallbackKey);
1735
+ const { fallbacks: _dropped, ...rest } = params;
1736
+ const retryParams = {
1737
+ ...rest,
1738
+ messages: toAnthropicMessages(downgradedMessages, cacheControl, { fallbackBlocks: false }).messages
1739
+ };
1740
+ return create(
1741
+ retryParams,
1742
+ toRequestOptions(betaHeaders.filter((b) => b !== SERVER_FALLBACK_BETA))
1743
+ );
1744
+ }
1471
1745
  };
1472
1746
  if (!useStreaming) {
1473
1747
  const nonStreamingClient = client.withOptions({ timeout: NON_STREAMING_TIMEOUT_MS });
1474
1748
  try {
1475
- const message = await nonStreamingClient.messages.create(
1476
- { ...params, stream: false },
1477
- requestOptions
1749
+ const message = await send(
1750
+ (p, o) => nonStreamingClient.messages.create(
1751
+ { ...p, stream: false },
1752
+ o
1753
+ )
1478
1754
  );
1479
1755
  yield* synthesizeEventsFromMessage(message);
1480
1756
  return messageToResponse(message);
@@ -1492,9 +1768,8 @@ async function* runStream(options) {
1492
1768
  const keepalive = { type: "keepalive" };
1493
1769
  let receivedAnyEvent = false;
1494
1770
  try {
1495
- const stream2 = await client.messages.create(
1496
- params,
1497
- requestOptions
1771
+ const stream2 = await send(
1772
+ (p, o) => client.messages.create(p, o)
1498
1773
  );
1499
1774
  for await (const event of stream2) {
1500
1775
  receivedAnyEvent = true;
@@ -1672,6 +1947,13 @@ async function* runStream(options) {
1672
1947
  if (usage?.output_tokens != null) {
1673
1948
  outputTokens = usage.output_tokens;
1674
1949
  }
1950
+ const totals = sumUsageIterations(usage);
1951
+ if (totals) {
1952
+ inputTokens = totals.inputTokens;
1953
+ outputTokens = totals.outputTokens;
1954
+ cacheRead = totals.cacheRead;
1955
+ cacheWrite = totals.cacheWrite;
1956
+ }
1675
1957
  yield keepalive;
1676
1958
  break;
1677
1959
  }
@@ -1801,10 +2083,11 @@ function messageToResponse(message) {
1801
2083
  }
1802
2084
  }
1803
2085
  const usage = message.usage;
1804
- const inputTokens = usage.input_tokens ?? 0;
1805
- const outputTokens = usage.output_tokens ?? 0;
1806
- const cacheRead = usage.cache_read_input_tokens;
1807
- const cacheWrite = usage.cache_creation_input_tokens;
2086
+ const totals = sumUsageIterations(usage);
2087
+ const inputTokens = totals?.inputTokens ?? usage.input_tokens ?? 0;
2088
+ const outputTokens = totals?.outputTokens ?? usage.output_tokens ?? 0;
2089
+ const cacheRead = totals?.cacheRead ?? usage.cache_read_input_tokens;
2090
+ const cacheWrite = totals?.cacheWrite ?? usage.cache_creation_input_tokens;
1808
2091
  return {
1809
2092
  message: {
1810
2093
  role: "assistant",
@@ -2510,7 +2793,7 @@ function extractRequestIdFromMessage(message) {
2510
2793
 
2511
2794
  // src/providers/openai-codex.ts
2512
2795
  var DEFAULT_BASE_URL = "https://chatgpt.com/backend-api";
2513
- var CODEX_CLIENT_VERSION = "0.155.1";
2796
+ var CODEX_CLIENT_VERSION = "0.159.1";
2514
2797
  var CODEX_REQUEST_COMPRESSION_MIN_BYTES = 16 * 1024;
2515
2798
  var zstdInitPromise;
2516
2799
  async function encodeCodexRequest(body) {
@@ -2556,7 +2839,7 @@ async function encodeCodexRequest(body) {
2556
2839
  }
2557
2840
  }
2558
2841
  function usesResponsesLite(model) {
2559
- return model.startsWith("gpt-5.6-") || model.startsWith("gpt-6-");
2842
+ return model.startsWith("gpt-5.6-") || model.startsWith("gpt-6-") || model.startsWith("gpt-6.");
2560
2843
  }
2561
2844
  function outputTextKey(itemId, contentIndex) {
2562
2845
  return `${itemId ?? ""}:${contentIndex ?? 0}`;
@@ -2678,7 +2961,7 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2678
2961
  if (response.status === 400 && message === `The '${options.model}' model is not supported when using Codex with a ChatGPT account.`) {
2679
2962
  hint = "This model is not available through your ChatGPT account. Switch to a model listed for OpenAI via the model selector, or check your ChatGPT usage limits.";
2680
2963
  } else if (response.status === 404 && text.includes("does not exist")) {
2681
- hint = "This model is not in OpenAI's current catalog for your ChatGPT account. Switch to GPT-6 Astra, GPT-6 Sol, or GPT-6 Luna via the model selector.";
2964
+ hint = "This model is not in OpenAI's current catalog for your ChatGPT account. Switch to GPT-6 Astra, GPT-6.1 Sol, or GPT-6 Luna via the model selector.";
2682
2965
  }
2683
2966
  throw new ProviderError("openai", message, {
2684
2967
  statusCode: response.status,
@@ -2692,6 +2975,8 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2692
2975
  const contentParts = [];
2693
2976
  let textAccum = "";
2694
2977
  const toolCalls = /* @__PURE__ */ new Map();
2978
+ const finishedToolCalls = /* @__PURE__ */ new Set();
2979
+ let terminal;
2695
2980
  const orderedItems = [];
2696
2981
  const outputItemTypes = /* @__PURE__ */ new Map();
2697
2982
  const outputTextByPart = /* @__PURE__ */ new Map();
@@ -2835,6 +3120,7 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2835
3120
  for (const [key, tc] of toolCalls) {
2836
3121
  if (key.endsWith(`|${itemId}`)) {
2837
3122
  tc.argsJson = argsStr;
3123
+ finishedToolCalls.add(key);
2838
3124
  break;
2839
3125
  }
2840
3126
  }
@@ -2860,6 +3146,7 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2860
3146
  const id = `${callId}|${itemId}`;
2861
3147
  const tc = toolCalls.get(id);
2862
3148
  if (tc) {
3149
+ finishedToolCalls.add(id);
2863
3150
  orderedItems.push({ kind: "tool", id });
2864
3151
  const args = parseToolArguments(tc.argsJson);
2865
3152
  yield {
@@ -2871,8 +3158,17 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2871
3158
  }
2872
3159
  }
2873
3160
  }
2874
- if (type === "response.completed" || type === "response.done") {
3161
+ if (type === "response.completed" || type === "response.done" || type === "response.incomplete") {
2875
3162
  const resp = event.response;
3163
+ if (type === "response.incomplete" || resp?.status === "incomplete") {
3164
+ const details = resp?.incomplete_details;
3165
+ terminal = {
3166
+ status: "incomplete",
3167
+ reason: typeof details?.reason === "string" ? details.reason : void 0
3168
+ };
3169
+ } else {
3170
+ terminal = { status: "completed" };
3171
+ }
2876
3172
  const usage = resp?.usage;
2877
3173
  if (usage) {
2878
3174
  cacheRead = usage.input_tokens_details?.cached_tokens ?? 0;
@@ -2882,6 +3178,25 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2882
3178
  }
2883
3179
  }
2884
3180
  }
3181
+ if (!terminal) {
3182
+ throw new ProviderError("openai", "Stream ended before completion (no response.completed).", {
3183
+ statusCode: 504
3184
+ });
3185
+ }
3186
+ const droppedToolCalls = [...toolCalls.keys()].filter((id) => !finishedToolCalls.has(id)).length;
3187
+ if (terminal.status === "completed") {
3188
+ for (const [id, tc] of toolCalls) {
3189
+ if (!finishedToolCalls.has(id)) {
3190
+ throw new ProviderError(
3191
+ "openai",
3192
+ `Codex reply completed with an unfinished tool call: ${tc.name} (${id}).`,
3193
+ { statusCode: 502 }
3194
+ );
3195
+ }
3196
+ }
3197
+ } else {
3198
+ providerDiag("codex_incomplete", { reason: terminal.reason ?? null, droppedToolCalls });
3199
+ }
2885
3200
  const seenTool = /* @__PURE__ */ new Set();
2886
3201
  let textInserted = false;
2887
3202
  for (const entry of orderedItems) {
@@ -2908,7 +3223,7 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2908
3223
  contentParts.push({ type: "text", text: textAccum });
2909
3224
  }
2910
3225
  for (const [id, tc] of toolCalls) {
2911
- if (seenTool.has(id)) continue;
3226
+ if (seenTool.has(id) || !finishedToolCalls.has(id)) continue;
2912
3227
  seenTool.add(id);
2913
3228
  contentParts.push({
2914
3229
  type: "tool_call",
@@ -2917,8 +3232,15 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2917
3232
  args: parseToolArguments(tc.argsJson)
2918
3233
  });
2919
3234
  }
3235
+ if (droppedToolCalls > 0) {
3236
+ let last = contentParts.at(-1);
3237
+ while (last?.type === "raw" && isEncryptedReasoning(last.data)) {
3238
+ contentParts.pop();
3239
+ last = contentParts.at(-1);
3240
+ }
3241
+ }
2920
3242
  const hasToolCalls = contentParts.some((p) => p.type === "tool_call");
2921
- const stopReason = hasToolCalls ? "tool_use" : "end_turn";
3243
+ const stopReason = terminal.status === "incomplete" ? incompleteStopReason(terminal.reason) : hasToolCalls ? "tool_use" : "end_turn";
2922
3244
  const streamResponse = {
2923
3245
  message: {
2924
3246
  role: "assistant",
@@ -2935,6 +3257,11 @@ async function* runStream3(options, retriedWithoutReasoning = false) {
2935
3257
  yield { type: "done", stopReason };
2936
3258
  return streamResponse;
2937
3259
  }
3260
+ function incompleteStopReason(reason) {
3261
+ if (reason === "max_output_tokens") return "max_tokens";
3262
+ if (reason === "content_filter") return "refusal";
3263
+ return "error";
3264
+ }
2938
3265
  async function* parseSSE(body) {
2939
3266
  for await (const event of readSseStream(body)) {
2940
3267
  const data = event.data.trim();
@@ -3794,6 +4121,41 @@ function sanitizeMessagesForWire(messages) {
3794
4121
  return sanitized ?? messages;
3795
4122
  }
3796
4123
 
4124
+ // src/utils/context-observation.ts
4125
+ function imageCount(message) {
4126
+ if (!message || !Array.isArray(message.content)) return 0;
4127
+ if (message.role === "user")
4128
+ return message.content.filter((part) => part.type === "image").length;
4129
+ if (message.role !== "tool") return 0;
4130
+ return message.content.reduce(
4131
+ (count, result) => count + (Array.isArray(result.content) ? result.content.filter((part) => part.type === "image").length : 0),
4132
+ 0
4133
+ );
4134
+ }
4135
+ function observePreparedContext(before, after, tools) {
4136
+ let imagesBefore = 0;
4137
+ let imagesAfter = 0;
4138
+ let firstImageDropMessage = null;
4139
+ for (let index = 0; index < before.length; index++) {
4140
+ const oldCount = imageCount(before[index]);
4141
+ const newCount = imageCount(after[index]);
4142
+ imagesBefore += oldCount;
4143
+ imagesAfter += newCount;
4144
+ if (firstImageDropMessage === null && newCount < oldCount) firstImageDropMessage = index;
4145
+ }
4146
+ return {
4147
+ messages: after,
4148
+ tools: tools.map((tool) => ({
4149
+ name: tool.name,
4150
+ description: tool.description,
4151
+ parameters: resolveToolSchema(tool)
4152
+ })),
4153
+ imagesBefore,
4154
+ imagesAfter,
4155
+ firstImageDropMessage
4156
+ };
4157
+ }
4158
+
3797
4159
  // src/stream.ts
3798
4160
  var GLM_CODING_BASE_URL = "https://api.z.ai/api/coding/paas/v4";
3799
4161
  var KIMI_CODE_USER_AGENT = `kimi-code-cli/${process.env.KIMI_CODE_VERSION ?? "1.0.11"}`;
@@ -3942,11 +4304,27 @@ function stream(options) {
3942
4304
  throw new VideoUnsupportedError();
3943
4305
  }
3944
4306
  const wireMessages = stripMessageProvenance(options.messages);
3945
- const messages = clampProviderContextImages(
3946
- sanitizeMessagesForWire(wireMessages),
3947
- options.provider,
3948
- options.supportsImages
4307
+ const sanitized = sanitizeMessagesForWire(wireMessages);
4308
+ const replayable = dropInvalidToolCalls(
4309
+ sanitized,
4310
+ toolCallNameRuleFor(options.provider, {
4311
+ accountId: options.accountId,
4312
+ baseUrl: options.baseUrl
4313
+ })
3949
4314
  );
4315
+ const messages = clampProviderContextImages(replayable, options.provider, options.supportsImages);
4316
+ if (options.onContextPrepared) {
4317
+ try {
4318
+ options.onContextPrepared(
4319
+ observePreparedContext(
4320
+ replayable === sanitized ? wireMessages : replayable,
4321
+ messages,
4322
+ options.tools ?? []
4323
+ )
4324
+ );
4325
+ } catch {
4326
+ }
4327
+ }
3950
4328
  return entry.stream(messages === options.messages ? options : { ...options, messages });
3951
4329
  }
3952
4330
  function stripMessageProvenance(messages) {
@@ -4070,6 +4448,8 @@ var CIRCULAR = "[CIRCULAR]";
4070
4448
  var SENSITIVE_NAME = /(?:^|[_-])(?:api[_-]?key|access[_-]?token|refresh[_-]?token|token|key|auth(?:orization)?|bearer|cookie|credential|private[_-]?key|password|passwd|secret)(?:$|[_-])/i;
4071
4449
  var ENV_SECRET_ASSIGNMENT = /\b((?:[A-Z0-9]+_)*(?:API_?KEY|ACCESS_TOKEN|REFRESH_TOKEN|TOKEN|KEY|AUTH|AUTHORIZATION|BEARER|CREDENTIALS?|PASSWORD|PASSWD|SECRET))\b(\s*[=:]\s*)(["']?)(?!\$|process\.env|os\.environ|import\.meta|env\.)(?=[^\s,"';}]*\d)([^\s,"';}=$][^\s,"';}]{7,})\3/g;
4072
4450
  var COMPACT_SECRET_ASSIGNMENT = /\b((?:[a-z0-9]+[_-])*(?:api[_-]?key|access[_-]?token|refresh[_-]?token|token|auth|authorization|credentials?|password|passwd|secret)|(?:[a-z0-9]+[_-])+key)=(["']?)(?!\$)([^\s,"'&;}=][^\s,"'&;}]{7,})\2/gi;
4451
+ var URL_USERINFO = /\b([a-z][a-z0-9+.-]*:\/\/)[^\s/?#@:"'`<>]*:[^\s/?#"'`<>]+@/gi;
4452
+ var URL_USERINFO_TO_FIRST_AT = /\b([a-z][a-z0-9+.-]*:\/\/)[^\s/@:]*:[^\s/@]+@/gi;
4073
4453
  function escaped(value) {
4074
4454
  return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
4075
4455
  }
@@ -4093,7 +4473,8 @@ function redactText(text, options = {}) {
4093
4473
  /-----BEGIN (?:[A-Z0-9 ]+ )?PRIVATE KEY-----[\s\S]*?-----END (?:[A-Z0-9 ]+ )?PRIVATE KEY-----/g,
4094
4474
  REDACTED
4095
4475
  );
4096
- result = result.replace(/\b([a-z][a-z0-9+.-]*:\/\/)[^\s/@:]+:[^\s/@]+@/gi, `$1${REDACTED}@`);
4476
+ result = result.replace(URL_USERINFO, `$1${REDACTED}@`);
4477
+ result = result.replace(URL_USERINFO_TO_FIRST_AT, `$1${REDACTED}@`);
4097
4478
  result = result.replace(
4098
4479
  /\b(authorization\s*[:=]\s*)(?:bearer|basic)\s+[^\s,;]+/gi,
4099
4480
  `$1${REDACTED}`
@@ -4354,14 +4735,17 @@ export {
4354
4735
  ProviderError,
4355
4736
  REDACTED as REDACTION_MARKER,
4356
4737
  StreamResult,
4738
+ TOOL_CALL_NAME_RULES,
4357
4739
  clampProviderContextImages,
4358
4740
  classifyProviderError,
4741
+ dropInvalidToolCalls,
4359
4742
  environmentSecrets,
4360
4743
  formatError,
4361
4744
  formatErrorForDisplay,
4362
4745
  hasLoneSurrogate,
4363
4746
  isHardBillingMessage,
4364
4747
  isUsageLimitError,
4748
+ isValidToolCallName,
4365
4749
  localWireModelId,
4366
4750
  palsuAssistantMessage,
4367
4751
  palsuText,
@@ -4380,6 +4764,7 @@ export {
4380
4764
  stream,
4381
4765
  toAnthropicMessages,
4382
4766
  toOpenAIMessages,
4383
- toWellFormedText
4767
+ toWellFormedText,
4768
+ toolCallNameRuleFor
4384
4769
  };
4385
4770
  //# sourceMappingURL=index.js.map