maxpool 1.5.63 → 1.5.64

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/package.json +1 -1
  2. package/src/server.js +190 -12
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.63",
3
+ "version": "1.5.64",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
package/src/server.js CHANGED
@@ -403,6 +403,13 @@ async function forwardRequest(
403
403
  ) {
404
404
  const configuredAttempts = Number(retryConfig.maxAttemptsPerRequest) || accountManager.accounts.length;
405
405
  const maxAttempts = Math.max(1, configuredAttempts);
406
+ // A body REPAIR (strip a block, convert a tool pair, downgrade effort) is not an
407
+ // account failover — it re-sends a FIXED body and consumes no account. Charging both
408
+ // to one budget sized by ACCOUNT COUNT meant a 1-account fleet could never repair
409
+ // anything and a 2-account fleet got exactly one repair, while the chain is now six
410
+ // deep. Each repair latches its own flag, so this budget is a backstop, not the bound.
411
+ const repairCount = requestInfo.repairCount || 0;
412
+ const canRepairBody = repairCount < 6 && !res.headersSent;
406
413
 
407
414
  // PRE-STRIP a session already known to carry provider-authored thinking. The client
408
415
  // resends the whole poisoned history every turn, so without this each turn pays another
@@ -1076,7 +1083,8 @@ async function forwardRequest(
1076
1083
  // text + tool_use preserved. Tried once per request (thinkingStripped guard); if it
1077
1084
  // still fails, the provider pin below is the fallback.
1078
1085
  if (isSignatureRejection && !requestInfo.thinkingStripped
1079
- && canRetryBufferedBody && retryCount + 1 < maxAttempts && !res.headersSent) {
1086
+ && canRetryBufferedBody && canRepairBody) {
1087
+ console.log(`[Maxpool] Anthropic rejected a block: ${describeRejectedBlock(body, errorBody)}`);
1080
1088
  const { body: cleanBody, removed, converted } = stripForeignThinkingBlocks(body);
1081
1089
  if (cleanBody) {
1082
1090
  // Latch it so EVERY later turn is stripped up front instead of re-paying this
@@ -1085,7 +1093,35 @@ async function forwardRequest(
1085
1093
  console.log(`[Maxpool] Recovering session on Claude: stripped ${removed} provider thinking block(s), converted ${converted} provider search block(s) to text`);
1086
1094
  return forwardRequest(
1087
1095
  req, res, cleanBody, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir,
1088
- retryConfig, queueConfig, { ...requestInfo, thinkingStripped: true },
1096
+ retryConfig, queueConfig, { ...requestInfo, thinkingStripped: true, repairCount: repairCount + 1 },
1097
+ canRetryBufferedBody, canQueueBufferedBody, excludedIndexes,
1098
+ );
1099
+ }
1100
+ }
1101
+
1102
+ // COORDINATE REPAIR (runs after the broad strip found nothing, or found the wrong
1103
+ // thing). Anthropic pointed at an exact block index; trust that over our own shape
1104
+ // model. Its own flag, so it still fires on a request whose broad strip already ran.
1105
+ if (isSignatureRejection && !requestInfo.rejectedBlockStripped && canRetryBufferedBody && canRepairBody) {
1106
+ const { body: fixedBody, removed, type } = stripRejectedBlockClass(body, errorBody);
1107
+ if (fixedBody) {
1108
+ // Latch the session ONLY for the class the pre-strip can actually repair up
1109
+ // front. `stripForeignThinkingBlocks` never touches `redacted_thinking`, so
1110
+ // latching on it would make every later turn pay a rejected round-trip that
1111
+ // the pre-strip cannot prevent — and would mislabel the give-up message as a
1112
+ // GLM/Kimi story. Same reason `thinkingStripped` is set only for `thinking`:
1113
+ // setting it here would bar the broad strip on the retry.
1114
+ const preStripCanRepeat = type === 'thinking';
1115
+ if (preStripCanRepeat) accountManager.markSessionThinkingContaminated?.(requestInfo.sessionKey);
1116
+ console.log(`[Maxpool] Recovering session on Claude: Anthropic rejected a "${type}" block by index; removed ${removed} block(s) of that type`);
1117
+ return forwardRequest(
1118
+ req, res, fixedBody, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir,
1119
+ retryConfig, queueConfig, {
1120
+ ...requestInfo,
1121
+ rejectedBlockStripped: true,
1122
+ thinkingStripped: preStripCanRepeat || requestInfo.thinkingStripped,
1123
+ repairCount: repairCount + 1,
1124
+ },
1089
1125
  canRetryBufferedBody, canQueueBufferedBody, excludedIndexes,
1090
1126
  );
1091
1127
  }
@@ -1147,9 +1183,24 @@ async function forwardRequest(
1147
1183
  // Say only what ACTUALLY happened. `thinkingStripped` is set only when the strip
1148
1184
  // ran; without it we found nothing provider-shaped, so claiming we stripped —
1149
1185
  // or blaming GLM/Kimi — would misdirect the user.
1150
- const what = requestInfo.thinkingStripped
1151
- ? 'This session ran on GLM/Kimi earlier, and Anthropic will not accept parts of what they wrote. Maxpool repaired what it could and retried on Claude, but Anthropic still rejected the history.'
1152
- : "Anthropic rejected part of this session's history that maxpool could not repair automatically.";
1186
+ // Log the shape on the way out: this is the ONE place a give-up is observable,
1187
+ // and without it the two surviving explanations (a block the strip cannot see
1188
+ // vs. a body over the retry buffer) are indistinguishable in the log.
1189
+ console.log(`[Maxpool] Unrepaired signature 400: ${describeRejectedBlock(body, errorBody)} bufferable=${canRetryBufferedBody} stripped=${!!requestInfo.thinkingStripped}`);
1190
+ // A body over the retry buffer bars EVERY repair above without a word — the user
1191
+ // then sees "could not repair" for a transcript maxpool never even tried to fix.
1192
+ // Name that separately so the reason is actionable rather than mysterious.
1193
+ // `peek` (not the full class-strip) because on THIS path the body may be the very
1194
+ // one just declared too large to rewrite — a full parse+rebuild there costs 15ms
1195
+ // and a discarded 4.8MB Buffer on a 9.6MB body, to read one string.
1196
+ const rejectedType = canRetryBufferedBody ? peekRejectedBlockType(body, errorBody) : null;
1197
+ const what = !canRetryBufferedBody
1198
+ ? `This session's history is too large for maxpool to rewrite automatically (over ${Math.round(retryConfig.maxRetryBufferBytes / (1024 * 1024))}MB). Run /compact and it will keep going.`
1199
+ : rejectedType && rejectedType !== 'thinking' && rejectedType !== 'redacted_thinking'
1200
+ ? `Anthropic rejected a "${rejectedType}" block in this session's history, which maxpool cannot remove without losing conversation content.`
1201
+ : requestInfo.thinkingStripped || requestInfo.rejectedBlockStripped
1202
+ ? 'This session ran on GLM/Kimi earlier, and Anthropic will not accept parts of what they wrote. Maxpool repaired what it could and retried on Claude, but Anthropic still rejected the history.'
1203
+ : "Anthropic rejected part of this session's history that maxpool could not repair automatically.";
1153
1204
  const hint = provs.length === 0
1154
1205
  ? ''
1155
1206
  : provs.every(a => a.enabled === false)
@@ -1160,7 +1211,12 @@ async function forwardRequest(
1160
1211
  type: 'error',
1161
1212
  error: {
1162
1213
  type: 'invalid_request_error',
1163
- message: `Start a new session to keep working — this one cannot continue. ${what}${hint}`,
1214
+ // The lead depends on whether the session is actually RECOVERABLE. Telling a
1215
+ // user "this one cannot continue" and then "run /compact and it will keep
1216
+ // going" is two mutually exclusive remedies in one sentence.
1217
+ message: canRetryBufferedBody
1218
+ ? `Start a new session to keep working — this one cannot continue. ${what}${hint}`
1219
+ : `${what}${hint}`,
1164
1220
  },
1165
1221
  });
1166
1222
  return;
@@ -1520,7 +1576,7 @@ function isContextLengthError(errorBody) {
1520
1576
  return /exceeded model token limit|maximum context length|context length exceeded|context window (?:size )?(?:exceeded|too)|prompt is too long|input is too long|reduce the length of|too many (?:input )?tokens|request too large/i.test(errorBody);
1521
1577
  }
1522
1578
 
1523
- export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, classifyEffortRejection, repairEffort, isCapacitySignalStatus, isStrippableThinkingBlock, stripForeignThinkingBlocks, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, isContextLengthError, streamResponse, startIdleRequestReaper };
1579
+ export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, classifyEffortRejection, repairEffort, isCapacitySignalStatus, isStrippableThinkingBlock, stripForeignThinkingBlocks, parseRejectedBlockPath, stripRejectedBlockClass, peekRejectedBlockType, describeRejectedBlock, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, isContextLengthError, streamResponse, startIdleRequestReaper };
1524
1580
 
1525
1581
  async function readErrorBody(upstreamRes, limitBytes = 64 * 1024) {
1526
1582
  if (!upstreamRes.body) return '';
@@ -1708,6 +1764,119 @@ function isStrippableThinkingBlock(block) {
1708
1764
  return block?.type === 'thinking';
1709
1765
  }
1710
1766
 
1767
+ /**
1768
+ * Anthropic's signature 400 names the EXACT block it rejected:
1769
+ * "messages.29.content.58: Invalid `signature` in `thinking` block"
1770
+ * That coordinate is GROUND TRUTH. Every other repair here depends on maxpool's own
1771
+ * model of what a provider-authored block looks like, and that model is what silently
1772
+ * failed — the role gate above meant `stripForeignThinkingBlocks` returned "nothing to
1773
+ * remove" for a block Anthropic had just pointed at by index.
1774
+ *
1775
+ * Returns { mi, ci } or null.
1776
+ */
1777
+ function parseRejectedBlockPath(errorBody) {
1778
+ const s = String(errorBody || '');
1779
+ const m = /messages\.(\d+)\.content\.(\d+)/.exec(s);
1780
+ if (!m) return null;
1781
+ // A NESTED path — `messages.29.content.58.content.3` — points INSIDE the block at
1782
+ // [58], not at it. Taking the outer coordinate would name the wrong block: the user
1783
+ // would be told a "tool_result" was rejected when a thinking block nested in it is
1784
+ // the real culprit, and a class-strip keyed on that type would be wrong too.
1785
+ if (/^\.content\./.test(s.slice(m.index + m[0].length))) return null;
1786
+ return { mi: Number(m[1]), ci: Number(m[2]) };
1787
+ }
1788
+
1789
+ /**
1790
+ * The rejected block's TYPE only — no parse-and-rebuild. Used on the give-up path,
1791
+ * where the body may be the very one we just declared too large to rewrite (measured:
1792
+ * a full stripRejectedBlockClass on a 9.6MB body costs 15ms and allocates a 4.8MB
1793
+ * Buffer that is discarded, because only `.type` is ever read).
1794
+ */
1795
+ function peekRejectedBlockType(body, errorBody) {
1796
+ const path = parseRejectedBlockPath(errorBody);
1797
+ if (!path) return null;
1798
+ try {
1799
+ const json = JSON.parse(Buffer.isBuffer(body) ? body.toString('utf8') : String(body));
1800
+ return json?.messages?.[path.mi]?.content?.[path.ci]?.type || null;
1801
+ } catch {
1802
+ return null;
1803
+ }
1804
+ }
1805
+
1806
+ /**
1807
+ * One line naming exactly what Anthropic rejected: coordinate, role, block type, and
1808
+ * whether the body was even parseable. Without it, the two remaining explanations for a
1809
+ * silent give-up (a block shape the strip cannot see vs. a body over the retry buffer)
1810
+ * are indistinguishable in the log — and once the repair starts working, the successful
1811
+ * path logs the same line as the already-working one, so the question becomes
1812
+ * unanswerable. Types and roles only; no transcript content.
1813
+ */
1814
+ function describeRejectedBlock(body, errorBody) {
1815
+ const path = parseRejectedBlockPath(errorBody);
1816
+ if (!path) return 'coordinate=unparsed';
1817
+ try {
1818
+ const json = JSON.parse(Buffer.isBuffer(body) ? body.toString('utf8') : String(body));
1819
+ const msg = json?.messages?.[path.mi];
1820
+ const block = msg?.content?.[path.ci];
1821
+ return `coordinate=messages.${path.mi}.content.${path.ci} role=${msg?.role ?? 'MISSING'} type=${block?.type ?? 'MISSING'} blocks=${Array.isArray(msg?.content) ? msg.content.length : 'n/a'}`;
1822
+ } catch {
1823
+ return `coordinate=messages.${path.mi}.content.${path.ci} body=UNPARSEABLE`;
1824
+ }
1825
+ }
1826
+
1827
+ /**
1828
+ * Last-resort repair driven by the upstream's own coordinate, for a rejected block no
1829
+ * shape heuristic here recognised. Removes every block sharing the rejected block's
1830
+ * TYPE, on any role — fixing the whole class in ONE round-trip rather than replaying
1831
+ * once per bad block (a 47-block transcript would otherwise cost 47 rejected requests).
1832
+ *
1833
+ * Restricted to the thinking family on purpose: `text` / `tool_use` / `tool_result`
1834
+ * carry conversation content and tool pairing, so removing them would corrupt the
1835
+ * transcript rather than repair it. A rejected block outside that family returns null
1836
+ * and the 400 surfaces with its real cause intact.
1837
+ *
1838
+ * Returns { body, removed, type } — `body` is null when nothing was safe to remove.
1839
+ */
1840
+ function stripRejectedBlockClass(body, errorBody) {
1841
+ const path = parseRejectedBlockPath(errorBody);
1842
+ if (!path) return { body: null, removed: 0, type: null };
1843
+ try {
1844
+ const json = JSON.parse(Buffer.isBuffer(body) ? body.toString('utf8') : String(body));
1845
+ if (!Array.isArray(json?.messages)) return { body: null, removed: 0, type: null };
1846
+ const target = json.messages[path.mi]?.content?.[path.ci];
1847
+ const type = target?.type;
1848
+ // Only the thinking family is safe to drop wholesale (verified 2026-07-25: a
1849
+ // history with thinking blocks removed replays 200 OK, text + tool_use preserved).
1850
+ if (type !== 'thinking' && type !== 'redacted_thinking') {
1851
+ return { body: null, removed: 0, type: type || null };
1852
+ }
1853
+ let removed = 0;
1854
+ const messages = [];
1855
+ for (const msg of json.messages) {
1856
+ if (!Array.isArray(msg?.content)) { messages.push(msg); continue; }
1857
+ const kept = msg.content.filter(b => {
1858
+ if (b?.type !== type) return true;
1859
+ removed++;
1860
+ return false;
1861
+ });
1862
+ if (kept.length === msg.content.length) { messages.push(msg); continue; }
1863
+ // A turn stripping empties is DROPPED — an empty content array is itself invalid,
1864
+ // and keeping the original would resend the exact body that just 400'd. Except
1865
+ // messages[0], which must survive as a `user` turn (see the same guard above).
1866
+ if (kept.length === 0) {
1867
+ if (messages.length === 0) { messages.push({ ...msg, content: [{ type: 'text', text: '(content removed)' }] }); }
1868
+ continue;
1869
+ }
1870
+ messages.push({ ...msg, content: kept });
1871
+ }
1872
+ if (!removed) return { body: null, removed: 0, type };
1873
+ json.messages = messages;
1874
+ return { body: Buffer.from(JSON.stringify(json)), removed, type };
1875
+ } catch {
1876
+ return { body: null, removed: 0, type: null };
1877
+ }
1878
+ }
1879
+
1711
1880
  /**
1712
1881
  * Recovery for a provider-contaminated transcript: drop the assistant `thinking` /
1713
1882
  * `redacted_thinking` blocks whose signature Anthropic can't validate, so the session
@@ -1847,10 +2016,12 @@ function stripForeignThinkingBlocks(body) {
1847
2016
  // assistant turn but its result can be carried on the following user turn.
1848
2017
  const tools = convertForeignServerTools(msg.content, foreignToolIds);
1849
2018
  converted += tools.converted;
1850
- if (msg.role !== 'assistant') {
1851
- messages.push(tools.converted ? { ...msg, content: tools.content } : msg);
1852
- continue;
1853
- }
2019
+ // Thinking blocks are stripped on EVERY role, not just `assistant`. Anthropic
2020
+ // validates the signature wherever the block sits, so a role gate here made the
2021
+ // repair silently find NOTHING to remove — which left `thinkingStripped` false,
2022
+ // barred the session latch, and surfaced the 400 to the user as "maxpool could
2023
+ // not repair automatically" while healthy Claude accounts sat idle. Measured
2024
+ // 2026-08-06: 315 signature 400s, the broad strip finding nothing on a subset.
1854
2025
  let localRemoved = 0;
1855
2026
  const kept = tools.content.filter(block => {
1856
2027
  if (!isStrippableThinkingBlock(block)) return true;
@@ -1866,7 +2037,14 @@ function stripForeignThinkingBlocks(body) {
1866
2037
  // leaving it would resend the exact body that just 400'd while reporting success
1867
2038
  // and burning the single recovery attempt. Verified against the live API — the
1868
2039
  // resulting consecutive user messages are accepted (200 OK).
1869
- if (kept.length === 0) continue;
2040
+ // EXCEPT messages[0]: Anthropic requires the first message to be role `user`, so
2041
+ // dropping it leaves an `assistant`-first transcript that is rejected outright.
2042
+ // Reachable only since the role gate was removed — before that, non-assistant
2043
+ // turns were never dropped at all.
2044
+ if (kept.length === 0) {
2045
+ if (messages.length === 0) { messages.push({ ...msg, content: [{ type: 'text', text: '(content removed)' }] }); }
2046
+ continue;
2047
+ }
1870
2048
  messages.push({ ...msg, content: kept });
1871
2049
  }
1872
2050
  if (!removed && !converted) return { body: null, removed: 0, converted: 0 };