maxpool 1.5.48 → 1.5.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.48",
3
+ "version": "1.5.49",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -1893,6 +1893,23 @@ export class AccountManager {
1893
1893
  console.log(`[Maxpool] Session "${sessionKey}" repaired — Claude routes re-enabled`);
1894
1894
  }
1895
1895
 
1896
+ /** Remember the effort level a MODEL actually accepted for this session, so later turns
1897
+ * apply it up front. The client resends the same rejected setting every turn, so without
1898
+ * this each turn pays a rejected round-trip AND charges a consecutive-failure to whichever
1899
+ * account ate it — deprioritising a perfectly healthy account for a client-side setting.
1900
+ * Keyed by model: opus-4-5 takes 'high' while opus-4-1 takes no effort at all. */
1901
+ markSessionEffort(sessionKey, model, effort) {
1902
+ if (!sessionKey || !model) return;
1903
+ const existing = this.sessionPolicies.get(sessionKey) || {};
1904
+ this.sessionPolicies.set(sessionKey, { ...existing, effortFix: { model, effort } });
1905
+ }
1906
+
1907
+ getSessionEffort(sessionKey, model) {
1908
+ if (!sessionKey || !model) return undefined;
1909
+ const fix = this.sessionPolicies.get(sessionKey)?.effortFix;
1910
+ return fix && fix.model === model ? fix : undefined;
1911
+ }
1912
+
1896
1913
  isSessionThinkingContaminated(sessionKey) {
1897
1914
  if (!sessionKey) return false;
1898
1915
  return Boolean(this.sessionPolicies.get(sessionKey)?.thinkingContaminated);
package/src/server.js CHANGED
@@ -357,6 +357,20 @@ async function forwardRequest(
357
357
  // PRE-STRIP a session already known to carry provider-authored thinking. The client
358
358
  // resends the whole poisoned history every turn, so without this each turn pays another
359
359
  // rejected round-trip before the reactive repair kicks in. Latched by the first repair.
360
+ // Apply the effort level this session's model already proved it accepts, so later turns
361
+ // skip the rejected round-trip entirely (the client resends the same setting every turn).
362
+ if (retryCount === 0 && !requestInfo.effortRepaired) {
363
+ const latched = accountManager.getSessionEffort?.(requestInfo.sessionKey, requestInfo.model);
364
+ if (latched) {
365
+ const pre = repairEffort(body, latched.effort ? 'downgrade' : 'drop',
366
+ latched.effort ? `Supported levels: ${latched.effort}` : '');
367
+ if (pre.body) {
368
+ body = pre.body;
369
+ requestInfo = { ...requestInfo, effortRepaired: true };
370
+ }
371
+ }
372
+ }
373
+
360
374
  // ALSO repairs a transcript maxpool predicts Anthropic will reject (a provider web
361
375
  // search). That prediction happens BEFORE any request is sent and bars every Claude
362
376
  // account, so the reactive repair in the 4xx handler could never be reached for it —
@@ -936,12 +950,27 @@ async function forwardRequest(
936
950
  // '^srvtoolu_…'`. Repairable by converting the pair to text (verified 200 OK) —
937
951
  // it used to fall through to a PERMANENT provider pin.
938
952
  || /server_tool_use\.id: String should match pattern/i.test(errorBody));
953
+ // LOG THE ACTUAL REASON. Previously a 4xx recorded only "HTTP 400" and the upstream
954
+ // message was never written anywhere, so a whole class of failures (e.g. a rejected
955
+ // effort level breaking every web search) was invisible in the log — you could not
956
+ // even grep for it. Truncated so a huge validation dump can't wall the file.
957
+ if (upstreamRes.status >= 400 && upstreamRes.status !== 429) {
958
+ const why = (() => {
959
+ try { return JSON.parse(errorBody)?.error?.message || errorBody; } catch { return errorBody; }
960
+ })();
961
+ console.log(`[Maxpool] ${upstreamRes.status} from "${account.name}": ${String(why).slice(0, 300)}`);
962
+ }
939
963
  const errorType = errorBody.includes('Invalid `signature` in `thinking` block')
940
964
  ? 'invalid_thinking_signature'
941
965
  : anthropicIncompat ? 'anthropic_incompatible_transcript'
942
966
  : providerTooSmall ? 'provider_context_too_small'
943
967
  : `HTTP ${upstreamRes.status}`;
944
- accountManager.releaseAccount(lease, { status: upstreamRes.status, error: errorType });
968
+ const effortMode = classifyEffortRejection(errorBody);
969
+ // A rejected effort level is a REQUEST-shaped fault, not an account-health signal —
970
+ // release neutral so it never charges a consecutive-failure to a healthy account.
971
+ accountManager.releaseAccount(lease, effortMode
972
+ ? { status: upstreamRes.status, error: errorType, neutral: true }
973
+ : { status: upstreamRes.status, error: errorType });
945
974
 
946
975
  if (logDir) {
947
976
  logSections.push(`=== RESPONSE ${upstreamRes.status} — non-retryable client error from "${account.name}" ===\n${formatHeaders(upstreamRes.headers)}`);
@@ -970,6 +999,25 @@ async function forwardRequest(
970
999
  }
971
1000
  }
972
1001
 
1002
+ // EFFORT REPAIR. A rejected output_config.effort is a hard failure of whatever the
1003
+ // client was doing (a web search, a tool call) — worth healing rather than surfacing.
1004
+ if (effortMode && !requestInfo.effortRepaired
1005
+ && canRetryBufferedBody && retryCount + 1 < maxAttempts && !res.headersSent) {
1006
+ const fix = repairEffort(body, effortMode, errorBody);
1007
+ if (fix.body) {
1008
+ // Latch it so LATER turns apply the working level up front: the client resends the
1009
+ // same rejected setting every turn, and each rejection would otherwise charge a
1010
+ // consecutive-failure to a healthy account and deprioritise it in the router.
1011
+ accountManager.markSessionEffort?.(requestInfo.sessionKey, requestInfo.model, fix.effort);
1012
+ console.log(`[Maxpool] "${requestInfo.model || 'model'}" rejected effort "${requestInfo.effort || 'xhigh'}" (via ${account.name}); retrying with ${fix.effort ? `effort "${fix.effort}"` : 'the effort setting removed'}`);
1013
+ return forwardRequest(
1014
+ req, res, fix.body, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir,
1015
+ retryConfig, queueConfig, { ...requestInfo, effortRepaired: true },
1016
+ canRetryBufferedBody, canQueueBufferedBody, excludedIndexes,
1017
+ );
1018
+ }
1019
+ }
1020
+
973
1021
  // RECOVER-ON-CLAUDE (preferred over the provider pin below): the transcript
974
1022
  // carries provider-authored thinking blocks whose signature Anthropic rejects.
975
1023
  // Strip exactly those blocks and retry on Claude, so a session that took even one
@@ -1375,7 +1423,7 @@ function isContextLengthError(errorBody) {
1375
1423
  return /exceeded model token limit|maximum context length|context length exceeded|context window (?:size )?(?:exceeded|too)|prompt is too long|input is too long|reduce the length of|too many (?:input )?tokens|request too large/i.test(errorBody);
1376
1424
  }
1377
1425
 
1378
- export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, isCapacitySignalStatus, isStrippableThinkingBlock, stripForeignThinkingBlocks, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, isContextLengthError, streamResponse, startIdleRequestReaper };
1426
+ export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, classifyEffortRejection, repairEffort, isCapacitySignalStatus, isStrippableThinkingBlock, stripForeignThinkingBlocks, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, isContextLengthError, streamResponse, startIdleRequestReaper };
1379
1427
 
1380
1428
  async function readErrorBody(upstreamRes, limitBytes = 64 * 1024) {
1381
1429
  if (!upstreamRes.body) return '';
@@ -1582,6 +1630,57 @@ function isStrippableThinkingBlock(block) {
1582
1630
  // But converting the pair into plain TEXT is accepted (200 OK) and keeps what the search
1583
1631
  // actually found, so the session survives with its information intact. This is what made
1584
1632
  // the web-search case look permanently unrepairable.
1633
+ // Claude Code can send an `output_config.effort` the target model won't take — usually
1634
+ // after the session's model changes (a resume/fallback) while the effort setting stays.
1635
+ // It is a HARD error: the tool call just fails, which is what killed the user's web
1636
+ // searches. Three shapes seen live 2026-07-26, all repairable:
1637
+ // "does not support effort level 'xhigh'. Supported levels: high, low, medium" -> downgrade
1638
+ // "'xhigh' is not supported when thinking is disabled … Use effort 'high' or below" -> downgrade
1639
+ // "does not support the effort parameter." -> drop it
1640
+ function classifyEffortRejection(errorBody) {
1641
+ if (!/effort/i.test(errorBody)) return null;
1642
+ if (/does not support the effort parameter/i.test(errorBody)) return 'drop';
1643
+ if (/does not support effort level|is not supported when thinking is disabled/i.test(errorBody)
1644
+ || /output_config\.effort: Input should be/i.test(errorBody)) { // invalid value from the client
1645
+ return 'downgrade';
1646
+ }
1647
+ return null;
1648
+ }
1649
+
1650
+ /** Rewrite the request's effort so the model accepts it. 'downgrade' picks the best level
1651
+ * the error itself advertises (falling back to 'high'); 'drop' removes the field. */
1652
+ function repairEffort(body, mode, errorBody = '') {
1653
+ try {
1654
+ const json = JSON.parse(Buffer.isBuffer(body) ? body.toString('utf8') : String(body));
1655
+ const cur = json?.output_config?.effort;
1656
+ if (!cur) return { body: null, effort: null };
1657
+ if (mode === 'drop') {
1658
+ delete json.output_config.effort;
1659
+ if (Object.keys(json.output_config).length === 0) delete json.output_config;
1660
+ return { body: Buffer.from(JSON.stringify(json)), effort: null };
1661
+ }
1662
+ // Prefer a level the error explicitly lists, else 'high' (what the message recommends).
1663
+ const rank = ['max', 'xhigh', 'high', 'medium', 'low'];
1664
+ const listed = /supported levels:\s*([a-z, ']+)/i.exec(errorBody)?.[1];
1665
+ let allowed = listed ? listed.split(',').map(x => x.trim().replace(/'/g, '').toLowerCase()).filter(Boolean) : [];
1666
+ // The other real shape names a ceiling instead of a list: "Use effort 'high' or below".
1667
+ const ceiling = /use effort '([a-z]+)' or below/i.exec(errorBody)?.[1]?.toLowerCase();
1668
+ if (!allowed.length && ceiling) allowed = rank.slice(rank.indexOf(ceiling)).filter(Boolean);
1669
+ let next = rank.find(r => allowed.includes(r));
1670
+ // Nothing usable advertised (or it names the level we already sent) — step strictly
1671
+ // BELOW the current level rather than giving up, so we never retry the same value.
1672
+ if (!next || next === cur) {
1673
+ const below = rank.slice(rank.indexOf(cur) + 1);
1674
+ next = below.find(r => !allowed.length || allowed.includes(r)) || below[0];
1675
+ }
1676
+ if (!next || next === cur) return { body: null, effort: null };
1677
+ json.output_config.effort = next;
1678
+ return { body: Buffer.from(JSON.stringify(json)), effort: next };
1679
+ } catch {
1680
+ return { body: null, effort: null };
1681
+ }
1682
+ }
1683
+
1585
1684
  /** Collect foreign server-tool ids across the WHOLE transcript first — a call sits on the
1586
1685
  * assistant turn but its result is often carried on the FOLLOWING user turn, so a
1587
1686
  * per-message scan would leave that result behind (and it alone still 400s). */