maxpool 1.5.48 → 1.5.50
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/account-manager.js +17 -0
- package/src/index.js +6 -1
- package/src/server.js +101 -2
package/package.json
CHANGED
package/src/account-manager.js
CHANGED
|
@@ -1893,6 +1893,23 @@ export class AccountManager {
|
|
|
1893
1893
|
console.log(`[Maxpool] Session "${sessionKey}" repaired — Claude routes re-enabled`);
|
|
1894
1894
|
}
|
|
1895
1895
|
|
|
1896
|
+
/** Remember the effort level a MODEL actually accepted for this session, so later turns
|
|
1897
|
+
* apply it up front. The client resends the same rejected setting every turn, so without
|
|
1898
|
+
* this each turn pays a rejected round-trip AND charges a consecutive-failure to whichever
|
|
1899
|
+
* account ate it — deprioritising a perfectly healthy account for a client-side setting.
|
|
1900
|
+
* Keyed by model: opus-4-5 takes 'high' while opus-4-1 takes no effort at all. */
|
|
1901
|
+
markSessionEffort(sessionKey, model, effort) {
|
|
1902
|
+
if (!sessionKey || !model) return;
|
|
1903
|
+
const existing = this.sessionPolicies.get(sessionKey) || {};
|
|
1904
|
+
this.sessionPolicies.set(sessionKey, { ...existing, effortFix: { model, effort } });
|
|
1905
|
+
}
|
|
1906
|
+
|
|
1907
|
+
getSessionEffort(sessionKey, model) {
|
|
1908
|
+
if (!sessionKey || !model) return undefined;
|
|
1909
|
+
const fix = this.sessionPolicies.get(sessionKey)?.effortFix;
|
|
1910
|
+
return fix && fix.model === model ? fix : undefined;
|
|
1911
|
+
}
|
|
1912
|
+
|
|
1896
1913
|
isSessionThinkingContaminated(sessionKey) {
|
|
1897
1914
|
if (!sessionKey) return false;
|
|
1898
1915
|
return Boolean(this.sessionPolicies.get(sessionKey)?.thinkingContaminated);
|
package/src/index.js
CHANGED
|
@@ -1150,7 +1150,12 @@ async function serverWorkerCommand() {
|
|
|
1150
1150
|
if (r && !r.hasUpdate) notifyUpdate('Already on the latest version');
|
|
1151
1151
|
};
|
|
1152
1152
|
if (tui) tui.checkNow = checkForUpdatesNow;
|
|
1153
|
-
|
|
1153
|
+
// 30 minutes, not 6 hours. At 6h a fix published minutes ago could sit unapplied until
|
|
1154
|
+
// the evening, which reads as "auto-update is broken" even when it works perfectly (it
|
|
1155
|
+
// did — a 4h gap between publish and pickup was the entire complaint). The check is one
|
|
1156
|
+
// cheap npm registry probe, and applying is a seamless reload, so a tighter cadence
|
|
1157
|
+
// costs almost nothing. Still env-tunable; still floored at 60s.
|
|
1158
|
+
const updateIntervalMs = Math.max(60_000, Number(process.env.MAXPOOL_UPDATE_CHECK_INTERVAL_MS) || 30 * 60 * 1000);
|
|
1154
1159
|
updateTimer = setInterval(() => {
|
|
1155
1160
|
// announce:false — the persistent TUI banner is the passive reminder; only real
|
|
1156
1161
|
// actions (installing / applying) log. runUpdateCheck's latch + hasLease gate prevent
|
package/src/server.js
CHANGED
|
@@ -357,6 +357,20 @@ async function forwardRequest(
|
|
|
357
357
|
// PRE-STRIP a session already known to carry provider-authored thinking. The client
|
|
358
358
|
// resends the whole poisoned history every turn, so without this each turn pays another
|
|
359
359
|
// rejected round-trip before the reactive repair kicks in. Latched by the first repair.
|
|
360
|
+
// Apply the effort level this session's model already proved it accepts, so later turns
|
|
361
|
+
// skip the rejected round-trip entirely (the client resends the same setting every turn).
|
|
362
|
+
if (retryCount === 0 && !requestInfo.effortRepaired) {
|
|
363
|
+
const latched = accountManager.getSessionEffort?.(requestInfo.sessionKey, requestInfo.model);
|
|
364
|
+
if (latched) {
|
|
365
|
+
const pre = repairEffort(body, latched.effort ? 'downgrade' : 'drop',
|
|
366
|
+
latched.effort ? `Supported levels: ${latched.effort}` : '');
|
|
367
|
+
if (pre.body) {
|
|
368
|
+
body = pre.body;
|
|
369
|
+
requestInfo = { ...requestInfo, effortRepaired: true };
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
|
|
360
374
|
// ALSO repairs a transcript maxpool predicts Anthropic will reject (a provider web
|
|
361
375
|
// search). That prediction happens BEFORE any request is sent and bars every Claude
|
|
362
376
|
// account, so the reactive repair in the 4xx handler could never be reached for it —
|
|
@@ -936,12 +950,27 @@ async function forwardRequest(
|
|
|
936
950
|
// '^srvtoolu_…'`. Repairable by converting the pair to text (verified 200 OK) —
|
|
937
951
|
// it used to fall through to a PERMANENT provider pin.
|
|
938
952
|
|| /server_tool_use\.id: String should match pattern/i.test(errorBody));
|
|
953
|
+
// LOG THE ACTUAL REASON. Previously a 4xx recorded only "HTTP 400" and the upstream
|
|
954
|
+
// message was never written anywhere, so a whole class of failures (e.g. a rejected
|
|
955
|
+
// effort level breaking every web search) was invisible in the log — you could not
|
|
956
|
+
// even grep for it. Truncated so a huge validation dump can't wall the file.
|
|
957
|
+
if (upstreamRes.status >= 400 && upstreamRes.status !== 429) {
|
|
958
|
+
const why = (() => {
|
|
959
|
+
try { return JSON.parse(errorBody)?.error?.message || errorBody; } catch { return errorBody; }
|
|
960
|
+
})();
|
|
961
|
+
console.log(`[Maxpool] ${upstreamRes.status} from "${account.name}": ${String(why).slice(0, 300)}`);
|
|
962
|
+
}
|
|
939
963
|
const errorType = errorBody.includes('Invalid `signature` in `thinking` block')
|
|
940
964
|
? 'invalid_thinking_signature'
|
|
941
965
|
: anthropicIncompat ? 'anthropic_incompatible_transcript'
|
|
942
966
|
: providerTooSmall ? 'provider_context_too_small'
|
|
943
967
|
: `HTTP ${upstreamRes.status}`;
|
|
944
|
-
|
|
968
|
+
const effortMode = classifyEffortRejection(errorBody);
|
|
969
|
+
// A rejected effort level is a REQUEST-shaped fault, not an account-health signal —
|
|
970
|
+
// release neutral so it never charges a consecutive-failure to a healthy account.
|
|
971
|
+
accountManager.releaseAccount(lease, effortMode
|
|
972
|
+
? { status: upstreamRes.status, error: errorType, neutral: true }
|
|
973
|
+
: { status: upstreamRes.status, error: errorType });
|
|
945
974
|
|
|
946
975
|
if (logDir) {
|
|
947
976
|
logSections.push(`=== RESPONSE ${upstreamRes.status} — non-retryable client error from "${account.name}" ===\n${formatHeaders(upstreamRes.headers)}`);
|
|
@@ -970,6 +999,25 @@ async function forwardRequest(
|
|
|
970
999
|
}
|
|
971
1000
|
}
|
|
972
1001
|
|
|
1002
|
+
// EFFORT REPAIR. A rejected output_config.effort is a hard failure of whatever the
|
|
1003
|
+
// client was doing (a web search, a tool call) — worth healing rather than surfacing.
|
|
1004
|
+
if (effortMode && !requestInfo.effortRepaired
|
|
1005
|
+
&& canRetryBufferedBody && retryCount + 1 < maxAttempts && !res.headersSent) {
|
|
1006
|
+
const fix = repairEffort(body, effortMode, errorBody);
|
|
1007
|
+
if (fix.body) {
|
|
1008
|
+
// Latch it so LATER turns apply the working level up front: the client resends the
|
|
1009
|
+
// same rejected setting every turn, and each rejection would otherwise charge a
|
|
1010
|
+
// consecutive-failure to a healthy account and deprioritise it in the router.
|
|
1011
|
+
accountManager.markSessionEffort?.(requestInfo.sessionKey, requestInfo.model, fix.effort);
|
|
1012
|
+
console.log(`[Maxpool] "${requestInfo.model || 'model'}" rejected effort "${requestInfo.effort || 'xhigh'}" (via ${account.name}); retrying with ${fix.effort ? `effort "${fix.effort}"` : 'the effort setting removed'}`);
|
|
1013
|
+
return forwardRequest(
|
|
1014
|
+
req, res, fix.body, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir,
|
|
1015
|
+
retryConfig, queueConfig, { ...requestInfo, effortRepaired: true },
|
|
1016
|
+
canRetryBufferedBody, canQueueBufferedBody, excludedIndexes,
|
|
1017
|
+
);
|
|
1018
|
+
}
|
|
1019
|
+
}
|
|
1020
|
+
|
|
973
1021
|
// RECOVER-ON-CLAUDE (preferred over the provider pin below): the transcript
|
|
974
1022
|
// carries provider-authored thinking blocks whose signature Anthropic rejects.
|
|
975
1023
|
// Strip exactly those blocks and retry on Claude, so a session that took even one
|
|
@@ -1375,7 +1423,7 @@ function isContextLengthError(errorBody) {
|
|
|
1375
1423
|
return /exceeded model token limit|maximum context length|context length exceeded|context window (?:size )?(?:exceeded|too)|prompt is too long|input is too long|reduce the length of|too many (?:input )?tokens|request too large/i.test(errorBody);
|
|
1376
1424
|
}
|
|
1377
1425
|
|
|
1378
|
-
export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, isCapacitySignalStatus, isStrippableThinkingBlock, stripForeignThinkingBlocks, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, isContextLengthError, streamResponse, startIdleRequestReaper };
|
|
1426
|
+
export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, classifyEffortRejection, repairEffort, isCapacitySignalStatus, isStrippableThinkingBlock, stripForeignThinkingBlocks, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, isContextLengthError, streamResponse, startIdleRequestReaper };
|
|
1379
1427
|
|
|
1380
1428
|
async function readErrorBody(upstreamRes, limitBytes = 64 * 1024) {
|
|
1381
1429
|
if (!upstreamRes.body) return '';
|
|
@@ -1582,6 +1630,57 @@ function isStrippableThinkingBlock(block) {
|
|
|
1582
1630
|
// But converting the pair into plain TEXT is accepted (200 OK) and keeps what the search
|
|
1583
1631
|
// actually found, so the session survives with its information intact. This is what made
|
|
1584
1632
|
// the web-search case look permanently unrepairable.
|
|
1633
|
+
// Claude Code can send an `output_config.effort` the target model won't take — usually
|
|
1634
|
+
// after the session's model changes (a resume/fallback) while the effort setting stays.
|
|
1635
|
+
// It is a HARD error: the tool call just fails, which is what killed the user's web
|
|
1636
|
+
// searches. Three shapes seen live 2026-07-26, all repairable:
|
|
1637
|
+
// "does not support effort level 'xhigh'. Supported levels: high, low, medium" -> downgrade
|
|
1638
|
+
// "'xhigh' is not supported when thinking is disabled … Use effort 'high' or below" -> downgrade
|
|
1639
|
+
// "does not support the effort parameter." -> drop it
|
|
1640
|
+
function classifyEffortRejection(errorBody) {
|
|
1641
|
+
if (!/effort/i.test(errorBody)) return null;
|
|
1642
|
+
if (/does not support the effort parameter/i.test(errorBody)) return 'drop';
|
|
1643
|
+
if (/does not support effort level|is not supported when thinking is disabled/i.test(errorBody)
|
|
1644
|
+
|| /output_config\.effort: Input should be/i.test(errorBody)) { // invalid value from the client
|
|
1645
|
+
return 'downgrade';
|
|
1646
|
+
}
|
|
1647
|
+
return null;
|
|
1648
|
+
}
|
|
1649
|
+
|
|
1650
|
+
/** Rewrite the request's effort so the model accepts it. 'downgrade' picks the best level
|
|
1651
|
+
* the error itself advertises (falling back to 'high'); 'drop' removes the field. */
|
|
1652
|
+
function repairEffort(body, mode, errorBody = '') {
|
|
1653
|
+
try {
|
|
1654
|
+
const json = JSON.parse(Buffer.isBuffer(body) ? body.toString('utf8') : String(body));
|
|
1655
|
+
const cur = json?.output_config?.effort;
|
|
1656
|
+
if (!cur) return { body: null, effort: null };
|
|
1657
|
+
if (mode === 'drop') {
|
|
1658
|
+
delete json.output_config.effort;
|
|
1659
|
+
if (Object.keys(json.output_config).length === 0) delete json.output_config;
|
|
1660
|
+
return { body: Buffer.from(JSON.stringify(json)), effort: null };
|
|
1661
|
+
}
|
|
1662
|
+
// Prefer a level the error explicitly lists, else 'high' (what the message recommends).
|
|
1663
|
+
const rank = ['max', 'xhigh', 'high', 'medium', 'low'];
|
|
1664
|
+
const listed = /supported levels:\s*([a-z, ']+)/i.exec(errorBody)?.[1];
|
|
1665
|
+
let allowed = listed ? listed.split(',').map(x => x.trim().replace(/'/g, '').toLowerCase()).filter(Boolean) : [];
|
|
1666
|
+
// The other real shape names a ceiling instead of a list: "Use effort 'high' or below".
|
|
1667
|
+
const ceiling = /use effort '([a-z]+)' or below/i.exec(errorBody)?.[1]?.toLowerCase();
|
|
1668
|
+
if (!allowed.length && ceiling) allowed = rank.slice(rank.indexOf(ceiling)).filter(Boolean);
|
|
1669
|
+
let next = rank.find(r => allowed.includes(r));
|
|
1670
|
+
// Nothing usable advertised (or it names the level we already sent) — step strictly
|
|
1671
|
+
// BELOW the current level rather than giving up, so we never retry the same value.
|
|
1672
|
+
if (!next || next === cur) {
|
|
1673
|
+
const below = rank.slice(rank.indexOf(cur) + 1);
|
|
1674
|
+
next = below.find(r => !allowed.length || allowed.includes(r)) || below[0];
|
|
1675
|
+
}
|
|
1676
|
+
if (!next || next === cur) return { body: null, effort: null };
|
|
1677
|
+
json.output_config.effort = next;
|
|
1678
|
+
return { body: Buffer.from(JSON.stringify(json)), effort: next };
|
|
1679
|
+
} catch {
|
|
1680
|
+
return { body: null, effort: null };
|
|
1681
|
+
}
|
|
1682
|
+
}
|
|
1683
|
+
|
|
1585
1684
|
/** Collect foreign server-tool ids across the WHOLE transcript first — a call sits on the
|
|
1586
1685
|
* assistant turn but its result is often carried on the FOLLOWING user turn, so a
|
|
1587
1686
|
* per-message scan would leave that result behind (and it alone still 400s). */
|