@link-assistant/hive-mind 2.4.0 → 2.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,11 @@
1
1
  # @link-assistant/hive-mind
2
2
 
3
+ ## 2.4.1
4
+
5
+ ### Patch Changes
6
+
7
+ - c412790: Improve model capacity fallback handling (Issue #2037): when the requested model is temporarily unavailable, every tool now retries the originally-requested model up to 5 times with exponential backoff before switching, then walks a fallback chain ordered by intelligence/size tier (e.g. `gpt-5.6-sol → gpt-5.6-terra → gpt-5.5 → gpt-5.4 → gpt-5.2`, skipping the smaller `gpt-5.6-luna` variant), keeps the mismatch warning informative rather than alarming, retries quickly after a capacity-driven model switch, and reports the fallback model's share of output tokens. Includes a case study reconstructing the timeline and root causes.
8
+
3
9
  ## 2.4.0
4
10
 
5
11
  ### Minor Changes
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@link-assistant/hive-mind",
3
- "version": "2.4.0",
3
+ "version": "2.4.1",
4
4
  "description": "AI-powered issue solver and hive mind for collaborative problem solving",
5
5
  "main": "src/hive.mjs",
6
6
  "type": "module",
package/src/agent.lib.mjs CHANGED
@@ -24,7 +24,7 @@ import semver from 'semver';
24
24
  import { agentModels, defaultModels, freeToBaseModelMap } from './models/index.mjs';
25
25
  import { checkPlaywrightMcpPackageAvailability, getAgentPlaywrightMcpDisableEnv } from './playwright-mcp.lib.mjs';
26
26
  import { createAgentTokenUsage, accumulateAgentStepFinishUsage, parseAgentTokenUsage } from './agent-token-usage.lib.mjs';
27
- import { classifyRetryableError, getRetryDelayMs, maybeSwitchToFallbackModel, waitWithCountdown } from './tool-retry.lib.mjs';
27
+ import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs';
28
28
  import { attachStreamingInput, finalizeBidirectionalHandler, setupBidirectionalHandler } from './bidirectional-interactive.lib.mjs';
29
29
 
30
30
  export { createAgentTokenUsage, accumulateAgentStepFinishUsage, parseAgentTokenUsage };
@@ -898,15 +898,22 @@ export const executeAgentCommand = async params => {
898
898
  const isRequestTimeoutRetry = retryableError.label === 'Request timeout';
899
899
  const maxRetries = isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutRetries : retryLimits.maxTransientErrorRetries;
900
900
  if (retryCount < maxRetries) {
901
- const delay = getRetryDelayMs({
901
+ if (sessionId && !argv.resume) argv.resume = sessionId;
902
+ // Issue #2037: retry the same model on capacity errors before falling back;
903
+ // after a capacity-driven model switch, retry quickly instead of waiting the
904
+ // full transient backoff — the new model may be available now.
905
+ const retryPlan = await prepareRetryAfterError({
906
+ tool: 'agent',
907
+ argv,
908
+ log,
909
+ errorMessage: retryableError.message,
902
910
  retryCount,
903
911
  initialDelayMs: isRequestTimeoutRetry ? retryLimits.initialRequestTimeoutDelayMs : retryLimits.initialTransientErrorDelayMs,
904
912
  maxDelayMs: isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutDelayMs : retryLimits.maxTransientErrorDelayMs,
905
913
  });
914
+ const delay = retryPlan.delay;
906
915
  const delayLabel = delay >= 60000 ? `${Math.round(delay / 60000)} min` : `${Math.round(delay / 1000)}s`;
907
916
  await log(`\n⚠️ ${retryableError.label} detected. Retry ${retryCount + 1}/${maxRetries} in ${delayLabel}${sessionId ? ' (session preserved)' : ''}...`, { level: 'warning' });
908
- if (sessionId && !argv.resume) argv.resume = sessionId;
909
- await maybeSwitchToFallbackModel({ tool: 'agent', argv, log, errorMessage: retryableError.message });
910
917
  await finalizeAgentBidirectionalHandler();
911
918
  await waitForRetryDelay(delay, log);
912
919
  await log('\n🔄 Retrying now...');
@@ -26,7 +26,7 @@ import { buildMcpConfigWithoutPlaywright, ensureClaudePlaywrightMcpServer } from
26
26
  import { resolveClaudeSessionToolFlags } from './useless-tools.lib.mjs';
27
27
  import { ensureClaudeQuietConfig } from './claude-quiet-config.lib.mjs';
28
28
  import { fetchModelInfo } from './model-info.lib.mjs';
29
- import { classifyRetryableError, logExecutionContext, maybeSwitchToFallbackModel, waitWithCountdown } from './tool-retry.lib.mjs';
29
+ import { classifyRetryableError, logExecutionContext, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs';
30
30
  import { resolveSubSessionSize } from './sub-session-size.lib.mjs'; // Issue #1706
31
31
  import { withAgentsMdAsClaudeMd } from './agents-md-claude-support.lib.mjs';
32
32
  import { deployHandoffSkill } from './handoff-skill.lib.mjs'; // Issue #1877
@@ -1212,7 +1212,11 @@ export const executeClaudeCommand = async params => {
1212
1212
  };
1213
1213
  }
1214
1214
  if (retryCount < maxRetries) {
1215
- const delay = Math.min(initialDelay * Math.pow(retryLimits.retryBackoffMultiplier, retryCount), maxDelay);
1215
+ // Activity timeout preserves session (work was started), startup timeout does not (no session created)
1216
+ if (!isStartupTimeout && sessionId && !argv.resume) argv.resume = sessionId;
1217
+ // Issue #2037: retry same model on capacity errors before falling back; a switch retries fast.
1218
+ const retryPlan = await prepareRetryAfterError({ tool: 'claude', argv, log, errorMessage: retryableLastError.message || lastMessage, retryCount, initialDelayMs: initialDelay, maxDelayMs: maxDelay });
1219
+ const delay = retryPlan.delay;
1216
1220
  const errorLabel = isStartupTimeout ? 'Stream startup timeout (Issue #1472/#1475)' : isActivityTimeout ? 'Stream activity timeout (Issue #1472)' : isRequestTimeout ? 'Request timeout' : retryableLastError.label || (isOverloadError || (lastMessage.includes('API Error: 500') && lastMessage.includes('Overloaded')) || (lastMessage.includes('API Error: 529') && lastMessage.includes('Overloaded')) ? `API overload (${lastMessage.includes('529') ? '529' : '500'})` : isInternalServerError || lastMessage.includes('Internal server error') ? 'Internal server error (500)' : isRateLimitError ? 'Server rate limited (429)' : '503 network error');
1217
1221
  const notRetryableHint = apiMarkedNotRetryable ? ' (API says not retryable — will stop early if no progress)' : '';
1218
1222
  const delayLabel = delay >= 60000 ? `${Math.round(delay / 60000)} min` : `${Math.round(delay / 1000)}s`;
@@ -1232,9 +1236,6 @@ export const executeClaudeCommand = async params => {
1232
1236
  await log(` Warning: Could not post force-kill comment to PR: ${commentError.message}`, { verbose: true });
1233
1237
  }
1234
1238
  }
1235
- // Activity timeout preserves session (work was started), startup timeout does not (no session created)
1236
- if (!isStartupTimeout && sessionId && !argv.resume) argv.resume = sessionId;
1237
- await maybeSwitchToFallbackModel({ tool: 'claude', argv, log, errorMessage: retryableLastError.message || lastMessage });
1238
1239
  await waitWithCountdown(delay, log);
1239
1240
  await log('\n🔄 Retrying now...');
1240
1241
  retryCount++;
@@ -1389,11 +1390,13 @@ export const executeClaudeCommand = async params => {
1389
1390
  const initialDelay = isTimeoutException ? retryLimits.initialRequestTimeoutDelayMs : retryLimits.initialTransientErrorDelayMs;
1390
1391
  const maxDelay = isTimeoutException ? retryLimits.maxRequestTimeoutDelayMs : retryLimits.maxTransientErrorDelayMs;
1391
1392
  if (retryCount < maxRetries) {
1392
- const delay = Math.min(initialDelay * Math.pow(retryLimits.retryBackoffMultiplier, retryCount), maxDelay);
1393
- const errorLabel = isTimeoutException ? 'Request timeout' : retryableException.label || (errorStr.includes('Overloaded') ? `API overload (${errorStr.includes('529') ? '529' : '500'})` : errorStr.includes('Internal server error') ? 'Internal server error (500)' : '503 network error');
1394
- await log(`\n⚠️ ${errorLabel} in exception. Retry ${retryCount + 1}/${maxRetries} in ${Math.round(delay / 60000)} min (session preserved)...`, { level: 'warning' });
1395
1393
  if (sessionId && !argv.resume) argv.resume = sessionId;
1396
- await maybeSwitchToFallbackModel({ tool: 'claude', argv, log, errorMessage: errorStr });
1394
+ // Issue #2037: retry same model on capacity errors before falling back; a switch retries fast.
1395
+ const retryPlan = await prepareRetryAfterError({ tool: 'claude', argv, log, errorMessage: errorStr, retryCount, initialDelayMs: initialDelay, maxDelayMs: maxDelay });
1396
+ const delay = retryPlan.delay;
1397
+ const errorLabel = isTimeoutException ? 'Request timeout' : retryableException.label || (errorStr.includes('Overloaded') ? `API overload (${errorStr.includes('529') ? '529' : '500'})` : errorStr.includes('Internal server error') ? 'Internal server error (500)' : '503 network error');
1398
+ const delayLabel = delay >= 60000 ? `${Math.round(delay / 60000)} min` : `${Math.round(delay / 1000)}s`;
1399
+ await log(`\n⚠️ ${errorLabel} in exception. Retry ${retryCount + 1}/${maxRetries} in ${delayLabel} (session preserved)...`, { level: 'warning' });
1397
1400
  await waitWithCountdown(delay, log);
1398
1401
  await log('\n🔄 Retrying now...');
1399
1402
  retryCount++;
package/src/codex.lib.mjs CHANGED
@@ -31,7 +31,7 @@ import { initProgressMonitoring } from './solve.progress-monitoring.lib.mjs';
31
31
  import { ensureCodexPlaywrightMcpServer, getCodexPlaywrightMcpDisableConfigArgs } from './playwright-mcp.lib.mjs';
32
32
  import { fetchModelInfo } from './model-info.lib.mjs';
33
33
  import { defaultModels } from './models/index.mjs';
34
- import { classifyRetryableError, getRetryDelayMs, maybeSwitchToFallbackModel, waitWithCountdown } from './tool-retry.lib.mjs';
34
+ import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs';
35
35
  import { parseSubSessionSize, buildCodexSubSessionSizeConfigArgs, buildCodexDisable1mContextConfigArgs } from './sub-session-size.lib.mjs'; // Issue #1706
36
36
  import { getCumulativeContextInputTokens } from './context-fill.lib.mjs';
37
37
  import { deployHandoffSkill } from './handoff-skill.lib.mjs'; // Issue #1877
@@ -1175,15 +1175,13 @@ export const executeCodexCommand = async params => {
1175
1175
  const isRequestTimeoutRetry = retryableError.label === 'Request timeout';
1176
1176
  const maxRetries = isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutRetries : retryLimits.maxTransientErrorRetries;
1177
1177
  if (retryCount < maxRetries) {
1178
- const delay = getRetryDelayMs({
1179
- retryCount,
1180
- initialDelayMs: isRequestTimeoutRetry ? retryLimits.initialRequestTimeoutDelayMs : retryLimits.initialTransientErrorDelayMs,
1181
- maxDelayMs: isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutDelayMs : retryLimits.maxTransientErrorDelayMs,
1182
- });
1178
+ if (sessionId && !argv.resume) argv.resume = sessionId;
1179
+ // Issue #2037: retry same model on capacity errors before falling back; a
1180
+ // capacity-driven switch retries fast, other transient errors use standard backoff.
1181
+ const retryPlan = await prepareRetryAfterError({ tool: 'codex', argv, log, errorMessage: retryableError.message, retryCount, initialDelayMs: isRequestTimeoutRetry ? retryLimits.initialRequestTimeoutDelayMs : retryLimits.initialTransientErrorDelayMs, maxDelayMs: isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutDelayMs : retryLimits.maxTransientErrorDelayMs });
1182
+ const delay = retryPlan.delay;
1183
1183
  const delayLabel = delay >= 60000 ? `${Math.round(delay / 60000)} min` : `${Math.round(delay / 1000)}s`;
1184
1184
  await log(`\n⚠️ ${retryableError.label} detected. Retry ${retryCount + 1}/${maxRetries} in ${delayLabel}${sessionId ? ' (session preserved)' : ''}...`, { level: 'warning' });
1185
- if (sessionId && !argv.resume) argv.resume = sessionId;
1186
- await maybeSwitchToFallbackModel({ tool: 'codex', argv, log, errorMessage: retryableError.message });
1187
1185
  await waitForRetryDelay(delay, log);
1188
1186
  await log('\n🔄 Retrying now...');
1189
1187
  retryCount++;
@@ -1222,15 +1220,13 @@ export const executeCodexCommand = async params => {
1222
1220
  const isRequestTimeoutRetry = retryableError.label === 'Request timeout';
1223
1221
  const maxRetries = isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutRetries : retryLimits.maxTransientErrorRetries;
1224
1222
  if (retryCount < maxRetries) {
1225
- const delay = getRetryDelayMs({
1226
- retryCount,
1227
- initialDelayMs: isRequestTimeoutRetry ? retryLimits.initialRequestTimeoutDelayMs : retryLimits.initialTransientErrorDelayMs,
1228
- maxDelayMs: isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutDelayMs : retryLimits.maxTransientErrorDelayMs,
1229
- });
1223
+ if (sessionId && !argv.resume) argv.resume = sessionId;
1224
+ // Issue #2037: retry same model on capacity errors before falling back; a
1225
+ // capacity-driven switch retries fast, other transient errors use standard backoff.
1226
+ const retryPlan = await prepareRetryAfterError({ tool: 'codex', argv, log, errorMessage: retryableError.message, retryCount, initialDelayMs: isRequestTimeoutRetry ? retryLimits.initialRequestTimeoutDelayMs : retryLimits.initialTransientErrorDelayMs, maxDelayMs: isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutDelayMs : retryLimits.maxTransientErrorDelayMs });
1227
+ const delay = retryPlan.delay;
1230
1228
  const delayLabel = delay >= 60000 ? `${Math.round(delay / 60000)} min` : `${Math.round(delay / 1000)}s`;
1231
1229
  await log(`\n⚠️ ${retryableError.label} detected. Retry ${retryCount + 1}/${maxRetries} in ${delayLabel}${sessionId ? ' (session preserved)' : ''}...`, { level: 'warning' });
1232
- if (sessionId && !argv.resume) argv.resume = sessionId;
1233
- await maybeSwitchToFallbackModel({ tool: 'codex', argv, log, errorMessage: retryableError.message });
1234
1230
  await waitForRetryDelay(delay, log);
1235
1231
  await log('\n🔄 Retrying now...');
1236
1232
  retryCount++;
@@ -127,6 +127,18 @@ export const retryLimits = {
127
127
  maxTransientErrorRetries: parseIntWithDefault('HIVE_MIND_MAX_TRANSIENT_ERROR_RETRIES', 10),
128
128
  initialTransientErrorDelayMs: parseIntWithDefault('HIVE_MIND_INITIAL_TRANSIENT_ERROR_DELAY_MS', 2 * 60 * 1000), // 2 minutes
129
129
  maxTransientErrorDelayMs: parseIntWithDefault('HIVE_MIND_MAX_TRANSIENT_ERROR_DELAY_MS', 30 * 60 * 1000), // 30 minutes
130
+ // Issue #2037: When a "model is at capacity" error triggers a switch to a *different*
131
+ // fallback model, the long transient backoff is wasteful — the different model is
132
+ // available now, so retry almost immediately instead of stalling for minutes.
133
+ modelSwitchRetryDelayMs: parseIntWithDefault('HIVE_MIND_MODEL_SWITCH_RETRY_DELAY_MS', 5 * 1000), // 5 seconds
134
+ // Issue #2037 (review): On a "model is at capacity" error, retry the *originally
135
+ // requested* model a few times with exponential backoff before falling back to a
136
+ // different (less-preferred) model. Capacity errors are often short-lived, so giving
137
+ // the preferred model several chances keeps the run on the model the user asked for.
138
+ // Only once these retries are exhausted do we step to the next-closest fallback model.
139
+ capacityRetriesBeforeFallback: parseIntWithDefault('HIVE_MIND_CAPACITY_RETRIES_BEFORE_FALLBACK', 5),
140
+ initialCapacityRetryDelayMs: parseIntWithDefault('HIVE_MIND_INITIAL_CAPACITY_RETRY_DELAY_MS', 15 * 1000), // 15 seconds
141
+ maxCapacityRetryDelayMs: parseIntWithDefault('HIVE_MIND_MAX_CAPACITY_RETRY_DELAY_MS', 4 * 60 * 1000), // 4 minutes
130
142
  // Request timeout retry configuration (Issue #1353)
131
143
  // Network timeouts need longer waits than API errors — Claude CLI already exhausted its own retries
132
144
  maxRequestTimeoutRetries: parseIntWithDefault('HIVE_MIND_MAX_REQUEST_TIMEOUT_RETRIES', 10),
@@ -19,7 +19,7 @@ const __geminiBuildSolveResumeCmd = (argv, sessionId, tempDir) => (sessionId &&
19
19
  import { sanitizeObjectStrings } from './unicode-sanitization.lib.mjs';
20
20
  import { defaultModels, geminiModels } from './models/index.mjs';
21
21
  import { checkPlaywrightMcpPackageAvailability } from './playwright-mcp.lib.mjs';
22
- import { classifyRetryableError, getRetryDelayMs, maybeSwitchToFallbackModel, waitWithCountdown } from './tool-retry.lib.mjs';
22
+ import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs';
23
23
  import { getCumulativeContextInputTokens, toTokenCount } from './context-fill.lib.mjs';
24
24
  import { getTerminalEventCompletionHealth } from './tool-run-health.lib.mjs'; // Issue #1990
25
25
 
@@ -522,14 +522,21 @@ export const executeGeminiCommand = async params => {
522
522
  const isRequestTimeoutRetry = retryableError.label === 'Request timeout';
523
523
  const maxRetries = isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutRetries : retryLimits.maxTransientErrorRetries;
524
524
  if (retryCount < maxRetries) {
525
- const delay = getRetryDelayMs({
525
+ // Issue #2037: retry the same model on capacity errors before falling back;
526
+ // after a capacity-driven model switch, retry quickly instead of waiting the
527
+ // full transient backoff — the new model may be available now.
528
+ const retryPlan = await prepareRetryAfterError({
529
+ tool: 'gemini',
530
+ argv,
531
+ log,
532
+ errorMessage: retryableError.message,
526
533
  retryCount,
527
534
  initialDelayMs: isRequestTimeoutRetry ? retryLimits.initialRequestTimeoutDelayMs : retryLimits.initialTransientErrorDelayMs,
528
535
  maxDelayMs: isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutDelayMs : retryLimits.maxTransientErrorDelayMs,
529
536
  });
537
+ const delay = retryPlan.delay;
530
538
  const delayLabel = delay >= 60000 ? `${Math.round(delay / 60000)} min` : `${Math.round(delay / 1000)}s`;
531
539
  await log(`\n⚠️ ${retryableError.label} detected. Retry ${retryCount + 1}/${maxRetries} in ${delayLabel}${sessionId ? ' (session preserved)' : ''}...`, { level: 'warning' });
532
- await maybeSwitchToFallbackModel({ tool: 'gemini', argv, log, errorMessage: retryableError.message });
533
540
  await waitForRetryDelay(delay, log);
534
541
  await log('\n🔄 Retrying now...');
535
542
  retryCount++;
@@ -388,6 +388,9 @@ export async function attachLogToGitHub(options) {
388
388
  }
389
389
  let totalCostUSD = publicPricingEstimate; // Issue #1225: token usage + actual model IDs
390
390
  let actualModelIds = null;
391
+ // Issue #2037 (review): per-model output-token map, used to report the share of
392
+ // output tokens produced by the fallback model in the "Models used:" section.
393
+ let modelUsageForComment = null;
391
394
  if (totalCostUSD === null && sessionId && tempDir && !errorMessage) {
392
395
  try {
393
396
  const { calculateSessionTokens } = await import('./claude.lib.mjs');
@@ -399,6 +402,7 @@ export async function attachLogToGitHub(options) {
399
402
  }
400
403
  if (tokenUsage.modelUsage && Object.keys(tokenUsage.modelUsage).length > 0) {
401
404
  actualModelIds = Object.keys(tokenUsage.modelUsage);
405
+ modelUsageForComment = tokenUsage.modelUsage;
402
406
  if (verbose) await log(` 🤖 Actual models used: ${actualModelIds.join(', ')}`, { verbose: true });
403
407
  }
404
408
  }
@@ -412,6 +416,7 @@ export async function attachLogToGitHub(options) {
412
416
  if (ids.length > 0 && (!actualModelIds || ids.length > actualModelIds.length)) {
413
417
  ids.sort((a, b) => (resultModelUsage[b]?.costUSD ?? 0) - (resultModelUsage[a]?.costUSD ?? 0));
414
418
  actualModelIds = ids;
419
+ if (!modelUsageForComment) modelUsageForComment = resultModelUsage;
415
420
  if (verbose) await log(` 🤖 Using result JSON modelUsage (${ids.length} models): ${ids.join(', ')}`, { verbose: true });
416
421
  }
417
422
  }
@@ -431,7 +436,7 @@ export async function attachLogToGitHub(options) {
431
436
  // Issue #1949: prefer an explicit thinkingInfo, otherwise derive it from argv
432
437
  // (e.g. "high (~24000 tokens)"). null when the run used the tool's default.
433
438
  const resolvedThinkingInfo = thinkingInfo ?? describeRequestedThinking(argv);
434
- modelInfoString = await getModelInfoForComment({ requestedModel, tool, pricingInfo, actualModelIds, thinkingInfo: resolvedThinkingInfo });
439
+ modelInfoString = await getModelInfoForComment({ requestedModel, tool, pricingInfo, actualModelIds, thinkingInfo: resolvedThinkingInfo, fallbackModel: argv?.fallbackModel ?? null, modelUsage: modelUsageForComment });
435
440
  if (verbose && modelInfoString) {
436
441
  await log(' 🤖 Model info fetched for comment', { verbose: true });
437
442
  }
@@ -367,9 +367,11 @@ export const getDefaultModelForTool = tool => {
367
367
 
368
368
  let cachedInstalledCodexModelsPromise = null;
369
369
  // Issue #2027: With gpt-5.6-sol as the preferred default, the fallback chain is only
370
- // consulted when Sol is absent from the local catalog. Prefer the previous stable
371
- // default (gpt-5.5) first, then the remaining GPT-5.6 preview tiers, then older models.
372
- const CODEX_DEFAULT_FALLBACK_CHAIN = ['gpt-5.5', 'openai.gpt-5.5', 'gpt-5.6-terra', 'gpt-5.6-luna', 'openai.gpt-5.6-sol', 'openai.gpt-5.6-terra', 'openai.gpt-5.6-luna', 'gpt-5.4', 'openai.gpt-5.4', 'gpt-5.5-mini', 'gpt-5.4-mini', 'gpt-5.3-codex', 'gpt-5.3-codex-spark', 'gpt-5.2', 'gpt-5.2-codex', 'gpt-5.5-nano', 'gpt-5.4-nano'];
370
+ // consulted when Sol is absent from the local catalog. Issue #2037 (review): order by
371
+ // intelligence / size tier (closest first), not by generation the flagship sibling
372
+ // `gpt-5.6-terra` is closer to Sol than the previous-generation `gpt-5.5`, which in turn
373
+ // is a larger, more capable model than the smaller GPT-5.6 `luna` tier.
374
+ const CODEX_DEFAULT_FALLBACK_CHAIN = ['gpt-5.6-terra', 'openai.gpt-5.6-terra', 'gpt-5.5', 'openai.gpt-5.5', 'gpt-5.4', 'openai.gpt-5.4', 'gpt-5.2', 'gpt-5.6-luna', 'openai.gpt-5.6-luna', 'openai.gpt-5.6-sol', 'gpt-5.5-mini', 'gpt-5.4-mini', 'gpt-5.3-codex', 'gpt-5.3-codex-spark', 'gpt-5.2-codex', 'gpt-5.5-nano', 'gpt-5.4-nano'];
373
375
 
374
376
  export const getInstalledCodexModels = async () => {
375
377
  if (!cachedInstalledCodexModelsPromise) {
@@ -1047,7 +1049,30 @@ const doesRequestedMatchActual = (requestedModel, actualModelId, tool) => {
1047
1049
  * @param {Array<{modelId: string, modelInfo: Object|null}>|null} options.modelsUsed - Actual models used from CLI JSON output
1048
1050
  * @returns {string} Formatted markdown string for model info section
1049
1051
  */
1050
- export const buildModelInfoString = ({ requestedModel = null, tool = null, pricingInfo = null, modelInfo = null, modelsUsed = null, thinkingInfo = null } = {}) => {
1052
+ /**
1053
+ * Compute the share (0-100) of total output tokens that a given model produced.
1054
+ * Used to report how much of the run actually ran on the fallback model, so the
1055
+ * PR/issue comment can manage expectations precisely (Issue #2037 review).
1056
+ * @param {Object|null} modelUsage - map of modelId -> { outputTokens } (or output_tokens)
1057
+ * @param {string} modelId - the model whose share to compute
1058
+ * @returns {number|null} integer percentage, or null when no output-token data
1059
+ */
1060
+ const computeOutputTokenSharePercent = (modelUsage, modelId) => {
1061
+ if (!modelUsage || typeof modelUsage !== 'object' || !modelId) return null;
1062
+ const target = normalizeForComparison(modelId);
1063
+ let total = 0;
1064
+ let matched = 0;
1065
+ for (const [id, usage] of Object.entries(modelUsage)) {
1066
+ const out = Number(usage?.outputTokens ?? usage?.output_tokens ?? 0) || 0;
1067
+ if (out <= 0) continue;
1068
+ total += out;
1069
+ if (normalizeForComparison(id) === target) matched += out;
1070
+ }
1071
+ if (total <= 0) return null;
1072
+ return Math.round((matched / total) * 100);
1073
+ };
1074
+
1075
+ export const buildModelInfoString = ({ requestedModel = null, tool = null, pricingInfo = null, modelInfo = null, modelsUsed = null, thinkingInfo = null, fallbackModel = null, modelUsage = null } = {}) => {
1051
1076
  const hasRequested = requestedModel !== null && requestedModel !== undefined;
1052
1077
  const hasModelsUsed = Array.isArray(modelsUsed) && modelsUsed.length > 0;
1053
1078
  const hasModelInfo = modelInfo !== null;
@@ -1090,12 +1115,30 @@ export const buildModelInfoString = ({ requestedModel = null, tool = null, prici
1090
1115
  const mainModelName = mainModelMeta?.name || mainModelId;
1091
1116
  const modelLabel = supportingEntries.length > 0 ? 'Main model' : 'Model';
1092
1117
 
1118
+ // Issue #2037: A mismatch between the requested model and the model that
1119
+ // actually ran happens when the run was downgraded to the configured fallback
1120
+ // model (e.g. Codex reported the requested `gpt-5.6-sol` was "at capacity", so
1121
+ // the retry loop switched to `gpt-5.6-terra`). Even though the fallback did its
1122
+ // job, the user did *not* get the model they asked for in full detail, so this
1123
+ // is still surfaced as a \u26A0\uFE0F warning (Issue #2037 review) \u2014 but a
1124
+ // clearer one that explains it was an automatic capacity fallback rather than an
1125
+ // unexplained mismatch. When output-token data is available we also report the
1126
+ // share of output tokens produced by the fallback model, so expectations are set
1127
+ // precisely.
1128
+ const matchesFallback = hasRequested && !mainMatches && fallbackModel ? doesRequestedMatchActual(fallbackModel, mainModelId, tool) : false;
1129
+
1093
1130
  if (mainMatches) {
1094
1131
  info += `\n- **${modelLabel}: ${mainModelName}** (\`${mainModelId}\`)`;
1095
1132
  } else {
1096
1133
  info += `\n- **${modelLabel}: ${mainModelName}** (\`${mainModelId}\`)`;
1097
1134
  if (hasRequested) {
1098
- info += `\n- \u26A0\uFE0F **Warning**: Main model \`${mainModelId}\` does not match requested model \`${requestedModel}\``;
1135
+ const sharePercent = computeOutputTokenSharePercent(modelUsage, mainModelId);
1136
+ const shareSuffix = sharePercent !== null ? ` (fallback model produced ${sharePercent}% of output tokens)` : '';
1137
+ if (matchesFallback) {
1138
+ info += `\n- \u26A0\uFE0F **Warning**: Requested model \`${requestedModel}\` was unavailable (at capacity); automatically fell back to \`${mainModelId}\`${shareSuffix}`;
1139
+ } else {
1140
+ info += `\n- \u26A0\uFE0F **Warning**: Main model \`${mainModelId}\` does not match requested model \`${requestedModel}\`${shareSuffix}`;
1141
+ }
1099
1142
  }
1100
1143
  }
1101
1144
 
@@ -1159,14 +1202,25 @@ export const defaultFallbackModels = {
1159
1202
  'claude-sonnet-5': 'sonnet-4-6',
1160
1203
  },
1161
1204
  codex: {
1162
- 'gpt-5.6-sol': 'gpt-5.5',
1205
+ // Issue #2037 (review): order fallbacks by *intelligence / size tier*, not by
1206
+ // generation. Within GPT-5.6, `sol` is the flagship and `terra` is the next tier
1207
+ // down; `luna` is a smaller/cheaper variant. When `gpt-5.6-sol` is at capacity the
1208
+ // closest replacement is `gpt-5.6-terra`, and the next-closest to `gpt-5.6-terra`
1209
+ // is the previous generation's flagship `gpt-5.5` (a larger, more capable model
1210
+ // than the smaller `gpt-5.6-luna`), then `gpt-5.5 -> gpt-5.4 -> gpt-5.2`, and so
1211
+ // on. So the flagship chain walks sol -> terra -> gpt-5.5 -> gpt-5.4 -> gpt-5.2
1212
+ // and never detours through the smaller `luna` tier. The smaller `luna` variant,
1213
+ // if requested directly, steps down to the previous full generation as well.
1214
+ 'gpt-5.6-sol': 'gpt-5.6-terra',
1163
1215
  'gpt-5.6-terra': 'gpt-5.5',
1164
1216
  'gpt-5.6-luna': 'gpt-5.5',
1165
- 'openai.gpt-5.6-sol': 'openai.gpt-5.5',
1217
+ 'openai.gpt-5.6-sol': 'openai.gpt-5.6-terra',
1166
1218
  'openai.gpt-5.6-terra': 'openai.gpt-5.5',
1167
1219
  'openai.gpt-5.6-luna': 'openai.gpt-5.5',
1168
1220
  'openai.gpt-5.5': 'openai.gpt-5.4',
1221
+ 'openai.gpt-5.4': 'openai.gpt-5.2',
1169
1222
  'gpt-5.5': 'gpt-5.4',
1223
+ 'gpt-5.4': 'gpt-5.2',
1170
1224
  },
1171
1225
  };
1172
1226
 
@@ -1189,7 +1243,7 @@ export const resolveDefaultFallbackModel = (tool, model) => {
1189
1243
  * @param {Array<string>|null} options.actualModelIds - Actual model IDs from CLI JSON output
1190
1244
  * @returns {Promise<string>} Formatted markdown model info section
1191
1245
  */
1192
- export const getModelInfoForComment = async ({ requestedModel = null, tool = null, pricingInfo = null, actualModelIds = null, thinkingInfo = null } = {}) => {
1246
+ export const getModelInfoForComment = async ({ requestedModel = null, tool = null, pricingInfo = null, actualModelIds = null, thinkingInfo = null, fallbackModel = null, modelUsage = null } = {}) => {
1193
1247
  let modelIds = [];
1194
1248
 
1195
1249
  if (Array.isArray(actualModelIds) && actualModelIds.length > 0) {
@@ -1221,5 +1275,7 @@ export const getModelInfoForComment = async ({ requestedModel = null, tool = nul
1221
1275
  modelInfo: modelsUsed.length === 0 ? firstModelInfo : null,
1222
1276
  modelsUsed: modelsUsed.length > 0 ? modelsUsed : null,
1223
1277
  thinkingInfo,
1278
+ fallbackModel,
1279
+ modelUsage,
1224
1280
  });
1225
1281
  };
@@ -23,7 +23,7 @@ import { opencodeModels, defaultModels } from './models/index.mjs';
23
23
  import { checkPlaywrightMcpPackageAvailability, getOpenCodePlaywrightMcpDisableEnv } from './playwright-mcp.lib.mjs';
24
24
  import { createAgentTokenUsage, accumulateAgentStepFinishUsage, parseAgentTokenUsage as parseOpenCodeTokenUsage } from './agent-token-usage.lib.mjs';
25
25
  import { calculateAgentPricing } from './agent.lib.mjs';
26
- import { classifyRetryableError, getRetryDelayMs, maybeSwitchToFallbackModel, waitWithCountdown } from './tool-retry.lib.mjs';
26
+ import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs';
27
27
 
28
28
  export { parseOpenCodeTokenUsage };
29
29
 
@@ -483,15 +483,22 @@ export const executeOpenCodeCommand = async params => {
483
483
  const isRequestTimeoutRetry = retryableError.label === 'Request timeout';
484
484
  const maxRetries = isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutRetries : retryLimits.maxTransientErrorRetries;
485
485
  if (retryCount < maxRetries) {
486
- const delay = getRetryDelayMs({
486
+ if (sessionId && !argv.resume) argv.resume = sessionId;
487
+ // Issue #2037: retry the same model on capacity errors before falling back;
488
+ // after a capacity-driven model switch, retry quickly instead of waiting the
489
+ // full transient backoff — the new model may be available now.
490
+ const retryPlan = await prepareRetryAfterError({
491
+ tool: 'opencode',
492
+ argv,
493
+ log,
494
+ errorMessage: retryableError.message,
487
495
  retryCount,
488
496
  initialDelayMs: isRequestTimeoutRetry ? retryLimits.initialRequestTimeoutDelayMs : retryLimits.initialTransientErrorDelayMs,
489
497
  maxDelayMs: isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutDelayMs : retryLimits.maxTransientErrorDelayMs,
490
498
  });
499
+ const delay = retryPlan.delay;
491
500
  const delayLabel = delay >= 60000 ? `${Math.round(delay / 60000)} min` : `${Math.round(delay / 1000)}s`;
492
501
  await log(`\n⚠️ ${retryableError.label} detected. Retry ${retryCount + 1}/${maxRetries} in ${delayLabel}${sessionId ? ' (session preserved)' : ''}...`, { level: 'warning' });
493
- if (sessionId && !argv.resume) argv.resume = sessionId;
494
- await maybeSwitchToFallbackModel({ tool: 'opencode', argv, log, errorMessage: retryableError.message });
495
502
  await waitForRetryDelay(delay, log);
496
503
  await log('\n🔄 Retrying now...');
497
504
  retryCount++;
package/src/qwen.lib.mjs CHANGED
@@ -20,7 +20,7 @@ import { detectUsageLimit, formatUsageLimitMessage } from './usage-limit.lib.mjs
20
20
  import { sanitizeObjectStrings } from './unicode-sanitization.lib.mjs';
21
21
  import { qwenModels, defaultModels } from './models/index.mjs';
22
22
  import { checkPlaywrightMcpPackageAvailability } from './playwright-mcp.lib.mjs';
23
- import { classifyRetryableError, getRetryDelayMs, maybeSwitchToFallbackModel, waitWithCountdown } from './tool-retry.lib.mjs';
23
+ import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs';
24
24
  import { getCumulativeContextInputTokens, getRestoredContextInputTokens, toTokenCount } from './context-fill.lib.mjs';
25
25
  import { getTerminalEventCompletionHealth } from './tool-run-health.lib.mjs'; // Issue #1990
26
26
 
@@ -599,15 +599,22 @@ export const executeQwenCommand = async params => {
599
599
  const isRequestTimeoutRetry = retryableError.label === 'Request timeout';
600
600
  const maxRetries = isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutRetries : retryLimits.maxTransientErrorRetries;
601
601
  if (retryCount < maxRetries) {
602
- const delay = getRetryDelayMs({
602
+ if (sessionId && !argv.resume) argv.resume = sessionId;
603
+ // Issue #2037: retry the same model on capacity errors before falling back;
604
+ // after a capacity-driven model switch, retry quickly instead of waiting the
605
+ // full transient backoff — the new model may be available now.
606
+ const retryPlan = await prepareRetryAfterError({
607
+ tool: 'qwen',
608
+ argv,
609
+ log,
610
+ errorMessage: retryableError.message,
603
611
  retryCount,
604
612
  initialDelayMs: isRequestTimeoutRetry ? retryLimits.initialRequestTimeoutDelayMs : retryLimits.initialTransientErrorDelayMs,
605
613
  maxDelayMs: isRequestTimeoutRetry ? retryLimits.maxRequestTimeoutDelayMs : retryLimits.maxTransientErrorDelayMs,
606
614
  });
615
+ const delay = retryPlan.delay;
607
616
  const delayLabel = delay >= 60000 ? `${Math.round(delay / 60000)} min` : `${Math.round(delay / 1000)}s`;
608
617
  await log(`\n⚠️ ${retryableError.label} detected. Retry ${retryCount + 1}/${maxRetries} in ${delayLabel}${sessionId ? ' (session preserved)' : ''}...`, { level: 'warning' });
609
- if (sessionId && !argv.resume) argv.resume = sessionId;
610
- await maybeSwitchToFallbackModel({ tool: 'qwen', argv, log, errorMessage: retryableError.message });
611
618
  await waitForRetryDelay(delay, log);
612
619
  await log('\n🔄 Retrying now...');
613
620
  retryCount++;
@@ -344,7 +344,7 @@ export const SOLVE_OPTION_DEFINITIONS = {
344
344
  },
345
345
  'fallback-model': {
346
346
  type: 'string',
347
- description: 'Fallback model to switch to on model capacity/overload errors (and, for Fable 5, on safety-classifier refusals). When supported, retries resume the same session with this model. Defaults: claude fable/claude-fable-5 -> opus (Opus 4.8); claude mythos-5/claude-mythos-5 -> fable; claude opus/opus-4-8 -> opus-4-7; claude opus-4-7 -> opus-4-6; codex gpt-5.6-sol/gpt-5.6-terra/gpt-5.6-luna -> gpt-5.5; codex gpt-5.5 -> gpt-5.4; all others unset.',
347
+ description: 'Fallback model to switch to on model capacity/overload errors (and, for Fable 5, on safety-classifier refusals). When supported, retries resume the same session with this model. An explicit value is pinned exactly; the built-in defaults form a chain that steps to the next-closest model on repeated capacity errors. Defaults: claude fable/claude-fable-5 -> opus (Opus 4.8); claude mythos-5/claude-mythos-5 -> fable; claude opus/opus-4-8 -> opus-4-7; claude opus-4-7 -> opus-4-6; codex gpt-5.6-sol -> gpt-5.6-terra -> gpt-5.6-luna -> gpt-5.5 -> gpt-5.4; all others unset.',
348
348
  default: undefined,
349
349
  },
350
350
  'sub-agent-model': {
@@ -1062,6 +1062,11 @@ export const parseArguments = async (yargs = getLinoYargsFactory(), hideBinFn =
1062
1062
  const defaultFallbackModel = resolveDefaultFallbackModel(argv.tool, argv.model);
1063
1063
  argv.fallbackModel = defaultFallbackModel || undefined;
1064
1064
  }
1065
+ // Issue #2037 (review): remember whether the fallback model was pinned by the user.
1066
+ // An explicit --fallback-model is honoured exactly and never walked past; an
1067
+ // implicit (default) fallback is allowed to step down the full default chain on
1068
+ // repeated capacity errors. See resolveConfiguredFallbackModel().
1069
+ argv._fallbackModelExplicit = fallbackModelExplicitlyProvided;
1065
1070
 
1066
1071
  // Validate mutual exclusivity of --claude-file and --gitkeep-file
1067
1072
  // Check if both are explicitly enabled (user passed both --claude-file and --gitkeep-file)
@@ -199,8 +199,30 @@ export const waitWithCountdown = async (delayMs, log) => {
199
199
  clearInterval(timer);
200
200
  };
201
201
 
202
- export const resolveConfiguredFallbackModel = ({ tool, currentModel, configuredFallbackModel = undefined } = {}) => {
203
- if (configuredFallbackModel) return configuredFallbackModel;
202
+ // Issue #2037 (review): Support a *multi-level* fallback chain (e.g.
203
+ // gpt-5.6-sol -> gpt-5.6-terra -> gpt-5.6-luna -> gpt-5.5 -> gpt-5.4) so repeated
204
+ // capacity errors keep stepping to the next-closest model instead of getting stuck
205
+ // on the first fallback. `configuredFallbackModel` (from --fallback-model, or the
206
+ // default resolved once at config time) is honoured only while it still differs from
207
+ // the current model; once the run has already switched onto it, we resolve the next
208
+ // hop from the default chain of the *current* model. An explicitly user-pinned
209
+ // fallback (`explicit: true`) is never walked past — the user chose that model on
210
+ // purpose, so it stays put.
211
+ export const resolveConfiguredFallbackModel = ({ tool, currentModel, configuredFallbackModel = undefined, explicit = false } = {}) => {
212
+ // A user-pinned fallback (--fallback-model) is honoured as-is and never walked
213
+ // past: the user chose that exact model on purpose. Return it while it still
214
+ // differs from the current model; once the run is already on it, stop switching.
215
+ if (explicit && configuredFallbackModel) {
216
+ const current = normalizeModelKey(resolveModelId(currentModel, tool));
217
+ const configured = normalizeModelKey(resolveModelId(configuredFallbackModel, tool));
218
+ if (configured && configured !== current) return configuredFallbackModel;
219
+ return null;
220
+ }
221
+ // Otherwise resolve the next hop from the default chain of the *current* model,
222
+ // so repeated capacity errors walk the whole chain
223
+ // (e.g. gpt-5.6-sol -> gpt-5.6-terra -> gpt-5.6-luna -> gpt-5.5 -> gpt-5.4).
224
+ // The auto-set argv.fallbackModel is intentionally ignored here — it only ever
225
+ // holds the first default hop and would otherwise pin the chain to one step.
204
226
  return resolveDefaultFallbackModel(tool, currentModel);
205
227
  };
206
228
 
@@ -237,6 +259,7 @@ export const maybeSwitchToFallbackModel = async ({ tool, argv, log, errorMessage
237
259
  tool,
238
260
  currentModel: argv?.model,
239
261
  configuredFallbackModel: argv?.fallbackModel,
262
+ explicit: argv?._fallbackModelExplicit === true,
240
263
  });
241
264
 
242
265
  const classification = classifyRetryableError(errorMessage);
@@ -262,7 +285,12 @@ export const maybeSwitchToFallbackModel = async ({ tool, argv, log, errorMessage
262
285
 
263
286
  const previousModel = argv.model;
264
287
  argv.model = fallbackModel;
265
- if (!argv.fallbackModel) argv.fallbackModel = fallbackModel;
288
+ // Issue #2037 (review): record the model we actually switched to as the current
289
+ // fallback target. For a multi-hop chain (sol -> terra -> luna -> ...) this keeps
290
+ // argv.fallbackModel pointing at the model that is now running, so the PR comment
291
+ // correctly reports it as the automatic capacity fallback. An explicit user pin
292
+ // already equals `fallbackModel` here, so this is a no-op in that case.
293
+ argv.fallbackModel = fallbackModel;
266
294
 
267
295
  if (typeof log === 'function') {
268
296
  // Issue #1949: show the resolved full model IDs so the switch is unambiguous,
@@ -278,12 +306,51 @@ export const maybeSwitchToFallbackModel = async ({ tool, argv, log, errorMessage
278
306
  };
279
307
  };
280
308
 
309
+ // Issue #2037 (review): Unified retry planner shared by every tool's retry loop.
310
+ // On a genuine "model is at capacity" error it first retries the *originally
311
+ // requested* model up to `capacityRetriesBeforeFallback` times with exponential
312
+ // backoff — capacity errors are often short-lived, so the preferred model gets
313
+ // several chances before we downgrade. Only once those same-model retries are
314
+ // exhausted does it switch to the next-closest fallback model (fast 5s retry).
315
+ // Non-capacity transient errors keep the current model and use the caller's
316
+ // standard backoff, exactly as before.
317
+ //
318
+ // The same-model capacity retry count is stored on `argv._capacityRetryCount` so it
319
+ // survives the recursive executeWithRetry calls without each tool tracking extra
320
+ // state. It resets to 0 whenever we actually switch models, so every model in the
321
+ // fallback chain gets its own batch of same-model retries before stepping down.
322
+ export const prepareRetryAfterError = async ({ tool, argv, log, errorMessage, retryCount, initialDelayMs, maxDelayMs } = {}) => {
323
+ const classification = classifyRetryableError(errorMessage);
324
+ const isCapacity = classification.isCapacity === true && !!argv?.model;
325
+ const capacityRetryCount = argv?._capacityRetryCount || 0;
326
+
327
+ if (isCapacity && capacityRetryCount < retryLimits.capacityRetriesBeforeFallback) {
328
+ if (argv) argv._capacityRetryCount = capacityRetryCount + 1;
329
+ const delay = getRetryDelayMs({
330
+ retryCount: capacityRetryCount,
331
+ initialDelayMs: retryLimits.initialCapacityRetryDelayMs,
332
+ maxDelayMs: retryLimits.maxCapacityRetryDelayMs,
333
+ });
334
+ if (typeof log === 'function') {
335
+ await log(` Model ${formatModelWithResolvedId(argv.model, tool)} at capacity — retrying same model (attempt ${capacityRetryCount + 1}/${retryLimits.capacityRetriesBeforeFallback}) before falling back`, { level: 'warning' });
336
+ }
337
+ return { delay, switched: false };
338
+ }
339
+
340
+ const switchResult = await maybeSwitchToFallbackModel({ tool, argv, log, errorMessage });
341
+ // A model switch starts a fresh batch of same-model retries for the new model.
342
+ if (switchResult?.switched && argv) argv._capacityRetryCount = 0;
343
+ const delay = switchResult?.switched ? retryLimits.modelSwitchRetryDelayMs : getRetryDelayMs({ retryCount, initialDelayMs, maxDelayMs });
344
+ return { delay, switched: switchResult?.switched === true };
345
+ };
346
+
281
347
  export default {
282
348
  classifyRetryableError,
283
349
  getRetryDelayMs,
284
350
  waitWithCountdown,
285
351
  resolveConfiguredFallbackModel,
286
352
  maybeSwitchToFallbackModel,
353
+ prepareRetryAfterError,
287
354
  formatModelWithResolvedId,
288
355
  logExecutionContext,
289
356
  };