mixdog 0.9.90 → 0.9.92

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/package.json +5 -1
  2. package/scripts/tool-overhead-microbench.mjs +60 -0
  3. package/src/agents/debugger/agent.json +1 -1
  4. package/src/agents/explore/agent.json +1 -1
  5. package/src/agents/heavy-worker/agent.json +1 -1
  6. package/src/agents/maintainer/agent.json +1 -1
  7. package/src/agents/reviewer/agent.json +1 -1
  8. package/src/agents/worker/agent.json +1 -1
  9. package/src/lib/rules-builder.cjs +5 -5
  10. package/src/output-styles/detailed.md +27 -0
  11. package/src/output-styles/extreme-minimal.md +8 -8
  12. package/src/output-styles/minimal.md +7 -10
  13. package/src/output-styles/simple.md +12 -20
  14. package/src/rules/lead/01-general.md +9 -1
  15. package/src/rules/lead/lead-brief.md +15 -14
  16. package/src/rules/shared/01-tool.md +32 -37
  17. package/src/runtime/agent/orchestrator/agent-runtime/commit-message-completion.mjs +67 -0
  18. package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +10 -0
  19. package/src/runtime/agent/orchestrator/providers/anthropic.mjs +14 -0
  20. package/src/runtime/agent/orchestrator/providers/openai-oauth-ws.mjs +10 -0
  21. package/src/runtime/agent/orchestrator/providers/retry-classifier.mjs +35 -0
  22. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +47 -1
  23. package/src/runtime/agent/orchestrator/session/loop/stop-hooks.mjs +9 -0
  24. package/src/runtime/agent/orchestrator/session/loop/stored-tool-args.mjs +28 -1
  25. package/src/runtime/agent/orchestrator/session/manager/session-lifecycle.mjs +20 -2
  26. package/src/runtime/agent/orchestrator/session/result-classification.mjs +28 -0
  27. package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +176 -3
  28. package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +74 -11
  29. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +6 -6
  30. package/src/runtime/agent/orchestrator/tools/builtin/list-tool.mjs +15 -3
  31. package/src/runtime/agent/orchestrator/tools/builtin/rg-runner.mjs +9 -0
  32. package/src/runtime/agent/orchestrator/tools/builtin/search-tool.mjs +16 -2
  33. package/src/runtime/agent/orchestrator/tools/builtin/shell-analysis.mjs +176 -16
  34. package/src/runtime/agent/orchestrator/tools/builtin/task-tool.mjs +5 -0
  35. package/src/runtime/agent/orchestrator/tools/lib/pwsh-standby-pool.mjs +30 -1
  36. package/src/runtime/agent/orchestrator/tools/patch/matcher.mjs +1 -1
  37. package/src/runtime/agent/orchestrator/tools/patch/orchestrator.mjs +96 -5
  38. package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +72 -1
  39. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +1 -1
  40. package/src/runtime/agent/orchestrator/tools/shell-command.mjs +48 -9
  41. package/src/runtime/channels/backends/discord.mjs +21 -1
  42. package/src/runtime/channels/tool-defs.mjs +1 -1
  43. package/src/runtime/memory/lib/trace-store.mjs +25 -3
  44. package/src/runtime/memory/tool-defs.mjs +1 -3
  45. package/src/runtime/search/tool-defs.mjs +2 -18
  46. package/src/runtime/shared/tool-execution-contract.mjs +1 -1
  47. package/src/session-runtime/output-styles.mjs +4 -6
  48. package/src/session-runtime/tool-defs.mjs +0 -1
  49. package/src/session-runtime/tool-surface.mjs +9 -0
  50. package/src/session-runtime/workflow.mjs +25 -8
  51. package/src/tui/dist/index.mjs +32 -1
  52. package/src/tui/engine/session-api.mjs +17 -0
  53. package/src/tui/engine/tui-steering-persist.mjs +24 -1
  54. package/src/workflows/default/WORKFLOW.md +7 -17
  55. package/src/workflows/solo/WORKFLOW.md +7 -5
  56. package/src/output-styles/default.md +0 -40
@@ -128,6 +128,12 @@ export {
128
128
  // this catches tight deterministic-failure loops (e.g. a command that errors
129
129
  // the same way every time) far earlier than 100 iterations.
130
130
  const REPEAT_FAIL_LIMIT = 3;
131
+ // Structured provider continuations (endTurn=false / pause_turn) are honored,
132
+ // but must not sustain an unbounded text-only loop: a lead session was
133
+ // observed burning a 30-minute agent budget (26K output tokens, zero tool
134
+ // calls) on back-to-back continuations. After this many continuations with no
135
+ // intervening tool batch, the current text is accepted as the final answer.
136
+ const PROVIDER_CONTINUATION_NO_TOOL_LIMIT = Math.max(1, Number(process.env.MIXDOG_PROVIDER_CONTINUATION_NO_TOOL_LIMIT) || 8);
131
137
  // A provider max-output stop is not a completed assistant turn, even when it
132
138
  // contains useful text. Preserve each partial in the provider transcript and
133
139
  // grant at most three direct continuations before surfacing a hard truncation.
@@ -384,6 +390,12 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
384
390
  // hard iteration cap remains the sole bound on how long a provider may
385
391
  // keep declaring "not done" inside one user turn.
386
392
  let _providerContinuationCount = 0;
393
+ // Continuations since the last executed tool batch — bounds the text-only
394
+ // continuation runaway (see PROVIDER_CONTINUATION_NO_TOOL_LIMIT).
395
+ let _continuationsSinceToolBatch = 0;
396
+ // Loop-level transport replays consumed this ask (see send-with-recovery
397
+ // TRANSPORT_RETRY_MAX): bounded per turn, reset only with a fresh ask.
398
+ let _transportRetriesUsed = 0;
387
399
  // Claude Code parity: queued prompt/task notifications are attached after a
388
400
  // tool batch, before the continuation provider send. Normal batches drain
389
401
  // up to 'next'; a Sleep-like tool grants a 'later' flush.
@@ -432,6 +444,7 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
432
444
  Math.floor(maxLoopIterations * 0.9),
433
445
  ];
434
446
  while (true) {
447
+ const _iterT0 = Date.now();
435
448
  throwIfAborted();
436
449
  if (iterations >= maxLoopIterations) {
437
450
  // Final-answer turn: instead of breaking mid-transcript (which
@@ -627,14 +640,21 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
627
640
  () => sendWithRecovery({
628
641
  provider, messages, model, sendTools, tools: sendTools, opts,
629
642
  sessionId, sessionRef, nextIteration, contextOverflowRetryUsed,
643
+ transportRetriesUsed: _transportRetriesUsed, signal,
630
644
  }),
631
645
  );
646
+ const _sendEndedAt = Date.now();
632
647
  if (_sendResult.action === 'retry') {
633
648
  delete opts.cacheBreakIntent;
634
649
  contextOverflowRetryUsed = true;
635
650
  reactiveOverflowRetryPending = true;
636
651
  continue;
637
652
  }
653
+ if (_sendResult.action === 'retry_transport') {
654
+ _transportRetriesUsed += 1;
655
+ delete opts.cacheBreakIntent;
656
+ continue;
657
+ }
638
658
  response = _sendResult.response;
639
659
  opts.onToolCall = undefined;
640
660
  delete opts.cacheBreakIntent;
@@ -821,11 +841,26 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
821
841
  const continuationSignal = !isOutputLimitStop && stopReason !== 'refusal'
822
842
  ? providerContinuationSignal(response)
823
843
  : null;
824
- if (continuationSignal && pushIntermediateAssistantResponse(response)) {
844
+ if (continuationSignal && _continuationsSinceToolBatch >= PROVIDER_CONTINUATION_NO_TOOL_LIMIT) {
845
+ // Text-only continuation runaway: stop honoring the signal and
846
+ // fall through to the terminal handling below, which accepts
847
+ // the current content as the final answer (or ends the loop).
848
+ process.stderr.write(`[loop] provider continuation cap ${PROVIDER_CONTINUATION_NO_TOOL_LIMIT} reached without tool calls (sess=${sessionId || 'unknown'}); accepting current text as final.\n`);
849
+ try {
850
+ appendAgentTrace({
851
+ sessionId,
852
+ iteration: iterations,
853
+ kind: 'steer',
854
+ payload: { tag: 'provider_continuation_no_tool_cap', count: _continuationsSinceToolBatch },
855
+ agent: sessionAgent || null,
856
+ });
857
+ } catch { /* best-effort */ }
858
+ } else if (continuationSignal && pushIntermediateAssistantResponse(response)) {
825
859
  if (hasContent && !suppressMidTurnText) {
826
860
  try { opts.onAssistantText?.(response.content); } catch { /* best-effort */ }
827
861
  }
828
862
  _providerContinuationCount += 1;
863
+ _continuationsSinceToolBatch += 1;
829
864
  _emptyNudgeStreak = 0;
830
865
  try {
831
866
  appendAgentTrace({
@@ -1005,6 +1040,7 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
1005
1040
  continue;
1006
1041
  }
1007
1042
  try { opts.onToolPhaseStarted?.(); } catch {}
1043
+ const _toolsT0 = Date.now();
1008
1044
  ({ dedupStubTotal: _dedupStubTotal, editCount: _editCount } = await processToolBatch({
1009
1045
  calls: _callsToExecute, messages, tools, cwd, sessionId, sessionRef, signal, opts,
1010
1046
  iterations, assistantTurnMsg: _assistantTurnMsg,
@@ -1017,7 +1053,17 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
1017
1053
  }));
1018
1054
  // Settle the stop hook on the batch that actually executed.
1019
1055
  _toolFailureStopHook.endBatch(_callsToExecute);
1056
+ // Loop-phase timing (diagnostics): where non-model time goes per
1057
+ // iteration — presend (repair/compact/snapshot), send (provider
1058
+ // round-trip incl. streaming), tools (batch execution). Gated by the
1059
+ // same env as [turn-timing] so bench runs opt in via -AgentEnv.
1060
+ if (process.env.MIXDOG_TURN_TIMING === '1') {
1061
+ try {
1062
+ process.stderr.write(`[loop-timing] iter=${nextIteration} presend=${sendStartedAt - _iterT0}ms send=${_sendEndedAt - sendStartedAt}ms tools=${Date.now() - _toolsT0}ms calls=${_callsToExecute.length}\n`);
1063
+ } catch { /* diagnostics only */ }
1064
+ }
1020
1065
  _toolBatchJustCompleted = true;
1066
+ _continuationsSinceToolBatch = 0;
1021
1067
  _lastToolBatchHadSleep = _callsToExecute.some(isSleepLikeToolCall);
1022
1068
  }
1023
1069
  // Classify WHY the loop ended so agent-tool can promote an empty/abnormal
@@ -14,6 +14,8 @@
14
14
 
15
15
  export const STOP_HOOK_SOURCE = 'tool-failure-stop-hook';
16
16
 
17
+ import { isInformationalShellExitOne } from '../result-classification.mjs';
18
+
17
19
  // Only a genuinely EXECUTED result resolves a failure, i.e. kind 'normal'.
18
20
  // Cache hits ('cache-hit' / 'scoped-cache-hit') replay an earlier result
19
21
  // without running anything, and dedup/guard skips ('skipped') execute nothing
@@ -59,6 +61,13 @@ export function createToolFailureStopHook() {
59
61
  // nothing was dispatched — they must not arm the hook.
60
62
  if (message.guardSkip === true) return;
61
63
  if (message.toolKind === 'error') {
64
+ // Informational exit-1 probes (grep-family no-match inside a
65
+ // compound command: useful stdout, blank stderr) stay 'error'
66
+ // for display/history, but blocking the terminal message over
67
+ // them forces a pointless re-verify turn — observed live
68
+ // (kv-store-grpc: /proc PID scan exit 1 → hook misfire, +2
69
+ // turns). Neutral here: neither arms nor clears.
70
+ if (isInformationalShellExitOne(message.content)) return;
62
71
  batchFailure = true;
63
72
  if (message.toolCallId) failedCallIds.add(message.toolCallId);
64
73
  } else if (EXECUTED_SUCCESS_TOOL_KINDS.has(message.toolKind)) {
@@ -28,6 +28,30 @@ const STORED_TOOL_ARG_LIMIT = 10_000;
28
28
  const STORED_TOOL_ARG_PREVIEW_HEAD = 360;
29
29
  const STORED_TOOL_ARG_PREVIEW_TAIL = 160;
30
30
 
31
+ // File paths a compacted patch touched, so the marker itself tells the model
32
+ // WHICH files to re-read instead of replaying the marker as patch input
33
+ // (measured: the compacted-placeholder resubmission was 39% of apply_patch
34
+ // failures). Marker contract: the returned text may not contain ']' or a
35
+ // newline, so bracket characters are stripped from paths.
36
+ function _compactedPatchTargets(value) {
37
+ const seen = new Set();
38
+ const add = (raw) => {
39
+ const p = String(raw || '').trim().replace(/[\[\]\r\n]/g, '');
40
+ if (p && p !== '/dev/null' && !seen.has(p)) seen.add(p);
41
+ };
42
+ const v4a = /^\*\*\*\s*(?:Update|Add|Delete) File:\s*(.+)$/gim;
43
+ for (let m; seen.size < 12 && (m = v4a.exec(value));) add(m[1]);
44
+ if (!seen.size) {
45
+ const uni = /^(?:\+\+\+|---)\s+(?:[ab]\/)?(\S+)/gm;
46
+ for (let m; seen.size < 12 && (m = uni.exec(value));) add(m[1]);
47
+ }
48
+ const all = [...seen];
49
+ const shown = all.slice(0, 4).map((p) => (p.length > 70 ? `…${p.slice(-70)}` : p));
50
+ if (!shown.length) return '';
51
+ const more = all.length > shown.length ? ` +${all.length - shown.length} more` : '';
52
+ return `${shown.join(', ')}${more}`;
53
+ }
54
+
31
55
  function compactStoredToolArgString(value, key = '', opts = {}) {
32
56
  if (typeof value !== 'string') return value;
33
57
  const isBody = STORED_TOOL_ARG_BODY_KEY_RE.test(key);
@@ -39,8 +63,11 @@ function compactStoredToolArgString(value, key = '', opts = {}) {
39
63
  // Body markers carry the recovery instruction inline: the compaction
40
64
  // detectors only require the `[mixdog compacted ...]` shape (no ']' or
41
65
  // newline inside), so the longer text stays fully compatible.
66
+ const targets = /^patch$/i.test(key) ? _compactedPatchTargets(value) : '';
42
67
  const marker = isBody
43
- ? `[mixdog compacted ${key}: ${value.length} chars, sha256:${hash}; already applied - do not copy; re-read the file and write a fresh patch]`
68
+ ? (targets
69
+ ? `[mixdog compacted ${key}: ${value.length} chars, sha256:${hash}; already applied to ${targets} - do not copy; re-read those files and write a fresh patch]`
70
+ : `[mixdog compacted ${key}: ${value.length} chars, sha256:${hash}; already applied - do not copy; re-read the file and write a fresh patch]`)
44
71
  : `[mixdog compacted ${key || 'string'}: ${value.length} chars, sha256:${hash}; do not copy]`;
45
72
  // Body args (patch / old_string / new_string / content / rewrite) are
46
73
  // apply_patch / edit inputs. Keeping a head/tail preview leaves real patch
@@ -318,7 +318,11 @@ export function createSession(opts) {
318
318
  const hasCallerAllow = Array.isArray(opts.schemaAllowedTools);
319
319
  const tools = finalizeSessionToolList(toolsForRouting, {
320
320
  schemaAllowedTools: hasCallerAllow ? opts.schemaAllowedTools : null,
321
- disallowedTools: hiddenAgent ? [...(Array.isArray(opts.disallowedTools) ? opts.disallowedTools : []), 'Skill'] : opts.disallowedTools,
321
+ disallowedTools: [
322
+ ...(Array.isArray(opts.disallowedTools) ? opts.disallowedTools : []),
323
+ ...(hiddenAgent ? ['Skill'] : []),
324
+ ...(!ownerIsAgent && workflowDisallowsAgentTool(opts.workflow) ? ['agent'] : []),
325
+ ],
322
326
  ownerIsAgent,
323
327
  resolvedAgent,
324
328
  });
@@ -499,6 +503,17 @@ const ACTIVE_OWNER_HB_FRESH_MS = 2 * 60 * 1000; // heartbeat freshness window
499
503
  const PREPARED_RESUME_LIMIT = 8;
500
504
  const _preparedResumes = new Map();
501
505
 
506
+ // A workflow that delegates to NOBODY (agents: declared empty — e.g. Solo)
507
+ // must not put the `agent` tool in the session tool list: policy rejects
508
+ // every call, so a schema-visible tool is a guaranteed error turn plus dead
509
+ // schema weight. Field source: workflowSummary() carries agentsConfigured /
510
+ // agents; older persisted sessions lack them and keep the tool (safe).
511
+ function workflowDisallowsAgentTool(workflow) {
512
+ return Boolean(workflow && typeof workflow === 'object'
513
+ && workflow.agentsConfigured === true
514
+ && Array.isArray(workflow.agents) && workflow.agents.length === 0);
515
+ }
516
+
502
517
  function _prepareResumeTools(session, preset) {
503
518
  const ownerIsAgent = isAgentOwner(session);
504
519
  const skills = ownerIsAgent ? [] : collectPromptSkillsCached(session.cwd);
@@ -521,7 +536,10 @@ function _prepareResumeTools(session, preset) {
521
536
  ownerIsAgent,
522
537
  tools: finalizeSessionToolList(toolsForRouting, {
523
538
  schemaAllowedTools: Array.isArray(session.schemaAllowedTools) ? session.schemaAllowedTools : null,
524
- disallowedTools: getHiddenAgent(session.agent || null) ? ['Skill'] : null,
539
+ disallowedTools: [
540
+ ...(getHiddenAgent(session.agent || null) ? ['Skill'] : []),
541
+ ...(!isAgentOwner(session) && workflowDisallowsAgentTool(session.workflow) ? ['agent'] : []),
542
+ ],
525
543
  ownerIsAgent,
526
544
  resolvedAgent: session.agent || null,
527
545
  }),
@@ -68,3 +68,31 @@ export function classifyResultKind(result, explicitSuccess = false) {
68
68
  }
69
69
  return 'normal';
70
70
  }
71
+
72
+ /**
73
+ * Informational shell exit-1: `Error: [shell-run-failed] [exit code: 1]`
74
+ * with a non-empty stdout body and NO stderr evidence (neither an inline
75
+ * `[stderr]` block nor a `[stderr: path]` spill). grep-family "no match"
76
+ * semantics inside compound probes (loops, `;`-chains, substitutions) land
77
+ * exactly here: the run produced useful output, wrote nothing to stderr,
78
+ * and exited 1 only because the final stage matched nothing. The static
79
+ * single-pipeline gate (bash-tool _isBenignSearchExitOne) deliberately
80
+ * refuses these ambiguous shapes, so the result stays toolKind 'error' —
81
+ * consumers that must not overreact to an informational failure (the turn
82
+ * stop hook) test this signature instead of reclassifying the result.
83
+ * Signals, timeouts, and other exit codes carry different status markers
84
+ * and never match; a destructive-warning prefix also disqualifies.
85
+ *
86
+ * @param {unknown} result
87
+ * @returns {boolean}
88
+ */
89
+ export function isInformationalShellExitOne(result) {
90
+ if (typeof result !== 'string') return false;
91
+ const trimmed = result.trimStart();
92
+ const header = /^error:\s*\[shell-run-failed\]\s*\[exit code: 1\]\s*\n/i.exec(trimmed);
93
+ if (!header) return false;
94
+ const payload = trimmed.slice(header[0].length).trim();
95
+ if (!payload || payload === '(no output)') return false;
96
+ if (payload.startsWith('[stderr') || payload.includes('\n[stderr')) return false;
97
+ return true;
98
+ }
@@ -5,7 +5,8 @@
5
5
  // pass), and unrecoverable errors throw. Behavior identical to the inline
6
6
  // try/catch it replaced.
7
7
  import { appendAgentTrace } from '../agent-trace.mjs';
8
- import { isContextOverflowError } from '../providers/retry-classifier.mjs';
8
+ import { classifyError, isContextOverflowError } from '../providers/retry-classifier.mjs';
9
+ import { setTimeout as sleepMs } from 'timers/promises';
9
10
  import { readStreamOutcome } from '../providers/lib/stream-outcome.mjs';
10
11
  import { resolveWorkerCompactPolicy } from './loop/compact-policy.mjs';
11
12
  import { agentContextOverflowError } from './loop/context-overflow.mjs';
@@ -35,21 +36,126 @@ function normalizedIncompleteUsage(raw) {
35
36
  };
36
37
  }
37
38
 
39
+ // Loop-level transport replay (codex retry_transport parity). The provider
40
+ // layer already retries transient failures with a ~15s total envelope
41
+ // (PROVIDER_RETRY_BACKOFF_MS); a real network blip (router/VPN flap — the
42
+ // 2026-08-02 17:37 incident also dropped the Discord gateway) outlasts it and
43
+ // used to surface as a failed turn. When the stream exposed NOTHING
44
+ // (replaySafe: no relayed text/reasoning, no dispatched tool call), re-sending
45
+ // the identical request is side-effect-free, so wait out the blip and retry
46
+ // the send at the loop level. Two attempts, 5s/15s — combined with the
47
+ // provider envelope this covers ~50s outages before failing honestly.
48
+ const TRANSPORT_RETRY_BACKOFF_MS = Object.freeze([5_000, 15_000]);
49
+ export const TRANSPORT_RETRY_MAX = TRANSPORT_RETRY_BACKOFF_MS.length;
50
+
38
51
  export async function sendWithRecovery(ctx) {
39
52
  const {
40
53
  provider, messages, model, sendTools, tools, opts,
41
54
  sessionId, sessionRef, nextIteration, contextOverflowRetryUsed,
55
+ transportRetriesUsed = 0, signal,
42
56
  } = ctx;
43
57
  let response;
58
+ // Bench-only turn timing (MIXDOG_TURN_TIMING=1): one stderr line per
59
+ // provider request — TTFT (first stream delta) and total stream time.
60
+ // Inert unless the env flag is set; used to profile harness vs model
61
+ // latency in Terminal-Bench runs.
62
+ const turnT0 = process.env.MIXDOG_TURN_TIMING === '1' ? Date.now() : 0;
63
+ let turnFirstDelta = 0;
64
+ let timedOpts = opts;
65
+ if (turnT0) {
66
+ const prevDelta = typeof opts?.onStreamDelta === 'function' ? opts.onStreamDelta : null;
67
+ timedOpts = {
68
+ ...(opts || {}),
69
+ onStreamDelta: (kind) => {
70
+ if (!turnFirstDelta) turnFirstDelta = Date.now();
71
+ if (prevDelta) prevDelta(kind);
72
+ },
73
+ };
74
+ }
75
+ const logTurnTiming = (status) => {
76
+ if (!turnT0) return;
77
+ const now = Date.now();
78
+ const ttft = turnFirstDelta ? turnFirstDelta - turnT0 : -1;
79
+ try {
80
+ console.error(`[turn-timing] status=${status} ttft=${ttft}ms total=${now - turnT0}ms model=${model}`);
81
+ } catch { /* logging must never break the send path */ }
82
+ };
83
+ // Loop-side exposure witness. Some providers throw truncation/stall
84
+ // errors that carry only partialContent — neither liveTextEmitted nor
85
+ // unsafeToRetry — so an outcome read from the ERROR alone can report
86
+ // replaySafe even though this very send already relayed text to the
87
+ // client through opts.onTextDelta, or dispatched a tool call through
88
+ // opts.onToolCall. Replaying such a send would duplicate output the user
89
+ // already saw (or re-run a side effect), so record what THIS send
90
+ // actually exposed and merge it into every outcome read below. The
91
+ // callbacks are wrapped in place and restored conditionally: the
92
+ // overflow-retry branch intentionally clears opts.onToolCall, and that
93
+ // clear must survive the restore.
94
+ const relayWitness = { textEmitted: false, toolCallsDispatched: 0 };
95
+ const prevOnTextDelta = typeof opts?.onTextDelta === 'function' ? opts.onTextDelta : null;
96
+ const prevOnToolCall = typeof opts?.onToolCall === 'function' ? opts.onToolCall : null;
97
+ const witnessedOnTextDelta = prevOnTextDelta
98
+ ? (...args) => {
99
+ if (typeof args[0] === 'string' && args[0].length > 0) relayWitness.textEmitted = true;
100
+ return prevOnTextDelta(...args);
101
+ }
102
+ : null;
103
+ const witnessedOnToolCall = prevOnToolCall
104
+ ? (...args) => {
105
+ relayWitness.toolCallsDispatched += 1;
106
+ return prevOnToolCall(...args);
107
+ }
108
+ : null;
109
+ if (opts) {
110
+ if (witnessedOnTextDelta) opts.onTextDelta = witnessedOnTextDelta;
111
+ if (witnessedOnToolCall) opts.onToolCall = witnessedOnToolCall;
112
+ }
113
+ if (timedOpts !== opts && timedOpts) {
114
+ if (witnessedOnTextDelta) timedOpts.onTextDelta = witnessedOnTextDelta;
115
+ if (witnessedOnToolCall) timedOpts.onToolCall = witnessedOnToolCall;
116
+ }
117
+ try {
44
118
  try {
45
- response = await provider.send(messages, model, sendTools.length ? sendTools : undefined, opts);
119
+ response = await provider.send(messages, model, sendTools.length ? sendTools : undefined, timedOpts);
120
+ logTurnTiming('ok');
46
121
  } catch (sendErr) {
122
+ logTurnTiming(`err:${sendErr?.code || sendErr?.name || 'unknown'}`);
47
123
  // Canonical stream outcome: ONE fail-closed read of what the
48
124
  // provider stream actually produced (terminal vs continuation,
49
125
  // observed text/reasoning, partial/complete/dispatched tool calls).
50
126
  // Every branch below consumes it instead of re-inferring safety
51
127
  // from provider-specific flags.
52
- const outcome = readStreamOutcome(sendErr);
128
+ const outcome = readStreamOutcome(sendErr, relayWitness);
129
+ // Text-only exposure retraction (cross-provider): a stream that
130
+ // died after relaying ONLY text — no dispatched or complete tool
131
+ // calls, no terminal — is replayable IF the ask owner retracts the
132
+ // exposed characters (onTextReset ack === true: the TUI truncates
133
+ // its live tail, the bench driver truncates its accumulator).
134
+ // This is the loop-level analogue of anthropic's
135
+ // recoverNonStreaming for providers WITHOUT a non-streaming
136
+ // fallback (gemini, openai-compat, openai WS) and for stalls that
137
+ // outlived the provider's in-place recovery. Observed live:
138
+ // make-mips-interpreter died with 31 exposed chars + a pending
139
+ // never-dispatched tool input and burned the whole trial.
140
+ const retractExposedTextForReplay = async () => {
141
+ if (outcome.terminalObserved === true) return false;
142
+ if (outcome.sideEffectDispatched === true) return false;
143
+ if (outcome.dispatchAmbiguous === true) return false;
144
+ if (Number(outcome.toolCallsDispatched) > 0) return false;
145
+ if (Number(outcome.toolCallsComplete) > 0) return false;
146
+ if (relayWitness.toolCallsDispatched > 0) return false;
147
+ if (typeof opts?.onTextReset !== 'function') return false;
148
+ const chars = Math.max(0, Number(outcome.textObservedChars) || 0)
149
+ || (typeof sendErr.partialContent === 'string' ? sendErr.partialContent.length : 0);
150
+ if (chars <= 0) return false;
151
+ let acked = false;
152
+ try {
153
+ acked = await opts.onTextReset({ chars, reason: 'loop-transport-retraction' }) === true;
154
+ } catch { acked = false; }
155
+ if (!acked) return false;
156
+ relayWitness.textEmitted = false;
157
+ return true;
158
+ };
53
159
  // Gemini REST/SDK reports MAX_TOKENS by throwing a typed
54
160
  // ProviderIncompleteError after preserving the streamed candidate.
55
161
  // Normalize only that exact, safe no-tool output-limit shape into a
@@ -101,6 +207,36 @@ export async function sendWithRecovery(ctx) {
101
207
  && sendErr.partialContent.trim().length > 0
102
208
  && outcome.toolCallsComplete === 0
103
209
  ) {
210
+ // Retractable shape: text-only exposure with the owner's
211
+ // acknowledgement replays on a fresh request instead of
212
+ // failing the turn. Non-acked (or tool-bearing) shapes keep
213
+ // the explicit-failure contract below unchanged.
214
+ if (
215
+ transportRetriesUsed < TRANSPORT_RETRY_MAX
216
+ && classifyError(sendErr) === 'transient'
217
+ && await retractExposedTextForReplay()
218
+ ) {
219
+ const waitMs = TRANSPORT_RETRY_BACKOFF_MS[transportRetriesUsed];
220
+ try {
221
+ process.stderr.write(
222
+ `[loop] exposed-text stall retracted (sess=${sessionId || 'unknown'} `
223
+ + `iter=${nextIteration} len=${sendErr.partialContent.length}); `
224
+ + `transport retry ${transportRetriesUsed + 1}/${TRANSPORT_RETRY_MAX} after ${waitMs}ms\n`,
225
+ );
226
+ } catch { /* best-effort */ }
227
+ try {
228
+ appendAgentTrace({
229
+ kind: 'exposed_text_retraction_retry',
230
+ sessionId: sessionId || null,
231
+ iteration: nextIteration,
232
+ attempt: transportRetriesUsed + 1,
233
+ waitMs,
234
+ partialContentLen: sendErr.partialContent.length,
235
+ });
236
+ } catch { /* best-effort */ }
237
+ await sleepMs(waitMs, undefined, signal ? { signal } : undefined);
238
+ return { action: 'retry_transport' };
239
+ }
104
240
  try {
105
241
  process.stderr.write(
106
242
  `[loop] final stream stalled with partial text (sess=${sessionId || 'unknown'} `
@@ -185,6 +321,35 @@ export async function sendWithRecovery(ctx) {
185
321
  };
186
322
  return { action: 'proceed', response };
187
323
  } else
324
+ // Clean transient transport failure with zero exposure: replay the
325
+ // send after a bounded wait instead of failing the turn.
326
+ if (
327
+ transportRetriesUsed < TRANSPORT_RETRY_MAX
328
+ && classifyError(sendErr) === 'transient'
329
+ && (outcome.replaySafe === true || await retractExposedTextForReplay())
330
+ ) {
331
+ const waitMs = TRANSPORT_RETRY_BACKOFF_MS[transportRetriesUsed];
332
+ try {
333
+ process.stderr.write(
334
+ `[loop] transient send failure with no observed output (sess=${sessionId || 'unknown'} `
335
+ + `iter=${nextIteration} code=${sendErr?.code ?? sendErr?.status ?? 'n/a'}); `
336
+ + `transport retry ${transportRetriesUsed + 1}/${TRANSPORT_RETRY_MAX} after ${waitMs}ms\n`,
337
+ );
338
+ } catch { /* best-effort */ }
339
+ try {
340
+ appendAgentTrace({
341
+ kind: 'transport_retry',
342
+ sessionId: sessionId || null,
343
+ iteration: nextIteration,
344
+ attempt: transportRetriesUsed + 1,
345
+ waitMs,
346
+ code: sendErr?.code ?? null,
347
+ status: sendErr?.status ?? null,
348
+ });
349
+ } catch { /* best-effort */ }
350
+ await sleepMs(waitMs, undefined, signal ? { signal } : undefined);
351
+ return { action: 'retry_transport' };
352
+ } else
188
353
  // Context-window-exceeded is a deterministic refusal from the API.
189
354
  // Recover context overflow reactively by compacting and retrying
190
355
  // in the same active turn. MixDog's proactive estimator can miss a
@@ -236,5 +401,13 @@ export async function sendWithRecovery(ctx) {
236
401
  messageTokensEst: estimateMessagesTokensSafe(messages),
237
402
  }, sendErr);
238
403
  }
404
+ } finally {
405
+ // Conditional restore: only unwind our own wrappers. An intentional
406
+ // opts.onToolCall = undefined (overflow-retry branch) stays cleared.
407
+ if (opts) {
408
+ if (witnessedOnTextDelta && opts.onTextDelta === witnessedOnTextDelta) opts.onTextDelta = prevOnTextDelta;
409
+ if (witnessedOnToolCall && opts.onToolCall === witnessedOnToolCall) opts.onToolCall = prevOnToolCall;
410
+ }
411
+ }
239
412
  return { action: 'proceed', response };
240
413
  }
@@ -1,5 +1,5 @@
1
1
  import { getAbortSignalForSession } from '../../session/abort-lookup.mjs';
2
- import { execShellCommand, stripAnsi } from '../shell-command.mjs';
2
+ import { acquireShellLeaseBounded, execShellCommand, stripAnsi } from '../shell-command.mjs';
3
3
  import { wrapCommandWithSnapshot } from '../shell-snapshot.mjs';
4
4
  import { getDestructiveCommandWarning } from '../destructive-warning.mjs';
5
5
  import { maybeRewriteWmicProcessCommand } from '../shell-policy.mjs';
@@ -21,6 +21,11 @@ import {
21
21
  } from './shell-jobs.mjs';
22
22
  import {
23
23
  analyzeShellCommandEffects,
24
+ buildPowerShellFilterTeePlan,
25
+ consumeFilterTeeCapture,
26
+ detectBlockedSleepPattern,
27
+ detectLongForegroundReason,
28
+ extractShellApplyPatchInvocation,
24
29
  foregroundLongCommandHint,
25
30
  isAutobackgroundingAllowed,
26
31
  preflightPowerShellHygiene,
@@ -241,7 +246,7 @@ export async function executeBashTool(args, workDir, options = {}) {
241
246
  const bashWorkDir = resolveSessionCwd(_sessionCwdKey, _hasExplicitCwd ? cwdResult.cwd : null, cwdResult.cwd);
242
247
  const _readStateScope = options?.readStateScope ?? options?.sessionId ?? null;
243
248
  const executionMode = resolveExecutionMode(args || {}, args?.run_in_background === true ? 'async' : 'sync');
244
- const runInBackground = executionMode === 'async';
249
+ let runInBackground = executionMode === 'async';
245
250
 
246
251
  // Run hard-block policy BEFORE branching into the persistent-shell tool.
247
252
  // The persistent path used to bypass the one-shot block scan because the
@@ -251,6 +256,21 @@ export async function executeBashTool(args, workDir, options = {}) {
251
256
  // decode + rm token guard; calling it here applies the same allowlist
252
257
  // to both persistent and stateless paths.
253
258
  const _rawCmd = String(args && args.command != null ? args.command : '');
259
+ // Codex-parity: `apply_patch` typed into the shell (heredoc/argument/bare
260
+ // patch forms) routes to the internal patch engine instead of failing as
261
+ // an unknown binary. Runs BEFORE the exec-policy scan so patch BODY lines
262
+ // (e.g. `+ rm -rf …`) are never misread as shell commands. Dynamic import
263
+ // avoids a bash-tool <-> patch/orchestrator module cycle.
264
+ if (_rawCmd) {
265
+ const _apCall = extractShellApplyPatchInvocation(_rawCmd);
266
+ if (_apCall?.error) {
267
+ return formatShellToolFailure(`${_apCall.error}. Call the apply_patch tool with the patch string instead of the shell.`);
268
+ }
269
+ if (_apCall?.patch) {
270
+ const { executePatchTool } = await import('../patch/orchestrator.mjs');
271
+ return executePatchTool('apply_patch', { patch: _apCall.patch }, bashWorkDir, options);
272
+ }
273
+ }
254
274
  if (_rawCmd) {
255
275
  // R5-③: persistent:true used to route into bash_session BEFORE the
256
276
  // stripQuotedAndHeredoc / extractShellCInner / unquote sweep ran
@@ -339,6 +359,29 @@ export async function executeBashTool(args, workDir, options = {}) {
339
359
  return formatShellToolFailure(_execPolicyBlock);
340
360
  }
341
361
 
362
+ // Sleep-chain auto-promotion: a leading `sleep N && …` / `Start-Sleep N; …`
363
+ // used to be DENIED preflight (CC detectBlockedSleepPattern parity).
364
+ // Measured over 10 days the deny fired 46× and every hit was a wasted
365
+ // turn, so the command is promoted to a background task instead — the
366
+ // exact remedy the deny message pointed at, without the failure. Only
367
+ // possible while background tasks are enabled; with them disabled the
368
+ // command runs foreground as before (the deny never fired there either).
369
+ const _bgTasksDisabled = /^(1|true|yes|on)$/i.test(
370
+ String(process.env.MIXDOG_SHELL_DISABLE_BACKGROUND_TASKS || '').trim(),
371
+ );
372
+ let autoAsyncReason = '';
373
+ if (!runInBackground && !_bgTasksDisabled) {
374
+ // Long-foreground shapes (watch-like dev servers/watchers, 30s+ sleeps
375
+ // anywhere in the chain) used to hard-fail via foregroundLongCommandHint
376
+ // (~22 wasted turns/14d measured). Promote them to a background task —
377
+ // the exact remedy the deny message pointed at — same as sleep chains.
378
+ const _blockedSleep = detectBlockedSleepPattern(command) || detectLongForegroundReason(command);
379
+ if (_blockedSleep) {
380
+ runInBackground = true;
381
+ autoAsyncReason = _blockedSleep;
382
+ }
383
+ }
384
+
342
385
  let shellEffects;
343
386
  let combinedBashAbort = null;
344
387
  try {
@@ -372,9 +415,6 @@ export async function executeBashTool(args, workDir, options = {}) {
372
415
  : DEFAULT_BASH_TIMEOUT_MS;
373
416
  const hasExplicitTimeout = typeof args.timeout === 'number' && args.timeout > 0;
374
417
  const timeoutMs = hasExplicitTimeout ? args.timeout : defaultTimeoutMs;
375
- const _bgTasksDisabled = /^(1|true|yes|on)$/i.test(
376
- String(process.env.MIXDOG_SHELL_DISABLE_BACKGROUND_TASKS || '').trim(),
377
- );
378
418
  const backgroundOnTimeout = !runInBackground
379
419
  && !_bgTasksDisabled
380
420
  && isAutobackgroundingAllowed(command, resolvedSpec.shellType);
@@ -429,11 +469,20 @@ export async function executeBashTool(args, workDir, options = {}) {
429
469
  scrubLoaderVars(spawnEnv);
430
470
  scrubRuntimeRootVars(spawnEnv);
431
471
  let wrappedCommand;
472
+ let _teePlan = null;
432
473
  // PowerShell UTF-8 prefix is PS-only: the Windows Git Bash path
433
474
  // (shellType==='posix') must NOT receive it. Snapshot wrapper stays
434
475
  // POSIX-host-only for now — no snapshot for Windows Git Bash initially.
435
476
  if (process.platform === 'win32' && shellType === 'powershell') {
436
- wrappedCommand = _prefixPowerShellUtf8(command);
477
+ // Filter-swallow rescue (sync path only): tee the unfiltered
478
+ // producer stream of an exactly-recognized filter pipeline so a
479
+ // failing run can attach the original output tail in THIS call
480
+ // instead of returning `[exit code: N]` + `(no output)`. Any
481
+ // ambiguity yields a null plan and the command runs untouched.
482
+ if (!runInBackground) {
483
+ try { _teePlan = buildPowerShellFilterTeePlan(command); } catch { _teePlan = null; }
484
+ }
485
+ wrappedCommand = _prefixPowerShellUtf8(_teePlan ? _teePlan.command : command);
437
486
  } else if (process.platform !== 'win32' && (shell.includes('bash') || shell.includes('zsh'))) {
438
487
  try {
439
488
  wrappedCommand = await wrapCommandWithSnapshot(shell, command);
@@ -451,8 +500,8 @@ export async function executeBashTool(args, workDir, options = {}) {
451
500
  let asyncLease = null;
452
501
  let job;
453
502
  try {
454
- asyncLease = await (options?.resourceAdmission || resourceAdmission).acquire('shell', {
455
- signal: combinedAsyncAbort.signal,
503
+ asyncLease = await acquireShellLeaseBounded(options?.resourceAdmission || resourceAdmission, {
504
+ abortSignal: combinedAsyncAbort.signal,
456
505
  label: String(command).replace(/\s+/g, ' ').slice(0, 120),
457
506
  dependency: 'detached',
458
507
  });
@@ -525,7 +574,10 @@ export async function executeBashTool(args, workDir, options = {}) {
525
574
  clientHostPid: options?.clientHostPid,
526
575
  });
527
576
  } catch { /* watcher arm is best-effort; never blocks the spawn */ }
528
- return _prependDestructiveWarning(command, renderBackgroundTask(task));
577
+ const _autoAsyncNote = autoAsyncReason
578
+ ? `[auto-async] ${autoAsyncReason} — promoted to a background task; act on its completion notification instead of blocking (do not poll).\n`
579
+ : '';
580
+ return _prependDestructiveWarning(command, _autoAsyncNote + renderBackgroundTask(task));
529
581
  } catch (error) {
530
582
  if (job?.jobId && !job.error) {
531
583
  try { killShellJob(job.jobId); } catch {}
@@ -643,6 +695,17 @@ export async function executeBashTool(args, workDir, options = {}) {
643
695
  const benignExitOne = _isBenignSearchExitOne(command, exitCode, signal, stderr);
644
696
  const shellRunFailed = !shellToolFailed && (!!signal || (exitCode !== 0 && exitCode !== null && !benignExitOne));
645
697
  const isReallyErrored = shellToolFailed || shellRunFailed;
698
+ // Filter-swallow rescue: the tee file is ALWAYS consumed (deleted)
699
+ // here; its tail is attached only when the run failed with an empty
700
+ // visible capture — the exact `(no output)` shape that previously
701
+ // cost the model extra diagnostic turns.
702
+ let _rescueNote = '';
703
+ if (_teePlan) {
704
+ const _rescueTail = consumeFilterTeeCapture(_teePlan.teePath);
705
+ if (isReallyErrored && _rescueTail && !stdout.trim() && !stderr.trim()) {
706
+ _rescueNote = `\n\n[filter-swallowed output rescue] the command failed but its trailing filter(s) matched nothing, so the visible output was empty. Unfiltered pipeline output (tail):\n${smartMiddleTruncate(_rescueTail)}`;
707
+ }
708
+ }
646
709
  const _driftNote = '';
647
710
  // Distinct timeout marker so callers see "killed by timeout after Nms"
648
711
  // vs an external signal (e.g. user Ctrl-C, OOM kill). result.timedOut
@@ -666,7 +729,7 @@ export async function executeBashTool(args, workDir, options = {}) {
666
729
  // cross-stream interleaving. Acceptable for most diagnostic
667
730
  // outputs; flag in shell-command if exact interleaving is required.
668
731
  const merged = stdout + stderr;
669
- if (statusMarker) return _prependDestructiveWarning(command, errorPrefix + smartMiddleTruncate(`${statusMarker}\n\n${merged || '(no output)'}`) + _driftNote);
732
+ if (statusMarker) return _prependDestructiveWarning(command, errorPrefix + smartMiddleTruncate(`${statusMarker}\n\n${merged || '(no output)'}`) + _rescueNote + _driftNote);
670
733
  return _prependDestructiveWarning(command, smartMiddleTruncate(merged || '(no output)') + _driftNote);
671
734
  }
672
735
  const truncatedStdout = smartMiddleTruncate(stdout);
@@ -688,7 +751,7 @@ export async function executeBashTool(args, workDir, options = {}) {
688
751
  const warningBlock = [
689
752
  wmicRewrite?.note || '',
690
753
  ].filter(Boolean).join('\n');
691
- const payload = `${body}${stderrBlock}${spillBlock}${_driftNote}`;
754
+ const payload = `${body}${stderrBlock}${spillBlock}${_rescueNote}${_driftNote}`;
692
755
  if (statusMarker) return _prependDestructiveWarning(command, _composeShellFailure(statusMarker, errorPrefix, warningBlock, payload));
693
756
  return _prependDestructiveWarning(command, warningBlock ? `${warningBlock}\n${payload}` : payload);
694
757
  }