mixdog 0.9.89 → 0.9.91

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/package.json +5 -1
  2. package/src/output-styles/detailed.md +27 -0
  3. package/src/output-styles/extreme-minimal.md +8 -8
  4. package/src/output-styles/minimal.md +7 -10
  5. package/src/output-styles/simple.md +13 -21
  6. package/src/rules/lead/01-general.md +9 -1
  7. package/src/rules/lead/lead-brief.md +15 -14
  8. package/src/rules/shared/01-tool.md +31 -37
  9. package/src/runtime/agent/orchestrator/agent-runtime/commit-message-completion.mjs +67 -0
  10. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +35 -1
  11. package/src/runtime/agent/orchestrator/session/loop/stored-tool-args.mjs +28 -1
  12. package/src/runtime/agent/orchestrator/session/manager/session-lifecycle.mjs +20 -2
  13. package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +116 -3
  14. package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +74 -11
  15. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +9 -10
  16. package/src/runtime/agent/orchestrator/tools/builtin/list-tool.mjs +15 -3
  17. package/src/runtime/agent/orchestrator/tools/builtin/rg-runner.mjs +9 -0
  18. package/src/runtime/agent/orchestrator/tools/builtin/search-tool.mjs +16 -2
  19. package/src/runtime/agent/orchestrator/tools/builtin/shell-analysis.mjs +176 -16
  20. package/src/runtime/agent/orchestrator/tools/code-graph-tool-defs.mjs +1 -1
  21. package/src/runtime/agent/orchestrator/tools/lib/pwsh-standby-pool.mjs +30 -1
  22. package/src/runtime/agent/orchestrator/tools/patch/matcher.mjs +1 -1
  23. package/src/runtime/agent/orchestrator/tools/patch/orchestrator.mjs +96 -5
  24. package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +72 -1
  25. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +1 -1
  26. package/src/runtime/agent/orchestrator/tools/shell-command.mjs +48 -9
  27. package/src/runtime/channels/backends/discord.mjs +21 -1
  28. package/src/runtime/channels/tool-defs.mjs +1 -1
  29. package/src/runtime/memory/lib/trace-store.mjs +25 -3
  30. package/src/runtime/memory/tool-defs.mjs +5 -5
  31. package/src/runtime/search/tool-defs.mjs +2 -18
  32. package/src/runtime/shared/tool-execution-contract.mjs +1 -1
  33. package/src/session-runtime/output-styles.mjs +7 -6
  34. package/src/session-runtime/tool-defs.mjs +0 -1
  35. package/src/session-runtime/tool-surface.mjs +9 -0
  36. package/src/session-runtime/workflow.mjs +7 -0
  37. package/src/standalone/explore-tool.mjs +2 -2
  38. package/src/workflows/default/WORKFLOW.md +2 -1
  39. package/src/workflows/solo/WORKFLOW.md +7 -5
  40. package/src/output-styles/default.md +0 -40
@@ -5,7 +5,8 @@
5
5
  // pass), and unrecoverable errors throw. Behavior identical to the inline
6
6
  // try/catch it replaced.
7
7
  import { appendAgentTrace } from '../agent-trace.mjs';
8
- import { isContextOverflowError } from '../providers/retry-classifier.mjs';
8
+ import { classifyError, isContextOverflowError } from '../providers/retry-classifier.mjs';
9
+ import { setTimeout as sleepMs } from 'timers/promises';
9
10
  import { readStreamOutcome } from '../providers/lib/stream-outcome.mjs';
10
11
  import { resolveWorkerCompactPolicy } from './loop/compact-policy.mjs';
11
12
  import { agentContextOverflowError } from './loop/context-overflow.mjs';
@@ -35,21 +36,96 @@ function normalizedIncompleteUsage(raw) {
35
36
  };
36
37
  }
37
38
 
39
+ // Loop-level transport replay (codex retry_transport parity). The provider
40
+ // layer already retries transient failures with a ~15s total envelope
41
+ // (PROVIDER_RETRY_BACKOFF_MS); a real network blip (router/VPN flap — the
42
+ // 2026-08-02 17:37 incident also dropped the Discord gateway) outlasts it and
43
+ // used to surface as a failed turn. When the stream exposed NOTHING
44
+ // (replaySafe: no relayed text/reasoning, no dispatched tool call), re-sending
45
+ // the identical request is side-effect-free, so wait out the blip and retry
46
+ // the send at the loop level. Two attempts, 5s/15s — combined with the
47
+ // provider envelope this covers ~50s outages before failing honestly.
48
+ const TRANSPORT_RETRY_BACKOFF_MS = Object.freeze([5_000, 15_000]);
49
+ export const TRANSPORT_RETRY_MAX = TRANSPORT_RETRY_BACKOFF_MS.length;
50
+
38
51
  export async function sendWithRecovery(ctx) {
39
52
  const {
40
53
  provider, messages, model, sendTools, tools, opts,
41
54
  sessionId, sessionRef, nextIteration, contextOverflowRetryUsed,
55
+ transportRetriesUsed = 0, signal,
42
56
  } = ctx;
43
57
  let response;
58
+ // Bench-only turn timing (MIXDOG_TURN_TIMING=1): one stderr line per
59
+ // provider request — TTFT (first stream delta) and total stream time.
60
+ // Inert unless the env flag is set; used to profile harness vs model
61
+ // latency in Terminal-Bench runs.
62
+ const turnT0 = process.env.MIXDOG_TURN_TIMING === '1' ? Date.now() : 0;
63
+ let turnFirstDelta = 0;
64
+ let timedOpts = opts;
65
+ if (turnT0) {
66
+ const prevDelta = typeof opts?.onStreamDelta === 'function' ? opts.onStreamDelta : null;
67
+ timedOpts = {
68
+ ...(opts || {}),
69
+ onStreamDelta: (kind) => {
70
+ if (!turnFirstDelta) turnFirstDelta = Date.now();
71
+ if (prevDelta) prevDelta(kind);
72
+ },
73
+ };
74
+ }
75
+ const logTurnTiming = (status) => {
76
+ if (!turnT0) return;
77
+ const now = Date.now();
78
+ const ttft = turnFirstDelta ? turnFirstDelta - turnT0 : -1;
79
+ try {
80
+ console.error(`[turn-timing] status=${status} ttft=${ttft}ms total=${now - turnT0}ms model=${model}`);
81
+ } catch { /* logging must never break the send path */ }
82
+ };
83
+ // Loop-side exposure witness. Some providers throw truncation/stall
84
+ // errors that carry only partialContent — neither liveTextEmitted nor
85
+ // unsafeToRetry — so an outcome read from the ERROR alone can report
86
+ // replaySafe even though this very send already relayed text to the
87
+ // client through opts.onTextDelta, or dispatched a tool call through
88
+ // opts.onToolCall. Replaying such a send would duplicate output the user
89
+ // already saw (or re-run a side effect), so record what THIS send
90
+ // actually exposed and merge it into every outcome read below. The
91
+ // callbacks are wrapped in place and restored conditionally: the
92
+ // overflow-retry branch intentionally clears opts.onToolCall, and that
93
+ // clear must survive the restore.
94
+ const relayWitness = { textEmitted: false, toolCallsDispatched: 0 };
95
+ const prevOnTextDelta = typeof opts?.onTextDelta === 'function' ? opts.onTextDelta : null;
96
+ const prevOnToolCall = typeof opts?.onToolCall === 'function' ? opts.onToolCall : null;
97
+ const witnessedOnTextDelta = prevOnTextDelta
98
+ ? (...args) => {
99
+ if (typeof args[0] === 'string' && args[0].length > 0) relayWitness.textEmitted = true;
100
+ return prevOnTextDelta(...args);
101
+ }
102
+ : null;
103
+ const witnessedOnToolCall = prevOnToolCall
104
+ ? (...args) => {
105
+ relayWitness.toolCallsDispatched += 1;
106
+ return prevOnToolCall(...args);
107
+ }
108
+ : null;
109
+ if (opts) {
110
+ if (witnessedOnTextDelta) opts.onTextDelta = witnessedOnTextDelta;
111
+ if (witnessedOnToolCall) opts.onToolCall = witnessedOnToolCall;
112
+ }
113
+ if (timedOpts !== opts && timedOpts) {
114
+ if (witnessedOnTextDelta) timedOpts.onTextDelta = witnessedOnTextDelta;
115
+ if (witnessedOnToolCall) timedOpts.onToolCall = witnessedOnToolCall;
116
+ }
117
+ try {
44
118
  try {
45
- response = await provider.send(messages, model, sendTools.length ? sendTools : undefined, opts);
119
+ response = await provider.send(messages, model, sendTools.length ? sendTools : undefined, timedOpts);
120
+ logTurnTiming('ok');
46
121
  } catch (sendErr) {
122
+ logTurnTiming(`err:${sendErr?.code || sendErr?.name || 'unknown'}`);
47
123
  // Canonical stream outcome: ONE fail-closed read of what the
48
124
  // provider stream actually produced (terminal vs continuation,
49
125
  // observed text/reasoning, partial/complete/dispatched tool calls).
50
126
  // Every branch below consumes it instead of re-inferring safety
51
127
  // from provider-specific flags.
52
- const outcome = readStreamOutcome(sendErr);
128
+ const outcome = readStreamOutcome(sendErr, relayWitness);
53
129
  // Gemini REST/SDK reports MAX_TOKENS by throwing a typed
54
130
  // ProviderIncompleteError after preserving the streamed candidate.
55
131
  // Normalize only that exact, safe no-tool output-limit shape into a
@@ -185,6 +261,35 @@ export async function sendWithRecovery(ctx) {
185
261
  };
186
262
  return { action: 'proceed', response };
187
263
  } else
264
+ // Clean transient transport failure with zero exposure: replay the
265
+ // send after a bounded wait instead of failing the turn.
266
+ if (
267
+ transportRetriesUsed < TRANSPORT_RETRY_MAX
268
+ && outcome.replaySafe === true
269
+ && classifyError(sendErr) === 'transient'
270
+ ) {
271
+ const waitMs = TRANSPORT_RETRY_BACKOFF_MS[transportRetriesUsed];
272
+ try {
273
+ process.stderr.write(
274
+ `[loop] transient send failure with no observed output (sess=${sessionId || 'unknown'} `
275
+ + `iter=${nextIteration} code=${sendErr?.code ?? sendErr?.status ?? 'n/a'}); `
276
+ + `transport retry ${transportRetriesUsed + 1}/${TRANSPORT_RETRY_MAX} after ${waitMs}ms\n`,
277
+ );
278
+ } catch { /* best-effort */ }
279
+ try {
280
+ appendAgentTrace({
281
+ kind: 'transport_retry',
282
+ sessionId: sessionId || null,
283
+ iteration: nextIteration,
284
+ attempt: transportRetriesUsed + 1,
285
+ waitMs,
286
+ code: sendErr?.code ?? null,
287
+ status: sendErr?.status ?? null,
288
+ });
289
+ } catch { /* best-effort */ }
290
+ await sleepMs(waitMs, undefined, signal ? { signal } : undefined);
291
+ return { action: 'retry_transport' };
292
+ } else
188
293
  // Context-window-exceeded is a deterministic refusal from the API.
189
294
  // Recover context overflow reactively by compacting and retrying
190
295
  // in the same active turn. MixDog's proactive estimator can miss a
@@ -236,5 +341,13 @@ export async function sendWithRecovery(ctx) {
236
341
  messageTokensEst: estimateMessagesTokensSafe(messages),
237
342
  }, sendErr);
238
343
  }
344
+ } finally {
345
+ // Conditional restore: only unwind our own wrappers. An intentional
346
+ // opts.onToolCall = undefined (overflow-retry branch) stays cleared.
347
+ if (opts) {
348
+ if (witnessedOnTextDelta && opts.onTextDelta === witnessedOnTextDelta) opts.onTextDelta = prevOnTextDelta;
349
+ if (witnessedOnToolCall && opts.onToolCall === witnessedOnToolCall) opts.onToolCall = prevOnToolCall;
350
+ }
351
+ }
239
352
  return { action: 'proceed', response };
240
353
  }
@@ -1,5 +1,5 @@
1
1
  import { getAbortSignalForSession } from '../../session/abort-lookup.mjs';
2
- import { execShellCommand, stripAnsi } from '../shell-command.mjs';
2
+ import { acquireShellLeaseBounded, execShellCommand, stripAnsi } from '../shell-command.mjs';
3
3
  import { wrapCommandWithSnapshot } from '../shell-snapshot.mjs';
4
4
  import { getDestructiveCommandWarning } from '../destructive-warning.mjs';
5
5
  import { maybeRewriteWmicProcessCommand } from '../shell-policy.mjs';
@@ -21,6 +21,11 @@ import {
21
21
  } from './shell-jobs.mjs';
22
22
  import {
23
23
  analyzeShellCommandEffects,
24
+ buildPowerShellFilterTeePlan,
25
+ consumeFilterTeeCapture,
26
+ detectBlockedSleepPattern,
27
+ detectLongForegroundReason,
28
+ extractShellApplyPatchInvocation,
24
29
  foregroundLongCommandHint,
25
30
  isAutobackgroundingAllowed,
26
31
  preflightPowerShellHygiene,
@@ -241,7 +246,7 @@ export async function executeBashTool(args, workDir, options = {}) {
241
246
  const bashWorkDir = resolveSessionCwd(_sessionCwdKey, _hasExplicitCwd ? cwdResult.cwd : null, cwdResult.cwd);
242
247
  const _readStateScope = options?.readStateScope ?? options?.sessionId ?? null;
243
248
  const executionMode = resolveExecutionMode(args || {}, args?.run_in_background === true ? 'async' : 'sync');
244
- const runInBackground = executionMode === 'async';
249
+ let runInBackground = executionMode === 'async';
245
250
 
246
251
  // Run hard-block policy BEFORE branching into the persistent-shell tool.
247
252
  // The persistent path used to bypass the one-shot block scan because the
@@ -251,6 +256,21 @@ export async function executeBashTool(args, workDir, options = {}) {
251
256
  // decode + rm token guard; calling it here applies the same allowlist
252
257
  // to both persistent and stateless paths.
253
258
  const _rawCmd = String(args && args.command != null ? args.command : '');
259
+ // Codex-parity: `apply_patch` typed into the shell (heredoc/argument/bare
260
+ // patch forms) routes to the internal patch engine instead of failing as
261
+ // an unknown binary. Runs BEFORE the exec-policy scan so patch BODY lines
262
+ // (e.g. `+ rm -rf …`) are never misread as shell commands. Dynamic import
263
+ // avoids a bash-tool <-> patch/orchestrator module cycle.
264
+ if (_rawCmd) {
265
+ const _apCall = extractShellApplyPatchInvocation(_rawCmd);
266
+ if (_apCall?.error) {
267
+ return formatShellToolFailure(`${_apCall.error}. Call the apply_patch tool with the patch string instead of the shell.`);
268
+ }
269
+ if (_apCall?.patch) {
270
+ const { executePatchTool } = await import('../patch/orchestrator.mjs');
271
+ return executePatchTool('apply_patch', { patch: _apCall.patch }, bashWorkDir, options);
272
+ }
273
+ }
254
274
  if (_rawCmd) {
255
275
  // R5-③: persistent:true used to route into bash_session BEFORE the
256
276
  // stripQuotedAndHeredoc / extractShellCInner / unquote sweep ran
@@ -339,6 +359,29 @@ export async function executeBashTool(args, workDir, options = {}) {
339
359
  return formatShellToolFailure(_execPolicyBlock);
340
360
  }
341
361
 
362
+ // Sleep-chain auto-promotion: a leading `sleep N && …` / `Start-Sleep N; …`
363
+ // used to be DENIED preflight (CC detectBlockedSleepPattern parity).
364
+ // Measured over 10 days the deny fired 46× and every hit was a wasted
365
+ // turn, so the command is promoted to a background task instead — the
366
+ // exact remedy the deny message pointed at, without the failure. Only
367
+ // possible while background tasks are enabled; with them disabled the
368
+ // command runs foreground as before (the deny never fired there either).
369
+ const _bgTasksDisabled = /^(1|true|yes|on)$/i.test(
370
+ String(process.env.MIXDOG_SHELL_DISABLE_BACKGROUND_TASKS || '').trim(),
371
+ );
372
+ let autoAsyncReason = '';
373
+ if (!runInBackground && !_bgTasksDisabled) {
374
+ // Long-foreground shapes (watch-like dev servers/watchers, 30s+ sleeps
375
+ // anywhere in the chain) used to hard-fail via foregroundLongCommandHint
376
+ // (~22 wasted turns/14d measured). Promote them to a background task —
377
+ // the exact remedy the deny message pointed at — same as sleep chains.
378
+ const _blockedSleep = detectBlockedSleepPattern(command) || detectLongForegroundReason(command);
379
+ if (_blockedSleep) {
380
+ runInBackground = true;
381
+ autoAsyncReason = _blockedSleep;
382
+ }
383
+ }
384
+
342
385
  let shellEffects;
343
386
  let combinedBashAbort = null;
344
387
  try {
@@ -372,9 +415,6 @@ export async function executeBashTool(args, workDir, options = {}) {
372
415
  : DEFAULT_BASH_TIMEOUT_MS;
373
416
  const hasExplicitTimeout = typeof args.timeout === 'number' && args.timeout > 0;
374
417
  const timeoutMs = hasExplicitTimeout ? args.timeout : defaultTimeoutMs;
375
- const _bgTasksDisabled = /^(1|true|yes|on)$/i.test(
376
- String(process.env.MIXDOG_SHELL_DISABLE_BACKGROUND_TASKS || '').trim(),
377
- );
378
418
  const backgroundOnTimeout = !runInBackground
379
419
  && !_bgTasksDisabled
380
420
  && isAutobackgroundingAllowed(command, resolvedSpec.shellType);
@@ -429,11 +469,20 @@ export async function executeBashTool(args, workDir, options = {}) {
429
469
  scrubLoaderVars(spawnEnv);
430
470
  scrubRuntimeRootVars(spawnEnv);
431
471
  let wrappedCommand;
472
+ let _teePlan = null;
432
473
  // PowerShell UTF-8 prefix is PS-only: the Windows Git Bash path
433
474
  // (shellType==='posix') must NOT receive it. Snapshot wrapper stays
434
475
  // POSIX-host-only for now — no snapshot for Windows Git Bash initially.
435
476
  if (process.platform === 'win32' && shellType === 'powershell') {
436
- wrappedCommand = _prefixPowerShellUtf8(command);
477
+ // Filter-swallow rescue (sync path only): tee the unfiltered
478
+ // producer stream of an exactly-recognized filter pipeline so a
479
+ // failing run can attach the original output tail in THIS call
480
+ // instead of returning `[exit code: N]` + `(no output)`. Any
481
+ // ambiguity yields a null plan and the command runs untouched.
482
+ if (!runInBackground) {
483
+ try { _teePlan = buildPowerShellFilterTeePlan(command); } catch { _teePlan = null; }
484
+ }
485
+ wrappedCommand = _prefixPowerShellUtf8(_teePlan ? _teePlan.command : command);
437
486
  } else if (process.platform !== 'win32' && (shell.includes('bash') || shell.includes('zsh'))) {
438
487
  try {
439
488
  wrappedCommand = await wrapCommandWithSnapshot(shell, command);
@@ -451,8 +500,8 @@ export async function executeBashTool(args, workDir, options = {}) {
451
500
  let asyncLease = null;
452
501
  let job;
453
502
  try {
454
- asyncLease = await (options?.resourceAdmission || resourceAdmission).acquire('shell', {
455
- signal: combinedAsyncAbort.signal,
503
+ asyncLease = await acquireShellLeaseBounded(options?.resourceAdmission || resourceAdmission, {
504
+ abortSignal: combinedAsyncAbort.signal,
456
505
  label: String(command).replace(/\s+/g, ' ').slice(0, 120),
457
506
  dependency: 'detached',
458
507
  });
@@ -525,7 +574,10 @@ export async function executeBashTool(args, workDir, options = {}) {
525
574
  clientHostPid: options?.clientHostPid,
526
575
  });
527
576
  } catch { /* watcher arm is best-effort; never blocks the spawn */ }
528
- return _prependDestructiveWarning(command, renderBackgroundTask(task));
577
+ const _autoAsyncNote = autoAsyncReason
578
+ ? `[auto-async] ${autoAsyncReason} — promoted to a background task; act on its completion notification instead of blocking (do not poll).\n`
579
+ : '';
580
+ return _prependDestructiveWarning(command, _autoAsyncNote + renderBackgroundTask(task));
529
581
  } catch (error) {
530
582
  if (job?.jobId && !job.error) {
531
583
  try { killShellJob(job.jobId); } catch {}
@@ -643,6 +695,17 @@ export async function executeBashTool(args, workDir, options = {}) {
643
695
  const benignExitOne = _isBenignSearchExitOne(command, exitCode, signal, stderr);
644
696
  const shellRunFailed = !shellToolFailed && (!!signal || (exitCode !== 0 && exitCode !== null && !benignExitOne));
645
697
  const isReallyErrored = shellToolFailed || shellRunFailed;
698
+ // Filter-swallow rescue: the tee file is ALWAYS consumed (deleted)
699
+ // here; its tail is attached only when the run failed with an empty
700
+ // visible capture — the exact `(no output)` shape that previously
701
+ // cost the model extra diagnostic turns.
702
+ let _rescueNote = '';
703
+ if (_teePlan) {
704
+ const _rescueTail = consumeFilterTeeCapture(_teePlan.teePath);
705
+ if (isReallyErrored && _rescueTail && !stdout.trim() && !stderr.trim()) {
706
+ _rescueNote = `\n\n[filter-swallowed output rescue] the command failed but its trailing filter(s) matched nothing, so the visible output was empty. Unfiltered pipeline output (tail):\n${smartMiddleTruncate(_rescueTail)}`;
707
+ }
708
+ }
646
709
  const _driftNote = '';
647
710
  // Distinct timeout marker so callers see "killed by timeout after Nms"
648
711
  // vs an external signal (e.g. user Ctrl-C, OOM kill). result.timedOut
@@ -666,7 +729,7 @@ export async function executeBashTool(args, workDir, options = {}) {
666
729
  // cross-stream interleaving. Acceptable for most diagnostic
667
730
  // outputs; flag in shell-command if exact interleaving is required.
668
731
  const merged = stdout + stderr;
669
- if (statusMarker) return _prependDestructiveWarning(command, errorPrefix + smartMiddleTruncate(`${statusMarker}\n\n${merged || '(no output)'}`) + _driftNote);
732
+ if (statusMarker) return _prependDestructiveWarning(command, errorPrefix + smartMiddleTruncate(`${statusMarker}\n\n${merged || '(no output)'}`) + _rescueNote + _driftNote);
670
733
  return _prependDestructiveWarning(command, smartMiddleTruncate(merged || '(no output)') + _driftNote);
671
734
  }
672
735
  const truncatedStdout = smartMiddleTruncate(stdout);
@@ -688,7 +751,7 @@ export async function executeBashTool(args, workDir, options = {}) {
688
751
  const warningBlock = [
689
752
  wmicRewrite?.note || '',
690
753
  ].filter(Boolean).join('\n');
691
- const payload = `${body}${stderrBlock}${spillBlock}${_driftNote}`;
754
+ const payload = `${body}${stderrBlock}${spillBlock}${_rescueNote}${_driftNote}`;
692
755
  if (statusMarker) return _prependDestructiveWarning(command, _composeShellFailure(statusMarker, errorPrefix, warningBlock, payload));
693
756
  return _prependDestructiveWarning(command, warningBlock ? `${warningBlock}\n${payload}` : payload);
694
757
  }
@@ -65,7 +65,7 @@ export const BUILTIN_TOOLS = [
65
65
  description: 'File path, or {path,offset,limit}[] regions. Pass real arrays, not JSON strings.',
66
66
  },
67
67
  offset: { type: 'number', minimum: 0, description: 'Lines to skip.' },
68
- limit: { type: 'number', minimum: 1, description: 'Max lines after offset.' },
68
+ limit: { type: 'number', minimum: 1, description: 'Max lines after offset. Defaults to 2000.' },
69
69
  },
70
70
  required: ['path'],
71
71
  },
@@ -74,8 +74,7 @@ export const BUILTIN_TOOLS = [
74
74
  name: 'shell',
75
75
  title: 'Mixdog Shell',
76
76
  annotations: { title: 'Mixdog Shell', readOnlyHint: false, destructiveHint: true, idempotentHint: false, openWorldHint: true, compressible: true },
77
- description: 'Run programs/change state; not file inspection. '
78
- + 'Combine order-dependent commands into one command. Use async for sleep/watch/dev loops.'
77
+ description: 'Runs a shell command and returns its output. Use async for sleep/watch/dev loops.'
79
78
  + `${_shellSyntaxCheat} ${TOOL_ASYNC_EXECUTION_CONTRACT}`,
80
79
  inputSchema: {
81
80
  type: 'object',
@@ -114,7 +113,7 @@ export const BUILTIN_TOOLS = [
114
113
  name: 'grep',
115
114
  title: 'Mixdog Grep',
116
115
  annotations: { title: 'Mixdog Grep', readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: false, compressible: true },
117
- description: 'Quoted/non-identifier literal or regex→grep, over verified scopes (project root counts as verified); guessed path fragment → find first. A nonzero content_with_context result resolves that search concept; only zero/error results may change tokens or scope. files_with_matches/count are existence modes. No path "." + guessed src/**.',
116
+ description: 'Content search by literal or regex over file/dir scopes.',
118
117
  inputSchema: {
119
118
  type: 'object',
120
119
  properties: {
@@ -140,7 +139,7 @@ export const BUILTIN_TOOLS = [
140
139
  description: 'Glob filter.',
141
140
  },
142
141
  output_mode: { type: 'string', enum: ['content_with_context', 'files_with_matches', 'count'], description: 'content_with_context (default); files_with_matches/count for existence.' },
143
- head_limit: { type: 'number', minimum: 0, description: 'Max results.' },
142
+ head_limit: { type: 'number', minimum: 0, description: 'Max results. Defaults to 250; 0 = unlimited.' },
144
143
  offset: { type: 'number', minimum: 0, description: 'Skip results for paging.' },
145
144
  '-C': { type: 'number', minimum: 0, description: 'Lines before/after each match.' },
146
145
  },
@@ -154,7 +153,7 @@ export const BUILTIN_TOOLS = [
154
153
  name: 'glob',
155
154
  title: 'Mixdog Glob',
156
155
  annotations: { title: 'Mixdog Glob', readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: false, compressible: true },
157
- description: 'Match exact glob patterns from verified base directories (project root is verified). Batch pattern[]/path[].',
156
+ description: 'Match exact glob patterns from base directories; batch pattern[]/path[].',
158
157
  inputSchema: {
159
158
  type: 'object',
160
159
  properties: {
@@ -172,7 +171,7 @@ export const BUILTIN_TOOLS = [
172
171
  ],
173
172
  description: 'Base directory(ies); path[] batches.',
174
173
  },
175
- head_limit: { type: 'number', description: 'Max entries.' },
174
+ head_limit: { type: 'number', description: 'Max entries. Defaults to 100; 0 = unlimited.' },
176
175
  offset: { type: 'number', description: 'Skip entries.' },
177
176
  },
178
177
  required: ['pattern'],
@@ -182,7 +181,7 @@ export const BUILTIN_TOOLS = [
182
181
  name: 'find',
183
182
  title: 'Mixdog Find Files',
184
183
  annotations: { title: 'Mixdog Find Files', readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: false, compressible: true },
185
- description: 'Fuzzy lookup only for unknown partial paths/names (dot dirs included); not for the project root or already-verified roots. Output paths are verified downstream. Not for file contents.',
184
+ description: 'Fuzzy lookup for partial paths/names (dot dirs included). Returns matching paths; not file contents.',
186
185
  inputSchema: {
187
186
  type: 'object',
188
187
  properties: {
@@ -194,7 +193,7 @@ export const BUILTIN_TOOLS = [
194
193
  description: 'Partial path/name words; query[] batches.',
195
194
  },
196
195
  path: { type: 'string', description: 'Base directory.' },
197
- head_limit: { type: 'number', description: 'Max paths.' },
196
+ head_limit: { type: 'number', description: 'Max paths. Defaults to 25.' },
198
197
  },
199
198
  required: ['query'],
200
199
  },
@@ -214,7 +213,7 @@ export const BUILTIN_TOOLS = [
214
213
  ],
215
214
  description: 'Directory; path[] batches.',
216
215
  },
217
- head_limit: { type: 'number', description: 'Max entries.' },
216
+ head_limit: { type: 'number', description: 'Max entries. Defaults to 200; 0 = no cap.' },
218
217
  offset: { type: 'number', description: 'Skip N entries for paging.' },
219
218
  },
220
219
  required: [],
@@ -40,6 +40,18 @@ const FIND_WALK_TIMEOUT_MS = 20_000;
40
40
  const LIST_WALK_TIMEOUT_MS = 20_000;
41
41
  const LIST_ABSOLUTE_CAP = 50_000;
42
42
 
43
+ // A/B override surface for the default result caps (stock: list/tree 200,
44
+ // fuzzy find 25). Env-gated so bench variants can match competitor-style
45
+ // generous caps without changing the shipped defaults.
46
+ function _listDefaultHeadLimit(fallback) {
47
+ const parsed = parseInt(process.env.MIXDOG_LIST_DEFAULT_HEAD_LIMIT ?? '', 10);
48
+ return parsed > 0 ? parsed : fallback;
49
+ }
50
+ function _findDefaultHeadLimit(fallback) {
51
+ const parsed = parseInt(process.env.MIXDOG_FIND_DEFAULT_HEAD_LIMIT ?? '', 10);
52
+ return parsed > 0 ? parsed : fallback;
53
+ }
54
+
43
55
  export async function executeListTool(args, workDir, options = {}) {
44
56
  args.path = coerceReadFamilyPathArg(args.path, workDir);
45
57
  if (Array.isArray(args.path)) {
@@ -108,7 +120,7 @@ export async function executeListTool(args, workDir, options = {}) {
108
120
  const hidden = Boolean(args.hidden);
109
121
  const sort = ['name', 'mtime', 'size'].includes(args.sort) ? args.sort : 'name';
110
122
  const typeFilter = ['any', 'file', 'dir'].includes(args.type) ? args.type : 'any';
111
- const headLimit = normalizeListHeadLimit(args.head_limit, 200);
123
+ const headLimit = normalizeListHeadLimit(args.head_limit, _listDefaultHeadLimit(200));
112
124
  const offset = typeof args.offset === 'number' && args.offset > 0 ? args.offset : 0;
113
125
  const needsGlobalStat = sort === 'mtime' || sort === 'size';
114
126
  const includeNoise = Boolean(args.include_noise);
@@ -246,7 +258,7 @@ export async function executeTreeTool(args, workDir, options = {}) {
246
258
  const inputPath = args.path || '.';
247
259
  const depth = Math.min(Math.max(parseInt(args.depth ?? 3, 10) || 3, 1), 6);
248
260
  const hidden = Boolean(args.hidden);
249
- const headLimit = normalizeListHeadLimit(args.head_limit, 200);
261
+ const headLimit = normalizeListHeadLimit(args.head_limit, _listDefaultHeadLimit(200));
250
262
  const offset = typeof args.offset === 'number' && args.offset > 0 ? args.offset : 0;
251
263
  const includeNoise = Boolean(args.include_noise);
252
264
  const _treeGuard = listGuardPath(inputPath);
@@ -589,7 +601,7 @@ export async function executeFuzzyFindTool(args, workDir, options = {}) {
589
601
  const includeNoise = Boolean(args.include_noise);
590
602
  // head_limit:0 means "no cap" per list semantics; default is intentionally
591
603
  // compact so ambiguous discovery does not dump a huge candidate list.
592
- const headLimit = normalizeListHeadLimit(args.head_limit, 25);
604
+ const headLimit = normalizeListHeadLimit(args.head_limit, _findDefaultHeadLimit(25));
593
605
  const depth = args.depth != null ? Math.max(parseInt(args.depth, 10) || 1, 1) : null;
594
606
  const cacheKey = buildListCacheKey({
595
607
  mode: 'fuzzy_find',
@@ -3,6 +3,7 @@ import { existsSync } from 'node:fs';
3
3
  import { resolve } from 'node:path';
4
4
  import { accessSync, constants, statSync } from 'node:fs';
5
5
  import { promisify } from 'node:util';
6
+ import { createRequire } from 'node:module';
6
7
  import os from 'node:os';
7
8
  import {
8
9
  acquire as acquireChildSpawnSlot,
@@ -181,6 +182,14 @@ export async function _resolveRgExecutable() {
181
182
  if (existsSync(candidate) && _usableRgCandidate(candidate, isWin)) return candidate;
182
183
  }
183
184
  }
185
+ // Vendored fallback: the optional @vscode/ripgrep dependency ships a
186
+ // platform rg binary with the package, so a machine without system
187
+ // ripgrep still gets a working search family (grep/find/explore). System
188
+ // rg on PATH always wins above; this only fills the gap.
189
+ try {
190
+ const { rgPath } = createRequire(import.meta.url)('@vscode/ripgrep');
191
+ if (rgPath && _usableRgCandidate(String(rgPath), isWin)) return String(rgPath);
192
+ } catch { /* optional dependency absent — continue to where/which */ }
184
193
  try {
185
194
  const cmd = isWin ? 'where' : 'which';
186
195
  const { stdout: out } = await execFileAsync(cmd, ['rg'], { encoding: 'utf8', windowsHide: true });
@@ -103,6 +103,20 @@ import {
103
103
  } from './lib/grep-output.mjs';
104
104
 
105
105
 
106
+ // Default grep result cap when head_limit is unspecified. 250 matches the
107
+ // Claude Code default (GrepTool DEFAULT_HEAD_LIMIT); the tool-result offload
108
+ // layer still bounds oversized results. MIXDOG_GREP_DEFAULT_HEAD_LIMIT
109
+ // overrides for A/B runs.
110
+ function _grepDefaultHeadLimit() {
111
+ const parsed = parseInt(process.env.MIXDOG_GREP_DEFAULT_HEAD_LIMIT ?? '', 10);
112
+ return parsed > 0 ? parsed : 250;
113
+ }
114
+ // Same A/B override surface for glob (stock default 100).
115
+ function _globDefaultHeadLimit() {
116
+ const parsed = parseInt(process.env.MIXDOG_GLOB_DEFAULT_HEAD_LIMIT ?? '', 10);
117
+ return parsed > 0 ? parsed : 100;
118
+ }
119
+
106
120
  export async function executeGrepTool(args, workDir, executeChildBuiltinTool, readStateScope = null, options = {}) {
107
121
  args = normalizeGrepArgs(args);
108
122
  args.path = coerceReadFamilyPathArg(args.path, workDir);
@@ -290,7 +304,7 @@ export async function executeGrepTool(args, workDir, executeChildBuiltinTool, re
290
304
  return `Error: invalid head_limit ${JSON.stringify(headLimitRaw)}; expected a non-negative integer (0 = unlimited)`;
291
305
  }
292
306
  const headLimit = headLimitCoerced === null
293
- ? 80
307
+ ? _grepDefaultHeadLimit()
294
308
  : (headLimitCoerced === 0 ? Infinity : headLimitCoerced);
295
309
  const offsetCoerced = coerceNonNegInt(args.offset);
296
310
  if (Number.isNaN(offsetCoerced)) {
@@ -988,7 +1002,7 @@ export async function executeGlobTool(args, workDir, options = {}) {
988
1002
  return `Error: invalid head_limit ${JSON.stringify(headLimitRaw)}; expected a non-negative integer (0 = unlimited)`;
989
1003
  }
990
1004
  const headLimit = headLimitCoerced === null
991
- ? 100
1005
+ ? _globDefaultHeadLimit()
992
1006
  : (headLimitCoerced === 0 ? Infinity : headLimitCoerced);
993
1007
  const offsetCoerced = coerceNonNegInt(args.offset);
994
1008
  if (Number.isNaN(offsetCoerced)) {