mixdog 0.9.91 → 0.9.93

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +147 -51
  2. package/package.json +6 -5
  3. package/scripts/code-graph-description-contract.mjs +6 -8
  4. package/scripts/tmp-cdp-errors.mjs +41 -0
  5. package/scripts/tmp-cdp-inspect.mjs +41 -0
  6. package/scripts/tool-overhead-microbench.mjs +60 -0
  7. package/scripts/tui-transcript-jitter-harness.mjs +2 -18
  8. package/src/agents/debugger/agent.json +1 -1
  9. package/src/agents/explore/agent.json +1 -1
  10. package/src/agents/heavy-worker/agent.json +1 -1
  11. package/src/agents/maintainer/agent.json +1 -1
  12. package/src/agents/reviewer/agent.json +1 -1
  13. package/src/agents/worker/agent.json +1 -1
  14. package/src/lib/rules-builder.cjs +5 -5
  15. package/src/output-styles/simple.md +1 -1
  16. package/src/rules/agent/00-core.md +1 -2
  17. package/src/rules/lead/01-general.md +1 -0
  18. package/src/rules/shared/01-tool.md +19 -21
  19. package/src/runtime/agent/orchestrator/agent-runtime/cache-strategy.mjs +16 -3
  20. package/src/runtime/agent/orchestrator/agent-trace.mjs +17 -0
  21. package/src/runtime/agent/orchestrator/context/collect.mjs +2 -1
  22. package/src/runtime/agent/orchestrator/providers/anthropic-effort.mjs +9 -1
  23. package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +90 -22
  24. package/src/runtime/agent/orchestrator/providers/anthropic.mjs +54 -19
  25. package/src/runtime/agent/orchestrator/providers/lib/anthropic-request-utils.mjs +18 -1
  26. package/src/runtime/agent/orchestrator/providers/openai-oauth-http-sse.mjs +44 -8
  27. package/src/runtime/agent/orchestrator/providers/openai-oauth-ws.mjs +10 -0
  28. package/src/runtime/agent/orchestrator/providers/openai-ws-events.mjs +3 -0
  29. package/src/runtime/agent/orchestrator/providers/openai-ws-stream.mjs +2 -0
  30. package/src/runtime/agent/orchestrator/providers/retry-classifier.mjs +35 -0
  31. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +34 -5
  32. package/src/runtime/agent/orchestrator/session/eager-dispatch.mjs +35 -29
  33. package/src/runtime/agent/orchestrator/session/loop/stop-hooks.mjs +9 -0
  34. package/src/runtime/agent/orchestrator/session/loop/stored-tool-args.mjs +11 -2
  35. package/src/runtime/agent/orchestrator/session/loop/tool-classify.mjs +5 -6
  36. package/src/runtime/agent/orchestrator/session/loop/tool-exec.mjs +31 -2
  37. package/src/runtime/agent/orchestrator/session/manager/compaction-runner.mjs +60 -0
  38. package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +60 -31
  39. package/src/runtime/agent/orchestrator/session/manager.mjs +1 -1
  40. package/src/runtime/agent/orchestrator/session/result-classification.mjs +28 -0
  41. package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +71 -2
  42. package/src/runtime/agent/orchestrator/session/store/listing.mjs +17 -0
  43. package/src/runtime/agent/orchestrator/session/store-summary-reader.mjs +101 -0
  44. package/src/runtime/agent/orchestrator/session/store.mjs +30 -0
  45. package/src/runtime/agent/orchestrator/session/tool-batch.mjs +19 -23
  46. package/src/runtime/agent/orchestrator/stall-policy.mjs +31 -21
  47. package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +4 -1
  48. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +12 -6
  49. package/src/runtime/agent/orchestrator/tools/builtin/task-tool.mjs +5 -0
  50. package/src/runtime/agent/orchestrator/tools/code-graph-tool-defs.mjs +4 -3
  51. package/src/runtime/agent/orchestrator/tools/lib/pwsh-standby-pool.mjs +47 -14
  52. package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +100 -0
  53. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +6 -5
  54. package/src/runtime/agent/orchestrator/tools/shell-command.mjs +9 -0
  55. package/src/runtime/agent/orchestrator/tools/shell-state.mjs +32 -2
  56. package/src/runtime/channels/backends/discord-gateway.mjs +6 -32
  57. package/src/runtime/channels/lib/inbound-handler.mjs +19 -3
  58. package/src/runtime/channels/lib/scheduler.mjs +51 -3
  59. package/src/runtime/channels/lib/worker-main.mjs +4 -0
  60. package/src/runtime/channels/tool-defs.mjs +4 -2
  61. package/src/runtime/memory/lib/query-handlers.mjs +11 -3
  62. package/src/runtime/memory/tool-defs.mjs +5 -5
  63. package/src/runtime/shared/channel-notification-routing.mjs +8 -2
  64. package/src/runtime/shared/llm/http-agent.mjs +11 -0
  65. package/src/session-runtime/lifecycle-api.mjs +26 -1
  66. package/src/session-runtime/output-styles.mjs +1 -4
  67. package/src/session-runtime/tool-catalog-data.mjs +5 -2
  68. package/src/session-runtime/workflow.mjs +25 -13
  69. package/src/standalone/agent-tool/tag-registry.mjs +5 -1
  70. package/src/standalone/explore-tool.mjs +1 -1
  71. package/src/tui/dist/index.mjs +100 -46
  72. package/src/tui/engine/agent-envelope.mjs +52 -3
  73. package/src/tui/engine/session-api.mjs +17 -0
  74. package/src/tui/engine/tui-steering-persist.mjs +24 -1
  75. package/src/tui/engine/turn.mjs +8 -9
  76. package/src/tui/engine.mjs +18 -41
  77. package/src/workflows/default/WORKFLOW.md +7 -18
  78. package/src/workflows/solo/WORKFLOW.md +0 -6
  79. package/src/workflows/solo-bench/WORKFLOW.md +17 -0
@@ -126,7 +126,7 @@ export {
126
126
  updateSessionManualTitle,
127
127
  flushSessionMetrics,
128
128
  } from './manager/session-crud.mjs';
129
- export { deleteSession } from './store.mjs';
129
+ export { deleteSession, listOwnedAgentSessionIds } from './store.mjs';
130
130
  // Read-only parsed-session access (desktop pane peek): no resume, no
131
131
  // ownership, no liveness side effects.
132
132
  export { loadSession } from './store.mjs';
@@ -68,3 +68,31 @@ export function classifyResultKind(result, explicitSuccess = false) {
68
68
  }
69
69
  return 'normal';
70
70
  }
71
+
72
+ /**
73
+ * Informational shell exit-1: `Error: [shell-run-failed] [exit code: 1]`
74
+ * with a non-empty stdout body and NO stderr evidence (neither an inline
75
+ * `[stderr]` block nor a `[stderr: path]` spill). grep-family "no match"
76
+ * semantics inside compound probes (loops, `;`-chains, substitutions) land
77
+ * exactly here: the run produced useful output, wrote nothing to stderr,
78
+ * and exited 1 only because the final stage matched nothing. The static
79
+ * single-pipeline gate (bash-tool _isBenignSearchExitOne) deliberately
80
+ * refuses these ambiguous shapes, so the result stays toolKind 'error' —
81
+ * consumers that must not overreact to an informational failure (the turn
82
+ * stop hook) test this signature instead of reclassifying the result.
83
+ * Signals, timeouts, and other exit codes carry different status markers
84
+ * and never match; a destructive-warning prefix also disqualifies.
85
+ *
86
+ * @param {unknown} result
87
+ * @returns {boolean}
88
+ */
89
+ export function isInformationalShellExitOne(result) {
90
+ if (typeof result !== 'string') return false;
91
+ const trimmed = result.trimStart();
92
+ const header = /^error:\s*\[shell-run-failed\]\s*\[exit code: 1\]\s*\n/i.exec(trimmed);
93
+ if (!header) return false;
94
+ const payload = trimmed.slice(header[0].length).trim();
95
+ if (!payload || payload === '(no output)') return false;
96
+ if (payload.startsWith('[stderr') || payload.includes('\n[stderr')) return false;
97
+ return true;
98
+ }
@@ -126,6 +126,36 @@ export async function sendWithRecovery(ctx) {
126
126
  // Every branch below consumes it instead of re-inferring safety
127
127
  // from provider-specific flags.
128
128
  const outcome = readStreamOutcome(sendErr, relayWitness);
129
+ // Text-only exposure retraction (cross-provider): a stream that
130
+ // died after relaying ONLY text — no dispatched or complete tool
131
+ // calls, no terminal — is replayable IF the ask owner retracts the
132
+ // exposed characters (onTextReset ack === true: the TUI truncates
133
+ // its live tail, the bench driver truncates its accumulator).
134
+ // This is the loop-level analogue of anthropic's
135
+ // recoverNonStreaming for providers WITHOUT a non-streaming
136
+ // fallback (gemini, openai-compat, openai WS) and for stalls that
137
+ // outlived the provider's in-place recovery. Observed live:
138
+ // make-mips-interpreter died with 31 exposed chars + a pending
139
+ // never-dispatched tool input and burned the whole trial.
140
+ const retractExposedTextForReplay = async () => {
141
+ if (outcome.terminalObserved === true) return false;
142
+ if (outcome.sideEffectDispatched === true) return false;
143
+ if (outcome.dispatchAmbiguous === true) return false;
144
+ if (Number(outcome.toolCallsDispatched) > 0) return false;
145
+ if (Number(outcome.toolCallsComplete) > 0) return false;
146
+ if (relayWitness.toolCallsDispatched > 0) return false;
147
+ if (typeof opts?.onTextReset !== 'function') return false;
148
+ const chars = Math.max(0, Number(outcome.textObservedChars) || 0)
149
+ || (typeof sendErr.partialContent === 'string' ? sendErr.partialContent.length : 0);
150
+ if (chars <= 0) return false;
151
+ let acked = false;
152
+ try {
153
+ acked = await opts.onTextReset({ chars, reason: 'loop-transport-retraction' }) === true;
154
+ } catch { acked = false; }
155
+ if (!acked) return false;
156
+ relayWitness.textEmitted = false;
157
+ return true;
158
+ };
129
159
  // Gemini REST/SDK reports MAX_TOKENS by throwing a typed
130
160
  // ProviderIncompleteError after preserving the streamed candidate.
131
161
  // Normalize only that exact, safe no-tool output-limit shape into a
@@ -177,6 +207,40 @@ export async function sendWithRecovery(ctx) {
177
207
  && sendErr.partialContent.trim().length > 0
178
208
  && outcome.toolCallsComplete === 0
179
209
  ) {
210
+ // Retractable shape: text-only exposure with the owner's
211
+ // acknowledgement replays on a fresh request instead of
212
+ // failing the turn. Non-acked (or tool-bearing) shapes keep
213
+ // the explicit-failure contract below unchanged.
214
+ // NOTE: no classifyError gate here — an exposed stall is
215
+ // stamped unsafeToRetry, which the general classifier reads as
216
+ // terminal, but that unsafety is exactly what the retraction
217
+ // removes (observed live: make-mips-interpreter failed twice
218
+ // because this guard demanded 'transient' and never fired).
219
+ if (
220
+ transportRetriesUsed < TRANSPORT_RETRY_MAX
221
+ && await retractExposedTextForReplay()
222
+ ) {
223
+ const waitMs = TRANSPORT_RETRY_BACKOFF_MS[transportRetriesUsed];
224
+ try {
225
+ process.stderr.write(
226
+ `[loop] exposed-text stall retracted (sess=${sessionId || 'unknown'} `
227
+ + `iter=${nextIteration} len=${sendErr.partialContent.length}); `
228
+ + `transport retry ${transportRetriesUsed + 1}/${TRANSPORT_RETRY_MAX} after ${waitMs}ms\n`,
229
+ );
230
+ } catch { /* best-effort */ }
231
+ try {
232
+ appendAgentTrace({
233
+ kind: 'exposed_text_retraction_retry',
234
+ sessionId: sessionId || null,
235
+ iteration: nextIteration,
236
+ attempt: transportRetriesUsed + 1,
237
+ waitMs,
238
+ partialContentLen: sendErr.partialContent.length,
239
+ });
240
+ } catch { /* best-effort */ }
241
+ await sleepMs(waitMs, undefined, signal ? { signal } : undefined);
242
+ return { action: 'retry_transport' };
243
+ }
180
244
  try {
181
245
  process.stderr.write(
182
246
  `[loop] final stream stalled with partial text (sess=${sessionId || 'unknown'} `
@@ -265,8 +329,13 @@ export async function sendWithRecovery(ctx) {
265
329
  // send after a bounded wait instead of failing the turn.
266
330
  if (
267
331
  transportRetriesUsed < TRANSPORT_RETRY_MAX
268
- && outcome.replaySafe === true
269
- && classifyError(sendErr) === 'transient'
332
+ && (
333
+ (outcome.replaySafe === true && classifyError(sendErr) === 'transient')
334
+ || (
335
+ (classifyError(sendErr) === 'transient' || outcome.stallObserved === true)
336
+ && await retractExposedTextForReplay()
337
+ )
338
+ )
270
339
  ) {
271
340
  const waitMs = TRANSPORT_RETRY_BACKOFF_MS[transportRetriesUsed];
272
341
  try {
@@ -64,6 +64,16 @@ const RESUMABLE_OPEN_MAX_COUNT = 300;
64
64
  // idle this long — see the blank-scratch branch in sweepStaleSessions.
65
65
  const BLANK_SCRATCH_MAX_AGE_MS = 60 * 60 * 1000; // 1h
66
66
 
67
+ /** Child-agent transcripts share their visible parent's retention boundary.
68
+ * Presence (including a tombstone or unreadable file) preserves the child;
69
+ * only proven parent absence releases it to ordinary cleanup. */
70
+ function retainedLinkedAgent(session) {
71
+ if (!session || !isAgentOwner(session)) return false;
72
+ const parentId = String(session.ownerSessionId || session.parentSessionId || '').trim();
73
+ if (!/^[A-Za-z0-9_-]+$/.test(parentId) || parentId === session.id) return false;
74
+ return probePath(sessionPath(parentId)).state !== PROBE_ABSENT;
75
+ }
76
+
67
77
 
68
78
  export function listStoredSessions(options = {}) {
69
79
  const dir = getStoreDir();
@@ -423,6 +433,13 @@ function* sweepStaleSessionSteps(ttlMs, options = {}) {
423
433
  // messages), so there is no second, divergent parse of this file.
424
434
  const actual = record.doc;
425
435
  const diskClosed = record.closed === true || actual.status === 'closed';
436
+ if (retainedLinkedAgent(actual)) {
437
+ // Parent-owned agent transcripts are task history, not
438
+ // ephemeral worker cache. This covers both new open sessions
439
+ // and tombstones created by older Mixdog versions.
440
+ remaining++;
441
+ continue;
442
+ }
426
443
  if (diskClosed) {
427
444
  // A shared store can be tombstoned by another process while
428
445
  // this process still owns an in-flight controller for the same
@@ -59,6 +59,106 @@ function cleanValue(value) {
59
59
  return String(value || '').trim();
60
60
  }
61
61
 
62
+ function archivedAgentNotification(content, sessionId) {
63
+ const text = typeof content === 'string' ? content : '';
64
+ const marker = '\n\nResult:\n';
65
+ const markerAt = text.indexOf(marker);
66
+ if (markerAt < 0 || !/The async agent task .* has finished \(/.test(text.slice(0, markerAt))) return null;
67
+ const lines = text.slice(markerAt + marker.length)
68
+ .split(/\r?\n/)
69
+ .map((line) => line.replace(/^>\s?/, ''));
70
+ const divider = lines.findIndex((line) => line.trim() === '');
71
+ if (divider < 0) return null;
72
+ const headers = new Map();
73
+ for (const line of lines.slice(0, divider)) {
74
+ const match = /^([A-Za-z][A-Za-z0-9_]*):\s*(.*?)\s*$/.exec(line);
75
+ if (match) headers.set(match[1], match[2]);
76
+ }
77
+ if (headers.get('surface') !== 'agent' || headers.get('sessionId') !== sessionId) return null;
78
+ const status = cleanValue(headers.get('status')).toLowerCase();
79
+ if (!/^(?:completed|failed|cancelled)$/.test(status)) return null;
80
+ const body = lines.slice(divider + 1).join('\n').trim();
81
+ if (!body) return null;
82
+ return {
83
+ body,
84
+ status,
85
+ tag: cleanValue(headers.get('tag') || headers.get('label')),
86
+ agent: cleanValue(headers.get('agent')),
87
+ provider: cleanValue(headers.get('provider')),
88
+ model: cleanValue(headers.get('model')),
89
+ effort: cleanValue(headers.get('effort')),
90
+ fast: cleanValue(headers.get('fast')).toLowerCase() === 'true',
91
+ finishedAt: Date.parse(cleanValue(headers.get('finished'))) || 0,
92
+ };
93
+ }
94
+
95
+ /** Legacy recovery for child transcripts already unlinked by terminal reaping.
96
+ * The parent owns one canonical body-carrying completion notification, so scan
97
+ * only files containing the exact child id and project the newest valid body. */
98
+ function readArchivedAgentResult(sessionId) {
99
+ const dir = join(dataDir(), 'sessions');
100
+ if (probePath(dir).state !== PROBE_PRESENT) return null;
101
+ let files;
102
+ try { files = readdirSync(dir).filter((file) => file.endsWith('.json')); }
103
+ catch { return null; }
104
+ let best = null;
105
+ for (const file of files) {
106
+ let raw;
107
+ try { raw = readFileSync(join(dir, file), 'utf8'); } catch { continue; }
108
+ if (!raw.includes(sessionId)) continue;
109
+ const record = readTopLevelLifecycleRecord(raw);
110
+ if (isLifecycleUnreadable(record) || record.id === sessionId) continue;
111
+ const parent = record.doc;
112
+ const messages = Array.isArray(parent.messages) ? parent.messages : [];
113
+ for (let index = messages.length - 1; index >= 0; index--) {
114
+ const message = messages[index];
115
+ if (message?.role !== 'user') continue;
116
+ const archived = archivedAgentNotification(message.content, sessionId);
117
+ if (!archived) continue;
118
+ const at = positiveNumber(
119
+ message?.meta?.transcript?.at,
120
+ archived.finishedAt || positiveNumber(parent.updatedAt),
121
+ );
122
+ if (best && best.at > at) continue;
123
+ best = { ...archived, at, parent };
124
+ break;
125
+ }
126
+ }
127
+ if (!best) return null;
128
+ const text = `# Archived agent result\n\n${best.body}`;
129
+ return {
130
+ sessionId,
131
+ items: [{
132
+ id: `archived-agent-result:${sessionId}`,
133
+ kind: 'assistant',
134
+ text,
135
+ status: best.status,
136
+ ...(best.at ? { at: best.at } : {}),
137
+ ...(best.model ? { model: best.model } : {}),
138
+ ...(best.provider ? { provider: best.provider } : {}),
139
+ ...(best.agent ? { agent: best.agent } : {}),
140
+ }],
141
+ provider: best.provider,
142
+ model: best.model,
143
+ effort: best.effort,
144
+ fast: best.fast,
145
+ cwd: cleanValue(best.parent.cwd),
146
+ desktopSession: desktopSession(best.parent.desktopSession, best.parent.cwd),
147
+ workflow: null,
148
+ stats: {
149
+ currentContextTokens: 0,
150
+ currentEstimatedContextTokens: 0,
151
+ currentContextSource: null,
152
+ },
153
+ contextWindow: null,
154
+ rawContextWindow: null,
155
+ displayContextWindow: null,
156
+ autoCompactTokenLimit: null,
157
+ archivedAgentResult: true,
158
+ readOnlyDetachedAgent: false,
159
+ };
160
+ }
161
+
62
162
  function activeAgentWorker(row) {
63
163
  const statuses = [row?.stage, row?.status]
64
164
  .map(cleanValue)
@@ -421,6 +521,7 @@ export async function readStoredSessionTranscript(id, options = {}) {
421
521
  // Same strict authority as the store, and the same fail-closed rule: an
422
522
  // absent, unreadable, ambiguous or foreign record yields no transcript.
423
523
  const read = readTextFile(join(dataDir(), 'sessions', `${sessionId}.json`));
524
+ if (read.state === PROBE_ABSENT) return readArchivedAgentResult(sessionId);
424
525
  if (read.state !== PROBE_PRESENT) return null;
425
526
  const record = readTopLevelLifecycleRecord(read.text);
426
527
  if (isLifecycleUnreadable(record) || record.id !== sessionId) return null;
@@ -1188,6 +1188,36 @@ export function loadSession(id) {
1188
1188
  return stored ? _ensureLifecycleFields(stored) : null;
1189
1189
  }
1190
1190
 
1191
+ /** Strictly enumerate child-agent session files linked to one visible parent.
1192
+ * Used only by explicit parent deletion; ordinary close/context switches keep
1193
+ * the relationship intact. */
1194
+ export function listOwnedAgentSessionIds(ownerSessionId) {
1195
+ const ownerId = String(ownerSessionId || '').trim();
1196
+ if (!/^[A-Za-z0-9_-]+$/.test(ownerId)) return [];
1197
+ const dir = getStoreDir();
1198
+ if (probePath(dir).state !== PROBE_PRESENT) return [];
1199
+ let files;
1200
+ try {
1201
+ files = readdirSync(dir).filter((file) => file.endsWith('.json'));
1202
+ } catch {
1203
+ return [];
1204
+ }
1205
+ const ids = [];
1206
+ for (const file of files) {
1207
+ const candidateId = file.slice(0, -5);
1208
+ if (!candidateId || candidateId === ownerId || !/^[A-Za-z0-9_-]+$/.test(candidateId)) continue;
1209
+ try {
1210
+ const record = readTopLevelLifecycleRecord(readFileSync(join(dir, file), 'utf8'));
1211
+ if (isLifecycleUnreadable(record) || record.id !== candidateId) continue;
1212
+ const session = record.doc;
1213
+ if (!isAgentOwner(session)) continue;
1214
+ const linkedOwner = String(session.ownerSessionId || session.parentSessionId || '').trim();
1215
+ if (linkedOwner === ownerId) ids.push(candidateId);
1216
+ } catch { /* unreadable/vanished records are never deletion targets */ }
1217
+ }
1218
+ return ids;
1219
+ }
1220
+
1191
1221
 
1192
1222
  export function deleteSession(id, options = {}) {
1193
1223
  // Keep caller probes and all vetoes ahead of the non-reentrant lock and
@@ -37,6 +37,7 @@ import {
37
37
  parseNativeToolSearchPayload,
38
38
  } from './loop/tool-helpers.mjs';
39
39
  import { restoreToolCallBodyForId } from './loop/stored-tool-args.mjs';
40
+ import { commitSessionCwdProbe } from '../tools/shell-state.mjs';
40
41
 
41
42
  function classifyToolReturn(value) {
42
43
  const normalized = normalizeToolEnvelope(value);
@@ -50,6 +51,7 @@ export async function processToolBatch(ctx) {
50
51
  crossTurnCalls, crossTurnCap, sessionAgent, steeringLadder,
51
52
  pushToolResultMessage, throwIfAborted, repeatFailLimit,
52
53
  } = ctx;
54
+ const executeToolFn = typeof ctx.executeToolFn === 'function' ? ctx.executeToolFn : executeTool;
53
55
  let dedupStubTotal = ctx.dedupStubTotal;
54
56
  let editCount = ctx.editCount;
55
57
  // Execute each tool and append results.
@@ -91,16 +93,11 @@ export async function processToolBatch(ctx) {
91
93
  // between two tool results of the same multi-tool turn (which would put a
92
94
  // user message between tool(A) and tool(B) and break provider pairing).
93
95
  const _batchNewMessages = [];
94
- // Ordered-mutation batch gate. apply_patch is a mutation tool (never
95
- // eager-dispatchable), so multiple apply_patch calls in ONE assistant
96
- // turn already execute serially in call-index order via this loop.
97
- // This flag records the FIRST ordered mutation whose execution failed
98
- // in this batch; every LATER apply_patch in the same batch is then
99
- // skipped (not executed) because its edits may depend on the failed
100
- // one and applying them against unchanged/partially-changed files
101
- // risks corrupt or misplaced writes. Only apply_patch is gated —
102
- // non-mutation tools (reads/grep/shell/...) keep their normal
103
- // parallelism and behavior. Reset per batch (per assistant turn).
96
+ // Ordered-mutation batch gate. Each apply_patch is an eager-dispatch
97
+ // barrier: side effects overlap within the current segment, but no
98
+ // later segment starts before the loop crosses the patch in model
99
+ // order. A failed patch prevents every unstarted later side effect;
100
+ // known read-only calls may still complete for diagnostics.
104
101
  let _orderedMutationFailed = null;
105
102
  for (let callIndex = 0; callIndex < calls.length; callIndex += 1) {
106
103
  const call = calls[callIndex];
@@ -196,22 +193,18 @@ export async function processToolBatch(ctx) {
196
193
  }
197
194
  }
198
195
  // Ordered-mutation skip: an earlier apply_patch in THIS batch failed,
199
- // so this later apply_patch is skipped rather than executed. Restore
200
- // its full patch body first (this call never ran) so the model can
201
- // re-issue it cleanly in a new turn instead of copying back a
202
- // `[mixdog compacted …]` placeholder. Emits a matching is_error
203
- // tool_result so the assistant tool_use is not orphaned.
204
- // Full-parallel caveat: a call that ALREADY started (pending eager
205
- // promise) cannot be "skipped" — its side effects are running, so
206
- // its real result must be consumed below. The gate only skips
207
- // calls that have not started (apply_patch never starts eagerly).
196
+ // so this later side-effect call is skipped rather than executed.
197
+ // Restore a later patch's full body first so it can be re-issued.
198
+ // `skipped` is deliberate: the call never started, so it is neither
199
+ // a tool failure nor a cancellation. The earlier patch remains the
200
+ // structural failure that drives retry/stop-hook behavior.
208
201
  if (_orderedMutationFailed && _isOrderedGateSkippable(call.name) && !pending.has(call.id)) {
209
202
  if (call?.id) restoreToolCallBodyForId(assistantTurnMsg, calls, call.id);
210
203
  pushToolResultMessage({
211
204
  role: 'tool',
212
- content: `Error: [ordered-mutation-skip] an earlier apply_patch in this same tool batch (call index ${_orderedMutationFailed.index + 1}) failed; ordered mutations in one batch are all-or-nothing after a failure, so this later \`${call.name}\` was NOT executed — it may depend on the failed mutation and running it now against unchanged/partially-changed state could corrupt files. Re-issue it in a new turn after resolving the earlier failure.`,
205
+ content: `[prerequisite-failed] an earlier apply_patch in this same tool batch (call index ${_orderedMutationFailed.index + 1}) failed, so this later \`${call.name}\` was skipped without starting. Re-issue it after resolving the patch failure.`,
213
206
  toolCallId: call.id,
214
- toolKind: 'error',
207
+ toolKind: 'skipped',
215
208
  });
216
209
  continue;
217
210
  }
@@ -320,7 +313,7 @@ export async function processToolBatch(ctx) {
320
313
  toolEndedAt = Date.now();
321
314
  _resultKind = 'error';
322
315
  } else {
323
- result = await executeTool(call.name, call.arguments, cwd, sessionId, sessionRef, { toolCallId: call.id, signal, notifyFn: opts.notifyFn, toolApprovalHook: opts.onToolApproval, iteration: iterations });
316
+ result = await executeToolFn(call.name, call.arguments, cwd, sessionId, sessionRef, { toolCallId: call.id, signal, notifyFn: opts.notifyFn, toolApprovalHook: opts.onToolApproval, iteration: iterations, deferShellCwdCommit: true });
324
317
  toolEndedAt = Date.now();
325
318
  // Boundary: tool-return string convention → structural kind.
326
319
  // The only prefix check in this codebase; downstream layers
@@ -342,6 +335,9 @@ export async function processToolBatch(ctx) {
342
335
  result = `Error: ${err instanceof Error ? err.message : String(err)}`;
343
336
  _resultKind = 'error';
344
337
  }
338
+ if (_isShellTool(call.name)) {
339
+ commitSessionCwdProbe(sessionId, call.id);
340
+ }
345
341
  // CENTRAL ENVELOPE NORMALIZE (general newMessages channel).
346
342
  // executeTool (serial + eager) and cache/error paths above all
347
343
  // funnel into `result`. Split ONCE here: downstream post-processing
@@ -506,7 +502,7 @@ export async function processToolBatch(ctx) {
506
502
  if (_isMutationTool(call.name)) {
507
503
  epoch.mutation += 1;
508
504
  // Record the first failed ordered mutation in this batch so any
509
- // LATER apply_patch is skipped by the gate at the top of the
505
+ // later side effect is skipped by the gate at the top of the
510
506
  // loop. Keyed on exec outcome (not post-processing): a mutation
511
507
  // whose write succeeded but post-processing threw still landed
512
508
  // on disk, so it must NOT block subsequent ordered patches.
@@ -130,22 +130,31 @@ export const PROVIDER_SSE_IDLE_TIMEOUT_MS = resolveTimeoutMs(
130
130
  // SEMANTIC progress (message/content/tool deltas) rather than raw keepalive
131
131
  // bytes (Anthropic `:ping`, comment frames). A truly silent stream — one that
132
132
  // emits no semantic event for this window — trips it; a live extended-thinking
133
- // stream (which emits thinking deltas) stays alive. Default 120s, floor 10s,
134
- // env-overridable and disablable via MIXDOG_ENABLE_STREAM_WATCHDOG=0.
133
+ // stream (which emits thinking deltas) stays alive. Floor 10s, env-overridable
134
+ // and disablable via MIXDOG_ENABLE_STREAM_WATCHDOG=0.
135
+ //
136
+ // 2026-07-05 trace audit: effort-mode (output_config.effort) claude streams
137
+ // can deliver NO deltas during the thinking phase — the whole thinking+text
138
+ // body flushes at the end (44/47 slow turns had stream_total-ttft < 2s;
139
+ // silent-window token rate a steady ~92 tok/s, i.e. live generation, not a
140
+ // wedge). So a "semantic-silent" stream is NOT necessarily dead.
141
+ //
142
+ // 2026-08-03 v3 postmortem: this window used to be DERIVED from the agent
143
+ // stall budget (warn − tick ≈ 285s). A harness that tightened STALL_TIMEOUT_S
144
+ // to 300s silently collapsed it to 135s, which deterministically beheaded
145
+ // live long-thinking turns (regex-chess / circuit-fibsqrt: ttft ~1.5s, then
146
+ // killed at exactly 135s of silent thinking, 4 attempts ≈ 552s per turn).
147
+ // Reference parity: Codex aborts only after a 300s single-gap silence
148
+ // (DEFAULT_STREAM_IDLE_TIMEOUT_MS); Claude Code ships its watchdog OFF by
149
+ // default and falls back to non-streaming when it does abort. The default is
150
+ // therefore a FIXED 300s, decoupled from the warn math. The only remaining
151
+ // coupling is the ordering guarantee: cap at (abort − tick) so the provider
152
+ // layer — which can retry or fall back non-streaming — always fires strictly
153
+ // before the agent stall watchdog's terminal abort.
135
154
  export const PROVIDER_SEMANTIC_IDLE_TIMEOUT_MS = resolveTimeoutMs(
136
155
  ['MIXDOG_PROVIDER_SEMANTIC_IDLE_TIMEOUT_MS', 'MIXDOG_PROVIDER_SSE_IDLE_TIMEOUT_MS'],
137
- // 2026-07-05 trace audit: effort-mode (output_config.effort) sonnet-5
138
- // streams deliver NO deltas during the thinking phase — the whole
139
- // thinking+text body flushes at the end (44/47 slow turns had
140
- // stream_total-ttft < 2s; silent-window token rate a steady ~92 tok/s,
141
- // i.e. live generation, not a wedge). Successful turns topped out at
142
- // ttft 171s while 13 kills sat at exactly ~183s fetch→fetch — the old
143
- // 180s window was beheading every turn whose silent thinking ran past
144
- // it, then retrying the whole thinking run from zero. Raised to the
145
- // policy ceiling (STALL_WARN - tick ≈ 285s); the agent stall watchdog
146
- // (worker 300s) remains the true-wedge backstop just above it.
147
- PROVIDER_MAX_BEFORE_WARN_MS,
148
- { minMs: 10_000, maxMs: STALL_WARN_MS },
156
+ 300_000,
157
+ { minMs: 10_000, maxMs: Math.max(10_000, STALL_ABORT_MS - STALL_TICK_MS) },
149
158
  );
150
159
 
151
160
  // Named terminal error for a mid-stream SEMANTIC idle abort. Distinct from a
@@ -246,15 +255,16 @@ export const PROVIDER_WS_FIRST_MEANINGFUL_TIMEOUT_MS = resolveTimeoutMs(
246
255
  { minMs: 10_000, maxMs: STALL_WARN_MS },
247
256
  );
248
257
 
249
- // WS semantic idle uses the same default ceiling as OpenAI HTTP/SSE semantic
250
- // idle. This is semantic progress only (not the WS inter-chunk byte timer), so
251
- // reasoning/text/tool deltas reset it while metadata cannot. Keep the default
252
- // at PROVIDER_MAX_BEFORE_WARN_MS (~285s), strictly below the 300s worker
253
- // watchdog, rather than rounding it to 300s.
258
+ // WS semantic idle uses the same default as the SSE semantic idle above
259
+ // (fixed 300s, decoupled from the agent stall-budget warn math — see that
260
+ // comment for the 135s-collapse regression). This is semantic progress only
261
+ // (not the WS inter-chunk byte timer), so reasoning/text/tool deltas reset it
262
+ // while metadata cannot. Capped at (abort − tick) so the provider layer fires
263
+ // strictly below the agent stall watchdog.
254
264
  export const PROVIDER_WS_SEMANTIC_IDLE_TIMEOUT_MS = resolveTimeoutMs(
255
265
  ['MIXDOG_PROVIDER_WS_SEMANTIC_IDLE_TIMEOUT_MS', 'MIXDOG_PROVIDER_WS_OUTPUT_IDLE_TIMEOUT_MS'],
256
- PROVIDER_MAX_BEFORE_WARN_MS,
257
- { minMs: 10_000, maxMs: PROVIDER_MAX_BEFORE_WARN_MS },
266
+ 300_000,
267
+ { minMs: 10_000, maxMs: Math.max(10_000, STALL_ABORT_MS - STALL_TICK_MS) },
258
268
  );
259
269
 
260
270
  // First retry has a small floor (250ms) instead of 0ms: an immediate reissue on
@@ -600,7 +600,10 @@ export async function executeBashTool(args, workDir, options = {}) {
600
600
  // so the exit code the model sees is unchanged.
601
601
  let syncCommand = wrappedCommand;
602
602
  try {
603
- const _stateFile = stateFilePath(_sessionCwdKey);
603
+ const _stateFile = stateFilePath(
604
+ _sessionCwdKey,
605
+ options?.deferShellCwdCommit === true ? options?.toolCallId : null,
606
+ );
604
607
  if (_stateFile) {
605
608
  syncCommand = (process.platform === 'win32' && shellType === 'powershell')
606
609
  ? wrapPowerShellWithCwdProbe(wrappedCommand, _stateFile)
@@ -23,9 +23,9 @@ function _shellMaxTimeoutMs() {
23
23
  return Math.max(parsed > 0 ? parsed : 600_000, _shellDefaultTimeoutMs());
24
24
  }
25
25
 
26
- // PowerShell-only syntax cheat, injected into the shell tool description when
27
- // the host default shell is PowerShell (win32). process.platform is fixed for
28
- // the process lifetime, so this is evaluated once at module load.
26
+ // PowerShell-only syntax cheat, kept next to the command argument when the host
27
+ // default shell is PowerShell (win32). process.platform is fixed for the
28
+ // process lifetime, so this is evaluated once at module load.
29
29
  const _shellSyntaxCheat =
30
30
  process.platform === 'win32'
31
31
  ? ' PowerShell: grep→Select-String, tail→Get-Content -Tail, head→Get-Content -TotalCount, /c/→C:\\, if && is unsupported use ;, $PID is reserved.'
@@ -68,18 +68,18 @@ export const BUILTIN_TOOLS = [
68
68
  limit: { type: 'number', minimum: 1, description: 'Max lines after offset. Defaults to 2000.' },
69
69
  },
70
70
  required: ['path'],
71
+ additionalProperties: false,
71
72
  },
72
73
  },
73
74
  {
74
75
  name: 'shell',
75
76
  title: 'Mixdog Shell',
76
77
  annotations: { title: 'Mixdog Shell', readOnlyHint: false, destructiveHint: true, idempotentHint: false, openWorldHint: true, compressible: true },
77
- description: 'Runs a shell command and returns its output. Use async for sleep/watch/dev loops.'
78
- + `${_shellSyntaxCheat} ${TOOL_ASYNC_EXECUTION_CONTRACT}`,
78
+ description: `Run a shell command. ${TOOL_ASYNC_EXECUTION_CONTRACT}`,
79
79
  inputSchema: {
80
80
  type: 'object',
81
81
  properties: {
82
- command: { type: 'string', description: 'Command.' },
82
+ command: { type: 'string', description: `Command.${_shellSyntaxCheat}` },
83
83
  cwd: { type: 'string', description: 'Working directory; persists across calls. Omit to reuse; absolute path changes it.' },
84
84
  timeout: {
85
85
  type: 'number',
@@ -92,6 +92,7 @@ export const BUILTIN_TOOLS = [
92
92
  shell: { type: 'string', enum: ['bash', 'powershell'], description: 'Force shell. Windows defaults to PowerShell; bash = Git Bash/POSIX.' },
93
93
  },
94
94
  required: ['command'],
95
+ additionalProperties: false,
95
96
  },
96
97
  },
97
98
  {
@@ -107,6 +108,7 @@ export const BUILTIN_TOOLS = [
107
108
  timeout_ms: { type: 'number', description: 'Wait timeout ms.' },
108
109
  },
109
110
  required: [],
111
+ additionalProperties: false,
110
112
  },
111
113
  },
112
114
  {
@@ -147,6 +149,7 @@ export const BUILTIN_TOOLS = [
147
149
  { required: ['pattern'] },
148
150
  { required: ['glob'] },
149
151
  ],
152
+ additionalProperties: false,
150
153
  },
151
154
  },
152
155
  {
@@ -175,6 +178,7 @@ export const BUILTIN_TOOLS = [
175
178
  offset: { type: 'number', description: 'Skip entries.' },
176
179
  },
177
180
  required: ['pattern'],
181
+ additionalProperties: false,
178
182
  },
179
183
  },
180
184
  {
@@ -196,6 +200,7 @@ export const BUILTIN_TOOLS = [
196
200
  head_limit: { type: 'number', description: 'Max paths. Defaults to 25.' },
197
201
  },
198
202
  required: ['query'],
203
+ additionalProperties: false,
199
204
  },
200
205
  },
201
206
  {
@@ -217,6 +222,7 @@ export const BUILTIN_TOOLS = [
217
222
  offset: { type: 'number', description: 'Skip N entries for paging.' },
218
223
  },
219
224
  required: [],
225
+ additionalProperties: false,
220
226
  },
221
227
  },
222
228
  ];
@@ -206,6 +206,11 @@ export async function executeTaskTool(args, options = {}) {
206
206
  // consumed synchronously, so no re-arm. Drop the persisted ctx here
207
207
  // or it leaks (cleanup only runs on a real watcher settle, which
208
208
  // never happens for a never-re-armed entry).
209
+ // EXCEPT when this wait's tool call was aborted: its result was
210
+ // discarded, so nobody consumed the outcome — re-arm so the
211
+ // watcher delivers the completion notification instead of
212
+ // swallowing it.
213
+ else if (options?.signal?.aborted) watchBackgroundShellJob(taskId);
209
214
  else clearShellJobNotifyCtx(taskId);
210
215
  }
211
216
  }
@@ -3,19 +3,20 @@ export const CODE_GRAPH_TOOL_DEFS = [
3
3
  name: 'code_graph',
4
4
  title: 'Code Graph',
5
5
  annotations: { title: 'Code Graph', readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: false, compressible: false, compressibleLossless: true },
6
- description: 'Repo code structure/flow over source files only. File modes take files[]; symbol modes find_symbol/symbol_search/search/references/callers/callees take symbols[]. Exact identifiers use find_symbol/references/callers/callees and keywords use symbol_search/search. Unsupported target arrays are omitted, never silently mixed.',
6
+ description: 'Repo code structure/flow over source files. File modes take files[]; symbol modes take symbols[] — exact identifiers via find_symbol/references/callers/callees, keywords via symbol_search/search. Unsupported target arrays are omitted, never mixed.',
7
7
  inputSchema: {
8
8
  type: 'object',
9
9
  properties: {
10
- mode: { type: 'string', enum: ['overview', 'imports', 'dependents', 'related', 'impact', 'symbols', 'find_symbol', 'symbol_search', 'search', 'references', 'callers', 'callees'], description: 'File modes={overview,imports,dependents,related,impact}; symbols with files→files[] file outline; symbol modes={find_symbol,symbol_search,search,references,callers,callees}; fileless symbols→symbol_search keywords.' },
10
+ mode: { type: 'string', enum: ['overview', 'imports', 'dependents', 'related', 'impact', 'symbols', 'find_symbol', 'symbol_search', 'search', 'references', 'callers', 'callees'], description: 'File modes={overview,imports,dependents,related,impact}; symbols with files[]=file outline; the rest are symbol modes.' },
11
11
  files: { anyOf: [{ type: 'string' }, { type: 'array', items: { type: 'string' }, minItems: 1 }], description: 'Source file path(s); supported targets only.' },
12
- symbols: { anyOf: [{ type: 'string' }, { type: 'array', items: { type: 'string' }, minItems: 1 }], description: 'Exact identifiers (find_symbol/references/callers/callees) or keywords (symbol_search/search); multiple exact symbols use one symbols[] call.' },
12
+ symbols: { anyOf: [{ type: 'string' }, { type: 'array', items: { type: 'string' }, minItems: 1 }], description: 'Exact identifiers or keywords; batch multiple in one symbols[].' },
13
13
  body: { type: 'boolean', description: 'Include body.' },
14
14
  limit: { type: 'number', minimum: 1, description: 'Max results.' },
15
15
  depth: { type: 'number', minimum: 1, maximum: 5, description: 'Caller depth.' },
16
16
  page: { type: 'number', minimum: 1, description: 'Caller page.' },
17
17
  },
18
18
  required: ['mode'],
19
+ additionalProperties: false,
19
20
  },
20
21
  },
21
22
  ];