mixdog 0.9.92 → 0.9.93

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/README.md +147 -51
  2. package/package.json +6 -5
  3. package/scripts/code-graph-description-contract.mjs +6 -8
  4. package/scripts/tmp-cdp-errors.mjs +41 -0
  5. package/scripts/tmp-cdp-inspect.mjs +41 -0
  6. package/scripts/tui-transcript-jitter-harness.mjs +2 -18
  7. package/src/rules/agent/00-core.md +1 -2
  8. package/src/rules/lead/01-general.md +1 -0
  9. package/src/rules/shared/01-tool.md +19 -22
  10. package/src/runtime/agent/orchestrator/agent-runtime/cache-strategy.mjs +16 -3
  11. package/src/runtime/agent/orchestrator/agent-trace.mjs +17 -0
  12. package/src/runtime/agent/orchestrator/context/collect.mjs +2 -1
  13. package/src/runtime/agent/orchestrator/providers/anthropic-effort.mjs +9 -1
  14. package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +79 -21
  15. package/src/runtime/agent/orchestrator/providers/anthropic.mjs +40 -19
  16. package/src/runtime/agent/orchestrator/providers/lib/anthropic-request-utils.mjs +18 -1
  17. package/src/runtime/agent/orchestrator/providers/openai-oauth-http-sse.mjs +44 -8
  18. package/src/runtime/agent/orchestrator/providers/openai-ws-events.mjs +3 -0
  19. package/src/runtime/agent/orchestrator/providers/openai-ws-stream.mjs +2 -0
  20. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +22 -5
  21. package/src/runtime/agent/orchestrator/session/eager-dispatch.mjs +35 -29
  22. package/src/runtime/agent/orchestrator/session/loop/stored-tool-args.mjs +11 -2
  23. package/src/runtime/agent/orchestrator/session/loop/tool-classify.mjs +5 -6
  24. package/src/runtime/agent/orchestrator/session/loop/tool-exec.mjs +31 -2
  25. package/src/runtime/agent/orchestrator/session/manager/compaction-runner.mjs +60 -0
  26. package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +60 -31
  27. package/src/runtime/agent/orchestrator/session/manager.mjs +1 -1
  28. package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +12 -3
  29. package/src/runtime/agent/orchestrator/session/store/listing.mjs +17 -0
  30. package/src/runtime/agent/orchestrator/session/store-summary-reader.mjs +101 -0
  31. package/src/runtime/agent/orchestrator/session/store.mjs +30 -0
  32. package/src/runtime/agent/orchestrator/session/tool-batch.mjs +19 -23
  33. package/src/runtime/agent/orchestrator/stall-policy.mjs +31 -21
  34. package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +4 -1
  35. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +12 -6
  36. package/src/runtime/agent/orchestrator/tools/code-graph-tool-defs.mjs +4 -3
  37. package/src/runtime/agent/orchestrator/tools/lib/pwsh-standby-pool.mjs +47 -14
  38. package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +100 -0
  39. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +6 -5
  40. package/src/runtime/agent/orchestrator/tools/shell-command.mjs +9 -0
  41. package/src/runtime/agent/orchestrator/tools/shell-state.mjs +32 -2
  42. package/src/runtime/channels/backends/discord-gateway.mjs +6 -32
  43. package/src/runtime/channels/lib/inbound-handler.mjs +19 -3
  44. package/src/runtime/channels/lib/scheduler.mjs +51 -3
  45. package/src/runtime/channels/lib/worker-main.mjs +4 -0
  46. package/src/runtime/channels/tool-defs.mjs +4 -2
  47. package/src/runtime/memory/lib/query-handlers.mjs +11 -3
  48. package/src/runtime/memory/tool-defs.mjs +5 -5
  49. package/src/runtime/shared/channel-notification-routing.mjs +8 -2
  50. package/src/runtime/shared/llm/http-agent.mjs +11 -0
  51. package/src/session-runtime/lifecycle-api.mjs +26 -1
  52. package/src/session-runtime/tool-catalog-data.mjs +5 -2
  53. package/src/session-runtime/workflow.mjs +7 -5
  54. package/src/standalone/agent-tool/tag-registry.mjs +5 -1
  55. package/src/standalone/explore-tool.mjs +1 -1
  56. package/src/tui/dist/index.mjs +68 -45
  57. package/src/tui/engine/agent-envelope.mjs +52 -3
  58. package/src/tui/engine/turn.mjs +8 -9
  59. package/src/tui/engine.mjs +18 -41
  60. package/src/workflows/solo/WORKFLOW.md +0 -6
  61. package/src/workflows/solo-bench/WORKFLOW.md +17 -0
@@ -37,6 +37,7 @@ import {
37
37
  parseNativeToolSearchPayload,
38
38
  } from './loop/tool-helpers.mjs';
39
39
  import { restoreToolCallBodyForId } from './loop/stored-tool-args.mjs';
40
+ import { commitSessionCwdProbe } from '../tools/shell-state.mjs';
40
41
 
41
42
  function classifyToolReturn(value) {
42
43
  const normalized = normalizeToolEnvelope(value);
@@ -50,6 +51,7 @@ export async function processToolBatch(ctx) {
50
51
  crossTurnCalls, crossTurnCap, sessionAgent, steeringLadder,
51
52
  pushToolResultMessage, throwIfAborted, repeatFailLimit,
52
53
  } = ctx;
54
+ const executeToolFn = typeof ctx.executeToolFn === 'function' ? ctx.executeToolFn : executeTool;
53
55
  let dedupStubTotal = ctx.dedupStubTotal;
54
56
  let editCount = ctx.editCount;
55
57
  // Execute each tool and append results.
@@ -91,16 +93,11 @@ export async function processToolBatch(ctx) {
91
93
  // between two tool results of the same multi-tool turn (which would put a
92
94
  // user message between tool(A) and tool(B) and break provider pairing).
93
95
  const _batchNewMessages = [];
94
- // Ordered-mutation batch gate. apply_patch is a mutation tool (never
95
- // eager-dispatchable), so multiple apply_patch calls in ONE assistant
96
- // turn already execute serially in call-index order via this loop.
97
- // This flag records the FIRST ordered mutation whose execution failed
98
- // in this batch; every LATER apply_patch in the same batch is then
99
- // skipped (not executed) because its edits may depend on the failed
100
- // one and applying them against unchanged/partially-changed files
101
- // risks corrupt or misplaced writes. Only apply_patch is gated —
102
- // non-mutation tools (reads/grep/shell/...) keep their normal
103
- // parallelism and behavior. Reset per batch (per assistant turn).
96
+ // Ordered-mutation batch gate. Each apply_patch is an eager-dispatch
97
+ // barrier: side effects overlap within the current segment, but no
98
+ // later segment starts before the loop crosses the patch in model
99
+ // order. A failed patch prevents every unstarted later side effect;
100
+ // known read-only calls may still complete for diagnostics.
104
101
  let _orderedMutationFailed = null;
105
102
  for (let callIndex = 0; callIndex < calls.length; callIndex += 1) {
106
103
  const call = calls[callIndex];
@@ -196,22 +193,18 @@ export async function processToolBatch(ctx) {
196
193
  }
197
194
  }
198
195
  // Ordered-mutation skip: an earlier apply_patch in THIS batch failed,
199
- // so this later apply_patch is skipped rather than executed. Restore
200
- // its full patch body first (this call never ran) so the model can
201
- // re-issue it cleanly in a new turn instead of copying back a
202
- // `[mixdog compacted …]` placeholder. Emits a matching is_error
203
- // tool_result so the assistant tool_use is not orphaned.
204
- // Full-parallel caveat: a call that ALREADY started (pending eager
205
- // promise) cannot be "skipped" — its side effects are running, so
206
- // its real result must be consumed below. The gate only skips
207
- // calls that have not started (apply_patch never starts eagerly).
196
+ // so this later side-effect call is skipped rather than executed.
197
+ // Restore a later patch's full body first so it can be re-issued.
198
+ // `skipped` is deliberate: the call never started, so it is neither
199
+ // a tool failure nor a cancellation. The earlier patch remains the
200
+ // structural failure that drives retry/stop-hook behavior.
208
201
  if (_orderedMutationFailed && _isOrderedGateSkippable(call.name) && !pending.has(call.id)) {
209
202
  if (call?.id) restoreToolCallBodyForId(assistantTurnMsg, calls, call.id);
210
203
  pushToolResultMessage({
211
204
  role: 'tool',
212
- content: `Error: [ordered-mutation-skip] an earlier apply_patch in this same tool batch (call index ${_orderedMutationFailed.index + 1}) failed; ordered mutations in one batch are all-or-nothing after a failure, so this later \`${call.name}\` was NOT executed — it may depend on the failed mutation and running it now against unchanged/partially-changed state could corrupt files. Re-issue it in a new turn after resolving the earlier failure.`,
205
+ content: `[prerequisite-failed] an earlier apply_patch in this same tool batch (call index ${_orderedMutationFailed.index + 1}) failed, so this later \`${call.name}\` was skipped without starting. Re-issue it after resolving the patch failure.`,
213
206
  toolCallId: call.id,
214
- toolKind: 'error',
207
+ toolKind: 'skipped',
215
208
  });
216
209
  continue;
217
210
  }
@@ -320,7 +313,7 @@ export async function processToolBatch(ctx) {
320
313
  toolEndedAt = Date.now();
321
314
  _resultKind = 'error';
322
315
  } else {
323
- result = await executeTool(call.name, call.arguments, cwd, sessionId, sessionRef, { toolCallId: call.id, signal, notifyFn: opts.notifyFn, toolApprovalHook: opts.onToolApproval, iteration: iterations });
316
+ result = await executeToolFn(call.name, call.arguments, cwd, sessionId, sessionRef, { toolCallId: call.id, signal, notifyFn: opts.notifyFn, toolApprovalHook: opts.onToolApproval, iteration: iterations, deferShellCwdCommit: true });
324
317
  toolEndedAt = Date.now();
325
318
  // Boundary: tool-return string convention → structural kind.
326
319
  // The only prefix check in this codebase; downstream layers
@@ -342,6 +335,9 @@ export async function processToolBatch(ctx) {
342
335
  result = `Error: ${err instanceof Error ? err.message : String(err)}`;
343
336
  _resultKind = 'error';
344
337
  }
338
+ if (_isShellTool(call.name)) {
339
+ commitSessionCwdProbe(sessionId, call.id);
340
+ }
345
341
  // CENTRAL ENVELOPE NORMALIZE (general newMessages channel).
346
342
  // executeTool (serial + eager) and cache/error paths above all
347
343
  // funnel into `result`. Split ONCE here: downstream post-processing
@@ -506,7 +502,7 @@ export async function processToolBatch(ctx) {
506
502
  if (_isMutationTool(call.name)) {
507
503
  epoch.mutation += 1;
508
504
  // Record the first failed ordered mutation in this batch so any
509
- // LATER apply_patch is skipped by the gate at the top of the
505
+ // later side effect is skipped by the gate at the top of the
510
506
  // loop. Keyed on exec outcome (not post-processing): a mutation
511
507
  // whose write succeeded but post-processing threw still landed
512
508
  // on disk, so it must NOT block subsequent ordered patches.
@@ -130,22 +130,31 @@ export const PROVIDER_SSE_IDLE_TIMEOUT_MS = resolveTimeoutMs(
130
130
  // SEMANTIC progress (message/content/tool deltas) rather than raw keepalive
131
131
  // bytes (Anthropic `:ping`, comment frames). A truly silent stream — one that
132
132
  // emits no semantic event for this window — trips it; a live extended-thinking
133
- // stream (which emits thinking deltas) stays alive. Default 120s, floor 10s,
134
- // env-overridable and disablable via MIXDOG_ENABLE_STREAM_WATCHDOG=0.
133
+ // stream (which emits thinking deltas) stays alive. Floor 10s, env-overridable
134
+ // and disablable via MIXDOG_ENABLE_STREAM_WATCHDOG=0.
135
+ //
136
+ // 2026-07-05 trace audit: effort-mode (output_config.effort) claude streams
137
+ // can deliver NO deltas during the thinking phase — the whole thinking+text
138
+ // body flushes at the end (44/47 slow turns had stream_total-ttft < 2s;
139
+ // silent-window token rate a steady ~92 tok/s, i.e. live generation, not a
140
+ // wedge). So a "semantic-silent" stream is NOT necessarily dead.
141
+ //
142
+ // 2026-08-03 v3 postmortem: this window used to be DERIVED from the agent
143
+ // stall budget (warn − tick ≈ 285s). A harness that tightened STALL_TIMEOUT_S
144
+ // to 300s silently collapsed it to 135s, which deterministically beheaded
145
+ // live long-thinking turns (regex-chess / circuit-fibsqrt: ttft ~1.5s, then
146
+ // killed at exactly 135s of silent thinking, 4 attempts ≈ 552s per turn).
147
+ // Reference parity: Codex aborts only after a 300s single-gap silence
148
+ // (DEFAULT_STREAM_IDLE_TIMEOUT_MS); Claude Code ships its watchdog OFF by
149
+ // default and falls back to non-streaming when it does abort. The default is
150
+ // therefore a FIXED 300s, decoupled from the warn math. The only remaining
151
+ // coupling is the ordering guarantee: cap at (abort − tick) so the provider
152
+ // layer — which can retry or fall back non-streaming — always fires strictly
153
+ // before the agent stall watchdog's terminal abort.
135
154
  export const PROVIDER_SEMANTIC_IDLE_TIMEOUT_MS = resolveTimeoutMs(
136
155
  ['MIXDOG_PROVIDER_SEMANTIC_IDLE_TIMEOUT_MS', 'MIXDOG_PROVIDER_SSE_IDLE_TIMEOUT_MS'],
137
- // 2026-07-05 trace audit: effort-mode (output_config.effort) sonnet-5
138
- // streams deliver NO deltas during the thinking phase — the whole
139
- // thinking+text body flushes at the end (44/47 slow turns had
140
- // stream_total-ttft < 2s; silent-window token rate a steady ~92 tok/s,
141
- // i.e. live generation, not a wedge). Successful turns topped out at
142
- // ttft 171s while 13 kills sat at exactly ~183s fetch→fetch — the old
143
- // 180s window was beheading every turn whose silent thinking ran past
144
- // it, then retrying the whole thinking run from zero. Raised to the
145
- // policy ceiling (STALL_WARN - tick ≈ 285s); the agent stall watchdog
146
- // (worker 300s) remains the true-wedge backstop just above it.
147
- PROVIDER_MAX_BEFORE_WARN_MS,
148
- { minMs: 10_000, maxMs: STALL_WARN_MS },
156
+ 300_000,
157
+ { minMs: 10_000, maxMs: Math.max(10_000, STALL_ABORT_MS - STALL_TICK_MS) },
149
158
  );
150
159
 
151
160
  // Named terminal error for a mid-stream SEMANTIC idle abort. Distinct from a
@@ -246,15 +255,16 @@ export const PROVIDER_WS_FIRST_MEANINGFUL_TIMEOUT_MS = resolveTimeoutMs(
246
255
  { minMs: 10_000, maxMs: STALL_WARN_MS },
247
256
  );
248
257
 
249
- // WS semantic idle uses the same default ceiling as OpenAI HTTP/SSE semantic
250
- // idle. This is semantic progress only (not the WS inter-chunk byte timer), so
251
- // reasoning/text/tool deltas reset it while metadata cannot. Keep the default
252
- // at PROVIDER_MAX_BEFORE_WARN_MS (~285s), strictly below the 300s worker
253
- // watchdog, rather than rounding it to 300s.
258
+ // WS semantic idle uses the same default as the SSE semantic idle above
259
+ // (fixed 300s, decoupled from the agent stall-budget warn math — see that
260
+ // comment for the 135s-collapse regression). This is semantic progress only
261
+ // (not the WS inter-chunk byte timer), so reasoning/text/tool deltas reset it
262
+ // while metadata cannot. Capped at (abort − tick) so the provider layer fires
263
+ // strictly below the agent stall watchdog.
254
264
  export const PROVIDER_WS_SEMANTIC_IDLE_TIMEOUT_MS = resolveTimeoutMs(
255
265
  ['MIXDOG_PROVIDER_WS_SEMANTIC_IDLE_TIMEOUT_MS', 'MIXDOG_PROVIDER_WS_OUTPUT_IDLE_TIMEOUT_MS'],
256
- PROVIDER_MAX_BEFORE_WARN_MS,
257
- { minMs: 10_000, maxMs: PROVIDER_MAX_BEFORE_WARN_MS },
266
+ 300_000,
267
+ { minMs: 10_000, maxMs: Math.max(10_000, STALL_ABORT_MS - STALL_TICK_MS) },
258
268
  );
259
269
 
260
270
  // First retry has a small floor (250ms) instead of 0ms: an immediate reissue on
@@ -600,7 +600,10 @@ export async function executeBashTool(args, workDir, options = {}) {
600
600
  // so the exit code the model sees is unchanged.
601
601
  let syncCommand = wrappedCommand;
602
602
  try {
603
- const _stateFile = stateFilePath(_sessionCwdKey);
603
+ const _stateFile = stateFilePath(
604
+ _sessionCwdKey,
605
+ options?.deferShellCwdCommit === true ? options?.toolCallId : null,
606
+ );
604
607
  if (_stateFile) {
605
608
  syncCommand = (process.platform === 'win32' && shellType === 'powershell')
606
609
  ? wrapPowerShellWithCwdProbe(wrappedCommand, _stateFile)
@@ -23,9 +23,9 @@ function _shellMaxTimeoutMs() {
23
23
  return Math.max(parsed > 0 ? parsed : 600_000, _shellDefaultTimeoutMs());
24
24
  }
25
25
 
26
- // PowerShell-only syntax cheat, injected into the shell tool description when
27
- // the host default shell is PowerShell (win32). process.platform is fixed for
28
- // the process lifetime, so this is evaluated once at module load.
26
+ // PowerShell-only syntax cheat, kept next to the command argument when the host
27
+ // default shell is PowerShell (win32). process.platform is fixed for the
28
+ // process lifetime, so this is evaluated once at module load.
29
29
  const _shellSyntaxCheat =
30
30
  process.platform === 'win32'
31
31
  ? ' PowerShell: grep→Select-String, tail→Get-Content -Tail, head→Get-Content -TotalCount, /c/→C:\\, if && is unsupported use ;, $PID is reserved.'
@@ -68,18 +68,18 @@ export const BUILTIN_TOOLS = [
68
68
  limit: { type: 'number', minimum: 1, description: 'Max lines after offset. Defaults to 2000.' },
69
69
  },
70
70
  required: ['path'],
71
+ additionalProperties: false,
71
72
  },
72
73
  },
73
74
  {
74
75
  name: 'shell',
75
76
  title: 'Mixdog Shell',
76
77
  annotations: { title: 'Mixdog Shell', readOnlyHint: false, destructiveHint: true, idempotentHint: false, openWorldHint: true, compressible: true },
77
- description: 'Runs a shell command and returns its output. Use async for sleep/watch/dev loops.'
78
- + `${_shellSyntaxCheat} ${TOOL_ASYNC_EXECUTION_CONTRACT}`,
78
+ description: `Run a shell command. ${TOOL_ASYNC_EXECUTION_CONTRACT}`,
79
79
  inputSchema: {
80
80
  type: 'object',
81
81
  properties: {
82
- command: { type: 'string', description: 'Command.' },
82
+ command: { type: 'string', description: `Command.${_shellSyntaxCheat}` },
83
83
  cwd: { type: 'string', description: 'Working directory; persists across calls. Omit to reuse; absolute path changes it.' },
84
84
  timeout: {
85
85
  type: 'number',
@@ -92,6 +92,7 @@ export const BUILTIN_TOOLS = [
92
92
  shell: { type: 'string', enum: ['bash', 'powershell'], description: 'Force shell. Windows defaults to PowerShell; bash = Git Bash/POSIX.' },
93
93
  },
94
94
  required: ['command'],
95
+ additionalProperties: false,
95
96
  },
96
97
  },
97
98
  {
@@ -107,6 +108,7 @@ export const BUILTIN_TOOLS = [
107
108
  timeout_ms: { type: 'number', description: 'Wait timeout ms.' },
108
109
  },
109
110
  required: [],
111
+ additionalProperties: false,
110
112
  },
111
113
  },
112
114
  {
@@ -147,6 +149,7 @@ export const BUILTIN_TOOLS = [
147
149
  { required: ['pattern'] },
148
150
  { required: ['glob'] },
149
151
  ],
152
+ additionalProperties: false,
150
153
  },
151
154
  },
152
155
  {
@@ -175,6 +178,7 @@ export const BUILTIN_TOOLS = [
175
178
  offset: { type: 'number', description: 'Skip entries.' },
176
179
  },
177
180
  required: ['pattern'],
181
+ additionalProperties: false,
178
182
  },
179
183
  },
180
184
  {
@@ -196,6 +200,7 @@ export const BUILTIN_TOOLS = [
196
200
  head_limit: { type: 'number', description: 'Max paths. Defaults to 25.' },
197
201
  },
198
202
  required: ['query'],
203
+ additionalProperties: false,
199
204
  },
200
205
  },
201
206
  {
@@ -217,6 +222,7 @@ export const BUILTIN_TOOLS = [
217
222
  offset: { type: 'number', description: 'Skip N entries for paging.' },
218
223
  },
219
224
  required: [],
225
+ additionalProperties: false,
220
226
  },
221
227
  },
222
228
  ];
@@ -3,19 +3,20 @@ export const CODE_GRAPH_TOOL_DEFS = [
3
3
  name: 'code_graph',
4
4
  title: 'Code Graph',
5
5
  annotations: { title: 'Code Graph', readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: false, compressible: false, compressibleLossless: true },
6
- description: 'Repo code structure/flow over source files only. File modes take files[]; symbol modes find_symbol/symbol_search/search/references/callers/callees take symbols[]. Exact identifiers use find_symbol/references/callers/callees and keywords use symbol_search/search. Unsupported target arrays are omitted, never silently mixed.',
6
+ description: 'Repo code structure/flow over source files. File modes take files[]; symbol modes take symbols[] — exact identifiers via find_symbol/references/callers/callees, keywords via symbol_search/search. Unsupported target arrays are omitted, never mixed.',
7
7
  inputSchema: {
8
8
  type: 'object',
9
9
  properties: {
10
- mode: { type: 'string', enum: ['overview', 'imports', 'dependents', 'related', 'impact', 'symbols', 'find_symbol', 'symbol_search', 'search', 'references', 'callers', 'callees'], description: 'File modes={overview,imports,dependents,related,impact}; symbols with files→files[] file outline; symbol modes={find_symbol,symbol_search,search,references,callers,callees}; fileless symbols→symbol_search keywords.' },
10
+ mode: { type: 'string', enum: ['overview', 'imports', 'dependents', 'related', 'impact', 'symbols', 'find_symbol', 'symbol_search', 'search', 'references', 'callers', 'callees'], description: 'File modes={overview,imports,dependents,related,impact}; symbols with files[]=file outline; the rest are symbol modes.' },
11
11
  files: { anyOf: [{ type: 'string' }, { type: 'array', items: { type: 'string' }, minItems: 1 }], description: 'Source file path(s); supported targets only.' },
12
- symbols: { anyOf: [{ type: 'string' }, { type: 'array', items: { type: 'string' }, minItems: 1 }], description: 'Exact identifiers (find_symbol/references/callers/callees) or keywords (symbol_search/search); multiple exact symbols use one symbols[] call.' },
12
+ symbols: { anyOf: [{ type: 'string' }, { type: 'array', items: { type: 'string' }, minItems: 1 }], description: 'Exact identifiers or keywords; batch multiple in one symbols[].' },
13
13
  body: { type: 'boolean', description: 'Include body.' },
14
14
  limit: { type: 'number', minimum: 1, description: 'Max results.' },
15
15
  depth: { type: 'number', minimum: 1, maximum: 5, description: 'Caller depth.' },
16
16
  page: { type: 'number', minimum: 1, description: 'Caller page.' },
17
17
  },
18
18
  required: ['mode'],
19
+ additionalProperties: false,
19
20
  },
20
21
  },
21
22
  ];
@@ -48,18 +48,22 @@ const EXPECTED_ARGS = Object.freeze(['-NoLogo', '-NoProfile', '-NonInteractive',
48
48
  // with the same `$?`→0/1 exit mapping pwsh -Command applies implicitly, then
49
49
  // emits `<<<MIXDOG_DONE_<nonce>:<code>>>>` on stdout and
50
50
  // `<<<MIXDOG_DONE_<nonce>>>>` on stderr (both streams provably drained).
51
- // Each command pipes through Microsoft.PowerShell.Core\Out-Default: the whole
52
- // marker loop is ONE top-level pipeline, so without an explicit per-command
53
- // formatting terminator Format-Table kept object output buffered PAST the done
54
- // sentinel and captures came back empty (`Get-ChildItem | Select-Object` →
55
- // no output). The per-command Out-Default flushes formatted output before the
56
- // sentinel is written, restoring one-shot `pwsh -Command` parity.
51
+ // Each command renders through Out-String -Stream and writes lines straight
52
+ // to [Console]::Out: host-side Out-Default table rendering proved ASYNC in
53
+ // long-lived standbys — formatted object output could flush AFTER the done
54
+ // sentinel (whole tables lost from the capture, then surfacing as stale
55
+ // bytes at the head of the NEXT command's capture). Rendering inside the
56
+ // pipeline and emitting via the SAME Console writer as the sentinel makes
57
+ // the ordering structural: every success-stream byte is on stdout before the
58
+ // sentinel line. Verified empirically: table shape/blank lines, long lines
59
+ // (no 120-col wrap), Write-Error rendering, and the `$?` exit mapping all
60
+ // match one-shot `pwsh -Command`.
57
61
  // stdin EOF exits the process with the LAST command's code — that is exactly
58
62
  // what background-promotion relies on for its exit-file. A user script that
59
63
  // calls `exit N` terminates the whole process with N (identical code to the
60
64
  // one-shot path); the caller settles via 'close' and the entry is discarded.
61
65
  function _standbyBootScript(nonce) {
62
- const runSentinel = `<<<MIXDOG_RUN_${nonce}>>>`;
66
+ const runPrefix = `<<<MIXDOG_RUN_${nonce}:`;
63
67
  const donePrefix = `<<<MIXDOG_DONE_${nonce}`;
64
68
  return [
65
69
  '$null = & {0}',
@@ -86,10 +90,15 @@ function _standbyBootScript(nonce) {
86
90
  '$__code = 0',
87
91
  'while ($true) {',
88
92
  ' $__buf = New-Object System.Text.StringBuilder',
93
+ " $__tok = ''",
89
94
  ' while ($true) {',
90
95
  ' $__line = $__sr.ReadLine()',
91
96
  ' if ($null -eq $__line) { exit $__code }',
92
- ` if ($__line -eq '${runSentinel}') { break }`,
97
+ // Per-COMMAND token: the RUN line carries a fresh token minted at
98
+ // take() time; DONE markers echo it back. A stale DONE from an
99
+ // earlier command on this entry can therefore NEVER match the
100
+ // current command's marker prefix (same-nonce offset hardening).
101
+ ` if ($__line.StartsWith('${runPrefix}') -and $__line.EndsWith('>>>')) { $__tok = $__line.Substring(${runPrefix.length}, $__line.Length - ${runPrefix.length} - 3); break }`,
93
102
  ' [void]$__buf.AppendLine($__line)',
94
103
  ' }',
95
104
  ' foreach ($__n in @(Get-ChildItem env: | ForEach-Object Name)) { if (-not $__baseEnv.ContainsKey($__n)) { Remove-Item ("env:$__n") -ErrorAction SilentlyContinue } }',
@@ -108,7 +117,7 @@ function _standbyBootScript(nonce) {
108
117
  ' $global:__MIXDOG_RC = 1',
109
118
  ' try {',
110
119
  " $__sc = [ScriptBlock]::Create($__buf.ToString() + [Environment]::NewLine + 'if ($?) { $global:__MIXDOG_RC = 0 } else { $global:__MIXDOG_RC = 1 }')",
111
- ' & $__sc | Microsoft.PowerShell.Core\\Out-Default',
120
+ ' & $__sc | Microsoft.PowerShell.Utility\\Out-String -Stream | Microsoft.PowerShell.Core\\ForEach-Object { [Console]::Out.WriteLine($_) }',
112
121
  ' } catch {',
113
122
  ' $global:__MIXDOG_RC = 1',
114
123
  " [Console]::Error.WriteLine('Exception: ' + $_.Exception.Message)",
@@ -116,8 +125,8 @@ function _standbyBootScript(nonce) {
116
125
  ' $__code = $global:__MIXDOG_RC',
117
126
  ' [Console]::Out.Flush()',
118
127
  ' [Console]::Error.Flush()',
119
- ` [Console]::Out.WriteLine('${donePrefix}:' + $__code + '>>>')`,
120
- ` [Console]::Error.WriteLine('${donePrefix}>>>')`,
128
+ ` [Console]::Out.WriteLine('${donePrefix}:' + $__tok + ':' + $__code + '>>>')`,
129
+ ` [Console]::Error.WriteLine('${donePrefix}:' + $__tok + '>>>')`,
121
130
  '}',
122
131
  ].join('\n');
123
132
  }
@@ -227,11 +236,19 @@ export function takePwshStandby({ shell, shellArgs, env }) {
227
236
  entry.child.stderr?.ref?.();
228
237
  } catch { /* best-effort */ }
229
238
  _topUp(shell, env, sig);
239
+ // Fresh token per command: DONE markers are only valid when they echo
240
+ // this token, so any straggler marker from a previous command on the
241
+ // same entry (same nonce) is inert for this capture.
242
+ const runToken = randomBytes(6).toString('hex');
230
243
  return {
231
244
  child: entry.child,
232
245
  nonce: entry.nonce,
233
- runSentinel: `<<<MIXDOG_RUN_${entry.nonce}>>>`,
234
- doneMarkerPrefix: `<<<MIXDOG_DONE_${entry.nonce}`,
246
+ runSentinel: `<<<MIXDOG_RUN_${entry.nonce}:${runToken}>>>`,
247
+ doneMarkerPrefix: `<<<MIXDOG_DONE_${entry.nonce}:${runToken}`,
248
+ // Entry-scoped prefix: lines matching this but NOT doneMarkerPrefix
249
+ // are stale markers from an earlier command — the caller drops them
250
+ // from the capture instead of surfacing sentinel noise.
251
+ staleMarkerPrefix: `<<<MIXDOG_DONE_${entry.nonce}`,
235
252
  run(spawnCommand, cwdForRun) {
236
253
  const esc = String(cwdForRun || '').replace(/'/g, "''");
237
254
  const prelude = esc
@@ -240,7 +257,7 @@ export function takePwshStandby({ shell, shellArgs, env }) {
240
257
  try {
241
258
  const body = prelude + String(spawnCommand ?? '');
242
259
  entry.child.stdin.write(body.endsWith('\n') ? body : `${body}\n`, 'utf8');
243
- entry.child.stdin.write(`<<<MIXDOG_RUN_${entry.nonce}>>>\n`, 'utf8');
260
+ entry.child.stdin.write(`<<<MIXDOG_RUN_${entry.nonce}:${runToken}>>>\n`, 'utf8');
244
261
  return true;
245
262
  } catch {
246
263
  // Feed failed (child died in the take→run window): kill so the
@@ -276,6 +293,22 @@ export function takePwshStandby({ shell, shellArgs, env }) {
276
293
  _killEntry(entry);
277
294
  return false;
278
295
  }
296
+ // Drain bytes a previous command left in the paused streams (late
297
+ // flushes that raced past the caller's detach). Without this they
298
+ // would surface at the HEAD of the next command's capture.
299
+ // Diagnostic (2026-08-04, intermittent empty-capture reports in a
300
+ // long-lived host): any stale byte seen here PROVES output escaped
301
+ // a finished command's capture on this entry — the exact precursor
302
+ // of a same-nonce marker offset corrupting the NEXT command. Rare
303
+ // by construction, so the log is always-on and one line.
304
+ let _staleBytes = 0;
305
+ try { let b; while ((b = c.stdout.read()) !== null) { _staleBytes += b.length; } } catch { /* best-effort */ }
306
+ try { let b; while ((b = c.stderr.read()) !== null) { _staleBytes += b.length; } } catch { /* best-effort */ }
307
+ if (_staleBytes > 0) {
308
+ try {
309
+ process.stderr.write(`[pwsh-standby] recycle drained ${_staleBytes}B stale output (pid=${c.pid}) — late flush escaped the previous command's capture\n`);
310
+ } catch { /* diagnostics only */ }
311
+ }
279
312
  entry.taken = false;
280
313
  entry.consumed = false;
281
314
  entry.spawnedAt = Date.now() - STANDBY_READY_AGE_MS;
@@ -208,6 +208,70 @@ function findContextTolerantWindow(sourceLines, oldLines, oldTags) {
208
208
  return windows.length === 1 ? { start: windows[0] } : null;
209
209
  }
210
210
 
211
+ // Last-resort recovery for a model that accidentally copied 1-2 unrelated
212
+ // unchanged context lines just outside the real edit. Trim only contiguous
213
+ // outer ' ' lines — never '-' or '+' lines — and accept the shortened old
214
+ // block only when it matches byte-exactly at exactly one place in the ENTIRE
215
+ // file. Requiring a deletion keeps this fallback off insertion-only hunks;
216
+ // exact whole-block matching proves every deletion line is current. Any
217
+ // competing trim plan or duplicate occurrence is ambiguous and stays a hard
218
+ // context miss.
219
+ function findOuterContextTrimmedWindow(sourceLines, hunk, stats) {
220
+ const body = (hunk?.lines || []).filter(
221
+ (line) => typeof line === 'string'
222
+ && line.length > 0
223
+ && !isV4AEndOfFileMarker(line)
224
+ && (line[0] === ' ' || line[0] === '-' || line[0] === '+'),
225
+ );
226
+ if (!body.some((line) => line[0] === '-')) return null;
227
+
228
+ let leadingAvailable = 0;
229
+ while (leadingAvailable < body.length && body[leadingAvailable][0] === ' ') leadingAvailable++;
230
+ let trailingAvailable = 0;
231
+ while (
232
+ trailingAvailable < body.length - leadingAvailable
233
+ && body[body.length - 1 - trailingAvailable][0] === ' '
234
+ ) trailingAvailable++;
235
+ if (leadingAvailable === 0 && trailingAvailable === 0) return null;
236
+
237
+ for (let totalTrimmed = 1; totalTrimmed <= 2; totalTrimmed++) {
238
+ const plans = [];
239
+ let ambiguous = false;
240
+ for (let leading = 0; leading <= Math.min(totalTrimmed, leadingAvailable); leading++) {
241
+ const trailing = totalTrimmed - leading;
242
+ if (trailing > trailingAvailable) continue;
243
+ const remainingBody = body.slice(leading, body.length - trailing);
244
+ if (!remainingBody.some((line) => line[0] === '-')) continue;
245
+ const oldLines = stats.oldLines.slice(leading, stats.oldLines.length - trailing);
246
+ const newLines = stats.newLines.slice(leading, stats.newLines.length - trailing);
247
+ if (oldLines.length === 0) continue;
248
+
249
+ const starts = [];
250
+ outer: for (let i = 0; i + oldLines.length <= sourceLines.length; i++) {
251
+ for (let k = 0; k < oldLines.length; k++) {
252
+ if (sourceLines[i + k] !== oldLines[k]) continue outer;
253
+ }
254
+ starts.push(i);
255
+ if (starts.length > 1) break;
256
+ }
257
+ if (starts.length > 1) {
258
+ ambiguous = true;
259
+ } else if (starts.length === 1) {
260
+ plans.push({
261
+ start: starts[0],
262
+ oldLines,
263
+ newLines,
264
+ leading,
265
+ trailing,
266
+ });
267
+ }
268
+ }
269
+ if (ambiguous || plans.length > 1) return null;
270
+ if (plans.length === 1) return plans[0];
271
+ }
272
+ return null;
273
+ }
274
+
211
275
  function resolveV4AHunkPosition(sourceLines, hunk, nextSearchLine, options = {}) {
212
276
  const stats = v4AHunkLineStats(hunk);
213
277
  if (stats.oldCount === 0 && stats.newCount === 0) return { skip: true };
@@ -223,6 +287,8 @@ function resolveV4AHunkPosition(sourceLines, hunk, nextSearchLine, options = {})
223
287
  let oldStartIdx;
224
288
  let trimmedTrailing = 0;
225
289
  let trimmedTrailingNew = 0;
290
+ let trimmedLeadingContext = 0;
291
+ let trimmedTrailingContext = 0;
226
292
  if (stats.oldCount === 0) {
227
293
  oldStartIdx = eofInsertionIndex(sourceLines);
228
294
  } else {
@@ -313,6 +379,20 @@ function resolveV4AHunkPosition(sourceLines, hunk, nextSearchLine, options = {})
313
379
  }
314
380
  }
315
381
  }
382
+ // A stray copied line outside the edit should not cost a failed tool turn.
383
+ // This fallback is deliberately narrower than the context-tolerance tier:
384
+ // fuzzy mode only, non-EOF, no newline-sentinel interaction, deletion hunks
385
+ // only, byte-exact remaining block, and one unique whole-file location.
386
+ if (oldStartIdx < 0 && fuzzy && !eof && trimmedTrailing === 0) {
387
+ const trimmed = findOuterContextTrimmedWindow(sourceLines, hunk, stats);
388
+ if (trimmed) {
389
+ oldStartIdx = trimmed.start;
390
+ oldLinesPattern = trimmed.oldLines;
391
+ newLinesPattern = trimmed.newLines;
392
+ trimmedLeadingContext = trimmed.leading;
393
+ trimmedTrailingContext = trimmed.trailing;
394
+ }
395
+ }
316
396
  if (oldStartIdx < 0) {
317
397
  const msg = `V4A hunk context not found: ${formatV4AHunkLocator(hunk)};${formatV4AContextMissHint(sourceLines, stats, anchorLine)}`;
318
398
  return { error: msg };
@@ -329,6 +409,8 @@ function resolveV4AHunkPosition(sourceLines, hunk, nextSearchLine, options = {})
329
409
  nextSearchLine: matchLen === 0 ? Math.max(0, nextSearchLine || 0) : oldStartIdx + matchLen,
330
410
  trimmedTrailing,
331
411
  trimmedTrailingNew,
412
+ trimmedLeadingContext,
413
+ trimmedTrailingContext,
332
414
  };
333
415
  }
334
416
 
@@ -764,6 +846,23 @@ export async function convertV4ASectionsToUnifiedPatch(sections, basePath, optio
764
846
  if (dropOldAt >= 0 && (!loc.trimmedTrailingNew || dropNewAt >= 0)) break;
765
847
  }
766
848
  }
849
+ const droppedOuterContext = new Set();
850
+ let leadingToDrop = loc.trimmedLeadingContext || 0;
851
+ for (let i = 0; i < hunk.lines.length && leadingToDrop > 0; i++) {
852
+ const line = hunk.lines[i];
853
+ if (isV4AEndOfFileMarker(line)) continue;
854
+ if (!line || line[0] !== ' ') break;
855
+ droppedOuterContext.add(i);
856
+ leadingToDrop--;
857
+ }
858
+ let trailingToDrop = loc.trimmedTrailingContext || 0;
859
+ for (let i = hunk.lines.length - 1; i >= 0 && trailingToDrop > 0; i--) {
860
+ const line = hunk.lines[i];
861
+ if (isV4AEndOfFileMarker(line)) continue;
862
+ if (!line || line[0] !== ' ') break;
863
+ droppedOuterContext.add(i);
864
+ trailingToDrop--;
865
+ }
767
866
  // Emit the body FIRST and derive the header counts from what was
768
867
  // actually emitted: the trailing-newline sentinel drop can remove a line
769
868
  // from one side only, so a header computed from the raw hunk stats can
@@ -775,6 +874,7 @@ export async function convertV4ASectionsToUnifiedPatch(sections, basePath, optio
775
874
  const srcEnd = loc.oldStartIdx + loc.matchLen;
776
875
  for (let i = 0; i < hunk.lines.length; i++) {
777
876
  const line = hunk.lines[i];
877
+ if (droppedOuterContext.has(i)) continue;
778
878
  if (isV4AEndOfFileMarker(line)) continue;
779
879
  const prefix = line[0];
780
880
  if (prefix === ' ' || prefix === '-') {
@@ -38,17 +38,16 @@ const APPLY_PATCH_FREEFORM_DESCRIPTION =
38
38
  // refs/codex/codex-rs/prompts/templates/apply_patch_tool_instructions.md,
39
39
  // adapted to the JSON `patch` argument.
40
40
  const APPLY_PATCH_JSON_DESCRIPTION = [
41
- 'Primary file-editing tool. Send `patch` as soon as the target and change are known.',
42
- 'Use this V4A envelope:',
41
+ 'Edit files with this V4A envelope:',
43
42
  '*** Begin Patch',
44
43
  '[file sections]',
45
44
  '*** End Patch',
46
45
  'Every section starts with exactly one header:',
47
- '- *** Add File: <path> then one or more +content lines.',
46
+ '- *** Add File: <path> then +content lines.',
48
47
  '- *** Delete File: <path> with nothing after it.',
49
48
  '- *** Update File: <path>, optionally followed by *** Move to: <new path>.',
50
- 'Updates contain hunks introduced by @@ or @@ <enclosing class/function>. Prefix every hunk line with space (context), - (remove), or + (add); a hunk at end of file may close with *** End of File.',
51
- 'Copy 3 context lines above and below verbatim from the newest tool output of the file (never retype from memory; after your own patch, use post-patch content). Do not duplicate overlapping context. If context is not unique, add one or more enclosing @@ headers.',
49
+ 'Update hunks open with @@ or @@ <enclosing class/function>; prefix every line with space (context), - (remove), or + (add); a hunk at end of file may close with *** End of File.',
50
+ 'Copy 3 context lines above and below verbatim from the newest tool output of the file (after your own patch, use post-patch content; never retype from memory). Do not duplicate overlapping context; if context is not unique, add enclosing @@ headers.',
52
51
  'Use project-relative paths. Every added file line needs +. Never submit a compacted-history marker; re-read and create a fresh patch.',
53
52
  ].join('\n');
54
53
 
@@ -68,8 +67,10 @@ export const PATCH_TOOL_DEFS = [
68
67
  type: 'object',
69
68
  properties: {
70
69
  patch: { type: 'string', description: 'The V4A patch text to apply (format and context rules in the tool description).' },
70
+ post_shell: { type: 'string', description: 'Verification command run when the patch applies; skipped on patch failure.' },
71
71
  },
72
72
  required: ['patch'],
73
+ additionalProperties: false,
73
74
  },
74
75
  },
75
76
  ];
@@ -492,6 +492,7 @@ export function execShellCommand({
492
492
  let _stderrData = (chunk) => taskOutput.writeStderr(chunk);
493
493
  if (_standby) {
494
494
  const donePrefix = _standby.doneMarkerPrefix;
495
+ const stalePrefix = _standby.staleMarkerPrefix;
495
496
  const _mk = { outDone: false, errDone: false, code: 1 };
496
497
  const tryComplete = () => {
497
498
  if (!_mk.outDone || !_mk.errDone) return;
@@ -515,6 +516,14 @@ export function execShellCommand({
515
516
  const line = tail.slice(0, idx + 1);
516
517
  tail = tail.slice(idx + 1);
517
518
  if (line.startsWith(donePrefix)) { onMarkerLine(line); continue; }
519
+ // Same-nonce marker with a DIFFERENT run token: a straggler
520
+ // from an earlier command on this standby entry. Drop it from
521
+ // the capture and log — this is the observable trace of the
522
+ // late-flush/offset condition the per-command token defuses.
523
+ if (stalePrefix && line.startsWith(stalePrefix)) {
524
+ try { process.stderr.write(`[pwsh-standby] dropped stale marker from a previous command (pid=${child.pid})\n`); } catch { /* diagnostics only */ }
525
+ continue;
526
+ }
518
527
  out += line;
519
528
  }
520
529
  if (out) write(out);