mixdog 0.9.91 → 0.9.93

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +147 -51
  2. package/package.json +6 -5
  3. package/scripts/code-graph-description-contract.mjs +6 -8
  4. package/scripts/tmp-cdp-errors.mjs +41 -0
  5. package/scripts/tmp-cdp-inspect.mjs +41 -0
  6. package/scripts/tool-overhead-microbench.mjs +60 -0
  7. package/scripts/tui-transcript-jitter-harness.mjs +2 -18
  8. package/src/agents/debugger/agent.json +1 -1
  9. package/src/agents/explore/agent.json +1 -1
  10. package/src/agents/heavy-worker/agent.json +1 -1
  11. package/src/agents/maintainer/agent.json +1 -1
  12. package/src/agents/reviewer/agent.json +1 -1
  13. package/src/agents/worker/agent.json +1 -1
  14. package/src/lib/rules-builder.cjs +5 -5
  15. package/src/output-styles/simple.md +1 -1
  16. package/src/rules/agent/00-core.md +1 -2
  17. package/src/rules/lead/01-general.md +1 -0
  18. package/src/rules/shared/01-tool.md +19 -21
  19. package/src/runtime/agent/orchestrator/agent-runtime/cache-strategy.mjs +16 -3
  20. package/src/runtime/agent/orchestrator/agent-trace.mjs +17 -0
  21. package/src/runtime/agent/orchestrator/context/collect.mjs +2 -1
  22. package/src/runtime/agent/orchestrator/providers/anthropic-effort.mjs +9 -1
  23. package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +90 -22
  24. package/src/runtime/agent/orchestrator/providers/anthropic.mjs +54 -19
  25. package/src/runtime/agent/orchestrator/providers/lib/anthropic-request-utils.mjs +18 -1
  26. package/src/runtime/agent/orchestrator/providers/openai-oauth-http-sse.mjs +44 -8
  27. package/src/runtime/agent/orchestrator/providers/openai-oauth-ws.mjs +10 -0
  28. package/src/runtime/agent/orchestrator/providers/openai-ws-events.mjs +3 -0
  29. package/src/runtime/agent/orchestrator/providers/openai-ws-stream.mjs +2 -0
  30. package/src/runtime/agent/orchestrator/providers/retry-classifier.mjs +35 -0
  31. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +34 -5
  32. package/src/runtime/agent/orchestrator/session/eager-dispatch.mjs +35 -29
  33. package/src/runtime/agent/orchestrator/session/loop/stop-hooks.mjs +9 -0
  34. package/src/runtime/agent/orchestrator/session/loop/stored-tool-args.mjs +11 -2
  35. package/src/runtime/agent/orchestrator/session/loop/tool-classify.mjs +5 -6
  36. package/src/runtime/agent/orchestrator/session/loop/tool-exec.mjs +31 -2
  37. package/src/runtime/agent/orchestrator/session/manager/compaction-runner.mjs +60 -0
  38. package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +60 -31
  39. package/src/runtime/agent/orchestrator/session/manager.mjs +1 -1
  40. package/src/runtime/agent/orchestrator/session/result-classification.mjs +28 -0
  41. package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +71 -2
  42. package/src/runtime/agent/orchestrator/session/store/listing.mjs +17 -0
  43. package/src/runtime/agent/orchestrator/session/store-summary-reader.mjs +101 -0
  44. package/src/runtime/agent/orchestrator/session/store.mjs +30 -0
  45. package/src/runtime/agent/orchestrator/session/tool-batch.mjs +19 -23
  46. package/src/runtime/agent/orchestrator/stall-policy.mjs +31 -21
  47. package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +4 -1
  48. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +12 -6
  49. package/src/runtime/agent/orchestrator/tools/builtin/task-tool.mjs +5 -0
  50. package/src/runtime/agent/orchestrator/tools/code-graph-tool-defs.mjs +4 -3
  51. package/src/runtime/agent/orchestrator/tools/lib/pwsh-standby-pool.mjs +47 -14
  52. package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +100 -0
  53. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +6 -5
  54. package/src/runtime/agent/orchestrator/tools/shell-command.mjs +9 -0
  55. package/src/runtime/agent/orchestrator/tools/shell-state.mjs +32 -2
  56. package/src/runtime/channels/backends/discord-gateway.mjs +6 -32
  57. package/src/runtime/channels/lib/inbound-handler.mjs +19 -3
  58. package/src/runtime/channels/lib/scheduler.mjs +51 -3
  59. package/src/runtime/channels/lib/worker-main.mjs +4 -0
  60. package/src/runtime/channels/tool-defs.mjs +4 -2
  61. package/src/runtime/memory/lib/query-handlers.mjs +11 -3
  62. package/src/runtime/memory/tool-defs.mjs +5 -5
  63. package/src/runtime/shared/channel-notification-routing.mjs +8 -2
  64. package/src/runtime/shared/llm/http-agent.mjs +11 -0
  65. package/src/session-runtime/lifecycle-api.mjs +26 -1
  66. package/src/session-runtime/output-styles.mjs +1 -4
  67. package/src/session-runtime/tool-catalog-data.mjs +5 -2
  68. package/src/session-runtime/workflow.mjs +25 -13
  69. package/src/standalone/agent-tool/tag-registry.mjs +5 -1
  70. package/src/standalone/explore-tool.mjs +1 -1
  71. package/src/tui/dist/index.mjs +100 -46
  72. package/src/tui/engine/agent-envelope.mjs +52 -3
  73. package/src/tui/engine/session-api.mjs +17 -0
  74. package/src/tui/engine/tui-steering-persist.mjs +24 -1
  75. package/src/tui/engine/turn.mjs +8 -9
  76. package/src/tui/engine.mjs +18 -41
  77. package/src/workflows/default/WORKFLOW.md +7 -18
  78. package/src/workflows/solo/WORKFLOW.md +0 -6
  79. package/src/workflows/solo-bench/WORKFLOW.md +17 -0
@@ -16,6 +16,7 @@
16
16
  import { randomBytes } from 'crypto';
17
17
  import { performance } from 'node:perf_hooks';
18
18
  import {
19
+ extractCacheWriteTokens,
19
20
  extractCachedTokens,
20
21
  appendAgentTrace,
21
22
  } from '../agent-trace.mjs';
@@ -1090,6 +1091,7 @@ export async function _streamResponse({
1090
1091
  inputTokens: u.input_tokens || 0,
1091
1092
  outputTokens: u.output_tokens || 0,
1092
1093
  cachedTokens: extractCachedTokens(u),
1094
+ cacheWriteTokens: extractCacheWriteTokens(u),
1093
1095
  // openai-oauth reports input_tokens as the total
1094
1096
  // prompt volume (cached portion is a subset, not
1095
1097
  // additive). Alias into the cross-provider
@@ -314,6 +314,41 @@ export function jitterDelayMs(ms, ratio = PROVIDER_RETRY_JITTER_RATIO, mode = 's
314
314
  return Math.max(0, Math.round(base + offset))
315
315
  }
316
316
 
317
+ // ── Stall-retry wall-clock budget (send-scoped) ──────────────────────────────
318
+ // Mid-stream 'stream_stalled' recoveries retry in place, which is right for a
319
+ // one-off blip but lets a chronically dying stream burn a whole task budget
320
+ // slowly (observed live: one send stretched 149s→298s→556s across stall
321
+ // retries before the agent deadline killed the task). Reference stacks bound
322
+ // this instead of retrying forever: Claude Code caps each request at ~300s
323
+ // wall clock (API_TIMEOUT_MS) and Codex kills a stream after one 300s silent
324
+ // gap (stream_idle_timeout). This guard is the equivalent for our in-place
325
+ // recovery: the clock starts at the FIRST stall of a send, and stall-classified
326
+ // retries are allowed only inside that window; past it the stall error
327
+ // surfaces so loop-level transport retry issues a FRESH request. Healthy
328
+ // streams never consult the clock (no stall → no budget reads), so long
329
+ // thinking/output can never trip it.
330
+ export const STREAM_STALL_RETRY_BUDGET_MS = (() => {
331
+ const v = Number(process.env.MIXDOG_STREAM_STALL_BUDGET_MS)
332
+ return Number.isFinite(v) && v > 0 ? Math.floor(v) : 300_000
333
+ })()
334
+
335
+ // One instance per provider send() call (NOT per attempt — the whole point is
336
+ // bounding the cross-attempt stall window). `now` is injectable for tests.
337
+ export function createStallRetryBudget(budgetMs = STREAM_STALL_RETRY_BUDGET_MS, now = Date.now) {
338
+ let firstStallAt = 0
339
+ return {
340
+ // Record a stall-classified retry candidate. Returns true while the
341
+ // send's stall window still has budget; false once exhausted (the caller
342
+ // surfaces the error instead of retrying in place).
343
+ allowStallRetry() {
344
+ const t = now()
345
+ if (!firstStallAt) firstStallAt = t
346
+ return (t - firstStallAt) <= budgetMs
347
+ },
348
+ get firstStallAt() { return firstStallAt },
349
+ }
350
+ }
351
+
317
352
  // ── Shared network-resilience interface ──────────────────────────────────────
318
353
  // One home for the logic shared across providers: mid-stream classifier
319
354
  // (WS + SSE), transport fallback predicate, stream-safety stamp latches,
@@ -227,7 +227,17 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
227
227
  // turn's push (deferBodies below) collapse to markers now — the model has
228
228
  // already seen them on that turn's follow-up send. Failed bodies stay
229
229
  // verbatim for retry.
230
- compactSettledToolCallBodies(messages);
230
+ // Out-of-loop transcript mutations (post-turn/manual compaction in
231
+ // manager/compaction-runner.mjs) run where no send opts exist; they park a
232
+ // one-shot intent on the session so the FIRST send of the next turn tags
233
+ // its expected cache break instead of an unexplained prefix mismatch.
234
+ if (!opts.cacheBreakIntent && typeof sessionRef?.pendingCacheBreakIntent === 'string') {
235
+ opts.cacheBreakIntent = sessionRef.pendingCacheBreakIntent;
236
+ delete sessionRef.pendingCacheBreakIntent;
237
+ }
238
+ if (compactSettledToolCallBodies(messages) && !opts.cacheBreakIntent) {
239
+ opts.cacheBreakIntent = 'deferred_body_compaction';
240
+ }
231
241
  // ---- Codex turn stop hook (refs/codex core/src/session/turn.rs:372-404) --
232
242
  // A no-tool assistant message is TERMINAL. Only a structured provider
233
243
  // follow-up signal (end_turn=false / pause_turn), pending input, tool
@@ -444,6 +454,7 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
444
454
  Math.floor(maxLoopIterations * 0.9),
445
455
  ];
446
456
  while (true) {
457
+ const _iterT0 = Date.now();
447
458
  throwIfAborted();
448
459
  if (iterations >= maxLoopIterations) {
449
460
  // Final-answer turn: instead of breaking mid-transcript (which
@@ -525,7 +536,10 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
525
536
  const _m = messages[_i];
526
537
  if (_m && _m.role === 'tool' && typeof _m.content === 'string' && _m.content.includes('⚠')) {
527
538
  const _stripped = stripSoftWarns(_m.content);
528
- if (_stripped !== _m.content) _m.content = _stripped;
539
+ if (_stripped !== _m.content) {
540
+ _m.content = _stripped;
541
+ if (!opts.cacheBreakIntent) opts.cacheBreakIntent = 'soft_warn_strip';
542
+ }
529
543
  }
530
544
  }
531
545
  sendTools = snapshotProviderRequestTools({
@@ -642,15 +656,18 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
642
656
  transportRetriesUsed: _transportRetriesUsed, signal,
643
657
  }),
644
658
  );
659
+ const _sendEndedAt = Date.now();
645
660
  if (_sendResult.action === 'retry') {
646
- delete opts.cacheBreakIntent;
661
+ // Keep opts.cacheBreakIntent: the failed send never consumed the
662
+ // tag, and the reactive-compact retry that follows IS the tagged
663
+ // transition — deleting it here made retry-side cache_break rows
664
+ // log intentional_transition: null.
647
665
  contextOverflowRetryUsed = true;
648
666
  reactiveOverflowRetryPending = true;
649
667
  continue;
650
668
  }
651
669
  if (_sendResult.action === 'retry_transport') {
652
670
  _transportRetriesUsed += 1;
653
- delete opts.cacheBreakIntent;
654
671
  continue;
655
672
  }
656
673
  response = _sendResult.response;
@@ -1019,7 +1036,9 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
1019
1036
  // Settle earlier deferred bodies before this turn's message lands:
1020
1037
  // every previous call already has its result row, so successful bodies
1021
1038
  // compact to markers while failed ones keep their full retry text.
1022
- compactSettledToolCallBodies(messages);
1039
+ if (compactSettledToolCallBodies(messages) && !opts.cacheBreakIntent) {
1040
+ opts.cacheBreakIntent = 'deferred_body_compaction';
1041
+ }
1023
1042
  messages.push(_assistantTurnMsg);
1024
1043
  try { opts.onAssistantMessageCommitted?.(_assistantTurnMsg); } catch {}
1025
1044
  const _callsToExecute = calls;
@@ -1038,6 +1057,7 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
1038
1057
  continue;
1039
1058
  }
1040
1059
  try { opts.onToolPhaseStarted?.(); } catch {}
1060
+ const _toolsT0 = Date.now();
1041
1061
  ({ dedupStubTotal: _dedupStubTotal, editCount: _editCount } = await processToolBatch({
1042
1062
  calls: _callsToExecute, messages, tools, cwd, sessionId, sessionRef, signal, opts,
1043
1063
  iterations, assistantTurnMsg: _assistantTurnMsg,
@@ -1050,6 +1070,15 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
1050
1070
  }));
1051
1071
  // Settle the stop hook on the batch that actually executed.
1052
1072
  _toolFailureStopHook.endBatch(_callsToExecute);
1073
+ // Loop-phase timing (diagnostics): where non-model time goes per
1074
+ // iteration — presend (repair/compact/snapshot), send (provider
1075
+ // round-trip incl. streaming), tools (batch execution). Gated by the
1076
+ // same env as [turn-timing] so bench runs opt in via -AgentEnv.
1077
+ if (process.env.MIXDOG_TURN_TIMING === '1') {
1078
+ try {
1079
+ process.stderr.write(`[loop-timing] iter=${nextIteration} presend=${sendStartedAt - _iterT0}ms send=${_sendEndedAt - sendStartedAt}ms tools=${Date.now() - _toolsT0}ms calls=${_callsToExecute.length}\n`);
1080
+ } catch { /* diagnostics only */ }
1081
+ }
1053
1082
  _toolBatchJustCompleted = true;
1054
1083
  _continuationsSinceToolBatch = 0;
1055
1084
  _lastToolBatchHadSleep = _callsToExecute.some(isSleepLikeToolCall);
@@ -1,18 +1,18 @@
1
1
  // Eager tool-dispatch controller, extracted from agent-loop.mjs. Owns the
2
2
  // per-turn pending promise map, the intra-turn in-flight signature set, and
3
- // the mutation epoch. FULL-PARALLEL policy: every tool call starts executing
4
- // the instant the provider streams its tool_use event (or at batch start),
5
- // shell/MCP/writes included — the model owns ordering by splitting dependent
6
- // work into separate turns. Only apply_patch (ordered mutation) waits for the
7
- // serial batch loop.
3
+ // the mutation epoch. Read-only calls may start while the provider is still
4
+ // streaming. Side-effect calls stream-start too until the first apply_patch
5
+ // appears in the stream (they are definitively in segment 0); after that they
6
+ // wait for their segment bounded by apply_patch barriers. Patches execute
7
+ // serially as barriers, so a failed patch can skip every later side effect.
8
8
  import { normalizeToolEnvelope } from './tool-envelope.mjs';
9
9
  import { isInvalidToolArgsMarker } from '../providers/openai-compat-stream.mjs';
10
- import { _intraTurnSig, _isMutationTool, _isReadTool, _isScopedCacheableTool, _isShellTool, _stripMcpPrefix } from './loop/tool-classify.mjs';
10
+ import { _intraTurnSig, _isMutationTool, _isOrderedGateSkippable, _isReadTool, _isScopedCacheableTool, _stripMcpPrefix } from './loop/tool-classify.mjs';
11
11
  import { tryReadCached, tryScopedToolCached } from './read-dedup.mjs';
12
12
  import { preDispatchDenyForSession } from './loop/pre-dispatch-deny.mjs';
13
13
  import { executeTool } from './loop/tool-exec.mjs';
14
14
  import { crossTurnSignature } from './loop/completion-guards.mjs';
15
- import { getToolKind, isParallelDispatchable, isToolCallDedupEligible } from './loop/tool-helpers.mjs';
15
+ import { getToolKind, isEagerDispatchable, isParallelDispatchable, isToolCallDedupEligible } from './loop/tool-helpers.mjs';
16
16
 
17
17
  export function createEagerDispatcher({
18
18
  tools, cwd, sessionId, sessionRef, signal, opts,
@@ -40,18 +40,11 @@ export function createEagerDispatcher({
40
40
  // resets at the turn boundary without leaking across getIterations().
41
41
  const _eagerInFlightSigs = new Map();
42
42
  const epoch = { mutation: 0 };
43
- // Patch→shell ordering insurance: a shell call that appears AFTER an
44
- // apply_patch in the same assistant turn must not eager-start before
45
- // that patch has executed (serial body runs both in call order).
46
- // Reads are already safe via the mutationEpoch re-execution gate;
47
- // only shell's side effects would consume pre-patch file state.
48
- let _streamSawMutation = false;
49
- const _hasEarlierMutation = (calls, index) => {
50
- for (let k = 0; k < index; k += 1) {
51
- if (_isMutationTool(calls[k]?.name)) return true;
52
- }
53
- return false;
54
- };
43
+ // True once an apply_patch tool_use has been seen in THIS turn's
44
+ // stream. Before that, every side-effect call streamed so far sits
45
+ // before the first patch in call order (segment 0), so it may start
46
+ // immediately — a patch that arrives later cannot gate an EARLIER call.
47
+ let _streamSeenMutation = false;
55
48
  const startEagerTool = (call) => {
56
49
  if (!call?.id || pending.has(call.id) || !isParallelDispatchable(call.name)) return null;
57
50
  // Never eager-execute a call whose arguments failed to parse
@@ -107,7 +100,7 @@ export function createEagerDispatcher({
107
100
  if (_dedupEligible) _eagerInFlightSigs.set(_sig, call.id);
108
101
  entry.promise = (async () => {
109
102
  try {
110
- return { ok: true, value: await executeToolFn(call.name, call.arguments, cwd, sessionId, sessionRef, { toolCallId: call.id, signal, notifyFn: opts.notifyFn, toolApprovalHook: opts.onToolApproval, iteration: getNextIteration() }) };
103
+ return { ok: true, value: await executeToolFn(call.name, call.arguments, cwd, sessionId, sessionRef, { toolCallId: call.id, signal, notifyFn: opts.notifyFn, toolApprovalHook: opts.onToolApproval, iteration: getNextIteration(), deferShellCwdCommit: true }) };
111
104
  } catch (error) {
112
105
  return { ok: false, error };
113
106
  }
@@ -168,16 +161,21 @@ export function createEagerDispatcher({
168
161
  return entry;
169
162
  };
170
163
  const startEagerRun = (calls, startIndex, dupSet) => {
164
+ const _nextOrderedMutationIndex = calls.findIndex(
165
+ (call, index) => index >= startIndex && _isMutationTool(call?.name),
166
+ );
171
167
  for (let j = startIndex; j < calls.length; j += 1) {
172
168
  const call = calls[j];
173
- // Full-parallel: only the ordered mutation (apply_patch) is
174
- // skipped — it executes in the serial batch body. No barrier:
175
- // later calls keep starting in parallel past it.
169
+ // Side effects may eager-start in the current segment but not
170
+ // across the next apply_patch barrier. Known read-only work
171
+ // may cross it and is protected by mutation-epoch re-execution.
176
172
  if (!call?.id || !isParallelDispatchable(call.name)) continue;
177
173
  if (dupSet && dupSet.has(call.id)) continue;
178
- // Patch→shell insurance: leave a shell that follows an
179
- // apply_patch to the serial body so it runs after the patch.
180
- if (_isShellTool(call.name) && _hasEarlierMutation(calls, j)) continue;
174
+ if (
175
+ _nextOrderedMutationIndex >= 0
176
+ && j >= _nextOrderedMutationIndex
177
+ && _isOrderedGateSkippable(call.name)
178
+ ) continue;
181
179
  // A null return here is NOT a state barrier. It means a
182
180
  // non-barrier stub — intra-turn in-flight dup, repeat-failure /
183
181
  // cross-turn dedup, pre-dispatch-deny, invalid-args, or a cache
@@ -188,9 +186,17 @@ export function createEagerDispatcher({
188
186
  }
189
187
  };
190
188
  const onToolCall = (call) => {
191
- if (_isMutationTool(call?.name)) { _streamSawMutation = true; return; }
192
- if (!isParallelDispatchable(call?.name)) return;
193
- if (_streamSawMutation && _isShellTool(call.name)) return;
189
+ if (_isMutationTool(call?.name)) {
190
+ _streamSeenMutation = true;
191
+ return;
192
+ }
193
+ // Declared read-only calls always overlap streaming (epoch guard
194
+ // re-executes them if a later patch lands). Side-effect calls may
195
+ // stream-start only while NO apply_patch has streamed yet: their
196
+ // call-order position is already fixed before the first barrier,
197
+ // matching exactly what startEagerRun would do post-batch. Once a
198
+ // patch has streamed, later side effects wait for their segment.
199
+ if (!isEagerDispatchable(call?.name, tools) && _streamSeenMutation) return;
194
200
  startEagerTool(call);
195
201
  };
196
202
  return { pending, epoch, startEagerTool, startEagerRun, onToolCall };
@@ -14,6 +14,8 @@
14
14
 
15
15
  export const STOP_HOOK_SOURCE = 'tool-failure-stop-hook';
16
16
 
17
+ import { isInformationalShellExitOne } from '../result-classification.mjs';
18
+
17
19
  // Only a genuinely EXECUTED result resolves a failure, i.e. kind 'normal'.
18
20
  // Cache hits ('cache-hit' / 'scoped-cache-hit') replay an earlier result
19
21
  // without running anything, and dedup/guard skips ('skipped') execute nothing
@@ -59,6 +61,13 @@ export function createToolFailureStopHook() {
59
61
  // nothing was dispatched — they must not arm the hook.
60
62
  if (message.guardSkip === true) return;
61
63
  if (message.toolKind === 'error') {
64
+ // Informational exit-1 probes (grep-family no-match inside a
65
+ // compound command: useful stdout, blank stderr) stay 'error'
66
+ // for display/history, but blocking the terminal message over
67
+ // them forces a pointless re-verify turn — observed live
68
+ // (kv-store-grpc: /proc PID scan exit 1 → hook misfire, +2
69
+ // turns). Neutral here: neither arms nor clears.
70
+ if (isInformationalShellExitOne(message.content)) return;
62
71
  batchFailure = true;
63
72
  if (message.toolCallId) failedCallIds.add(message.toolCallId);
64
73
  } else if (EXECUTED_SUCCESS_TOOL_KINDS.has(message.toolKind)) {
@@ -59,6 +59,9 @@ function compactStoredToolArgString(value, key = '', opts = {}) {
59
59
  const isLong = isBody || STORED_TOOL_ARG_LONG_KEY_RE.test(key);
60
60
  const limit = isLong ? STORED_TOOL_ARG_LIMIT : Infinity;
61
61
  if (value.length <= limit) return value;
62
+ // A marker is about to replace verbatim text — report the mutation so
63
+ // sweep callers can tag the resulting prefix-cache break.
64
+ try { opts.onCompacted?.(); } catch { /* observability only */ }
62
65
  const hash = createHash('sha256').update(value).digest('hex').slice(0, 16);
63
66
  // Body markers carry the recovery instruction inline: the compaction
64
67
  // detectors only require the `[mixdog compacted ...]` shape (no ']' or
@@ -115,14 +118,19 @@ export function compactToolCallsForHistory(calls, opts = {}) {
115
118
  // contract as restoreToolCallBodyForId);
116
119
  // - a call with no result row yet (current batch / interrupted turn) is
117
120
  // left untouched.
121
+ // Returns true when at least one body was actually collapsed to a marker
122
+ // (i.e. the transcript prefix changed), so callers can tag the intentional
123
+ // cache break instead of logging an unexplained input_prefix_mismatch.
118
124
  export function compactSettledToolCallBodies(messages) {
119
- if (!Array.isArray(messages)) return;
125
+ if (!Array.isArray(messages)) return false;
120
126
  const resultKinds = new Map();
121
127
  for (const message of messages) {
122
128
  if (message?.role === 'tool' && message.toolCallId) {
123
129
  resultKinds.set(message.toolCallId, message.toolKind || 'normal');
124
130
  }
125
131
  }
132
+ let changed = false;
133
+ const sweepOpts = { onCompacted: () => { changed = true; } };
126
134
  for (const message of messages) {
127
135
  if (message?.role !== 'assistant' || !Array.isArray(message.toolCalls)) continue;
128
136
  for (const call of message.toolCalls) {
@@ -130,9 +138,10 @@ export function compactSettledToolCallBodies(messages) {
130
138
  if (!call.arguments || typeof call.arguments !== 'object') continue;
131
139
  const kind = call.id ? resultKinds.get(call.id) : undefined;
132
140
  if (kind === undefined || kind === 'error') continue;
133
- call.arguments = compactStoredToolArgValue(call.arguments);
141
+ call.arguments = compactStoredToolArgValue(call.arguments, '', 0, sweepOpts);
134
142
  }
135
143
  }
144
+ return changed;
136
145
  }
137
146
 
138
147
  // Restore retry-safe long command/script text for ONE failed tool call inside a
@@ -15,12 +15,11 @@ export function _isMutationTool(name) {
15
15
  const n = _stripMcpPrefix(name);
16
16
  return n === 'apply_patch';
17
17
  }
18
- // Side-effect-free read-only tools that stay parallel even after an earlier
19
- // ordered mutation failed in the same batch. Everything NOT in this set is
20
- // treated as ordered-gate-skippable (see _isOrderedGateSkippable): apply_patch,
21
- // shell/bash_session, write/edit-style tools, and any (non-mixdog) MCP tool
22
- // whose effects are unknown. Kept separate from _isMutationTool, which stays
23
- // apply_patch-only for epoch-mutation counting and eager-dispatch gating.
18
+ // Side-effect-free read-only tools that may eager-start across an upcoming
19
+ // apply_patch barrier and may keep running after an earlier patch fails.
20
+ // Everything NOT in this set waits for its side-effect segment and is skipped
21
+ // after a failed patch. Kept separate from _isMutationTool, which stays
22
+ // apply_patch-only for epoch counting.
24
23
  const ORDERED_GATE_SAFE_READONLY_TOOLS = new Set([
25
24
  'read',
26
25
  'find',
@@ -226,8 +226,37 @@ export async function executeTool(name, args, cwd, callerSessionId, sessionRef,
226
226
  return result;
227
227
  }
228
228
  if (name === 'apply_patch') {
229
- const patchArgs = typeof args === 'string' ? { patch: args } : args;
230
- return executePatchTool(name, patchArgs, cwd, { sessionId: callerSessionId, toolCallId: executeOpts.toolCallId || null });
229
+ const patchArgs = typeof args === 'string' ? { patch: args } : { ...(args || {}) };
230
+ // post_shell: optional verification command executed through the normal
231
+ // one-shot shell path ONLY after the patch applies cleanly; a failed
232
+ // patch skips it. Patch text and shell output return as ONE tool
233
+ // result, so patch+verify costs a single call. Runtime-only knob:
234
+ // stripped before executePatchTool sees the args.
235
+ const postShell = typeof patchArgs.post_shell === 'string' && patchArgs.post_shell.trim()
236
+ ? patchArgs.post_shell.trim()
237
+ : null;
238
+ delete patchArgs.post_shell;
239
+ const patchResult = await executePatchTool(name, patchArgs, cwd, {
240
+ sessionId: callerSessionId,
241
+ toolCallId: executeOpts.toolCallId || null,
242
+ });
243
+ if (!postShell) return patchResult;
244
+ const patchNorm = normalizeToolEnvelope(patchResult);
245
+ const patchText = typeof patchNorm.result === 'string' ? patchNorm.result : String(patchNorm.result ?? '');
246
+ // Text-based failure detection only: legacy string returns normalize
247
+ // to explicitSuccess:false even on success, so that flag is unusable here.
248
+ const patchFailed = /^Error[\s:[]/.test(patchText.trimStart());
249
+ if (patchFailed) {
250
+ return `${patchText}\n--- post_shell skipped: patch failed ---`;
251
+ }
252
+ const shellRes = await executeBuiltinTool('shell', { command: postShell }, cwd, completionToolOpts);
253
+ const shellNorm = normalizeToolEnvelope(shellRes);
254
+ const shellText = typeof shellNorm.result === 'string' ? shellNorm.result : String(shellNorm.result ?? '');
255
+ const shellFailed = /^Error[\s:[]/.test(shellText.trimStart());
256
+ const header = shellFailed
257
+ ? '--- post_shell FAILED (patch is applied; fix and re-verify) ---'
258
+ : '--- post_shell ---';
259
+ return `${patchText}\n\n${header}\n${shellText}`;
231
260
  }
232
261
  if (isBuiltinTool(name)) {
233
262
  // clientHostPid threaded for the same per-terminal job-scope reason as
@@ -25,6 +25,7 @@ import {
25
25
  compactTypeForSession,
26
26
  } from './context-meta.mjs';
27
27
  import { resolveSemanticSummaryModel } from '../loop/compact-policy.mjs';
28
+ import { traceAgentCompact, messagePrefixHash } from '../../agent-trace.mjs';
28
29
  import { uncachedInputTokensForProvider } from './usage-metrics.mjs';
29
30
  import { pruneOffloadSession } from '../tool-result-offload.mjs';
30
31
  import { _getPendingMessagesForSession } from './pending-messages.mjs';
@@ -302,6 +303,7 @@ export async function runSessionCompaction(session, opts = {}) {
302
303
  semanticCompact: false,
303
304
  };
304
305
  const budget = targetBudgetTokens;
306
+ const compactStartedAt = Date.now();
305
307
  try { await opts.onStageChange?.('compacting'); } catch { /* best-effort */ }
306
308
  const provider = opts.provider || getProvider(session.provider) || null;
307
309
  let compacted;
@@ -486,6 +488,31 @@ export async function runSessionCompaction(session, opts = {}) {
486
488
  lastRecallFastTrackError: recallFastTrackError?.message || null,
487
489
  lastError: compactError?.message || semanticCompactError?.message || recallFastTrackError?.message || String(compactError || semanticCompactError || recallFastTrackError || 'compact failed'),
488
490
  };
491
+ // compact_meta parity with the loop's pre-send pass: the out-of-loop
492
+ // (post-turn/manual) compaction failure was previously invisible to
493
+ // trace analytics.
494
+ traceAgentCompact({
495
+ sessionId: opts.sessionId || session.id || null,
496
+ stage: mode === 'auto' ? 'post_turn' : 'manual',
497
+ trigger: mode,
498
+ compact_type: compactType,
499
+ compact_changed: false,
500
+ before_count: messages.length,
501
+ after_count: messages.length,
502
+ context_window: positiveContextWindow(session.contextWindow) || null,
503
+ budget_tokens: boundary,
504
+ boundary_tokens: boundary,
505
+ target_budget_tokens: budget,
506
+ reserve_tokens: reserveTokens,
507
+ pressure_tokens: pressureTokens,
508
+ trigger_tokens: triggerTokens,
509
+ message_tokens_est: beforeMessageTokens,
510
+ duration_ms: Date.now() - compactStartedAt,
511
+ provider: session.provider || null,
512
+ model: session.model || null,
513
+ error: session.compaction.lastError,
514
+ error_code: 'compact_failed',
515
+ });
489
516
  return {
490
517
  changed: false,
491
518
  error: session.compaction.lastError,
@@ -574,6 +601,39 @@ export async function runSessionCompaction(session, opts = {}) {
574
601
  compactCount: (session.compaction?.compactCount || 0) + (changed ? 1 : 0),
575
602
  };
576
603
  if (changed) invalidateProviderContextBaseline(session);
604
+ // Observability parity with the loop's pre-send pass: record the
605
+ // out-of-loop mutation as compact_meta and park a one-shot intent so the
606
+ // next turn's first send tags its cache break instead of logging an
607
+ // unexplained input_prefix_mismatch (observed live: a 403k→10k post-turn
608
+ // compact traced as intentional_transition: null with no compact_meta).
609
+ let beforePrefixHash = null;
610
+ try { beforePrefixHash = messagePrefixHash(messages); } catch { /* best-effort */ }
611
+ traceAgentCompact({
612
+ sessionId: pruneSessionId || null,
613
+ stage: mode === 'auto' ? 'post_turn' : 'manual',
614
+ trigger: mode,
615
+ compact_type: compactType,
616
+ compact_changed: changed,
617
+ input_prefix_hash: beforePrefixHash,
618
+ before_count: messages.length,
619
+ after_count: compacted.length,
620
+ before_bytes: beforeEncoded ? Buffer.byteLength(beforeEncoded, 'utf8') : null,
621
+ after_bytes: afterEncoded ? Buffer.byteLength(afterEncoded, 'utf8') : null,
622
+ context_window: positiveContextWindow(session.contextWindow) || null,
623
+ budget_tokens: boundary,
624
+ boundary_tokens: boundary,
625
+ target_budget_tokens: budget,
626
+ reserve_tokens: reserveTokens,
627
+ pressure_tokens: pressureTokens,
628
+ trigger_tokens: triggerTokens,
629
+ message_tokens_est: beforeMessageTokens,
630
+ duration_ms: Date.now() - compactStartedAt,
631
+ provider: session.provider || null,
632
+ model: session.model || null,
633
+ });
634
+ if (changed) {
635
+ session.pendingCacheBreakIntent = mode === 'auto' ? 'post_turn_compaction' : 'manual_compaction';
636
+ }
577
637
  return {
578
638
  changed,
579
639
  reason: unchangedReason,
@@ -19,16 +19,41 @@ const PENDING_MESSAGES_MODE = 0o600;
19
19
  const PENDING_ORPHAN_TTL_MS = 7 * 24 * 60 * 60 * 1000;
20
20
  const PENDING_ORPHAN_GRACE_MS = 60 * 60 * 1000;
21
21
  // Replay window for genuine user/steering entries. A cross-surface submit is
22
- // meant for a LIVE owner; if nothing picked it up within this window, firing
23
- // it into a session resumed hours or days later reads as a surprise
24
- // self-injection (user report: stale spool messages replayed on re-entry).
25
- // Crash+restart recovery within the window still replays normally.
22
+ // meant for a LIVE owner; entries that predate this process and exceeded the
23
+ // window are still DELIVERED (CC parity, reference messageQueueManager.ts:
24
+ // queued user input is never silently discarded), but annotated with an
25
+ // explicit late-delivery header so a session resumed hours later reads them
26
+ // as clearly-late input instead of a surprise self-injection.
26
27
  const STALE_USER_INJECTION_TTL_MS = 30 * 60 * 1000;
27
28
 
29
+ // Ownership epoch for the staleness gate below: entries enqueued while THIS
30
+ // process was already alive were aimed at a live owner that still exists.
31
+ const _PENDING_PROCESS_START_MS = Date.now();
32
+
28
33
  function isStaleUserInjection(entry, now = Date.now()) {
29
34
  if (isCompletionNotificationEntry(entry)) return false;
30
35
  const enqueuedAt = Number(entry?.enqueuedAt) || 0;
31
- return enqueuedAt > 0 && (now - enqueuedAt) > STALE_USER_INJECTION_TTL_MS;
36
+ if (enqueuedAt <= 0) return false;
37
+ // A submit that arrived AFTER this owner process booted is CURRENT input
38
+ // the owner was merely too busy to take yet (observed: remote sends
39
+ // silently discarded after 30m while the owner ground through long
40
+ // turns). It must deliver regardless of age. The stale window only
41
+ // guards entries that PREDATE this process — the resumed-session replay
42
+ // case the TTL was built for (surprise self-injection on re-entry).
43
+ if (enqueuedAt >= _PENDING_PROCESS_START_MS) return false;
44
+ return (now - enqueuedAt) > STALE_USER_INJECTION_TTL_MS;
45
+ }
46
+
47
+ // CC parity: never silently discard queued user input. A stale entry is
48
+ // delivered with this explicit age-annotated header so neither the user nor
49
+ // the model mistakes it for fresh input.
50
+ function lateDeliveryText(text, entry, now = Date.now()) {
51
+ const value = String(text ?? '');
52
+ if (!value.trim()) return value;
53
+ const enqueuedAt = Number(entry?.enqueuedAt) || 0;
54
+ const ageMinutes = Math.max(1, Math.round((now - enqueuedAt) / 60000));
55
+ const age = ageMinutes >= 120 ? `~${Math.round(ageMinutes / 60)}h` : `~${ageMinutes}m`;
56
+ return `[late delivery: queued ${age} ago, before the current session owner started]\n${value}`;
32
57
  }
33
58
  // Marker for deferred agent/tool *completion* notifications. Such entries must
34
59
  // never be replayed into a later turn on session resume (out-of-order delivery
@@ -765,7 +790,6 @@ export function hydratePendingMessages(sessionId, options = {}) {
765
790
  let hydrated = [];
766
791
  let alreadyDelivered = [];
767
792
  let staleLedgerEntries = [];
768
- const staleUserEntries = [];
769
793
  const ledgerSession = loadSession(sessionId);
770
794
  // Durable lifecycle epoch this hydration started under. Publishing claims the
771
795
  // durable entries into THIS process's memory, so it must be revalidated
@@ -797,16 +821,19 @@ export function hydratePendingMessages(sessionId, options = {}) {
797
821
  return false;
798
822
  }
799
823
  if (!id || inDelivery.has(id) || acked.has(id)) return false;
800
- // Stale genuine user/steering entries must not surprise-inject
801
- // into a session resumed long after they were queued (user:
802
- // "re-entering the session suddenly injected old messages").
803
- // Completion entries keep their own resume-drop policy.
804
- if (isStaleUserInjection(entry)) {
805
- staleUserEntries.push(entry);
806
- return false;
807
- }
808
824
  return true;
809
825
  });
826
+ // CC parity: stale genuine user/steering entries DELIVER with a
827
+ // late-delivery header instead of being silently dropped (the old
828
+ // behavior discarded them; user report: remote/steering sends
829
+ // silently ignored around owner restarts). Completion entries
830
+ // keep their own resume-drop policy (drain discards them).
831
+ for (const entry of hydrated) {
832
+ if (!isStaleUserInjection(entry)) continue;
833
+ if (typeof entry.message === 'string') {
834
+ entry.message = lateDeliveryText(entry.message, entry);
835
+ }
836
+ }
810
837
  // Read-only claim: durable data remains until successful delivery
811
838
  // acknowledges these exact ids. A crash here therefore redelivers.
812
839
  return undefined;
@@ -828,13 +855,6 @@ export function hydratePendingMessages(sessionId, options = {}) {
828
855
  runPendingTestHook('hydrate:betweenCleanups', { sessionId });
829
856
  if (pendingLifecycleInvalidated(sessionId, startToken)) return 0;
830
857
  }
831
- if (staleUserEntries.length > 0) {
832
- try { process.stderr.write(`[session] dropped ${staleUserEntries.length} stale queued message(s) (older than ${Math.round(STALE_USER_INJECTION_TTL_MS / 60000)}m) sessionId=${sessionId}\n`); } catch {}
833
- const cleaned = await acknowledgePendingMessages(sessionId, staleUserEntries, { expectedToken: startToken });
834
- if (cleaned) cleanupConfirmed.push(...staleUserEntries);
835
- runPendingTestHook('hydrate:betweenCleanups', { sessionId });
836
- if (pendingLifecycleInvalidated(sessionId, startToken)) return 0;
837
- }
838
858
  if (cleanupConfirmed.length > 0) {
839
859
  try {
840
860
  // One session save prunes both IDs whose replay spool was removed
@@ -968,7 +988,19 @@ function modelVisiblePendingMessages(messages) {
968
988
  }
969
989
 
970
990
  export function _mergePendingMessageEntries(entries) {
971
- const normalized = (Array.isArray(entries) ? entries : [])
991
+ // CC-parity delivery priority (reference messageQueueManager.ts: user
992
+ // input enqueues at 'next', task notifications at 'later', and dequeue
993
+ // always serves 'next' first). Our single merged turn message is the
994
+ // analogue of that dequeue order: genuine user/steering entries are
995
+ // merged BEFORE deferred completion notifications so queued user input
996
+ // is never buried under system notification text. FIFO is preserved
997
+ // within each group (stable partition).
998
+ const source = Array.isArray(entries) ? entries : [];
999
+ const ordered = [
1000
+ ...source.filter((entry) => !isCompletionNotificationEntry(entry)),
1001
+ ...source.filter(isCompletionNotificationEntry),
1002
+ ];
1003
+ const normalized = ordered
972
1004
  .map(normalizePendingMessageEntry)
973
1005
  .filter(Boolean);
974
1006
  if (normalized.length === 0) return null;
@@ -1106,7 +1138,6 @@ export function drainForeignUserInjections(sessionId) {
1106
1138
  }
1107
1139
  } catch { /* ledger unavailable — in-memory sets still guard */ }
1108
1140
  const taken = [];
1109
- let droppedStale = 0;
1110
1141
  // The mtime memo may ONLY be armed by a scan that actually reached a
1111
1142
  // lifecycle-valid decision under the spool lock. Arming it upfront meant a
1112
1143
  // drain refused in-lock (a close/detach/reopen landing in the window) still
@@ -1137,14 +1168,15 @@ export function drainForeignUserInjections(sessionId) {
1137
1168
  && !isCompletionNotificationEntry(entry)
1138
1169
  && !isLegacyUnmarkedCompletionNotification(text)
1139
1170
  && text && !isInternalRuntimeNotificationText(text);
1140
- // Stale foreign submits are removed WITHOUT injecting: they
1141
- // were aimed at a live owner that no longer exists (user:
1142
- // stale spool messages fired on session re-entry days later).
1143
- if (foreignUser && isStaleUserInjection(entry)) droppedStale += 1;
1171
+ // CC parity: stale foreign submits still deliver, carrying a
1172
+ // late-delivery header instead of being silently removed
1173
+ // (user report: remote sends silently discarded around owner
1174
+ // restarts).
1175
+ if (foreignUser && isStaleUserInjection(entry)) taken.push(lateDeliveryText(text, entry));
1144
1176
  else if (foreignUser) taken.push(text);
1145
1177
  else kept.push(entry);
1146
1178
  }
1147
- if (taken.length === 0 && droppedStale === 0) return undefined;
1179
+ if (taken.length === 0) return undefined;
1148
1180
  if (kept.length > 0) next.sessions[sessionId] = kept;
1149
1181
  else {
1150
1182
  delete next.sessions[sessionId];
@@ -1158,9 +1190,6 @@ export function drainForeignUserInjections(sessionId) {
1158
1190
  return [];
1159
1191
  }
1160
1192
  if (lifecycleDecided) _rememberForeignSpoolScan(sessionId, mtime);
1161
- if (droppedStale > 0) {
1162
- try { process.stderr.write(`[session] dropped ${droppedStale} stale foreign submit(s) (older than ${Math.round(STALE_USER_INJECTION_TTL_MS / 60000)}m) sessionId=${sessionId}\n`); } catch {}
1163
- }
1164
1193
  return taken;
1165
1194
  }
1166
1195