mixdog 0.9.92 → 0.9.93

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/README.md +147 -51
  2. package/package.json +6 -5
  3. package/scripts/code-graph-description-contract.mjs +6 -8
  4. package/scripts/tmp-cdp-errors.mjs +41 -0
  5. package/scripts/tmp-cdp-inspect.mjs +41 -0
  6. package/scripts/tui-transcript-jitter-harness.mjs +2 -18
  7. package/src/rules/agent/00-core.md +1 -2
  8. package/src/rules/lead/01-general.md +1 -0
  9. package/src/rules/shared/01-tool.md +19 -22
  10. package/src/runtime/agent/orchestrator/agent-runtime/cache-strategy.mjs +16 -3
  11. package/src/runtime/agent/orchestrator/agent-trace.mjs +17 -0
  12. package/src/runtime/agent/orchestrator/context/collect.mjs +2 -1
  13. package/src/runtime/agent/orchestrator/providers/anthropic-effort.mjs +9 -1
  14. package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +79 -21
  15. package/src/runtime/agent/orchestrator/providers/anthropic.mjs +40 -19
  16. package/src/runtime/agent/orchestrator/providers/lib/anthropic-request-utils.mjs +18 -1
  17. package/src/runtime/agent/orchestrator/providers/openai-oauth-http-sse.mjs +44 -8
  18. package/src/runtime/agent/orchestrator/providers/openai-ws-events.mjs +3 -0
  19. package/src/runtime/agent/orchestrator/providers/openai-ws-stream.mjs +2 -0
  20. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +22 -5
  21. package/src/runtime/agent/orchestrator/session/eager-dispatch.mjs +35 -29
  22. package/src/runtime/agent/orchestrator/session/loop/stored-tool-args.mjs +11 -2
  23. package/src/runtime/agent/orchestrator/session/loop/tool-classify.mjs +5 -6
  24. package/src/runtime/agent/orchestrator/session/loop/tool-exec.mjs +31 -2
  25. package/src/runtime/agent/orchestrator/session/manager/compaction-runner.mjs +60 -0
  26. package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +60 -31
  27. package/src/runtime/agent/orchestrator/session/manager.mjs +1 -1
  28. package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +12 -3
  29. package/src/runtime/agent/orchestrator/session/store/listing.mjs +17 -0
  30. package/src/runtime/agent/orchestrator/session/store-summary-reader.mjs +101 -0
  31. package/src/runtime/agent/orchestrator/session/store.mjs +30 -0
  32. package/src/runtime/agent/orchestrator/session/tool-batch.mjs +19 -23
  33. package/src/runtime/agent/orchestrator/stall-policy.mjs +31 -21
  34. package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +4 -1
  35. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +12 -6
  36. package/src/runtime/agent/orchestrator/tools/code-graph-tool-defs.mjs +4 -3
  37. package/src/runtime/agent/orchestrator/tools/lib/pwsh-standby-pool.mjs +47 -14
  38. package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +100 -0
  39. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +6 -5
  40. package/src/runtime/agent/orchestrator/tools/shell-command.mjs +9 -0
  41. package/src/runtime/agent/orchestrator/tools/shell-state.mjs +32 -2
  42. package/src/runtime/channels/backends/discord-gateway.mjs +6 -32
  43. package/src/runtime/channels/lib/inbound-handler.mjs +19 -3
  44. package/src/runtime/channels/lib/scheduler.mjs +51 -3
  45. package/src/runtime/channels/lib/worker-main.mjs +4 -0
  46. package/src/runtime/channels/tool-defs.mjs +4 -2
  47. package/src/runtime/memory/lib/query-handlers.mjs +11 -3
  48. package/src/runtime/memory/tool-defs.mjs +5 -5
  49. package/src/runtime/shared/channel-notification-routing.mjs +8 -2
  50. package/src/runtime/shared/llm/http-agent.mjs +11 -0
  51. package/src/session-runtime/lifecycle-api.mjs +26 -1
  52. package/src/session-runtime/tool-catalog-data.mjs +5 -2
  53. package/src/session-runtime/workflow.mjs +7 -5
  54. package/src/standalone/agent-tool/tag-registry.mjs +5 -1
  55. package/src/standalone/explore-tool.mjs +1 -1
  56. package/src/tui/dist/index.mjs +68 -45
  57. package/src/tui/engine/agent-envelope.mjs +52 -3
  58. package/src/tui/engine/turn.mjs +8 -9
  59. package/src/tui/engine.mjs +18 -41
  60. package/src/workflows/solo/WORKFLOW.md +0 -6
  61. package/src/workflows/solo-bench/WORKFLOW.md +17 -0
@@ -38,6 +38,22 @@ function extractCachedTokens(usage) {
38
38
  return 0;
39
39
  }
40
40
 
41
+ function extractCacheWriteTokens(usage) {
42
+ const candidates = [
43
+ usage?.input_tokens_details?.cache_write_tokens,
44
+ usage?.prompt_tokens_details?.cache_write_tokens,
45
+ usage?.inputTokensDetails?.cacheWriteTokens,
46
+ usage?.promptTokensDetails?.cacheWriteTokens,
47
+ usage?.cache_write_tokens,
48
+ usage?.cacheWriteTokens,
49
+ ];
50
+ for (const value of candidates) {
51
+ const n = Number(value);
52
+ if (Number.isFinite(n)) return n;
53
+ }
54
+ return 0;
55
+ }
56
+
41
57
  // Lightweight fingerprint of the conversation prefix. Hashes the first 4096
42
58
  // characters of JSON.stringify(messages) — enough to detect prefix mutation
43
59
  // across iterations (which invalidates the provider prompt cache) without
@@ -267,6 +283,7 @@ export {
267
283
  appendAgentTrace,
268
284
  drainAgentTrace,
269
285
  estimateProviderPayloadBytes,
286
+ extractCacheWriteTokens,
270
287
  extractCachedTokens,
271
288
  messagePrefixHash,
272
289
  traceAgentFetch,
@@ -275,7 +275,7 @@ export function buildSkillToolEnvelope(name, content, skillDir) {
275
275
  };
276
276
  }
277
277
 
278
- function compactSkillManifestText(value, max = 180) {
278
+ function compactSkillManifestText(value, max = 100) {
279
279
  const text = String(value || '').replace(/\s+/g, ' ').trim();
280
280
  return text.length > max ? `${text.slice(0, Math.max(1, max - 3))}...` : text;
281
281
  }
@@ -528,6 +528,7 @@ export function buildSkillToolDefs(skills, { ownerIsAgentSession = false } = {})
528
528
  name: { type: 'string', description: 'Skill name' },
529
529
  },
530
530
  required: ['name'],
531
+ additionalProperties: false,
531
532
  },
532
533
  },
533
534
  ];
@@ -208,7 +208,15 @@ export function applyAnthropicEffortToBody(
208
208
  // modelSupportsEffort() allowlist so older models never receive it.
209
209
  // Set unconditionally (independent of `normalized`) so effort-capable
210
210
  // turns always carry adaptive thinking + round-trip signatures.
211
- body.thinking = { type: 'adaptive', display: 'summarized' };
211
+ // MIXDOG_ANTHROPIC_THINKING_DISPLAY=omitted (operator/bench knob):
212
+ // CC-parity mode — no thinking blocks come back, so nothing is
213
+ // replayed into later requests (saves the 1h cache-write + re-read on
214
+ // accumulated thinking) at the cost of losing visible reasoning and
215
+ // cross-iteration thinking continuity. Default stays summarized.
216
+ const display = (process.env.MIXDOG_ANTHROPIC_THINKING_DISPLAY || '').trim() === 'omitted'
217
+ ? 'omitted'
218
+ : 'summarized';
219
+ body.thinking = { type: 'adaptive', display };
212
220
  // Adaptive/4.7+ models reject any non-default sampling param with a 400.
213
221
  delete body.temperature;
214
222
  delete body.top_p;
@@ -64,6 +64,18 @@ import {
64
64
  stampAnthropicStreamOutcome,
65
65
  } from './anthropic-sse.mjs';
66
66
  import { buildAnthropicBetaHeaders, supportsAnthropicFastMode } from './anthropic-betas.mjs';
67
+ import { gzipSync } from 'node:zlib';
68
+
69
+ // Request-body gzip gate (see the fetch site below). Env kill-switch
70
+ // (MIXDOG_ANTHROPIC_REQ_GZIP=0) plus a process-wide latch flipped on the
71
+ // first 400 response to a compressed request. Small bodies skip compression:
72
+ // below ~8KB the CPU + header cost outweighs the upload saving.
73
+ const ANTHROPIC_REQ_GZIP_MIN_BYTES = 8 * 1024;
74
+ let _anthropicReqGzipLatch = false;
75
+ function _anthropicReqGzipDisabled() {
76
+ return _anthropicReqGzipLatch || process.env.MIXDOG_ANTHROPIC_REQ_GZIP === '0';
77
+ }
78
+ function _disableAnthropicReqGzip() { _anthropicReqGzipLatch = true; }
67
79
  import {
68
80
  applyAnthropicEffortToBody,
69
81
  effortValuesForModel,
@@ -654,7 +666,16 @@ export class AnthropicOAuthProvider {
654
666
  // provider-visible cache breakpoint off the cached one — the
655
667
  // exact COLD-turn bug this change fixes. Order is fixed:
656
668
  // build → sanitize (once) → mark → JSON.stringify.
657
- const response = await fetch(API_URL, {
669
+ // Request-body gzip (probe-verified 2026-08-04: /v1/messages
670
+ // returns 200 for Content-Encoding: gzip, 400 for zstd). Large
671
+ // turn bodies (system prompt + history, typically 50-100KB+)
672
+ // compress ~5-10x, trimming upload time off every call's
673
+ // header wait. Latch OFF process-wide on the first 400 seen on
674
+ // a compressed request and retry that attempt uncompressed, so
675
+ // a server-side behavior change can never wedge the session.
676
+ const rawBody = Buffer.from(JSON.stringify(requestBody));
677
+ const useGzip = !_anthropicReqGzipDisabled() && rawBody.length >= ANTHROPIC_REQ_GZIP_MIN_BYTES;
678
+ const sendAttempt = (gz) => fetch(API_URL, {
658
679
  method: 'POST',
659
680
  headers: {
660
681
  'Authorization': `Bearer ${accessToken}`,
@@ -669,11 +690,18 @@ export class AnthropicOAuthProvider {
669
690
  'user-agent': `claude-cli/${resolveCliVersion()} (external, sdk-cli)`,
670
691
  'x-app': 'cli',
671
692
  'Content-Type': 'application/json',
693
+ ...(gz ? { 'Content-Encoding': 'gzip' } : {}),
672
694
  },
673
- body: JSON.stringify(requestBody),
695
+ body: gz ? gzipSync(rawBody) : rawBody,
674
696
  signal: controller.signal,
675
697
  dispatcher: getLlmDispatcher(),
676
698
  });
699
+ let response = await sendAttempt(useGzip);
700
+ if (useGzip && response.status === 400) {
701
+ _disableAnthropicReqGzip();
702
+ try { await response.arrayBuffer(); } catch { /* drain best-effort */ }
703
+ response = await sendAttempt(false);
704
+ }
677
705
 
678
706
  traceAgentFetch({
679
707
  sessionId,
@@ -780,25 +808,13 @@ export class AnthropicOAuthProvider {
780
808
  // clock starting at the first stall (see createStallRetryBudget).
781
809
  const stallRetryBudget = createStallRetryBudget();
782
810
 
783
- const recoverNonStreaming = async (midState, streamingError, controller) => {
784
- const exposedChars = Number(midState?.emittedTextChars) || 0;
785
- if (!onTextReset || exposedChars <= 0
786
- || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
787
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
788
- throw streamingError;
789
- }
790
- let resetAccepted = false;
791
- try {
792
- resetAccepted = await onTextReset({
793
- chars: exposedChars,
794
- reason: 'anthropic-streaming-fallback',
795
- }) === true;
796
- } catch {}
797
- if (!resetAccepted) {
798
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
799
- throw streamingError;
800
- }
801
- try { controller?.abort?.(streamingError); } catch {}
811
+ // Core non-streaming re-issue: abort the dead stream and repeat the
812
+ // SAME request with stream:false. Shared by the exposed-text recovery
813
+ // (which must first get the owner's onTextReset acknowledgement) and
814
+ // the no-exposure stall fallback below (trivially safe — nothing was
815
+ // relayed or dispatched, so there is nothing to withdraw or replay).
816
+ const issueNonStreamingFallback = async (controller, abortReason) => {
817
+ try { controller?.abort?.(abortReason); } catch {}
802
818
  try { onStageChange?.('requesting', { transport: 'non-streaming-fallback' }); } catch {}
803
819
  let fallback = await requestWithRetry(creds.accessToken, { ...body, stream: false });
804
820
  if (fallback.response.status === 401) {
@@ -825,6 +841,27 @@ export class AnthropicOAuthProvider {
825
841
  }
826
842
  };
827
843
 
844
+ const recoverNonStreaming = async (midState, streamingError, controller) => {
845
+ const exposedChars = Number(midState?.emittedTextChars) || 0;
846
+ if (!onTextReset || exposedChars <= 0
847
+ || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
848
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
849
+ throw streamingError;
850
+ }
851
+ let resetAccepted = false;
852
+ try {
853
+ resetAccepted = await onTextReset({
854
+ chars: exposedChars,
855
+ reason: 'anthropic-streaming-fallback',
856
+ }) === true;
857
+ } catch {}
858
+ if (!resetAccepted) {
859
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
860
+ throw streamingError;
861
+ }
862
+ return issueNonStreamingFallback(controller, streamingError);
863
+ };
864
+
828
865
  try {
829
866
  for (let attemptIndex = 0; attemptIndex <= MAX_MIDSTREAM_RETRIES; attemptIndex++) {
830
867
  let response, controller, cancelHandler;
@@ -1057,6 +1094,27 @@ export class AnthropicOAuthProvider {
1057
1094
  continue;
1058
1095
  }
1059
1096
  const classifier = _classifyMidstreamError(err, midState);
1097
+ // CC-parity stall recovery (2026-08-03 v3 postmortem): a
1098
+ // stalled stream that exposed NOTHING (no text/thinking
1099
+ // relayed, no tool emitted) is re-issued NON-STREAMING instead
1100
+ // of retrying the same streaming shape. Effort-mode models can
1101
+ // legitimately think in silence past any streaming idle
1102
+ // window; an in-place streaming retry re-runs the same silent
1103
+ // generation into the same timer (observed live: deterministic
1104
+ // 4×~138s beheading, ~552s per turn), while the non-streaming
1105
+ // transport simply waits for the full body (bounded by
1106
+ // PROVIDER_NONSTREAM_TOTAL_TIMEOUT_MS). Claude Code does
1107
+ // exactly this on its watchdog aborts. Replay is trivially
1108
+ // safe here — nothing was relayed or dispatched.
1109
+ if (classifier === 'stream_stalled'
1110
+ && _outcome?.replayUnsafe !== true
1111
+ && !midState.emittedText
1112
+ && !midState.emittedToolCall
1113
+ && !midState.partialToolCall
1114
+ && !midState.emittedThinking) {
1115
+ try { process.stderr.write('[anthropic-oauth] stream stalled with no exposure — retrying non-streaming\n'); } catch {}
1116
+ return await issueNonStreamingFallback(controller, err);
1117
+ }
1060
1118
  if (classifier === 'stream_stalled' && !stallRetryBudget.allowStallRetry()) {
1061
1119
  try { process.stderr.write(`[anthropic-oauth] stall retry budget exhausted (${STREAM_STALL_RETRY_BUDGET_MS}ms since first stall) — surfacing for fresh-request retry\n`); } catch {}
1062
1120
  try { controller?.abort?.(err); } catch { /* best-effort teardown */ }
@@ -320,25 +320,11 @@ export class AnthropicProvider {
320
320
  };
321
321
  };
322
322
 
323
- const recoverNonStreaming = async (midState, streamingError, streamController) => {
324
- const exposedChars = Number(midState?.emittedTextChars) || 0;
325
- if (!onTextReset || exposedChars <= 0
326
- || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
327
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
328
- throw streamingError;
329
- }
330
- let resetAccepted = false;
331
- try {
332
- resetAccepted = await onTextReset({
333
- chars: exposedChars,
334
- reason: 'anthropic-streaming-fallback',
335
- }) === true;
336
- } catch {}
337
- if (!resetAccepted) {
338
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
339
- throw streamingError;
340
- }
341
- try { streamController.abort?.(streamingError); } catch {}
323
+ // Core non-streaming re-issue shared by the exposed-text recovery
324
+ // (onTextReset-gated) and the no-exposure stall fallback (trivially
325
+ // safe — nothing was relayed or dispatched). Mirrors anthropic-oauth.
326
+ const issueNonStreamingFallback = async (streamController, abortReason) => {
327
+ try { streamController.abort?.(abortReason); } catch {}
342
328
  try { onStageChange?.('requesting', { transport: 'non-streaming-fallback' }); } catch {}
343
329
  const nonStreamingParams = { ...params, stream: false };
344
330
  const message = await withRetry(
@@ -362,6 +348,27 @@ export class AnthropicProvider {
362
348
  return buildReturnFromParse(normalizeAnthropicNonStreamingResponse(message, useModel));
363
349
  };
364
350
 
351
+ const recoverNonStreaming = async (midState, streamingError, streamController) => {
352
+ const exposedChars = Number(midState?.emittedTextChars) || 0;
353
+ if (!onTextReset || exposedChars <= 0
354
+ || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
355
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
356
+ throw streamingError;
357
+ }
358
+ let resetAccepted = false;
359
+ try {
360
+ resetAccepted = await onTextReset({
361
+ chars: exposedChars,
362
+ reason: 'anthropic-streaming-fallback',
363
+ }) === true;
364
+ } catch {}
365
+ if (!resetAccepted) {
366
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
367
+ throw streamingError;
368
+ }
369
+ return issueNonStreamingFallback(streamController, streamingError);
370
+ };
371
+
365
372
  try {
366
373
  for (let attemptIndex = 0; attemptIndex <= MAX_MIDSTREAM_RETRIES; attemptIndex++) {
367
374
  const streamController = createAbortController();
@@ -597,6 +604,20 @@ export class AnthropicProvider {
597
604
  continue;
598
605
  }
599
606
  const classifier = _classifyMidstreamError(err, midState);
607
+ // CC-parity stall recovery (ported from anthropic-oauth,
608
+ // 2026-08-03): a stalled stream that exposed NOTHING is
609
+ // re-issued non-streaming instead of retrying the same
610
+ // streaming shape into the same idle window. Replay is
611
+ // trivially safe — nothing was relayed or dispatched.
612
+ if (classifier === 'stream_stalled'
613
+ && _outcome?.replayUnsafe !== true
614
+ && !midState.emittedText
615
+ && !midState.emittedToolCall
616
+ && !midState.partialToolCall
617
+ && !midState.emittedThinking) {
618
+ try { process.stderr.write(`[${this.name}] stream stalled with no exposure — retrying non-streaming\n`); } catch {}
619
+ return await issueNonStreamingFallback(streamController, err);
620
+ }
600
621
  if (classifier === 'stream_stalled' && !stallRetryBudget.allowStallRetry()) {
601
622
  try {
602
623
  process.stderr.write(
@@ -268,10 +268,27 @@ export function applyAnthropicCacheMarkers(sanitizedMessages, {
268
268
  const tailIdx = sanitizedMessages.length - 1;
269
269
  return hasUserText(sanitizedMessages[tailIdx]) ? tailIdx : -1;
270
270
  };
271
+ // True-tip anchor: when the request ends with a PERSISTED user text turn
272
+ // (the current prompt — it re-appears verbatim in every later request's
273
+ // prefix), mark it first. Without this, mid-session turn-first requests
274
+ // (multi-turn only; single-turn sessions are already covered by
275
+ // firstRequestUserPromptIdx) leave the fresh prompt unmarked: its tokens
276
+ // bill once at $5/M uncached, then again as a cache write when a later
277
+ // anchor advances past them — cost bounded by the prompt's size, so it
278
+ // matters for large pasted prompts. (2026-08-03 A/B note: session totals'
279
+ // totalUncachedInputTokens = input + cacheWrite by design, see
280
+ // uncachedInputTokensForProvider; billing-uncached input measured via
281
+ // usage.json was already ~0 on single-turn bench tasks before and after
282
+ // this change.) Synthetic system-reminder tails are excluded by
283
+ // hasUserText, so per-call volatile content still never keys the cache.
284
+ const currentTailUserIdx = () => {
285
+ const tailIdx = sanitizedMessages.length - 1;
286
+ return hasUserText(sanitizedMessages[tailIdx]) ? tailIdx : -1;
287
+ };
271
288
  if (messageTtl !== null) {
272
289
  const slots = Math.max(0, Math.min(4, Number(messageSlots) || 0));
273
290
  const marked = new Set();
274
- const candidates = [latestToolResultTailIdx(), previousUserTextAnchorIdx(), firstRequestUserPromptIdx()];
291
+ const candidates = [currentTailUserIdx(), latestToolResultTailIdx(), previousUserTextAnchorIdx(), firstRequestUserPromptIdx()];
275
292
  for (const idx of candidates) {
276
293
  if (slots <= 0) break;
277
294
  if (idx < 0 || marked.has(idx)) continue;
@@ -7,7 +7,10 @@
7
7
  * (scripts/openai-oauth-http-sse-toolcall-smoke.mjs) and fallback headers.
8
8
  */
9
9
  import { randomBytes } from 'crypto';
10
+ import zlib from 'node:zlib';
10
11
  import {
12
+ extractCacheWriteTokens,
13
+ extractCachedTokens,
11
14
  traceAgentFetch,
12
15
  traceAgentSse,
13
16
  traceAgentUsage,
@@ -47,6 +50,21 @@ const CODEX_REQUEST_MAX_RETRIES = 4;
47
50
  const CODEX_REQUEST_BACKOFF_MS = Object.freeze([200, 400, 800, 1600]);
48
51
  const CODEX_RETRY_JITTER_RATIO = 0.1;
49
52
 
53
+ // Request-body zstd gate (see sendViaHttpSse). Namespace import: on Node
54
+ // runtimes without zlib zstd bindings the named export would fail at module
55
+ // load, so the call site typeof-guards zlib.zstdCompressSync instead.
56
+ const OPENAI_REQ_ZSTD_MIN_BYTES = 8 * 1024;
57
+ let _openaiReqZstdLatch = false;
58
+ function _openaiReqZstdDisabled() {
59
+ return _openaiReqZstdLatch || process.env.MIXDOG_OPENAI_REQ_ZSTD === '0';
60
+ }
61
+ function _disableOpenaiReqZstd() { _openaiReqZstdLatch = true; }
62
+ function _zstdHeaders(headers, bodyForSend) {
63
+ return bodyForSend.encoding
64
+ ? { ...headers, 'Content-Encoding': bodyForSend.encoding }
65
+ : headers;
66
+ }
67
+
50
68
  export function _envFlag(name, fallback = true) {
51
69
  const raw = process.env[name];
52
70
  if (raw == null || raw === '') return fallback;
@@ -72,11 +90,6 @@ function _parseJsonObject(value) {
72
90
  }
73
91
  }
74
92
 
75
- function _extractCachedTokens(usage) {
76
- const details = usage?.input_tokens_details || usage?.prompt_tokens_details || {};
77
- return Number(details.cached_tokens ?? details.cached ?? usage?.cached_tokens ?? 0) || 0;
78
- }
79
-
80
93
  function _sseEventsFromBuffer(buffer) {
81
94
  const frames = [];
82
95
  let rest = buffer.replace(/\r\n/g, '\n');
@@ -231,6 +244,19 @@ export async function sendViaHttpSse({
231
244
  const responsesUrl = auth?.type === 'openai-direct'
232
245
  ? OPENAI_DIRECT_RESPONSES_URL
233
246
  : CODEX_RESPONSES_URL;
247
+ // Request-body zstd (codex parity: core client.rs
248
+ // responses_request_compression enables zstd for the codex backend on the
249
+ // OpenAI provider — the server decompresses Content-Encoding: zstd).
250
+ // openai-direct is excluded: only the codex backend is verified. Env
251
+ // kill-switch plus a process-wide latch flipped on the first 400 seen on
252
+ // a compressed request, which then replays that attempt uncompressed.
253
+ const _rawReqBytes = Buffer.from(JSON.stringify(body));
254
+ let _reqBodyForSend = auth?.type !== 'openai-direct'
255
+ && !_openaiReqZstdDisabled()
256
+ && typeof zlib.zstdCompressSync === 'function'
257
+ && _rawReqBytes.length >= OPENAI_REQ_ZSTD_MIN_BYTES
258
+ ? { bytes: zlib.zstdCompressSync(_rawReqBytes), encoding: 'zstd' }
259
+ : { bytes: _rawReqBytes, encoding: null };
234
260
  let response;
235
261
  for (let attempt = 0; attempt <= CODEX_REQUEST_MAX_RETRIES; attempt++) {
236
262
  const headerTimeout = createTimeoutSignal(
@@ -253,8 +279,8 @@ export async function sendViaHttpSse({
253
279
  } catch {}
254
280
  response = await fetchFn(responsesUrl, {
255
281
  method: 'POST',
256
- headers,
257
- body: JSON.stringify(body),
282
+ headers: _zstdHeaders(headers, _reqBodyForSend),
283
+ body: _reqBodyForSend.bytes,
258
284
  signal: headerTimeout.signal,
259
285
  dispatcher: getLlmDispatcher(),
260
286
  });
@@ -267,6 +293,15 @@ export async function sendViaHttpSse({
267
293
  headerTimeout.cleanup();
268
294
  }
269
295
 
296
+ // zstd rejection fallback: a 400 on a compressed request latches
297
+ // compression OFF process-wide and replays this attempt uncompressed.
298
+ if (response && response.status === 400 && _reqBodyForSend.encoding) {
299
+ _disableOpenaiReqZstd();
300
+ _reqBodyForSend = { bytes: _rawReqBytes, encoding: null };
301
+ await response.arrayBuffer().catch(() => {});
302
+ response = undefined;
303
+ continue;
304
+ }
270
305
  const retryableStatus = response && response.status >= 500 && response.status <= 599;
271
306
  // Typed transient transport failures only (errno / SDK connection
272
307
  // type). An unknown pre-response failure throws immediately instead of
@@ -783,7 +818,8 @@ export async function sendViaHttpSse({
783
818
  usage = {
784
819
  inputTokens: resp.usage.input_tokens || 0,
785
820
  outputTokens: resp.usage.output_tokens || 0,
786
- cachedTokens: _extractCachedTokens(resp.usage),
821
+ cachedTokens: extractCachedTokens(resp.usage),
822
+ cacheWriteTokens: extractCacheWriteTokens(resp.usage),
787
823
  promptTokens: resp.usage.input_tokens || 0,
788
824
  raw: serviceTier ? { ...resp.usage, service_tier: serviceTier } : resp.usage,
789
825
  };
@@ -39,6 +39,9 @@ export function _combineUsageWithWarmup(actual, warmup, { separateMainContext =
39
39
  inputTokens: _usageNum(actual.inputTokens) + _usageNum(warmup.inputTokens),
40
40
  outputTokens: _usageNum(actual.outputTokens) + _usageNum(warmup.outputTokens),
41
41
  cachedTokens: _usageNum(actual.cachedTokens) + _usageNum(warmup.cachedTokens),
42
+ ...(actual.cacheWriteTokens != null || warmup.cacheWriteTokens != null
43
+ ? { cacheWriteTokens: _usageNum(actual.cacheWriteTokens) + _usageNum(warmup.cacheWriteTokens) }
44
+ : {}),
42
45
  promptTokens: _usageNum(actual.promptTokens) + _usageNum(warmup.promptTokens),
43
46
  warmupInputTokens: _usageNum(warmup.inputTokens),
44
47
  warmupCachedTokens: _usageNum(warmup.cachedTokens),
@@ -16,6 +16,7 @@
16
16
  import { randomBytes } from 'crypto';
17
17
  import { performance } from 'node:perf_hooks';
18
18
  import {
19
+ extractCacheWriteTokens,
19
20
  extractCachedTokens,
20
21
  appendAgentTrace,
21
22
  } from '../agent-trace.mjs';
@@ -1090,6 +1091,7 @@ export async function _streamResponse({
1090
1091
  inputTokens: u.input_tokens || 0,
1091
1092
  outputTokens: u.output_tokens || 0,
1092
1093
  cachedTokens: extractCachedTokens(u),
1094
+ cacheWriteTokens: extractCacheWriteTokens(u),
1093
1095
  // openai-oauth reports input_tokens as the total
1094
1096
  // prompt volume (cached portion is a subset, not
1095
1097
  // additive). Alias into the cross-provider
@@ -227,7 +227,17 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
227
227
  // turn's push (deferBodies below) collapse to markers now — the model has
228
228
  // already seen them on that turn's follow-up send. Failed bodies stay
229
229
  // verbatim for retry.
230
- compactSettledToolCallBodies(messages);
230
+ // Out-of-loop transcript mutations (post-turn/manual compaction in
231
+ // manager/compaction-runner.mjs) run where no send opts exist; they park a
232
+ // one-shot intent on the session so the FIRST send of the next turn tags
233
+ // its expected cache break instead of an unexplained prefix mismatch.
234
+ if (!opts.cacheBreakIntent && typeof sessionRef?.pendingCacheBreakIntent === 'string') {
235
+ opts.cacheBreakIntent = sessionRef.pendingCacheBreakIntent;
236
+ delete sessionRef.pendingCacheBreakIntent;
237
+ }
238
+ if (compactSettledToolCallBodies(messages) && !opts.cacheBreakIntent) {
239
+ opts.cacheBreakIntent = 'deferred_body_compaction';
240
+ }
231
241
  // ---- Codex turn stop hook (refs/codex core/src/session/turn.rs:372-404) --
232
242
  // A no-tool assistant message is TERMINAL. Only a structured provider
233
243
  // follow-up signal (end_turn=false / pause_turn), pending input, tool
@@ -526,7 +536,10 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
526
536
  const _m = messages[_i];
527
537
  if (_m && _m.role === 'tool' && typeof _m.content === 'string' && _m.content.includes('⚠')) {
528
538
  const _stripped = stripSoftWarns(_m.content);
529
- if (_stripped !== _m.content) _m.content = _stripped;
539
+ if (_stripped !== _m.content) {
540
+ _m.content = _stripped;
541
+ if (!opts.cacheBreakIntent) opts.cacheBreakIntent = 'soft_warn_strip';
542
+ }
530
543
  }
531
544
  }
532
545
  sendTools = snapshotProviderRequestTools({
@@ -645,14 +658,16 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
645
658
  );
646
659
  const _sendEndedAt = Date.now();
647
660
  if (_sendResult.action === 'retry') {
648
- delete opts.cacheBreakIntent;
661
+ // Keep opts.cacheBreakIntent: the failed send never consumed the
662
+ // tag, and the reactive-compact retry that follows IS the tagged
663
+ // transition — deleting it here made retry-side cache_break rows
664
+ // log intentional_transition: null.
649
665
  contextOverflowRetryUsed = true;
650
666
  reactiveOverflowRetryPending = true;
651
667
  continue;
652
668
  }
653
669
  if (_sendResult.action === 'retry_transport') {
654
670
  _transportRetriesUsed += 1;
655
- delete opts.cacheBreakIntent;
656
671
  continue;
657
672
  }
658
673
  response = _sendResult.response;
@@ -1021,7 +1036,9 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
1021
1036
  // Settle earlier deferred bodies before this turn's message lands:
1022
1037
  // every previous call already has its result row, so successful bodies
1023
1038
  // compact to markers while failed ones keep their full retry text.
1024
- compactSettledToolCallBodies(messages);
1039
+ if (compactSettledToolCallBodies(messages) && !opts.cacheBreakIntent) {
1040
+ opts.cacheBreakIntent = 'deferred_body_compaction';
1041
+ }
1025
1042
  messages.push(_assistantTurnMsg);
1026
1043
  try { opts.onAssistantMessageCommitted?.(_assistantTurnMsg); } catch {}
1027
1044
  const _callsToExecute = calls;
@@ -1,18 +1,18 @@
1
1
  // Eager tool-dispatch controller, extracted from agent-loop.mjs. Owns the
2
2
  // per-turn pending promise map, the intra-turn in-flight signature set, and
3
- // the mutation epoch. FULL-PARALLEL policy: every tool call starts executing
4
- // the instant the provider streams its tool_use event (or at batch start),
5
- // shell/MCP/writes included — the model owns ordering by splitting dependent
6
- // work into separate turns. Only apply_patch (ordered mutation) waits for the
7
- // serial batch loop.
3
+ // the mutation epoch. Read-only calls may start while the provider is still
4
+ // streaming. Side-effect calls stream-start too until the first apply_patch
5
+ // appears in the stream (they are definitively in segment 0); after that they
6
+ // wait for their segment bounded by apply_patch barriers. Patches execute
7
+ // serially as barriers, so a failed patch can skip every later side effect.
8
8
  import { normalizeToolEnvelope } from './tool-envelope.mjs';
9
9
  import { isInvalidToolArgsMarker } from '../providers/openai-compat-stream.mjs';
10
- import { _intraTurnSig, _isMutationTool, _isReadTool, _isScopedCacheableTool, _isShellTool, _stripMcpPrefix } from './loop/tool-classify.mjs';
10
+ import { _intraTurnSig, _isMutationTool, _isOrderedGateSkippable, _isReadTool, _isScopedCacheableTool, _stripMcpPrefix } from './loop/tool-classify.mjs';
11
11
  import { tryReadCached, tryScopedToolCached } from './read-dedup.mjs';
12
12
  import { preDispatchDenyForSession } from './loop/pre-dispatch-deny.mjs';
13
13
  import { executeTool } from './loop/tool-exec.mjs';
14
14
  import { crossTurnSignature } from './loop/completion-guards.mjs';
15
- import { getToolKind, isParallelDispatchable, isToolCallDedupEligible } from './loop/tool-helpers.mjs';
15
+ import { getToolKind, isEagerDispatchable, isParallelDispatchable, isToolCallDedupEligible } from './loop/tool-helpers.mjs';
16
16
 
17
17
  export function createEagerDispatcher({
18
18
  tools, cwd, sessionId, sessionRef, signal, opts,
@@ -40,18 +40,11 @@ export function createEagerDispatcher({
40
40
  // resets at the turn boundary without leaking across getIterations().
41
41
  const _eagerInFlightSigs = new Map();
42
42
  const epoch = { mutation: 0 };
43
- // Patch→shell ordering insurance: a shell call that appears AFTER an
44
- // apply_patch in the same assistant turn must not eager-start before
45
- // that patch has executed (serial body runs both in call order).
46
- // Reads are already safe via the mutationEpoch re-execution gate;
47
- // only shell's side effects would consume pre-patch file state.
48
- let _streamSawMutation = false;
49
- const _hasEarlierMutation = (calls, index) => {
50
- for (let k = 0; k < index; k += 1) {
51
- if (_isMutationTool(calls[k]?.name)) return true;
52
- }
53
- return false;
54
- };
43
+ // True once an apply_patch tool_use has been seen in THIS turn's
44
+ // stream. Before that, every side-effect call streamed so far sits
45
+ // before the first patch in call order (segment 0), so it may start
46
+ // immediately — a patch that arrives later cannot gate an EARLIER call.
47
+ let _streamSeenMutation = false;
55
48
  const startEagerTool = (call) => {
56
49
  if (!call?.id || pending.has(call.id) || !isParallelDispatchable(call.name)) return null;
57
50
  // Never eager-execute a call whose arguments failed to parse
@@ -107,7 +100,7 @@ export function createEagerDispatcher({
107
100
  if (_dedupEligible) _eagerInFlightSigs.set(_sig, call.id);
108
101
  entry.promise = (async () => {
109
102
  try {
110
- return { ok: true, value: await executeToolFn(call.name, call.arguments, cwd, sessionId, sessionRef, { toolCallId: call.id, signal, notifyFn: opts.notifyFn, toolApprovalHook: opts.onToolApproval, iteration: getNextIteration() }) };
103
+ return { ok: true, value: await executeToolFn(call.name, call.arguments, cwd, sessionId, sessionRef, { toolCallId: call.id, signal, notifyFn: opts.notifyFn, toolApprovalHook: opts.onToolApproval, iteration: getNextIteration(), deferShellCwdCommit: true }) };
111
104
  } catch (error) {
112
105
  return { ok: false, error };
113
106
  }
@@ -168,16 +161,21 @@ export function createEagerDispatcher({
168
161
  return entry;
169
162
  };
170
163
  const startEagerRun = (calls, startIndex, dupSet) => {
164
+ const _nextOrderedMutationIndex = calls.findIndex(
165
+ (call, index) => index >= startIndex && _isMutationTool(call?.name),
166
+ );
171
167
  for (let j = startIndex; j < calls.length; j += 1) {
172
168
  const call = calls[j];
173
- // Full-parallel: only the ordered mutation (apply_patch) is
174
- // skipped — it executes in the serial batch body. No barrier:
175
- // later calls keep starting in parallel past it.
169
+ // Side effects may eager-start in the current segment but not
170
+ // across the next apply_patch barrier. Known read-only work
171
+ // may cross it and is protected by mutation-epoch re-execution.
176
172
  if (!call?.id || !isParallelDispatchable(call.name)) continue;
177
173
  if (dupSet && dupSet.has(call.id)) continue;
178
- // Patch→shell insurance: leave a shell that follows an
179
- // apply_patch to the serial body so it runs after the patch.
180
- if (_isShellTool(call.name) && _hasEarlierMutation(calls, j)) continue;
174
+ if (
175
+ _nextOrderedMutationIndex >= 0
176
+ && j >= _nextOrderedMutationIndex
177
+ && _isOrderedGateSkippable(call.name)
178
+ ) continue;
181
179
  // A null return here is NOT a state barrier. It means a
182
180
  // non-barrier stub — intra-turn in-flight dup, repeat-failure /
183
181
  // cross-turn dedup, pre-dispatch-deny, invalid-args, or a cache
@@ -188,9 +186,17 @@ export function createEagerDispatcher({
188
186
  }
189
187
  };
190
188
  const onToolCall = (call) => {
191
- if (_isMutationTool(call?.name)) { _streamSawMutation = true; return; }
192
- if (!isParallelDispatchable(call?.name)) return;
193
- if (_streamSawMutation && _isShellTool(call.name)) return;
189
+ if (_isMutationTool(call?.name)) {
190
+ _streamSeenMutation = true;
191
+ return;
192
+ }
193
+ // Declared read-only calls always overlap streaming (epoch guard
194
+ // re-executes them if a later patch lands). Side-effect calls may
195
+ // stream-start only while NO apply_patch has streamed yet: their
196
+ // call-order position is already fixed before the first barrier,
197
+ // matching exactly what startEagerRun would do post-batch. Once a
198
+ // patch has streamed, later side effects wait for their segment.
199
+ if (!isEagerDispatchable(call?.name, tools) && _streamSeenMutation) return;
194
200
  startEagerTool(call);
195
201
  };
196
202
  return { pending, epoch, startEagerTool, startEagerRun, onToolCall };