mixdog 0.9.91 → 0.9.93

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +147 -51
  2. package/package.json +6 -5
  3. package/scripts/code-graph-description-contract.mjs +6 -8
  4. package/scripts/tmp-cdp-errors.mjs +41 -0
  5. package/scripts/tmp-cdp-inspect.mjs +41 -0
  6. package/scripts/tool-overhead-microbench.mjs +60 -0
  7. package/scripts/tui-transcript-jitter-harness.mjs +2 -18
  8. package/src/agents/debugger/agent.json +1 -1
  9. package/src/agents/explore/agent.json +1 -1
  10. package/src/agents/heavy-worker/agent.json +1 -1
  11. package/src/agents/maintainer/agent.json +1 -1
  12. package/src/agents/reviewer/agent.json +1 -1
  13. package/src/agents/worker/agent.json +1 -1
  14. package/src/lib/rules-builder.cjs +5 -5
  15. package/src/output-styles/simple.md +1 -1
  16. package/src/rules/agent/00-core.md +1 -2
  17. package/src/rules/lead/01-general.md +1 -0
  18. package/src/rules/shared/01-tool.md +19 -21
  19. package/src/runtime/agent/orchestrator/agent-runtime/cache-strategy.mjs +16 -3
  20. package/src/runtime/agent/orchestrator/agent-trace.mjs +17 -0
  21. package/src/runtime/agent/orchestrator/context/collect.mjs +2 -1
  22. package/src/runtime/agent/orchestrator/providers/anthropic-effort.mjs +9 -1
  23. package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +90 -22
  24. package/src/runtime/agent/orchestrator/providers/anthropic.mjs +54 -19
  25. package/src/runtime/agent/orchestrator/providers/lib/anthropic-request-utils.mjs +18 -1
  26. package/src/runtime/agent/orchestrator/providers/openai-oauth-http-sse.mjs +44 -8
  27. package/src/runtime/agent/orchestrator/providers/openai-oauth-ws.mjs +10 -0
  28. package/src/runtime/agent/orchestrator/providers/openai-ws-events.mjs +3 -0
  29. package/src/runtime/agent/orchestrator/providers/openai-ws-stream.mjs +2 -0
  30. package/src/runtime/agent/orchestrator/providers/retry-classifier.mjs +35 -0
  31. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +34 -5
  32. package/src/runtime/agent/orchestrator/session/eager-dispatch.mjs +35 -29
  33. package/src/runtime/agent/orchestrator/session/loop/stop-hooks.mjs +9 -0
  34. package/src/runtime/agent/orchestrator/session/loop/stored-tool-args.mjs +11 -2
  35. package/src/runtime/agent/orchestrator/session/loop/tool-classify.mjs +5 -6
  36. package/src/runtime/agent/orchestrator/session/loop/tool-exec.mjs +31 -2
  37. package/src/runtime/agent/orchestrator/session/manager/compaction-runner.mjs +60 -0
  38. package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +60 -31
  39. package/src/runtime/agent/orchestrator/session/manager.mjs +1 -1
  40. package/src/runtime/agent/orchestrator/session/result-classification.mjs +28 -0
  41. package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +71 -2
  42. package/src/runtime/agent/orchestrator/session/store/listing.mjs +17 -0
  43. package/src/runtime/agent/orchestrator/session/store-summary-reader.mjs +101 -0
  44. package/src/runtime/agent/orchestrator/session/store.mjs +30 -0
  45. package/src/runtime/agent/orchestrator/session/tool-batch.mjs +19 -23
  46. package/src/runtime/agent/orchestrator/stall-policy.mjs +31 -21
  47. package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +4 -1
  48. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +12 -6
  49. package/src/runtime/agent/orchestrator/tools/builtin/task-tool.mjs +5 -0
  50. package/src/runtime/agent/orchestrator/tools/code-graph-tool-defs.mjs +4 -3
  51. package/src/runtime/agent/orchestrator/tools/lib/pwsh-standby-pool.mjs +47 -14
  52. package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +100 -0
  53. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +6 -5
  54. package/src/runtime/agent/orchestrator/tools/shell-command.mjs +9 -0
  55. package/src/runtime/agent/orchestrator/tools/shell-state.mjs +32 -2
  56. package/src/runtime/channels/backends/discord-gateway.mjs +6 -32
  57. package/src/runtime/channels/lib/inbound-handler.mjs +19 -3
  58. package/src/runtime/channels/lib/scheduler.mjs +51 -3
  59. package/src/runtime/channels/lib/worker-main.mjs +4 -0
  60. package/src/runtime/channels/tool-defs.mjs +4 -2
  61. package/src/runtime/memory/lib/query-handlers.mjs +11 -3
  62. package/src/runtime/memory/tool-defs.mjs +5 -5
  63. package/src/runtime/shared/channel-notification-routing.mjs +8 -2
  64. package/src/runtime/shared/llm/http-agent.mjs +11 -0
  65. package/src/session-runtime/lifecycle-api.mjs +26 -1
  66. package/src/session-runtime/output-styles.mjs +1 -4
  67. package/src/session-runtime/tool-catalog-data.mjs +5 -2
  68. package/src/session-runtime/workflow.mjs +25 -13
  69. package/src/standalone/agent-tool/tag-registry.mjs +5 -1
  70. package/src/standalone/explore-tool.mjs +1 -1
  71. package/src/tui/dist/index.mjs +100 -46
  72. package/src/tui/engine/agent-envelope.mjs +52 -3
  73. package/src/tui/engine/session-api.mjs +17 -0
  74. package/src/tui/engine/tui-steering-persist.mjs +24 -1
  75. package/src/tui/engine/turn.mjs +8 -9
  76. package/src/tui/engine.mjs +18 -41
  77. package/src/workflows/default/WORKFLOW.md +7 -18
  78. package/src/workflows/solo/WORKFLOW.md +0 -6
  79. package/src/workflows/solo-bench/WORKFLOW.md +17 -0
@@ -2,7 +2,7 @@
2
2
  name: simple
3
3
  title: Simple
4
4
  description: Outcome-first concise handoffs for coding work
5
- aliases: concise, handoff, default
5
+ aliases: concise, handoff
6
6
  keep-coding-instructions: true
7
7
  ---
8
8
 
@@ -1,7 +1,6 @@
1
1
  # Agent Constraints
2
2
 
3
- - Agent communication is English. One turn is one batch; include every
4
- compatible read-only call.
3
+ - Agent communication is English.
5
4
  - Call tools immediately: no preamble/progress; text only in final handoff.
6
5
  - Final handoff is fragments: outcome, key `file:line`, verification
7
6
  command+result, material risk/blocker. Never repeat the brief, process,
@@ -8,6 +8,7 @@
8
8
  validated target paths — never `~`, a root, or unresolved variables/globs;
9
9
  report material deletions with recoverability.
10
10
  - Act proactively; ask only for decisions.
11
+ - Build only what the task requires; trust internal and framework guarantees.
11
12
  - Mid-task input: a replacement supersedes current work, an addition folds
12
13
  into it, a status question gets a brief answer while work continues; after
13
14
  context compaction continue from the summary — never restart or redo
@@ -1,21 +1,18 @@
1
1
  # Tool Use
2
2
 
3
- - Before the first call, gather every known facet in one tool message; one
4
- shortest route per facet: broad/uncertain→`explore` (roles without it:
5
- `find`); partial path/name→`find`; verified root+wildcard→`glob`;
6
- quoted/non-identifier literal or regex→`grep`; exact code identifier/
7
- relation→`code_graph` before grep; known file/span→`read` directly without
8
- `grep`; verified directory→`list`; known edit→`apply_patch` (span already
9
- seen; else `read`/`grep` first); program/state change→`shell`; web/current
10
- external info→`search`.
11
- - Shortest total calls, maximum batching — every turn: all independent calls
12
- in one concurrent message (shell included); combine variants/symbols/
13
- scopes/paths/queries per call; same-file regions as one real
14
- `{path,offset,limit}` array; graph targets as arrays; `explore` facets in
15
- one `query[]` (max 8, no rephrased duplicates); all new edits in one patch.
16
- Distinct facets, not alternative routes; sequential singles only for a
17
- genuinely dependent step (unconditional follow-ups are not dependent —
18
- batch them); only apply_patch executes in order.
3
+ - Before the first call, gather every known facet — environment, capability,
4
+ artifact, and failure checks — in one bounded tool message. Use one shortest
5
+ route per facet: broad/uncertain→`explore` (roles without it: `find`);
6
+ partial path/name→`find`; verified root+wildcard→`glob`;
7
+ quoted/non-identifier literal or regex→`grep`; exact code
8
+ identifier/relation→`code_graph` before grep; known file/span→`read`
9
+ directly without `grep`; verified directory→`list`; known
10
+ edit→`apply_patch`; program/state change→`shell`; web/current external
11
+ info→`search`.
12
+ - Shortest total calls, maximum batching — every turn: every determined call
13
+ in one concurrent message (mix `shell` in), merged per tool — one `shell`
14
+ chain, one `read`, one `apply_patch` carrying full verification in
15
+ `post_shell`; a later turn only for steps needing unseen output.
19
16
  - Verified paths: project root, session cwd, user-provided, tool-returned.
20
17
  `find` first for guessed path/name fragments; on ENOENT, find the basename.
21
18
  Retry `EXPLORATION_FAILED` once with changed tokens.
@@ -23,11 +20,12 @@
23
20
  nonzero `content_with_context` result is final — act on it (inspecting it
24
21
  via read/code_graph is valid); only zero/error results justify changed
25
22
  tokens or scope. Don't re-locate, re-verify, or reread returned spans.
26
- - `apply_patch` is the primary edit tool: send the patch as soon as target
27
- path and new content are known. Hunk context comes verbatim from the newest
28
- tool output of that span (`read`/`grep`/your own patch — post-patch content
29
- after edits), never retyped from memory; one look-up beats a failed patch.
30
- A same-turn shell after `apply_patch` runs once the patch lands.
23
+ - Verify changes in proportion to risk with one decisive batched boundary
24
+ probe. A pass is final; on failure, fix and rerun only what failed. Keep
25
+ optional diagnostics non-fatal; report verified vs assumed.
26
+ - `apply_patch` is the primary edit tool: once target path and new content are
27
+ known, include the patch in the current tool batch, hunk context verbatim
28
+ from the newest tool output of that span (post-patch content after edits).
31
29
  - After starting or receiving a background task, end the turn — its
32
30
  completion notification resumes the work. Never poll, sleep-loop, or block;
33
31
  explicit wait only for a result the current turn cannot proceed without.
@@ -120,18 +120,31 @@ export function resolveCacheStrategy(agent, { autoClear } = {}) {
120
120
  if (isOneShotMaintenanceAgent(agent)) {
121
121
  return { tools: 'none', system: 'none', tier3: 'none', messages: 'none' };
122
122
  }
123
+ // Operator override for the BP4 (messages-tail) TTL. Short-lived
124
+ // rapid-turn deployments (bench-style: session dies in <15min) never
125
+ // benefit from a 1h tail across its premium window, so '5m' trades the
126
+ // 2x write premium (1h, $10/M) down to 1.25x ($6.25/M) and deactivates
127
+ // the 1h volatile-content anchor guard. Product defaults below stay
128
+ // untouched when the env is unset.
129
+ const envMessagesTtl = (process.env.MIXDOG_CACHE_MESSAGES_TTL || '').trim();
130
+ const applyEnv = (strategy) => {
131
+ if (envMessagesTtl === '1h' || envMessagesTtl === '5m' || envMessagesTtl === 'none') {
132
+ return { ...strategy, messages: envMessagesTtl };
133
+ }
134
+ return strategy;
135
+ };
123
136
  if (getHiddenAgent(agent)) {
124
- return { tools: 'none', system: '1h', tier3: '1h', messages: '1h' };
137
+ return applyEnv({ tools: 'none', system: '1h', tier3: '1h', messages: '1h' });
125
138
  }
126
139
  if (agent && agent !== 'lead') {
127
140
  // Public (non-hidden, non-lead) agents keep the flat 1h tail — only
128
141
  // the Lead session's tail is linked to autoClear.
129
- return { tools: 'none', system: '1h', tier3: '1h', messages: '1h' };
142
+ return applyEnv({ tools: 'none', system: '1h', tier3: '1h', messages: '1h' });
130
143
  }
131
144
  // Lead session (agent === 'lead', or no agent — raw/CLI callers default
132
145
  // to Lead behavior): message tail TTL is linked to autoClear (see
133
146
  // resolveLeadMessagesTtl).
134
- return { tools: 'none', system: '1h', tier3: '1h', messages: resolveLeadMessagesTtl(autoClear) };
147
+ return applyEnv({ tools: 'none', system: '1h', tier3: '1h', messages: resolveLeadMessagesTtl(autoClear) });
135
148
  }
136
149
 
137
150
  /**
@@ -38,6 +38,22 @@ function extractCachedTokens(usage) {
38
38
  return 0;
39
39
  }
40
40
 
41
+ function extractCacheWriteTokens(usage) {
42
+ const candidates = [
43
+ usage?.input_tokens_details?.cache_write_tokens,
44
+ usage?.prompt_tokens_details?.cache_write_tokens,
45
+ usage?.inputTokensDetails?.cacheWriteTokens,
46
+ usage?.promptTokensDetails?.cacheWriteTokens,
47
+ usage?.cache_write_tokens,
48
+ usage?.cacheWriteTokens,
49
+ ];
50
+ for (const value of candidates) {
51
+ const n = Number(value);
52
+ if (Number.isFinite(n)) return n;
53
+ }
54
+ return 0;
55
+ }
56
+
41
57
  // Lightweight fingerprint of the conversation prefix. Hashes the first 4096
42
58
  // characters of JSON.stringify(messages) — enough to detect prefix mutation
43
59
  // across iterations (which invalidates the provider prompt cache) without
@@ -267,6 +283,7 @@ export {
267
283
  appendAgentTrace,
268
284
  drainAgentTrace,
269
285
  estimateProviderPayloadBytes,
286
+ extractCacheWriteTokens,
270
287
  extractCachedTokens,
271
288
  messagePrefixHash,
272
289
  traceAgentFetch,
@@ -275,7 +275,7 @@ export function buildSkillToolEnvelope(name, content, skillDir) {
275
275
  };
276
276
  }
277
277
 
278
- function compactSkillManifestText(value, max = 180) {
278
+ function compactSkillManifestText(value, max = 100) {
279
279
  const text = String(value || '').replace(/\s+/g, ' ').trim();
280
280
  return text.length > max ? `${text.slice(0, Math.max(1, max - 3))}...` : text;
281
281
  }
@@ -528,6 +528,7 @@ export function buildSkillToolDefs(skills, { ownerIsAgentSession = false } = {})
528
528
  name: { type: 'string', description: 'Skill name' },
529
529
  },
530
530
  required: ['name'],
531
+ additionalProperties: false,
531
532
  },
532
533
  },
533
534
  ];
@@ -208,7 +208,15 @@ export function applyAnthropicEffortToBody(
208
208
  // modelSupportsEffort() allowlist so older models never receive it.
209
209
  // Set unconditionally (independent of `normalized`) so effort-capable
210
210
  // turns always carry adaptive thinking + round-trip signatures.
211
- body.thinking = { type: 'adaptive', display: 'summarized' };
211
+ // MIXDOG_ANTHROPIC_THINKING_DISPLAY=omitted (operator/bench knob):
212
+ // CC-parity mode — no thinking blocks come back, so nothing is
213
+ // replayed into later requests (saves the 1h cache-write + re-read on
214
+ // accumulated thinking) at the cost of losing visible reasoning and
215
+ // cross-iteration thinking continuity. Default stays summarized.
216
+ const display = (process.env.MIXDOG_ANTHROPIC_THINKING_DISPLAY || '').trim() === 'omitted'
217
+ ? 'omitted'
218
+ : 'summarized';
219
+ body.thinking = { type: 'adaptive', display };
212
220
  // Adaptive/4.7+ models reject any non-default sampling param with a 400.
213
221
  delete body.temperature;
214
222
  delete body.top_p;
@@ -50,8 +50,10 @@ import {
50
50
  anthropicRequestTimeoutMs,
51
51
  classifyError,
52
52
  anthropicMaxAttempts,
53
+ createStallRetryBudget,
53
54
  midstreamBackoffFor,
54
55
  retryAfterMsFromError,
56
+ STREAM_STALL_RETRY_BUDGET_MS,
55
57
  withRetry,
56
58
  } from './retry-classifier.mjs';
57
59
  import {
@@ -62,6 +64,18 @@ import {
62
64
  stampAnthropicStreamOutcome,
63
65
  } from './anthropic-sse.mjs';
64
66
  import { buildAnthropicBetaHeaders, supportsAnthropicFastMode } from './anthropic-betas.mjs';
67
+ import { gzipSync } from 'node:zlib';
68
+
69
+ // Request-body gzip gate (see the fetch site below). Env kill-switch
70
+ // (MIXDOG_ANTHROPIC_REQ_GZIP=0) plus a process-wide latch flipped on the
71
+ // first 400 response to a compressed request. Small bodies skip compression:
72
+ // below ~8KB the CPU + header cost outweighs the upload saving.
73
+ const ANTHROPIC_REQ_GZIP_MIN_BYTES = 8 * 1024;
74
+ let _anthropicReqGzipLatch = false;
75
+ function _anthropicReqGzipDisabled() {
76
+ return _anthropicReqGzipLatch || process.env.MIXDOG_ANTHROPIC_REQ_GZIP === '0';
77
+ }
78
+ function _disableAnthropicReqGzip() { _anthropicReqGzipLatch = true; }
65
79
  import {
66
80
  applyAnthropicEffortToBody,
67
81
  effortValuesForModel,
@@ -652,7 +666,16 @@ export class AnthropicOAuthProvider {
652
666
  // provider-visible cache breakpoint off the cached one — the
653
667
  // exact COLD-turn bug this change fixes. Order is fixed:
654
668
  // build → sanitize (once) → mark → JSON.stringify.
655
- const response = await fetch(API_URL, {
669
+ // Request-body gzip (probe-verified 2026-08-04: /v1/messages
670
+ // returns 200 for Content-Encoding: gzip, 400 for zstd). Large
671
+ // turn bodies (system prompt + history, typically 50-100KB+)
672
+ // compress ~5-10x, trimming upload time off every call's
673
+ // header wait. Latch OFF process-wide on the first 400 seen on
674
+ // a compressed request and retry that attempt uncompressed, so
675
+ // a server-side behavior change can never wedge the session.
676
+ const rawBody = Buffer.from(JSON.stringify(requestBody));
677
+ const useGzip = !_anthropicReqGzipDisabled() && rawBody.length >= ANTHROPIC_REQ_GZIP_MIN_BYTES;
678
+ const sendAttempt = (gz) => fetch(API_URL, {
656
679
  method: 'POST',
657
680
  headers: {
658
681
  'Authorization': `Bearer ${accessToken}`,
@@ -667,11 +690,18 @@ export class AnthropicOAuthProvider {
667
690
  'user-agent': `claude-cli/${resolveCliVersion()} (external, sdk-cli)`,
668
691
  'x-app': 'cli',
669
692
  'Content-Type': 'application/json',
693
+ ...(gz ? { 'Content-Encoding': 'gzip' } : {}),
670
694
  },
671
- body: JSON.stringify(requestBody),
695
+ body: gz ? gzipSync(rawBody) : rawBody,
672
696
  signal: controller.signal,
673
697
  dispatcher: getLlmDispatcher(),
674
698
  });
699
+ let response = await sendAttempt(useGzip);
700
+ if (useGzip && response.status === 400) {
701
+ _disableAnthropicReqGzip();
702
+ try { await response.arrayBuffer(); } catch { /* drain best-effort */ }
703
+ response = await sendAttempt(false);
704
+ }
675
705
 
676
706
  traceAgentFetch({
677
707
  sessionId,
@@ -774,26 +804,17 @@ export class AnthropicOAuthProvider {
774
804
  const MAX_MIDSTREAM_RETRIES = ANTHROPIC_MAX_MIDSTREAM_RETRIES;
775
805
  let firstAttemptError = null;
776
806
  let firstAttemptClassifier = null;
777
-
778
- const recoverNonStreaming = async (midState, streamingError, controller) => {
779
- const exposedChars = Number(midState?.emittedTextChars) || 0;
780
- if (!onTextReset || exposedChars <= 0
781
- || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
782
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
783
- throw streamingError;
784
- }
785
- let resetAccepted = false;
786
- try {
787
- resetAccepted = await onTextReset({
788
- chars: exposedChars,
789
- reason: 'anthropic-streaming-fallback',
790
- }) === true;
791
- } catch {}
792
- if (!resetAccepted) {
793
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
794
- throw streamingError;
795
- }
796
- try { controller?.abort?.(streamingError); } catch {}
807
+ // Send-scoped stall window: in-place stall retries share one wall
808
+ // clock starting at the first stall (see createStallRetryBudget).
809
+ const stallRetryBudget = createStallRetryBudget();
810
+
811
+ // Core non-streaming re-issue: abort the dead stream and repeat the
812
+ // SAME request with stream:false. Shared by the exposed-text recovery
813
+ // (which must first get the owner's onTextReset acknowledgement) and
814
+ // the no-exposure stall fallback below (trivially safe — nothing was
815
+ // relayed or dispatched, so there is nothing to withdraw or replay).
816
+ const issueNonStreamingFallback = async (controller, abortReason) => {
817
+ try { controller?.abort?.(abortReason); } catch {}
797
818
  try { onStageChange?.('requesting', { transport: 'non-streaming-fallback' }); } catch {}
798
819
  let fallback = await requestWithRetry(creds.accessToken, { ...body, stream: false });
799
820
  if (fallback.response.status === 401) {
@@ -820,6 +841,27 @@ export class AnthropicOAuthProvider {
820
841
  }
821
842
  };
822
843
 
844
+ const recoverNonStreaming = async (midState, streamingError, controller) => {
845
+ const exposedChars = Number(midState?.emittedTextChars) || 0;
846
+ if (!onTextReset || exposedChars <= 0
847
+ || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
848
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
849
+ throw streamingError;
850
+ }
851
+ let resetAccepted = false;
852
+ try {
853
+ resetAccepted = await onTextReset({
854
+ chars: exposedChars,
855
+ reason: 'anthropic-streaming-fallback',
856
+ }) === true;
857
+ } catch {}
858
+ if (!resetAccepted) {
859
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
860
+ throw streamingError;
861
+ }
862
+ return issueNonStreamingFallback(controller, streamingError);
863
+ };
864
+
823
865
  try {
824
866
  for (let attemptIndex = 0; attemptIndex <= MAX_MIDSTREAM_RETRIES; attemptIndex++) {
825
867
  let response, controller, cancelHandler;
@@ -1052,6 +1094,32 @@ export class AnthropicOAuthProvider {
1052
1094
  continue;
1053
1095
  }
1054
1096
  const classifier = _classifyMidstreamError(err, midState);
1097
+ // CC-parity stall recovery (2026-08-03 v3 postmortem): a
1098
+ // stalled stream that exposed NOTHING (no text/thinking
1099
+ // relayed, no tool emitted) is re-issued NON-STREAMING instead
1100
+ // of retrying the same streaming shape. Effort-mode models can
1101
+ // legitimately think in silence past any streaming idle
1102
+ // window; an in-place streaming retry re-runs the same silent
1103
+ // generation into the same timer (observed live: deterministic
1104
+ // 4×~138s beheading, ~552s per turn), while the non-streaming
1105
+ // transport simply waits for the full body (bounded by
1106
+ // PROVIDER_NONSTREAM_TOTAL_TIMEOUT_MS). Claude Code does
1107
+ // exactly this on its watchdog aborts. Replay is trivially
1108
+ // safe here — nothing was relayed or dispatched.
1109
+ if (classifier === 'stream_stalled'
1110
+ && _outcome?.replayUnsafe !== true
1111
+ && !midState.emittedText
1112
+ && !midState.emittedToolCall
1113
+ && !midState.partialToolCall
1114
+ && !midState.emittedThinking) {
1115
+ try { process.stderr.write('[anthropic-oauth] stream stalled with no exposure — retrying non-streaming\n'); } catch {}
1116
+ return await issueNonStreamingFallback(controller, err);
1117
+ }
1118
+ if (classifier === 'stream_stalled' && !stallRetryBudget.allowStallRetry()) {
1119
+ try { process.stderr.write(`[anthropic-oauth] stall retry budget exhausted (${STREAM_STALL_RETRY_BUDGET_MS}ms since first stall) — surfacing for fresh-request retry\n`); } catch {}
1120
+ try { controller?.abort?.(err); } catch { /* best-effort teardown */ }
1121
+ throw err;
1122
+ }
1055
1123
  if (classifier && attemptIndex < MAX_MIDSTREAM_RETRIES) {
1056
1124
  firstAttemptError = err;
1057
1125
  firstAttemptClassifier = classifier;
@@ -8,8 +8,10 @@ import {
8
8
  anthropicMaxAttempts,
9
9
  anthropicRequestTimeoutMs,
10
10
  classifyError,
11
+ createStallRetryBudget,
11
12
  midstreamBackoffFor,
12
13
  sleepWithAbort,
14
+ STREAM_STALL_RETRY_BUDGET_MS,
13
15
  withRetry,
14
16
  retryAfterMsFromError,
15
17
  } from './retry-classifier.mjs';
@@ -249,6 +251,9 @@ export class AnthropicProvider {
249
251
  const MAX_MIDSTREAM_RETRIES = ANTHROPIC_MAX_MIDSTREAM_RETRIES;
250
252
  let firstAttemptError = null;
251
253
  let firstAttemptClassifier = null;
254
+ // Send-scoped stall window: in-place stall retries share one wall
255
+ // clock starting at the first stall (see createStallRetryBudget).
256
+ const stallRetryBudget = createStallRetryBudget();
252
257
 
253
258
  const buildReturnFromParse = (parseResult) => {
254
259
  const usageRaw = parseResult.usage?.raw || null;
@@ -315,25 +320,11 @@ export class AnthropicProvider {
315
320
  };
316
321
  };
317
322
 
318
- const recoverNonStreaming = async (midState, streamingError, streamController) => {
319
- const exposedChars = Number(midState?.emittedTextChars) || 0;
320
- if (!onTextReset || exposedChars <= 0
321
- || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
322
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
323
- throw streamingError;
324
- }
325
- let resetAccepted = false;
326
- try {
327
- resetAccepted = await onTextReset({
328
- chars: exposedChars,
329
- reason: 'anthropic-streaming-fallback',
330
- }) === true;
331
- } catch {}
332
- if (!resetAccepted) {
333
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
334
- throw streamingError;
335
- }
336
- try { streamController.abort?.(streamingError); } catch {}
323
+ // Core non-streaming re-issue shared by the exposed-text recovery
324
+ // (onTextReset-gated) and the no-exposure stall fallback (trivially
325
+ // safe — nothing was relayed or dispatched). Mirrors anthropic-oauth.
326
+ const issueNonStreamingFallback = async (streamController, abortReason) => {
327
+ try { streamController.abort?.(abortReason); } catch {}
337
328
  try { onStageChange?.('requesting', { transport: 'non-streaming-fallback' }); } catch {}
338
329
  const nonStreamingParams = { ...params, stream: false };
339
330
  const message = await withRetry(
@@ -357,6 +348,27 @@ export class AnthropicProvider {
357
348
  return buildReturnFromParse(normalizeAnthropicNonStreamingResponse(message, useModel));
358
349
  };
359
350
 
351
+ const recoverNonStreaming = async (midState, streamingError, streamController) => {
352
+ const exposedChars = Number(midState?.emittedTextChars) || 0;
353
+ if (!onTextReset || exposedChars <= 0
354
+ || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
355
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
356
+ throw streamingError;
357
+ }
358
+ let resetAccepted = false;
359
+ try {
360
+ resetAccepted = await onTextReset({
361
+ chars: exposedChars,
362
+ reason: 'anthropic-streaming-fallback',
363
+ }) === true;
364
+ } catch {}
365
+ if (!resetAccepted) {
366
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
367
+ throw streamingError;
368
+ }
369
+ return issueNonStreamingFallback(streamController, streamingError);
370
+ };
371
+
360
372
  try {
361
373
  for (let attemptIndex = 0; attemptIndex <= MAX_MIDSTREAM_RETRIES; attemptIndex++) {
362
374
  const streamController = createAbortController();
@@ -592,6 +604,29 @@ export class AnthropicProvider {
592
604
  continue;
593
605
  }
594
606
  const classifier = _classifyMidstreamError(err, midState);
607
+ // CC-parity stall recovery (ported from anthropic-oauth,
608
+ // 2026-08-03): a stalled stream that exposed NOTHING is
609
+ // re-issued non-streaming instead of retrying the same
610
+ // streaming shape into the same idle window. Replay is
611
+ // trivially safe — nothing was relayed or dispatched.
612
+ if (classifier === 'stream_stalled'
613
+ && _outcome?.replayUnsafe !== true
614
+ && !midState.emittedText
615
+ && !midState.emittedToolCall
616
+ && !midState.partialToolCall
617
+ && !midState.emittedThinking) {
618
+ try { process.stderr.write(`[${this.name}] stream stalled with no exposure — retrying non-streaming\n`); } catch {}
619
+ return await issueNonStreamingFallback(streamController, err);
620
+ }
621
+ if (classifier === 'stream_stalled' && !stallRetryBudget.allowStallRetry()) {
622
+ try {
623
+ process.stderr.write(
624
+ `[${this.name}] stall retry budget exhausted (${STREAM_STALL_RETRY_BUDGET_MS}ms since first stall) — surfacing for fresh-request retry\n`,
625
+ );
626
+ } catch {}
627
+ try { streamController.abort?.(err); } catch {}
628
+ throw err;
629
+ }
595
630
  if (classifier && attemptIndex < MAX_MIDSTREAM_RETRIES) {
596
631
  firstAttemptError = err;
597
632
  firstAttemptClassifier = classifier;
@@ -268,10 +268,27 @@ export function applyAnthropicCacheMarkers(sanitizedMessages, {
268
268
  const tailIdx = sanitizedMessages.length - 1;
269
269
  return hasUserText(sanitizedMessages[tailIdx]) ? tailIdx : -1;
270
270
  };
271
+ // True-tip anchor: when the request ends with a PERSISTED user text turn
272
+ // (the current prompt — it re-appears verbatim in every later request's
273
+ // prefix), mark it first. Without this, mid-session turn-first requests
274
+ // (multi-turn only; single-turn sessions are already covered by
275
+ // firstRequestUserPromptIdx) leave the fresh prompt unmarked: its tokens
276
+ // bill once at $5/M uncached, then again as a cache write when a later
277
+ // anchor advances past them — cost bounded by the prompt's size, so it
278
+ // matters for large pasted prompts. (2026-08-03 A/B note: session totals'
279
+ // totalUncachedInputTokens = input + cacheWrite by design, see
280
+ // uncachedInputTokensForProvider; billing-uncached input measured via
281
+ // usage.json was already ~0 on single-turn bench tasks before and after
282
+ // this change.) Synthetic system-reminder tails are excluded by
283
+ // hasUserText, so per-call volatile content still never keys the cache.
284
+ const currentTailUserIdx = () => {
285
+ const tailIdx = sanitizedMessages.length - 1;
286
+ return hasUserText(sanitizedMessages[tailIdx]) ? tailIdx : -1;
287
+ };
271
288
  if (messageTtl !== null) {
272
289
  const slots = Math.max(0, Math.min(4, Number(messageSlots) || 0));
273
290
  const marked = new Set();
274
- const candidates = [latestToolResultTailIdx(), previousUserTextAnchorIdx(), firstRequestUserPromptIdx()];
291
+ const candidates = [currentTailUserIdx(), latestToolResultTailIdx(), previousUserTextAnchorIdx(), firstRequestUserPromptIdx()];
275
292
  for (const idx of candidates) {
276
293
  if (slots <= 0) break;
277
294
  if (idx < 0 || marked.has(idx)) continue;
@@ -7,7 +7,10 @@
7
7
  * (scripts/openai-oauth-http-sse-toolcall-smoke.mjs) and fallback headers.
8
8
  */
9
9
  import { randomBytes } from 'crypto';
10
+ import zlib from 'node:zlib';
10
11
  import {
12
+ extractCacheWriteTokens,
13
+ extractCachedTokens,
11
14
  traceAgentFetch,
12
15
  traceAgentSse,
13
16
  traceAgentUsage,
@@ -47,6 +50,21 @@ const CODEX_REQUEST_MAX_RETRIES = 4;
47
50
  const CODEX_REQUEST_BACKOFF_MS = Object.freeze([200, 400, 800, 1600]);
48
51
  const CODEX_RETRY_JITTER_RATIO = 0.1;
49
52
 
53
+ // Request-body zstd gate (see sendViaHttpSse). Namespace import: on Node
54
+ // runtimes without zlib zstd bindings the named export would fail at module
55
+ // load, so the call site typeof-guards zlib.zstdCompressSync instead.
56
+ const OPENAI_REQ_ZSTD_MIN_BYTES = 8 * 1024;
57
+ let _openaiReqZstdLatch = false;
58
+ function _openaiReqZstdDisabled() {
59
+ return _openaiReqZstdLatch || process.env.MIXDOG_OPENAI_REQ_ZSTD === '0';
60
+ }
61
+ function _disableOpenaiReqZstd() { _openaiReqZstdLatch = true; }
62
+ function _zstdHeaders(headers, bodyForSend) {
63
+ return bodyForSend.encoding
64
+ ? { ...headers, 'Content-Encoding': bodyForSend.encoding }
65
+ : headers;
66
+ }
67
+
50
68
  export function _envFlag(name, fallback = true) {
51
69
  const raw = process.env[name];
52
70
  if (raw == null || raw === '') return fallback;
@@ -72,11 +90,6 @@ function _parseJsonObject(value) {
72
90
  }
73
91
  }
74
92
 
75
- function _extractCachedTokens(usage) {
76
- const details = usage?.input_tokens_details || usage?.prompt_tokens_details || {};
77
- return Number(details.cached_tokens ?? details.cached ?? usage?.cached_tokens ?? 0) || 0;
78
- }
79
-
80
93
  function _sseEventsFromBuffer(buffer) {
81
94
  const frames = [];
82
95
  let rest = buffer.replace(/\r\n/g, '\n');
@@ -231,6 +244,19 @@ export async function sendViaHttpSse({
231
244
  const responsesUrl = auth?.type === 'openai-direct'
232
245
  ? OPENAI_DIRECT_RESPONSES_URL
233
246
  : CODEX_RESPONSES_URL;
247
+ // Request-body zstd (codex parity: core client.rs
248
+ // responses_request_compression enables zstd for the codex backend on the
249
+ // OpenAI provider — the server decompresses Content-Encoding: zstd).
250
+ // openai-direct is excluded: only the codex backend is verified. Env
251
+ // kill-switch plus a process-wide latch flipped on the first 400 seen on
252
+ // a compressed request, which then replays that attempt uncompressed.
253
+ const _rawReqBytes = Buffer.from(JSON.stringify(body));
254
+ let _reqBodyForSend = auth?.type !== 'openai-direct'
255
+ && !_openaiReqZstdDisabled()
256
+ && typeof zlib.zstdCompressSync === 'function'
257
+ && _rawReqBytes.length >= OPENAI_REQ_ZSTD_MIN_BYTES
258
+ ? { bytes: zlib.zstdCompressSync(_rawReqBytes), encoding: 'zstd' }
259
+ : { bytes: _rawReqBytes, encoding: null };
234
260
  let response;
235
261
  for (let attempt = 0; attempt <= CODEX_REQUEST_MAX_RETRIES; attempt++) {
236
262
  const headerTimeout = createTimeoutSignal(
@@ -253,8 +279,8 @@ export async function sendViaHttpSse({
253
279
  } catch {}
254
280
  response = await fetchFn(responsesUrl, {
255
281
  method: 'POST',
256
- headers,
257
- body: JSON.stringify(body),
282
+ headers: _zstdHeaders(headers, _reqBodyForSend),
283
+ body: _reqBodyForSend.bytes,
258
284
  signal: headerTimeout.signal,
259
285
  dispatcher: getLlmDispatcher(),
260
286
  });
@@ -267,6 +293,15 @@ export async function sendViaHttpSse({
267
293
  headerTimeout.cleanup();
268
294
  }
269
295
 
296
+ // zstd rejection fallback: a 400 on a compressed request latches
297
+ // compression OFF process-wide and replays this attempt uncompressed.
298
+ if (response && response.status === 400 && _reqBodyForSend.encoding) {
299
+ _disableOpenaiReqZstd();
300
+ _reqBodyForSend = { bytes: _rawReqBytes, encoding: null };
301
+ await response.arrayBuffer().catch(() => {});
302
+ response = undefined;
303
+ continue;
304
+ }
270
305
  const retryableStatus = response && response.status >= 500 && response.status <= 599;
271
306
  // Typed transient transport failures only (errno / SDK connection
272
307
  // type). An unknown pre-response failure throws immediately instead of
@@ -783,7 +818,8 @@ export async function sendViaHttpSse({
783
818
  usage = {
784
819
  inputTokens: resp.usage.input_tokens || 0,
785
820
  outputTokens: resp.usage.output_tokens || 0,
786
- cachedTokens: _extractCachedTokens(resp.usage),
821
+ cachedTokens: extractCachedTokens(resp.usage),
822
+ cacheWriteTokens: extractCacheWriteTokens(resp.usage),
787
823
  promptTokens: resp.usage.input_tokens || 0,
788
824
  raw: serviceTier ? { ...resp.usage, service_tier: serviceTier } : resp.usage,
789
825
  };
@@ -36,9 +36,11 @@ import {
36
36
  classifyHandshakeError,
37
37
  classifyMidstreamError,
38
38
  createStreamSafetyStamps,
39
+ createStallRetryBudget,
39
40
  jitterDelayMs,
40
41
  MIDSTREAM_RETRY_POLICY,
41
42
  sleepWithAbort,
43
+ STREAM_STALL_RETRY_BUDGET_MS,
42
44
  } from './retry-classifier.mjs';
43
45
  import { stampStreamOutcome, STREAM_TRANSPORTS } from './lib/stream-outcome.mjs';
44
46
  import {
@@ -459,6 +461,9 @@ export async function sendViaWebSocket({
459
461
  const MAX_MIDSTREAM_RETRIES = MIDSTREAM_WS_TRANSIENT_RETRY_LIMIT;
460
462
  let firstAttemptError = null;
461
463
  let firstAttemptClassifier = null;
464
+ // Send-scoped stall window: in-place stall retries share one wall clock
465
+ // starting at the first stall (see createStallRetryBudget).
466
+ const stallRetryBudget = createStallRetryBudget();
462
467
  // A generate:false prewarm is billable even if its main request later
463
468
  // retries on a fresh socket or falls back to HTTP. Retain one completed
464
469
  // result across the whole logical send and attach it to terminal errors.
@@ -1006,6 +1011,11 @@ export async function sendViaWebSocket({
1006
1011
  const classifier = err?.unsafeToRetry === true
1007
1012
  ? null
1008
1013
  : _classifyMidstreamError(err, midState);
1014
+ if (classifier === 'stream_stalled' && !stallRetryBudget.allowStallRetry()) {
1015
+ try { process.stderr.write(`[openai-oauth] stall retry budget exhausted (${STREAM_STALL_RETRY_BUDGET_MS}ms since first stall) — surfacing for fresh-request retry\n`); } catch {}
1016
+ emitSendSpan('error');
1017
+ throw _stampTool(_stampLiveText(err));
1018
+ }
1009
1019
  const retryLimit = classifier ? _midstreamRetryLimit(classifier) : 0;
1010
1020
  if (classifier && attemptIndex < retryLimit) {
1011
1021
  // Retry-eligible: stash the first-attempt error, emit progress,
@@ -39,6 +39,9 @@ export function _combineUsageWithWarmup(actual, warmup, { separateMainContext =
39
39
  inputTokens: _usageNum(actual.inputTokens) + _usageNum(warmup.inputTokens),
40
40
  outputTokens: _usageNum(actual.outputTokens) + _usageNum(warmup.outputTokens),
41
41
  cachedTokens: _usageNum(actual.cachedTokens) + _usageNum(warmup.cachedTokens),
42
+ ...(actual.cacheWriteTokens != null || warmup.cacheWriteTokens != null
43
+ ? { cacheWriteTokens: _usageNum(actual.cacheWriteTokens) + _usageNum(warmup.cacheWriteTokens) }
44
+ : {}),
42
45
  promptTokens: _usageNum(actual.promptTokens) + _usageNum(warmup.promptTokens),
43
46
  warmupInputTokens: _usageNum(warmup.inputTokens),
44
47
  warmupCachedTokens: _usageNum(warmup.cachedTokens),