mixdog 0.9.92 → 0.9.94

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/README.md +147 -51
  2. package/package.json +6 -5
  3. package/scripts/code-graph-description-contract.mjs +6 -8
  4. package/scripts/tmp-cdp-errors.mjs +41 -0
  5. package/scripts/tmp-cdp-inspect.mjs +41 -0
  6. package/scripts/tui-transcript-jitter-harness.mjs +2 -18
  7. package/src/rules/agent/00-core.md +1 -2
  8. package/src/rules/agent/30-explorer.md +22 -16
  9. package/src/rules/lead/01-general.md +1 -0
  10. package/src/rules/shared/01-tool.md +27 -25
  11. package/src/runtime/agent/orchestrator/agent-runtime/cache-strategy.mjs +16 -3
  12. package/src/runtime/agent/orchestrator/agent-trace-format.mjs +22 -6
  13. package/src/runtime/agent/orchestrator/agent-trace.mjs +17 -0
  14. package/src/runtime/agent/orchestrator/context/collect.mjs +8 -3
  15. package/src/runtime/agent/orchestrator/providers/anthropic-effort.mjs +9 -1
  16. package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +79 -21
  17. package/src/runtime/agent/orchestrator/providers/anthropic.mjs +40 -19
  18. package/src/runtime/agent/orchestrator/providers/lib/anthropic-request-utils.mjs +18 -1
  19. package/src/runtime/agent/orchestrator/providers/oauth-usage.mjs +75 -15
  20. package/src/runtime/agent/orchestrator/providers/openai-oauth-http-sse.mjs +44 -8
  21. package/src/runtime/agent/orchestrator/providers/openai-ws-events.mjs +3 -0
  22. package/src/runtime/agent/orchestrator/providers/openai-ws-stream.mjs +2 -0
  23. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +22 -5
  24. package/src/runtime/agent/orchestrator/session/cache/scoped-cache.mjs +42 -2
  25. package/src/runtime/agent/orchestrator/session/eager-dispatch.mjs +35 -29
  26. package/src/runtime/agent/orchestrator/session/loop/stored-tool-args.mjs +11 -2
  27. package/src/runtime/agent/orchestrator/session/loop/tool-classify.mjs +5 -6
  28. package/src/runtime/agent/orchestrator/session/loop/tool-exec.mjs +31 -2
  29. package/src/runtime/agent/orchestrator/session/manager/compaction-runner.mjs +60 -0
  30. package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +60 -31
  31. package/src/runtime/agent/orchestrator/session/manager.mjs +1 -1
  32. package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +12 -3
  33. package/src/runtime/agent/orchestrator/session/store/listing.mjs +17 -0
  34. package/src/runtime/agent/orchestrator/session/store-summary-reader.mjs +101 -0
  35. package/src/runtime/agent/orchestrator/session/store.mjs +30 -0
  36. package/src/runtime/agent/orchestrator/session/tool-batch.mjs +119 -109
  37. package/src/runtime/agent/orchestrator/session/tool-result-offload.mjs +98 -3
  38. package/src/runtime/agent/orchestrator/stall-policy.mjs +31 -21
  39. package/src/runtime/agent/orchestrator/tools/bash-session.mjs +3 -3
  40. package/src/runtime/agent/orchestrator/tools/builtin/arg-guard.mjs +3 -3
  41. package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +8 -2
  42. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +16 -10
  43. package/src/runtime/agent/orchestrator/tools/builtin/fuzzy-match.mjs +12 -3
  44. package/src/runtime/agent/orchestrator/tools/builtin/grep-formatting.mjs +22 -0
  45. package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-context-expander.mjs +491 -0
  46. package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-output.mjs +91 -13
  47. package/src/runtime/agent/orchestrator/tools/builtin/list-tool.mjs +90 -27
  48. package/src/runtime/agent/orchestrator/tools/builtin/path-utils.mjs +6 -1
  49. package/src/runtime/agent/orchestrator/tools/builtin/read-batch.mjs +1 -1
  50. package/src/runtime/agent/orchestrator/tools/builtin/read-single-tool.mjs +19 -9
  51. package/src/runtime/agent/orchestrator/tools/builtin/read-streaming.mjs +7 -2
  52. package/src/runtime/agent/orchestrator/tools/builtin/read-tool.mjs +32 -4
  53. package/src/runtime/agent/orchestrator/tools/builtin/search-builders.mjs +16 -1
  54. package/src/runtime/agent/orchestrator/tools/builtin/search-tool.mjs +543 -19
  55. package/src/runtime/agent/orchestrator/tools/builtin/shell-analysis.mjs +5 -2
  56. package/src/runtime/agent/orchestrator/tools/builtin/shell-output.mjs +3 -3
  57. package/src/runtime/agent/orchestrator/tools/builtin/tool-output-limit.mjs +48 -0
  58. package/src/runtime/agent/orchestrator/tools/builtin.mjs +71 -1
  59. package/src/runtime/agent/orchestrator/tools/code-graph/build.mjs +4 -2
  60. package/src/runtime/agent/orchestrator/tools/code-graph/dispatch.mjs +51 -2
  61. package/src/runtime/agent/orchestrator/tools/code-graph/search-references.mjs +6 -17
  62. package/src/runtime/agent/orchestrator/tools/code-graph/search.mjs +2 -4
  63. package/src/runtime/agent/orchestrator/tools/code-graph-tool-defs.mjs +4 -3
  64. package/src/runtime/agent/orchestrator/tools/lib/pwsh-standby-pool.mjs +47 -14
  65. package/src/runtime/agent/orchestrator/tools/patch/dispatch.mjs +3 -3
  66. package/src/runtime/agent/orchestrator/tools/patch/orchestrator.mjs +3 -3
  67. package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +100 -0
  68. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +6 -5
  69. package/src/runtime/agent/orchestrator/tools/shell-command.mjs +9 -0
  70. package/src/runtime/agent/orchestrator/tools/shell-exec-output.mjs +1 -1
  71. package/src/runtime/agent/orchestrator/tools/shell-state.mjs +32 -2
  72. package/src/runtime/channels/backends/discord-gateway.mjs +6 -32
  73. package/src/runtime/channels/lib/inbound-handler.mjs +19 -3
  74. package/src/runtime/channels/lib/scheduler.mjs +51 -3
  75. package/src/runtime/channels/lib/worker-main.mjs +4 -0
  76. package/src/runtime/channels/tool-defs.mjs +4 -2
  77. package/src/runtime/memory/lib/query-handlers.mjs +11 -3
  78. package/src/runtime/memory/lib/tool-call-handler.mjs +16 -1
  79. package/src/runtime/memory/tool-defs.mjs +5 -5
  80. package/src/runtime/shared/background-tasks.mjs +10 -3
  81. package/src/runtime/shared/channel-notification-routing.mjs +8 -2
  82. package/src/runtime/shared/child-spawn-gate.mjs +50 -26
  83. package/src/runtime/shared/llm/http-agent.mjs +11 -0
  84. package/src/runtime/shared/task-notification-envelope.mjs +11 -2
  85. package/src/runtime/shared/tool-card-model.mjs +6 -2
  86. package/src/runtime/shared/tool-surface.mjs +7 -2
  87. package/src/session-runtime/lifecycle-api.mjs +26 -1
  88. package/src/session-runtime/provider-usage.mjs +26 -2
  89. package/src/session-runtime/tool-catalog-data.mjs +5 -2
  90. package/src/session-runtime/workflow.mjs +7 -5
  91. package/src/standalone/agent-tool/tag-registry.mjs +5 -1
  92. package/src/standalone/explore-tool.mjs +2 -2
  93. package/src/tui/app/use-transcript-window.mjs +7 -1
  94. package/src/tui/components/Spinner.jsx +18 -9
  95. package/src/tui/dist/index.mjs +179 -70
  96. package/src/tui/engine/agent-envelope.mjs +52 -3
  97. package/src/tui/engine/live-share.mjs +23 -3
  98. package/src/tui/engine/session-api.mjs +7 -0
  99. package/src/tui/engine/turn.mjs +85 -13
  100. package/src/tui/engine.mjs +32 -47
  101. package/src/tui/index.jsx +7 -0
  102. package/src/workflows/solo/WORKFLOW.md +0 -6
  103. package/src/workflows/solo-bench/WORKFLOW.md +17 -0
@@ -1,34 +1,36 @@
1
1
  # Tool Use
2
2
 
3
- - Before the first call, gather every known facet in one tool message; one
4
- shortest route per facet: broad/uncertain→`explore` (roles without it:
5
- `find`); partial path/name→`find`; verified root+wildcard→`glob`;
6
- quoted/non-identifier literal or regex→`grep`; exact code identifier/
7
- relation→`code_graph` before grep; known file/span→`read` directly without
8
- `grep`; verified directory→`list`; known edit→`apply_patch` (span already
9
- seen; else `read`/`grep` first); program/state change→`shell`; web/current
10
- external info→`search`.
11
- - Shortest total calls, maximum batching — every turn: all independent calls
12
- in one concurrent message (shell included); combine variants/symbols/
13
- scopes/paths/queries per call; same-file regions as one real
14
- `{path,offset,limit}` array; graph targets as arrays; `explore` facets in
15
- one `query[]` (max 8, no rephrased duplicates); all new edits in one patch.
16
- Distinct facets, not alternative routes; sequential singles only for a
17
- step whose arguments require the previous result — fixed follow-ups
18
- (pinned installs, known writes) go in the same batch; only apply_patch
19
- executes in order.
3
+ - Before the first call, gather every known facet — environment, capability,
4
+ artifact, failure checks — in one bounded tool message, one shortest route
5
+ per facet: broad/uncertain→`explore` (roles without it: `find`); known
6
+ name fragment→`find`; verified root+wildcard→`glob`; text/code→`grep`;
7
+ symbol body/relation→`code_graph`; known file/span→`read`, not `grep`;
8
+ verified directory→`list`; known edit→`apply_patch`; program/state
9
+ change→`shell`; web/current info→`search`.
10
+ - A turn is a plan, not a step: emit every already-determined call in one
11
+ concurrent message, merged per tool — one `shell` chain (`&&`/`;`), one
12
+ `read`, one `apply_patch` with verification in `post_shell`. In-message
13
+ order is guaranteed — edits land before the shell that checks them — so
14
+ produce and its check always ride one message, never a follow-up turn.
15
+ Distinct facets only — never two routes per facet. The archetype is two
16
+ turns — one message observes through the dedicated tools (`shell` beside
17
+ them, not instead of them), one chain produces and proves itself; a new
18
+ turn exists only at a true data dependency.
20
19
  - Verified paths: project root, session cwd, user-provided, tool-returned.
21
20
  `find` first for guessed path/name fragments; on ENOENT, find the basename.
22
21
  Retry `EXPLORATION_FAILED` once with changed tokens.
23
22
  - Stop when evidence covers the deliverable: a returned `path:line` or
24
- nonzero `content_with_context` result is final — act on it (inspecting it
25
- via read/code_graph is valid); only zero/error results justify changed
26
- tokens or scope. Don't re-locate, re-verify, or reread returned spans.
27
- - `apply_patch` is the primary edit tool: send the patch as soon as target
28
- path and new content are known. Hunk context comes verbatim from the newest
29
- tool output of that span (`read`/`grep`/your own patch — post-patch content
30
- after edits), never retyped from memory; one look-up beats a failed patch.
31
- A same-turn shell after `apply_patch` runs once the patch lands.
23
+ nonzero `content_with_context` result is final for its returned range. Read
24
+ is allowed for new/uncovered lines; do not call read when grep/read already
25
+ fully covers the requested range. Only zero/error results justify new scope.
26
+ - Verify in proportion to risk, appended to the producing chain (`shell`
27
+ tail or `post_shell`) — one decisive boundary probe covering its failure
28
+ modes. A pass is final — observed matching output IS the verification,
29
+ never re-checked in a later turn; on failure fix and rerun only what
30
+ failed. Optional diagnostics non-fatal; report verified vs assumed.
31
+ - `apply_patch` is the primary edit tool: once target path and new content are
32
+ known, include the patch in the current tool batch, hunk context verbatim
33
+ from the newest tool output of that span (post-patch content after edits).
32
34
  - After starting or receiving a background task, end the turn — its
33
35
  completion notification resumes the work. Never poll, sleep-loop, or block;
34
36
  explicit wait only for a result the current turn cannot proceed without.
@@ -120,18 +120,31 @@ export function resolveCacheStrategy(agent, { autoClear } = {}) {
120
120
  if (isOneShotMaintenanceAgent(agent)) {
121
121
  return { tools: 'none', system: 'none', tier3: 'none', messages: 'none' };
122
122
  }
123
+ // Operator override for the BP4 (messages-tail) TTL. Short-lived
124
+ // rapid-turn deployments (bench-style: session dies in <15min) never
125
+ // benefit from a 1h tail across its premium window, so '5m' trades the
126
+ // 2x write premium (1h, $10/M) down to 1.25x ($6.25/M) and deactivates
127
+ // the 1h volatile-content anchor guard. Product defaults below stay
128
+ // untouched when the env is unset.
129
+ const envMessagesTtl = (process.env.MIXDOG_CACHE_MESSAGES_TTL || '').trim();
130
+ const applyEnv = (strategy) => {
131
+ if (envMessagesTtl === '1h' || envMessagesTtl === '5m' || envMessagesTtl === 'none') {
132
+ return { ...strategy, messages: envMessagesTtl };
133
+ }
134
+ return strategy;
135
+ };
123
136
  if (getHiddenAgent(agent)) {
124
- return { tools: 'none', system: '1h', tier3: '1h', messages: '1h' };
137
+ return applyEnv({ tools: 'none', system: '1h', tier3: '1h', messages: '1h' });
125
138
  }
126
139
  if (agent && agent !== 'lead') {
127
140
  // Public (non-hidden, non-lead) agents keep the flat 1h tail — only
128
141
  // the Lead session's tail is linked to autoClear.
129
- return { tools: 'none', system: '1h', tier3: '1h', messages: '1h' };
142
+ return applyEnv({ tools: 'none', system: '1h', tier3: '1h', messages: '1h' });
130
143
  }
131
144
  // Lead session (agent === 'lead', or no agent — raw/CLI callers default
132
145
  // to Lead behavior): message tail TTL is linked to autoClear (see
133
146
  // resolveLeadMessagesTtl).
134
- return { tools: 'none', system: '1h', tier3: '1h', messages: resolveLeadMessagesTtl(autoClear) };
147
+ return applyEnv({ tools: 'none', system: '1h', tier3: '1h', messages: resolveLeadMessagesTtl(autoClear) });
135
148
  }
136
149
 
137
150
  /**
@@ -1,6 +1,6 @@
1
1
  import { createHash } from 'crypto';
2
2
  import { countJsonNextCalls } from './tools/next-call-utils.mjs';
3
- import { splitGrepLinePrefix } from './tools/builtin/grep-formatting.mjs';
3
+ import { parseGrepContextHeader, splitGrepLinePrefix } from './tools/builtin/grep-formatting.mjs';
4
4
  import {
5
5
  appendAgentTrace,
6
6
  normalizeSessionId,
@@ -236,7 +236,27 @@ export function parseGrepCoverage(resultText, toolName, toolArgs, resultKind) {
236
236
  const out = [];
237
237
  const seen = new Set();
238
238
  let sectionPath = null;
239
+ let rawSourceLinesRemaining = 0;
240
+ const addLine = (path, lineNo) => {
241
+ if (!path || !Number.isInteger(lineNo) || lineNo < 1 || out.length >= GREP_COVERAGE_MAX) return;
242
+ const key = `${path}\0${lineNo}`;
243
+ if (seen.has(key)) return;
244
+ seen.add(key);
245
+ out.push({ path: String(path).replace(/\\/g, '/'), line: lineNo });
246
+ };
239
247
  for (const line of String(resultText ?? '').split(/\r?\n/)) {
248
+ if (rawSourceLinesRemaining > 0) {
249
+ rawSourceLinesRemaining--;
250
+ continue;
251
+ }
252
+ const header = parseGrepContextHeader(line);
253
+ if (header) {
254
+ for (let lineNo = header.startLine; lineNo <= header.endLine && out.length < GREP_COVERAGE_MAX; lineNo++) {
255
+ addLine(header.path, lineNo);
256
+ }
257
+ rawSourceLinesRemaining = header.sourceLineCount;
258
+ continue;
259
+ }
240
260
  const section = line.match(/^# grep (.+)$/);
241
261
  if (section) {
242
262
  if (!section[1].startsWith('pattern:')) sectionPath = section[1];
@@ -252,11 +272,7 @@ export function parseGrepCoverage(resultText, toolName, toolArgs, resultKind) {
252
272
  const path = split?.path || (omitted ? toolArgs.path : null) || (sectionOmitted ? sectionPath : null);
253
273
  const lineNo = split?.lineNo || (omitted ? Number(omitted[1]) : null)
254
274
  || (sectionOmitted ? Number(sectionOmitted[1]) : null);
255
- if (!path || !Number.isInteger(lineNo) || lineNo < 1) continue;
256
- const key = `${path}\0${lineNo}`;
257
- if (seen.has(key)) continue;
258
- seen.add(key);
259
- out.push({ path: String(path).replace(/\\/g, '/'), line: lineNo });
275
+ addLine(path, lineNo);
260
276
  if (out.length >= GREP_COVERAGE_MAX) break;
261
277
  }
262
278
  return out.length ? out : null;
@@ -38,6 +38,22 @@ function extractCachedTokens(usage) {
38
38
  return 0;
39
39
  }
40
40
 
41
+ function extractCacheWriteTokens(usage) {
42
+ const candidates = [
43
+ usage?.input_tokens_details?.cache_write_tokens,
44
+ usage?.prompt_tokens_details?.cache_write_tokens,
45
+ usage?.inputTokensDetails?.cacheWriteTokens,
46
+ usage?.promptTokensDetails?.cacheWriteTokens,
47
+ usage?.cache_write_tokens,
48
+ usage?.cacheWriteTokens,
49
+ ];
50
+ for (const value of candidates) {
51
+ const n = Number(value);
52
+ if (Number.isFinite(n)) return n;
53
+ }
54
+ return 0;
55
+ }
56
+
41
57
  // Lightweight fingerprint of the conversation prefix. Hashes the first 4096
42
58
  // characters of JSON.stringify(messages) — enough to detect prefix mutation
43
59
  // across iterations (which invalidates the provider prompt cache) without
@@ -267,6 +283,7 @@ export {
267
283
  appendAgentTrace,
268
284
  drainAgentTrace,
269
285
  estimateProviderPayloadBytes,
286
+ extractCacheWriteTokens,
270
287
  extractCachedTokens,
271
288
  messagePrefixHash,
272
289
  traceAgentFetch,
@@ -275,7 +275,7 @@ export function buildSkillToolEnvelope(name, content, skillDir) {
275
275
  };
276
276
  }
277
277
 
278
- function compactSkillManifestText(value, max = 180) {
278
+ function compactSkillManifestText(value, max = 100) {
279
279
  const text = String(value || '').replace(/\s+/g, ' ').trim();
280
280
  return text.length > max ? `${text.slice(0, Math.max(1, max - 3))}...` : text;
281
281
  }
@@ -374,6 +374,11 @@ function sanitizeMcpInstructionText(text, max = MCP_INSTRUCTION_MAX_CHARS) {
374
374
  /**
375
375
  * Per-server MCP initialize instructions for deferred-pool tools only.
376
376
  * Empty when no instructions or no matching deferred MCP tools → omit block.
377
+ * Emits ONLY the server heading + instruction body: the per-server tool names
378
+ * are deliberately NOT repeated here — every pool tool is already listed once
379
+ * (with its description) in <available-deferred-tools>, and re-listing ~30
380
+ * names per server doubled the MCP share of the BP1 prefix (2026-08-05 audit).
381
+ * Server membership stays evident from the mcp__<server>__ name prefix.
377
382
  */
378
383
  function buildMcpInstructionsManifest(mcpServerInstructions, poolNames) {
379
384
  const map = mcpServerInstructions && typeof mcpServerInstructions === 'object'
@@ -400,8 +405,7 @@ function buildMcpInstructionsManifest(mcpServerInstructions, poolNames) {
400
405
  for (const server of servers) {
401
406
  const safeServer = sanitizeMcpManifestServerName(server);
402
407
  const body = sanitizeMcpInstructionText(map[server]);
403
- const tools = [...toolsByServer.get(server)].sort((a, b) => a.localeCompare(b));
404
- lines.push(`## ${safeServer}`, body, ...tools.map((tool) => `- ${tool}`));
408
+ lines.push(`## ${safeServer}`, body);
405
409
  }
406
410
  lines.push('</mcp-instructions>');
407
411
  return lines.join('\n');
@@ -528,6 +532,7 @@ export function buildSkillToolDefs(skills, { ownerIsAgentSession = false } = {})
528
532
  name: { type: 'string', description: 'Skill name' },
529
533
  },
530
534
  required: ['name'],
535
+ additionalProperties: false,
531
536
  },
532
537
  },
533
538
  ];
@@ -208,7 +208,15 @@ export function applyAnthropicEffortToBody(
208
208
  // modelSupportsEffort() allowlist so older models never receive it.
209
209
  // Set unconditionally (independent of `normalized`) so effort-capable
210
210
  // turns always carry adaptive thinking + round-trip signatures.
211
- body.thinking = { type: 'adaptive', display: 'summarized' };
211
+ // MIXDOG_ANTHROPIC_THINKING_DISPLAY=omitted (operator/bench knob):
212
+ // CC-parity mode — no thinking blocks come back, so nothing is
213
+ // replayed into later requests (saves the 1h cache-write + re-read on
214
+ // accumulated thinking) at the cost of losing visible reasoning and
215
+ // cross-iteration thinking continuity. Default stays summarized.
216
+ const display = (process.env.MIXDOG_ANTHROPIC_THINKING_DISPLAY || '').trim() === 'omitted'
217
+ ? 'omitted'
218
+ : 'summarized';
219
+ body.thinking = { type: 'adaptive', display };
212
220
  // Adaptive/4.7+ models reject any non-default sampling param with a 400.
213
221
  delete body.temperature;
214
222
  delete body.top_p;
@@ -64,6 +64,18 @@ import {
64
64
  stampAnthropicStreamOutcome,
65
65
  } from './anthropic-sse.mjs';
66
66
  import { buildAnthropicBetaHeaders, supportsAnthropicFastMode } from './anthropic-betas.mjs';
67
+ import { gzipSync } from 'node:zlib';
68
+
69
+ // Request-body gzip gate (see the fetch site below). Env kill-switch
70
+ // (MIXDOG_ANTHROPIC_REQ_GZIP=0) plus a process-wide latch flipped on the
71
+ // first 400 response to a compressed request. Small bodies skip compression:
72
+ // below ~8KB the CPU + header cost outweighs the upload saving.
73
+ const ANTHROPIC_REQ_GZIP_MIN_BYTES = 8 * 1024;
74
+ let _anthropicReqGzipLatch = false;
75
+ function _anthropicReqGzipDisabled() {
76
+ return _anthropicReqGzipLatch || process.env.MIXDOG_ANTHROPIC_REQ_GZIP === '0';
77
+ }
78
+ function _disableAnthropicReqGzip() { _anthropicReqGzipLatch = true; }
67
79
  import {
68
80
  applyAnthropicEffortToBody,
69
81
  effortValuesForModel,
@@ -654,7 +666,16 @@ export class AnthropicOAuthProvider {
654
666
  // provider-visible cache breakpoint off the cached one — the
655
667
  // exact COLD-turn bug this change fixes. Order is fixed:
656
668
  // build → sanitize (once) → mark → JSON.stringify.
657
- const response = await fetch(API_URL, {
669
+ // Request-body gzip (probe-verified 2026-08-04: /v1/messages
670
+ // returns 200 for Content-Encoding: gzip, 400 for zstd). Large
671
+ // turn bodies (system prompt + history, typically 50-100KB+)
672
+ // compress ~5-10x, trimming upload time off every call's
673
+ // header wait. Latch OFF process-wide on the first 400 seen on
674
+ // a compressed request and retry that attempt uncompressed, so
675
+ // a server-side behavior change can never wedge the session.
676
+ const rawBody = Buffer.from(JSON.stringify(requestBody));
677
+ const useGzip = !_anthropicReqGzipDisabled() && rawBody.length >= ANTHROPIC_REQ_GZIP_MIN_BYTES;
678
+ const sendAttempt = (gz) => fetch(API_URL, {
658
679
  method: 'POST',
659
680
  headers: {
660
681
  'Authorization': `Bearer ${accessToken}`,
@@ -669,11 +690,18 @@ export class AnthropicOAuthProvider {
669
690
  'user-agent': `claude-cli/${resolveCliVersion()} (external, sdk-cli)`,
670
691
  'x-app': 'cli',
671
692
  'Content-Type': 'application/json',
693
+ ...(gz ? { 'Content-Encoding': 'gzip' } : {}),
672
694
  },
673
- body: JSON.stringify(requestBody),
695
+ body: gz ? gzipSync(rawBody) : rawBody,
674
696
  signal: controller.signal,
675
697
  dispatcher: getLlmDispatcher(),
676
698
  });
699
+ let response = await sendAttempt(useGzip);
700
+ if (useGzip && response.status === 400) {
701
+ _disableAnthropicReqGzip();
702
+ try { await response.arrayBuffer(); } catch { /* drain best-effort */ }
703
+ response = await sendAttempt(false);
704
+ }
677
705
 
678
706
  traceAgentFetch({
679
707
  sessionId,
@@ -780,25 +808,13 @@ export class AnthropicOAuthProvider {
780
808
  // clock starting at the first stall (see createStallRetryBudget).
781
809
  const stallRetryBudget = createStallRetryBudget();
782
810
 
783
- const recoverNonStreaming = async (midState, streamingError, controller) => {
784
- const exposedChars = Number(midState?.emittedTextChars) || 0;
785
- if (!onTextReset || exposedChars <= 0
786
- || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
787
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
788
- throw streamingError;
789
- }
790
- let resetAccepted = false;
791
- try {
792
- resetAccepted = await onTextReset({
793
- chars: exposedChars,
794
- reason: 'anthropic-streaming-fallback',
795
- }) === true;
796
- } catch {}
797
- if (!resetAccepted) {
798
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
799
- throw streamingError;
800
- }
801
- try { controller?.abort?.(streamingError); } catch {}
811
+ // Core non-streaming re-issue: abort the dead stream and repeat the
812
+ // SAME request with stream:false. Shared by the exposed-text recovery
813
+ // (which must first get the owner's onTextReset acknowledgement) and
814
+ // the no-exposure stall fallback below (trivially safe — nothing was
815
+ // relayed or dispatched, so there is nothing to withdraw or replay).
816
+ const issueNonStreamingFallback = async (controller, abortReason) => {
817
+ try { controller?.abort?.(abortReason); } catch {}
802
818
  try { onStageChange?.('requesting', { transport: 'non-streaming-fallback' }); } catch {}
803
819
  let fallback = await requestWithRetry(creds.accessToken, { ...body, stream: false });
804
820
  if (fallback.response.status === 401) {
@@ -825,6 +841,27 @@ export class AnthropicOAuthProvider {
825
841
  }
826
842
  };
827
843
 
844
+ const recoverNonStreaming = async (midState, streamingError, controller) => {
845
+ const exposedChars = Number(midState?.emittedTextChars) || 0;
846
+ if (!onTextReset || exposedChars <= 0
847
+ || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
848
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
849
+ throw streamingError;
850
+ }
851
+ let resetAccepted = false;
852
+ try {
853
+ resetAccepted = await onTextReset({
854
+ chars: exposedChars,
855
+ reason: 'anthropic-streaming-fallback',
856
+ }) === true;
857
+ } catch {}
858
+ if (!resetAccepted) {
859
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
860
+ throw streamingError;
861
+ }
862
+ return issueNonStreamingFallback(controller, streamingError);
863
+ };
864
+
828
865
  try {
829
866
  for (let attemptIndex = 0; attemptIndex <= MAX_MIDSTREAM_RETRIES; attemptIndex++) {
830
867
  let response, controller, cancelHandler;
@@ -1057,6 +1094,27 @@ export class AnthropicOAuthProvider {
1057
1094
  continue;
1058
1095
  }
1059
1096
  const classifier = _classifyMidstreamError(err, midState);
1097
+ // CC-parity stall recovery (2026-08-03 v3 postmortem): a
1098
+ // stalled stream that exposed NOTHING (no text/thinking
1099
+ // relayed, no tool emitted) is re-issued NON-STREAMING instead
1100
+ // of retrying the same streaming shape. Effort-mode models can
1101
+ // legitimately think in silence past any streaming idle
1102
+ // window; an in-place streaming retry re-runs the same silent
1103
+ // generation into the same timer (observed live: deterministic
1104
+ // 4×~138s beheading, ~552s per turn), while the non-streaming
1105
+ // transport simply waits for the full body (bounded by
1106
+ // PROVIDER_NONSTREAM_TOTAL_TIMEOUT_MS). Claude Code does
1107
+ // exactly this on its watchdog aborts. Replay is trivially
1108
+ // safe here — nothing was relayed or dispatched.
1109
+ if (classifier === 'stream_stalled'
1110
+ && _outcome?.replayUnsafe !== true
1111
+ && !midState.emittedText
1112
+ && !midState.emittedToolCall
1113
+ && !midState.partialToolCall
1114
+ && !midState.emittedThinking) {
1115
+ try { process.stderr.write('[anthropic-oauth] stream stalled with no exposure — retrying non-streaming\n'); } catch {}
1116
+ return await issueNonStreamingFallback(controller, err);
1117
+ }
1060
1118
  if (classifier === 'stream_stalled' && !stallRetryBudget.allowStallRetry()) {
1061
1119
  try { process.stderr.write(`[anthropic-oauth] stall retry budget exhausted (${STREAM_STALL_RETRY_BUDGET_MS}ms since first stall) — surfacing for fresh-request retry\n`); } catch {}
1062
1120
  try { controller?.abort?.(err); } catch { /* best-effort teardown */ }
@@ -320,25 +320,11 @@ export class AnthropicProvider {
320
320
  };
321
321
  };
322
322
 
323
- const recoverNonStreaming = async (midState, streamingError, streamController) => {
324
- const exposedChars = Number(midState?.emittedTextChars) || 0;
325
- if (!onTextReset || exposedChars <= 0
326
- || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
327
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
328
- throw streamingError;
329
- }
330
- let resetAccepted = false;
331
- try {
332
- resetAccepted = await onTextReset({
333
- chars: exposedChars,
334
- reason: 'anthropic-streaming-fallback',
335
- }) === true;
336
- } catch {}
337
- if (!resetAccepted) {
338
- try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
339
- throw streamingError;
340
- }
341
- try { streamController.abort?.(streamingError); } catch {}
323
+ // Core non-streaming re-issue shared by the exposed-text recovery
324
+ // (onTextReset-gated) and the no-exposure stall fallback (trivially
325
+ // safe — nothing was relayed or dispatched). Mirrors anthropic-oauth.
326
+ const issueNonStreamingFallback = async (streamController, abortReason) => {
327
+ try { streamController.abort?.(abortReason); } catch {}
342
328
  try { onStageChange?.('requesting', { transport: 'non-streaming-fallback' }); } catch {}
343
329
  const nonStreamingParams = { ...params, stream: false };
344
330
  const message = await withRetry(
@@ -362,6 +348,27 @@ export class AnthropicProvider {
362
348
  return buildReturnFromParse(normalizeAnthropicNonStreamingResponse(message, useModel));
363
349
  };
364
350
 
351
+ const recoverNonStreaming = async (midState, streamingError, streamController) => {
352
+ const exposedChars = Number(midState?.emittedTextChars) || 0;
353
+ if (!onTextReset || exposedChars <= 0
354
+ || midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
355
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
356
+ throw streamingError;
357
+ }
358
+ let resetAccepted = false;
359
+ try {
360
+ resetAccepted = await onTextReset({
361
+ chars: exposedChars,
362
+ reason: 'anthropic-streaming-fallback',
363
+ }) === true;
364
+ } catch {}
365
+ if (!resetAccepted) {
366
+ try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
367
+ throw streamingError;
368
+ }
369
+ return issueNonStreamingFallback(streamController, streamingError);
370
+ };
371
+
365
372
  try {
366
373
  for (let attemptIndex = 0; attemptIndex <= MAX_MIDSTREAM_RETRIES; attemptIndex++) {
367
374
  const streamController = createAbortController();
@@ -597,6 +604,20 @@ export class AnthropicProvider {
597
604
  continue;
598
605
  }
599
606
  const classifier = _classifyMidstreamError(err, midState);
607
+ // CC-parity stall recovery (ported from anthropic-oauth,
608
+ // 2026-08-03): a stalled stream that exposed NOTHING is
609
+ // re-issued non-streaming instead of retrying the same
610
+ // streaming shape into the same idle window. Replay is
611
+ // trivially safe — nothing was relayed or dispatched.
612
+ if (classifier === 'stream_stalled'
613
+ && _outcome?.replayUnsafe !== true
614
+ && !midState.emittedText
615
+ && !midState.emittedToolCall
616
+ && !midState.partialToolCall
617
+ && !midState.emittedThinking) {
618
+ try { process.stderr.write(`[${this.name}] stream stalled with no exposure — retrying non-streaming\n`); } catch {}
619
+ return await issueNonStreamingFallback(streamController, err);
620
+ }
600
621
  if (classifier === 'stream_stalled' && !stallRetryBudget.allowStallRetry()) {
601
622
  try {
602
623
  process.stderr.write(
@@ -268,10 +268,27 @@ export function applyAnthropicCacheMarkers(sanitizedMessages, {
268
268
  const tailIdx = sanitizedMessages.length - 1;
269
269
  return hasUserText(sanitizedMessages[tailIdx]) ? tailIdx : -1;
270
270
  };
271
+ // True-tip anchor: when the request ends with a PERSISTED user text turn
272
+ // (the current prompt — it re-appears verbatim in every later request's
273
+ // prefix), mark it first. Without this, mid-session turn-first requests
274
+ // (multi-turn only; single-turn sessions are already covered by
275
+ // firstRequestUserPromptIdx) leave the fresh prompt unmarked: its tokens
276
+ // bill once at $5/M uncached, then again as a cache write when a later
277
+ // anchor advances past them — cost bounded by the prompt's size, so it
278
+ // matters for large pasted prompts. (2026-08-03 A/B note: session totals'
279
+ // totalUncachedInputTokens = input + cacheWrite by design, see
280
+ // uncachedInputTokensForProvider; billing-uncached input measured via
281
+ // usage.json was already ~0 on single-turn bench tasks before and after
282
+ // this change.) Synthetic system-reminder tails are excluded by
283
+ // hasUserText, so per-call volatile content still never keys the cache.
284
+ const currentTailUserIdx = () => {
285
+ const tailIdx = sanitizedMessages.length - 1;
286
+ return hasUserText(sanitizedMessages[tailIdx]) ? tailIdx : -1;
287
+ };
271
288
  if (messageTtl !== null) {
272
289
  const slots = Math.max(0, Math.min(4, Number(messageSlots) || 0));
273
290
  const marked = new Set();
274
- const candidates = [latestToolResultTailIdx(), previousUserTextAnchorIdx(), firstRequestUserPromptIdx()];
291
+ const candidates = [currentTailUserIdx(), latestToolResultTailIdx(), previousUserTextAnchorIdx(), firstRequestUserPromptIdx()];
275
292
  for (const idx of candidates) {
276
293
  if (slots <= 0) break;
277
294
  if (idx < 0 || marked.has(idx)) continue;
@@ -18,6 +18,10 @@ const FETCH_TIMEOUT_MS = 4500;
18
18
  const WARN_TTL_MS = 5 * 60_000;
19
19
  const CODEX_RESET_CREDITS_URL = 'https://chatgpt.com/backend-api/wham/rate-limit-reset-credits';
20
20
  const CODEX_RESET_CONSUME_URL = `${CODEX_RESET_CREDITS_URL}/consume`;
21
+ // Redeeming a reset credit is an explicit user action, not a poll: it gets the
22
+ // generous budget the orca client uses (REDEEM_BACKEND_TIMEOUT_MS) so a slow
23
+ // backend cannot abort a request the server is already applying.
24
+ const CODEX_REDEEM_TIMEOUT_MS = 30_000;
21
25
 
22
26
  const memoryCache = new Map();
23
27
  const inflight = new Map();
@@ -99,6 +103,38 @@ try {
99
103
  // Embedded runtimes may not expose process lifecycle hooks.
100
104
  }
101
105
 
106
+ /** Drops every cached usage snapshot of one provider (memory, queued disk
107
+ * writes and the persisted routes). A mutation that changes quota state
108
+ * server-side — redeeming a Codex reset credit — must not keep serving the
109
+ * pre-mutation meters from a 60s/10min cache. */
110
+ export function invalidateOAuthUsageSnapshots(provider) {
111
+ const providerOnly = String(provider || '').toLowerCase();
112
+ if (!providerOnly) return;
113
+ const routePrefix = `${providerOnly}\u0001`;
114
+ const owned = (key) => key === providerOnly || String(key).startsWith(routePrefix);
115
+ for (const key of [...memoryCache.keys()]) {
116
+ if (owned(key)) memoryCache.delete(key);
117
+ }
118
+ for (const key of [...pendingDiskSnapshots.keys()]) {
119
+ if (owned(key)) pendingDiskSnapshots.delete(key);
120
+ }
121
+ try {
122
+ updateJsonAtomicSync(cachePath(), (curRaw) => {
123
+ const cur = curRaw && typeof curRaw === 'object' ? curRaw : {};
124
+ const routes = cur.routes && typeof cur.routes === 'object' ? cur.routes : {};
125
+ return {
126
+ version: 1,
127
+ updatedAt: Date.now(),
128
+ routes: Object.fromEntries(
129
+ Object.entries(routes).filter(([key]) => !owned(key)),
130
+ ),
131
+ };
132
+ }, { compact: true, fsync: false, fsyncDir: false });
133
+ } catch {
134
+ // Usage display must never break the reset path.
135
+ }
136
+ }
137
+
102
138
  function isContentfulSnapshot(snapshot) {
103
139
  return !!snapshot
104
140
  && typeof snapshot === 'object'
@@ -253,11 +289,15 @@ function normalizeOpenAICodexResetCredits(data, accountId = '') {
253
289
  .filter((value) => Number.isFinite(value) && value > 0);
254
290
  const nextExpiresAt = resetAtMs(data.next_expires_at ?? data.nextExpiresAt)
255
291
  || (expiryCandidates.length ? Math.min(...expiryCandidates) : null);
292
+ // Identity of the OFFER, not of one payload shape: the detail endpoint and
293
+ // the counts embedded in /wham/usage describe the same credits with
294
+ // different fields, so hashing the raw rows made the same offer produce two
295
+ // revisions — and the desktop scopes its durable idempotency key by
296
+ // revision. Count + soonest expiry is what a user is offered.
256
297
  const offerRevision = `v1:${createHash('sha256').update(JSON.stringify({
257
298
  accountId,
258
299
  availableCount,
259
300
  nextExpiresAt,
260
- credits,
261
301
  })).digest('hex')}`;
262
302
  return {
263
303
  availableCount,
@@ -290,6 +330,32 @@ function codexResetOutcome(code) {
290
330
  throw new Error(`Unknown Codex reset outcome: ${cleanString(code) || 'missing'}`);
291
331
  }
292
332
 
333
+ async function postOpenAICodexResetConsume(auth, idempotencyKey) {
334
+ return await fetch(CODEX_RESET_CONSUME_URL, {
335
+ ...fetchOptions({
336
+ ...codexHeaders(auth),
337
+ 'Content-Type': 'application/json',
338
+ }, CODEX_REDEEM_TIMEOUT_MS),
339
+ method: 'POST',
340
+ body: JSON.stringify({ redeem_request_id: idempotencyKey }),
341
+ });
342
+ }
343
+
344
+ async function redeemOpenAICodexResetCredit(auth, idempotencyKey) {
345
+ // A transport failure (abort, dropped socket) leaves the outcome unknown
346
+ // while the credit may already be spent. redeem_request_id makes the request
347
+ // idempotent, so ONE replay turns that unknown into the server's real answer
348
+ // instead of reporting "could not be confirmed" over a consumed credit.
349
+ let response;
350
+ try {
351
+ response = await postOpenAICodexResetConsume(auth, idempotencyKey);
352
+ } catch {
353
+ response = await postOpenAICodexResetConsume(auth, idempotencyKey);
354
+ }
355
+ if (!response.ok) throw new Error(`Codex reset failed: HTTP ${response.status}`);
356
+ return codexResetOutcome((await response.json())?.code);
357
+ }
358
+
293
359
  export async function consumeOpenAICodexResetCredit(providerObj, options = {}) {
294
360
  const expectedOfferRevision = cleanString(options?.expectedOfferRevision);
295
361
  const idempotencyKey = cleanString(options?.idempotencyKey);
@@ -301,20 +367,14 @@ export async function consumeOpenAICodexResetCredit(providerObj, options = {}) {
301
367
  }
302
368
  const auth = await resolveOpenAICodexAuth(providerObj);
303
369
  if (!auth) throw new Error('Codex is not signed in');
304
- const current = await fetchOpenAICodexResetCreditsWithAuth(auth);
305
- if (!current || current.availableCount < 1 || current.offerRevision !== expectedOfferRevision) {
306
- return { status: 'offerChanged', resetCredits: current };
307
- }
308
- const response = await fetch(CODEX_RESET_CONSUME_URL, {
309
- ...fetchOptions({
310
- ...codexHeaders(auth),
311
- 'Content-Type': 'application/json',
312
- }, 15_000),
313
- method: 'POST',
314
- body: JSON.stringify({ redeem_request_id: idempotencyKey }),
315
- });
316
- if (!response.ok) throw new Error(`Codex reset failed: HTTP ${response.status}`);
317
- const outcome = codexResetOutcome((await response.json())?.code);
370
+ // The SERVER decides the outcome (orca parity): redeem_request_id makes the
371
+ // call idempotent and `already_redeemed`/`no_credit` are real answers. The
372
+ // old client-side offer gate ran before every attempt, so retrying an
373
+ // unconfirmed redeem — whose credit was already spent, hence a changed
374
+ // revision — could only ever report "offer changed" and never the truth.
375
+ const outcome = await redeemOpenAICodexResetCredit(auth, idempotencyKey);
376
+ // Quota meters just changed server-side; cached snapshots are now wrong.
377
+ invalidateOAuthUsageSnapshots('openai-oauth');
318
378
  const resetCredits = await fetchOpenAICodexResetCreditsWithAuth(auth).catch(() => null);
319
379
  return { outcome, resetCredits };
320
380
  }