mixdog 0.9.94 → 0.9.95

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (161) hide show
  1. package/NOTICE.md +72 -0
  2. package/package.json +16 -6
  3. package/scripts/lib/isolated-root-cleanup.mjs +19 -0
  4. package/scripts/run-suite.mjs +100 -0
  5. package/src/lib/rules-builder.cjs +4 -3
  6. package/src/output-styles/detailed.md +13 -15
  7. package/src/output-styles/extreme-minimal.md +7 -11
  8. package/src/output-styles/minimal.md +4 -7
  9. package/src/output-styles/simple.md +11 -13
  10. package/src/rules/lead/01-general.md +1 -2
  11. package/src/rules/lead/lead-brief.md +4 -5
  12. package/src/rules/lead/lead-tool.md +3 -2
  13. package/src/rules/shared/01-tool.md +28 -34
  14. package/src/runtime/agent/orchestrator/agent-runtime/cache-strategy.mjs +2 -2
  15. package/src/runtime/agent/orchestrator/agent-trace-format.mjs +1 -1
  16. package/src/runtime/agent/orchestrator/agent-trace-io.mjs +4 -0
  17. package/src/runtime/agent/orchestrator/agent-trace.mjs +29 -0
  18. package/src/runtime/agent/orchestrator/mcp/client.mjs +3 -3
  19. package/src/runtime/agent/orchestrator/providers/anthropic-effort.mjs +97 -7
  20. package/src/runtime/agent/orchestrator/providers/anthropic-model-resolve.mjs +11 -4
  21. package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +2 -3
  22. package/src/runtime/agent/orchestrator/providers/anthropic.mjs +1 -1
  23. package/src/runtime/agent/orchestrator/providers/codex-client-meta.mjs +4 -6
  24. package/src/runtime/agent/orchestrator/providers/lib/stream-outcome.mjs +1 -1
  25. package/src/runtime/agent/orchestrator/providers/openai-codex-metadata.mjs +8 -8
  26. package/src/runtime/agent/orchestrator/providers/openai-oauth-http-sse.mjs +7 -8
  27. package/src/runtime/agent/orchestrator/providers/openai-oauth-ws.mjs +2 -2
  28. package/src/runtime/agent/orchestrator/providers/openai-responses-payload.mjs +14 -16
  29. package/src/runtime/agent/orchestrator/providers/openai-ws-headers.mjs +5 -4
  30. package/src/runtime/agent/orchestrator/providers/openai-ws-pool.mjs +10 -10
  31. package/src/runtime/agent/orchestrator/providers/openai-ws-stream.mjs +2 -3
  32. package/src/runtime/agent/orchestrator/providers/registry.mjs +51 -4
  33. package/src/runtime/agent/orchestrator/providers/retry-classifier.mjs +15 -15
  34. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +12 -32
  35. package/src/runtime/agent/orchestrator/session/cache/read-cache.mjs +7 -0
  36. package/src/runtime/agent/orchestrator/session/context-compaction-policy.mjs +10 -3
  37. package/src/runtime/agent/orchestrator/session/context-utils.mjs +10 -11
  38. package/src/runtime/agent/orchestrator/session/eager-dispatch.mjs +1 -0
  39. package/src/runtime/agent/orchestrator/session/loop/compact-policy.mjs +3 -3
  40. package/src/runtime/agent/orchestrator/session/loop/completion-guards.mjs +0 -14
  41. package/src/runtime/agent/orchestrator/session/loop/stop-hooks.mjs +9 -8
  42. package/src/runtime/agent/orchestrator/session/manager/ask-session.mjs +7 -4
  43. package/src/runtime/agent/orchestrator/session/manager/context-meta.mjs +1 -1
  44. package/src/runtime/agent/orchestrator/session/manager/idle-cleanup.mjs +1 -1
  45. package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +40 -13
  46. package/src/runtime/agent/orchestrator/session/manager/session-close.mjs +3 -0
  47. package/src/runtime/agent/orchestrator/session/manager/session-lifecycle.mjs +8 -1
  48. package/src/runtime/agent/orchestrator/session/manager/turn-interruption.mjs +5 -9
  49. package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +3 -3
  50. package/src/runtime/agent/orchestrator/session/tool-batch.mjs +3 -6
  51. package/src/runtime/agent/orchestrator/session/tool-result-offload.mjs +1 -1
  52. package/src/runtime/agent/orchestrator/stall-policy.mjs +2 -3
  53. package/src/runtime/agent/orchestrator/tools/builtin/arg-guard.mjs +22 -0
  54. package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +6 -1
  55. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +1 -1
  56. package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-context-expander.mjs +1 -1
  57. package/src/runtime/agent/orchestrator/tools/builtin/path-utils.mjs +14 -0
  58. package/src/runtime/agent/orchestrator/tools/builtin/read-constants.mjs +4 -4
  59. package/src/runtime/agent/orchestrator/tools/builtin/read-image-resize.mjs +2 -2
  60. package/src/runtime/agent/orchestrator/tools/builtin/read-snapshot-runtime.mjs +2 -1
  61. package/src/runtime/agent/orchestrator/tools/builtin/read-special-files.mjs +3 -3
  62. package/src/runtime/agent/orchestrator/tools/builtin/rg-runner.mjs +1 -1
  63. package/src/runtime/agent/orchestrator/tools/builtin/search-tool.mjs +3 -4
  64. package/src/runtime/agent/orchestrator/tools/builtin/shell-analysis.mjs +3 -3
  65. package/src/runtime/agent/orchestrator/tools/builtin/shell-job-paths.mjs +12 -0
  66. package/src/runtime/agent/orchestrator/tools/builtin/shell-job-spawn.mjs +10 -3
  67. package/src/runtime/agent/orchestrator/tools/builtin/shell-jobs.mjs +35 -5
  68. package/src/runtime/agent/orchestrator/tools/builtin/snapshot-store.mjs +98 -0
  69. package/src/runtime/agent/orchestrator/tools/builtin.mjs +2 -2
  70. package/src/runtime/agent/orchestrator/tools/code-graph/build.mjs +3 -1
  71. package/src/runtime/agent/orchestrator/tools/code-graph/dispatch.mjs +53 -12
  72. package/src/runtime/agent/orchestrator/tools/code-graph/project-root.mjs +47 -2
  73. package/src/runtime/agent/orchestrator/tools/code-graph/trusted-roots.mjs +3 -1
  74. package/src/runtime/agent/orchestrator/tools/env-scrub.mjs +9 -2
  75. package/src/runtime/agent/orchestrator/tools/patch/dispatch.mjs +2 -2
  76. package/src/runtime/agent/orchestrator/tools/patch/matcher.mjs +1 -1
  77. package/src/runtime/agent/orchestrator/tools/patch/native-server.mjs +57 -2
  78. package/src/runtime/agent/orchestrator/tools/patch/orchestrator.mjs +148 -15
  79. package/src/runtime/agent/orchestrator/tools/patch/parsing.mjs +5 -1
  80. package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +109 -13
  81. package/src/runtime/agent/orchestrator/tools/patch-manifest.json +10 -10
  82. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +4 -5
  83. package/src/runtime/agent/orchestrator/tools/shell-command.mjs +4 -0
  84. package/src/runtime/channels/backends/discord.mjs +5 -14
  85. package/src/runtime/channels/backends/telegram.mjs +0 -5
  86. package/src/runtime/channels/lib/inbound-handler.mjs +0 -1
  87. package/src/runtime/channels/lib/output-forwarder.mjs +24 -5
  88. package/src/runtime/channels/lib/scheduler.mjs +1 -1
  89. package/src/runtime/channels/lib/worker-main.mjs +1 -1
  90. package/src/runtime/media/renditions.mjs +21 -2
  91. package/src/runtime/memory/tool-defs.mjs +6 -6
  92. package/src/runtime/shared/atomic-file.mjs +53 -0
  93. package/src/runtime/shared/background-tasks.mjs +6 -3
  94. package/src/runtime/shared/resource-admission.mjs +40 -1
  95. package/src/runtime/shared/tool-status.mjs +10 -1
  96. package/src/runtime/shared/tool-surface.mjs +7 -7
  97. package/src/runtime/shared/turn-snapshot.mjs +385 -21
  98. package/src/runtime/shared/turn-worktree-snapshot.mjs +540 -0
  99. package/src/session-runtime/context-status.mjs +9 -3
  100. package/src/session-runtime/lifecycle-api.mjs +35 -5
  101. package/src/session-runtime/mcp-glue.mjs +11 -6
  102. package/src/session-runtime/provider-models.mjs +94 -43
  103. package/src/session-runtime/runtime-core.mjs +46 -8
  104. package/src/session-runtime/runtime-tunables.mjs +4 -0
  105. package/src/session-runtime/self-update.mjs +33 -4
  106. package/src/session-runtime/session-lifecycle.mjs +2 -0
  107. package/src/session-runtime/session-text.mjs +2 -1
  108. package/src/session-runtime/session-turn-api.mjs +106 -22
  109. package/src/session-runtime/workflow.mjs +11 -4
  110. package/src/standalone/agent-tool.mjs +4 -4
  111. package/src/standalone/backend-daemon.mjs +570 -0
  112. package/src/standalone/channel-daemon-transport.mjs +141 -1
  113. package/src/standalone/channel-worker.mjs +3 -2
  114. package/src/standalone/engine-daemon-client.mjs +894 -0
  115. package/src/standalone/engine-daemon-local-bridge.mjs +20 -0
  116. package/src/standalone/engine-daemon-protocol.mjs +33 -0
  117. package/src/standalone/engine-daemon-service.mjs +864 -0
  118. package/src/standalone/engine-daemon-transport.mjs +603 -0
  119. package/src/tui/App.jsx +62 -47
  120. package/src/tui/app/app-format.mjs +4 -2
  121. package/src/tui/app/app-view.jsx +5 -0
  122. package/src/tui/app/channel-pickers.mjs +7 -6
  123. package/src/tui/app/core-memory-picker.mjs +4 -4
  124. package/src/tui/app/extension-pickers.mjs +20 -18
  125. package/src/tui/app/maintenance-pickers.mjs +27 -27
  126. package/src/tui/app/onboarding-steps.mjs +23 -18
  127. package/src/tui/app/prompt-submit.mjs +13 -2
  128. package/src/tui/app/route-pickers.mjs +13 -8
  129. package/src/tui/app/settings-picker.mjs +55 -58
  130. package/src/tui/app/slash-dispatch.mjs +32 -28
  131. package/src/tui/app/transcript-window.mjs +19 -0
  132. package/src/tui/app/usage-context-panels.mjs +22 -7
  133. package/src/tui/app/use-mouse-input.mjs +25 -3
  134. package/src/tui/app/use-prompt-queue-history.mjs +31 -16
  135. package/src/tui/app/use-transcript-scroll.mjs +30 -8
  136. package/src/tui/app/use-transcript-window.mjs +15 -0
  137. package/src/tui/app/use-welcome-prompt-hint.mjs +2 -2
  138. package/src/tui/components/PromptInput.jsx +10 -0
  139. package/src/tui/components/Spinner.jsx +89 -95
  140. package/src/tui/components/TextEntryPanel.jsx +14 -0
  141. package/src/tui/components/ToolExecution.jsx +2 -2
  142. package/src/tui/dist/index.mjs +1534 -9801
  143. package/src/tui/engine/agent-job-feed.mjs +2 -2
  144. package/src/tui/engine/live-share.mjs +76 -3
  145. package/src/tui/engine/session-api-ext.mjs +23 -1
  146. package/src/tui/engine/session-api.mjs +81 -41
  147. package/src/tui/engine/session-flow.mjs +43 -4
  148. package/src/tui/engine/tool-card-results.mjs +6 -0
  149. package/src/tui/engine/turn.mjs +8 -1
  150. package/src/tui/engine-local-session.mjs +1108 -0
  151. package/src/tui/engine.mjs +16 -1065
  152. package/src/tui/index.jsx +40 -2
  153. package/src/tui/markdown/stream-fence.mjs +1 -1
  154. package/src/tui/spinner-meta.mjs +80 -0
  155. package/src/tui/spinner-verbs.mjs +35 -0
  156. package/src/ui/statusline-segments.mjs +43 -10
  157. package/src/ui/statusline.mjs +10 -1
  158. package/scripts/tmp-cdp-errors.mjs +0 -41
  159. package/scripts/tmp-cdp-inspect.mjs +0 -41
  160. package/src/runtime/agent/orchestrator/session/loop/steering-ladder.mjs +0 -176
  161. package/src/standalone/channel-daemon.mjs +0 -226
@@ -17,6 +17,67 @@ function _sortEffortLevels(levels) {
17
17
  return [...levels].sort((a, b) => rank(a) - rank(b));
18
18
  }
19
19
 
20
+ // Control flags that ride the capabilities.effort map alongside real levels
21
+ // (`{supported:true, low:true, …}`). They are NOT selectable effort levels;
22
+ // letting them through minted a bogus "supported" level that got persisted
23
+ // into the on-disk catalog and offered in the UI.
24
+ const CONTROL_EFFORT_KEYS = new Set(['supported', 'enabled', 'default']);
25
+
26
+ // ── Catalog-first capability index ───────────────────────────────────────
27
+ // The provider catalog already carries what each model advertises
28
+ // (capabilities.effort → reasoningOptions:[{type:'effort',values:[…]}]), so it
29
+ // — not a hardcoded regex ladder — is the source of truth for what we put on
30
+ // the wire. anthropic-model-resolve feeds this whenever the in-memory catalog
31
+ // moves. The regex ladder below stays as the fallback for ids the catalog does
32
+ // not know (offline start, api-key provider, first run before /v1/models).
33
+ const _catalogEffortLevels = new Map();
34
+
35
+ function _catalogKeys(model) {
36
+ const id = normalizeModelId(model);
37
+ if (!id) return [];
38
+ // A dated id and its bare version alias describe the SAME model — index
39
+ // both so `claude-opus-5` and `claude-opus-5-20260101` resolve alike.
40
+ const undated = id.replace(/-\d{8}$/, '');
41
+ return undated !== id ? [id, undated] : [id];
42
+ }
43
+
44
+ /**
45
+ * Replace the catalog-derived capability index. Records without a
46
+ * `reasoningOptions` array are treated as UNKNOWN (skipped, so the regex
47
+ * fallback still applies) rather than as "no effort" — offline/static fallback
48
+ * lists carry no capability data and must not disable effort.
49
+ */
50
+ export function setModelEffortCapabilities(models) {
51
+ _catalogEffortLevels.clear();
52
+ if (!Array.isArray(models)) return;
53
+ for (const model of models) {
54
+ if (!model?.id || !Array.isArray(model.reasoningOptions)) continue;
55
+ const option = model.reasoningOptions.find(
56
+ (entry) => String(entry?.type || '').trim().toLowerCase() === 'effort',
57
+ );
58
+ const levels = new Set(
59
+ (Array.isArray(option?.values) ? option.values : [])
60
+ .map((value) => String(value || '').trim().toLowerCase())
61
+ .filter((value) => value && !CONTROL_EFFORT_KEYS.has(value)),
62
+ );
63
+ for (const key of _catalogKeys(model.id)) {
64
+ const existing = _catalogEffortLevels.get(key);
65
+ // A bare alias shared by several records keeps the richer entry.
66
+ if (existing?.size && levels.size === 0) continue;
67
+ _catalogEffortLevels.set(key, levels);
68
+ }
69
+ }
70
+ }
71
+
72
+ /** Advertised levels for `model`, or null when the catalog has no record. */
73
+ function _catalogLevels(model) {
74
+ for (const key of _catalogKeys(model)) {
75
+ const levels = _catalogEffortLevels.get(key);
76
+ if (levels) return levels;
77
+ }
78
+ return null;
79
+ }
80
+
20
81
  export const EFFORT_BETA_HEADER = 'effort-2025-11-24';
21
82
 
22
83
  export const LEGACY_EFFORT_BUDGET = Object.freeze({
@@ -40,7 +101,10 @@ function parseClaudeVersion(model) {
40
101
  if (triple) {
41
102
  return { family: triple[1], major: Number(triple[2]), minor: Number(triple[3]) };
42
103
  }
43
- const pair = m.match(/^claude-(sonnet|fable)-(\d+)(?:$|[-@])/);
104
+ // Bare version ids carry family + major only (claude-opus-5). opus/haiku
105
+ // were missing here, so `claude-opus-5` parsed to null and fell through
106
+ // every capability check as an unknown shape.
107
+ const pair = m.match(/^claude-(sonnet|fable|opus|haiku)-(\d+)(?:$|[-@])/);
44
108
  if (pair) {
45
109
  return { family: pair[1], major: Number(pair[2]), minor: null };
46
110
  }
@@ -64,6 +128,9 @@ function isLegacyAnthropicReasoningModel(model) {
64
128
  if (parsed.family === 'sonnet' || parsed.family === 'opus') {
65
129
  if (parsed.major < 4) return true;
66
130
  if (parsed.major === 4 && parsed.minor !== null && parsed.minor < 6) return true;
131
+ // Bare major, no minor (claude-opus-4 / claude-sonnet-4): adaptive
132
+ // effort only ships from 5 up, so 4.x aliases stay manual-thinking.
133
+ if (parsed.minor === null && parsed.major < 5) return true;
67
134
  return false;
68
135
  }
69
136
  return false;
@@ -72,13 +139,22 @@ function isLegacyAnthropicReasoningModel(model) {
72
139
  // @[MODEL LAUNCH]: extend allowlist when new Claude models ship with effort support.
73
140
  export function modelSupportsEffort(model) {
74
141
  if (isEnvTruthy(process.env.MIXDOG_ANTHROPIC_ALWAYS_ENABLE_EFFORT)) return true;
142
+ const advertised = _catalogLevels(model);
143
+ if (advertised) return advertised.size > 0;
144
+ return _regexSupportsEffort(model);
145
+ }
146
+
147
+ // Fallback ladder for ids the catalog does not carry. Also used by
148
+ // effortValuesForModel, which BUILDS the catalog records and therefore must
149
+ // never consult the index it feeds.
150
+ function _regexSupportsEffort(model) {
75
151
  const m = normalizeModelId(model);
76
152
  if (!m.includes('claude')) return false;
77
153
  if (isLegacyAnthropicReasoningModel(model)) return false;
78
154
  if (m.includes('opus-4-6') || m.includes('sonnet-4-6')) return true;
79
155
  if (m.includes('sonnet-5') || m.includes('fable-5')) return true;
80
156
  if (/^claude-opus-4-(6|7|8)(?:$|[-@])/.test(m)) return true;
81
- if (/^claude-opus-5-/.test(m)) return true;
157
+ if (/^claude-opus-5(?:$|[-@])/.test(m)) return true;
82
158
  // Fallthrough for not-yet-enumerated modern models: only grant effort when
83
159
  // parseClaudeVersion resolves a family with major>=4 (and not a manual-
84
160
  // thinking legacy already excluded above). A bare `startsWith('claude-')`
@@ -98,6 +174,12 @@ export function modelSupportsEffort(model) {
98
174
  // supported by: Fable 5, Mythos 5, Opus 4.8, Opus 4.7, Sonnet 5.
99
175
  // NOT Opus 4.6 / Sonnet 4.6 (those support max but not xhigh).
100
176
  export function modelSupportsXhighEffort(model) {
177
+ const advertised = _catalogLevels(model);
178
+ if (advertised) return advertised.has('xhigh');
179
+ return _regexSupportsXhighEffort(model);
180
+ }
181
+
182
+ function _regexSupportsXhighEffort(model) {
101
183
  const m = normalizeModelId(model);
102
184
  if (/^claude-opus-4-(7|8)(?:$|[-@])/.test(m)) return true;
103
185
  if (/^claude-opus-5(?:$|[-@])/.test(m)) return true;
@@ -111,7 +193,14 @@ export function modelSupportsXhighEffort(model) {
111
193
  // @[MODEL LAUNCH]: extend when new Opus models support max effort.
112
194
  // Max list = xhigh list PLUS Opus 4.6, Sonnet 4.6, Opus 4.5.
113
195
  export function modelSupportsMaxEffort(model) {
114
- if (modelSupportsXhighEffort(model)) return true;
196
+ const advertised = _catalogLevels(model);
197
+ // xhigh support implies max support (kept from the ladder below).
198
+ if (advertised) return advertised.has('max') || advertised.has('xhigh');
199
+ return _regexSupportsMaxEffort(model);
200
+ }
201
+
202
+ function _regexSupportsMaxEffort(model) {
203
+ if (_regexSupportsXhighEffort(model)) return true;
115
204
  const m = normalizeModelId(model);
116
205
  if (/^claude-opus-4-(5|6)(?:$|[-@])/.test(m)) return true;
117
206
  if (/^claude-sonnet-4-6(?:$|[-@])/.test(m)) return true;
@@ -209,7 +298,7 @@ export function applyAnthropicEffortToBody(
209
298
  // Set unconditionally (independent of `normalized`) so effort-capable
210
299
  // turns always carry adaptive thinking + round-trip signatures.
211
300
  // MIXDOG_ANTHROPIC_THINKING_DISPLAY=omitted (operator/bench knob):
212
- // CC-parity mode — no thinking blocks come back, so nothing is
301
+ // thinking blocks are omitted entirely, so nothing is
213
302
  // replayed into later requests (saves the 1h cache-write + re-read on
214
303
  // accumulated thinking) at the cost of losing visible reasoning and
215
304
  // cross-iteration thinking continuity. Default stays summarized.
@@ -254,7 +343,8 @@ export function effortValuesForModel(capabilities, modelId) {
254
343
  // advertises (xhigh, max, or anything future), no hardcoded allowlist.
255
344
  if (effort !== true && typeof effort === 'object') {
256
345
  const advertised = Object.keys(effort).filter(
257
- (level) => effort[level] === true || effort[level]?.supported === true,
346
+ (level) => !CONTROL_EFFORT_KEYS.has(String(level).trim().toLowerCase())
347
+ && (effort[level] === true || effort[level]?.supported === true),
258
348
  );
259
349
  if (advertised.length) return _sortEffortLevels(advertised);
260
350
  // Object with no per-level flags: fall through to the boolean/supported
@@ -265,10 +355,10 @@ export function effortValuesForModel(capabilities, modelId) {
265
355
  // the model supports effort but doesn't enumerate levels, so derive the
266
356
  // set from the known level ladder, gated by the top-tier model check.
267
357
  let levels = [...EFFORT_LEVELS];
268
- if (!modelSupportsMaxEffort(modelId)) {
358
+ if (!_regexSupportsMaxEffort(modelId)) {
269
359
  levels = levels.filter((level) => level !== 'max');
270
360
  }
271
- if (!modelSupportsXhighEffort(modelId)) {
361
+ if (!_regexSupportsXhighEffort(modelId)) {
272
362
  levels = levels.filter((level) => level !== 'xhigh');
273
363
  }
274
364
  return levels;
@@ -8,20 +8,22 @@
8
8
  import { enrichModels } from './model-catalog.mjs';
9
9
  import { sanitizeModelList } from './model-list-sanitize.mjs';
10
10
  import { makeModelCache } from './model-cache.mjs';
11
- import { effortValuesForModel } from './anthropic-effort.mjs';
11
+ import { effortValuesForModel, setModelEffortCapabilities } from './anthropic-effort.mjs';
12
12
 
13
13
  // Disk-backed cache so repeated process starts (cron, tool calls) don't
14
14
  // hammer /v1/models. 24h TTL matches the upstream client cadence.
15
15
  const MODEL_CACHE_TTL_MS = 24 * 60 * 60_000;
16
16
  // Bump when the on-disk cache shape changes so stale-shape entries are
17
17
  // discarded instead of misread.
18
- const ANTHROPIC_MODEL_CACHE_SCHEMA_VERSION = 1;
18
+ // v2: effort level lists no longer carry the `supported` control flag as a
19
+ // selectable level, so v1 caches are discarded rather than replayed.
20
+ const ANTHROPIC_MODEL_CACHE_SCHEMA_VERSION = 2;
19
21
 
20
22
  const _modelCache = makeModelCache({
21
23
  fileName: 'anthropic-oauth-models.json',
22
24
  ttlMs: MODEL_CACHE_TTL_MS,
23
25
  version: ANTHROPIC_MODEL_CACHE_SCHEMA_VERSION,
24
- onSave: (m) => { _inMemoryCatalog = Array.isArray(m) ? m.slice() : null; },
26
+ onSave: (m) => { _setInMemoryCatalog(m); },
25
27
  });
26
28
 
27
29
  // Async wrappers so callers can keep awaiting; the shared cache CRUD is sync.
@@ -42,6 +44,9 @@ let _inMemoryCatalog = null;
42
44
  // listModels() warm path, so expose a setter instead of the raw binding.
43
45
  export function _setInMemoryCatalog(models) {
44
46
  _inMemoryCatalog = Array.isArray(models) ? models.slice() : null;
47
+ // The request builder resolves effort capability from the catalog, so the
48
+ // capability index moves with the mirror — one write, one source of truth.
49
+ setModelEffortCapabilities(_inMemoryCatalog || []);
45
50
  }
46
51
 
47
52
  export function _catalogHas(id) {
@@ -185,7 +190,9 @@ export function _catalogOutputTokens(model) {
185
190
  try {
186
191
  if (!Array.isArray(_inMemoryCatalog)) {
187
192
  const cached = _modelCache.loadSync();
188
- if (Array.isArray(cached)) _inMemoryCatalog = cached.slice();
193
+ // Route the lazy warm through the setter: a direct assignment left
194
+ // the effort-capability index empty even though the mirror was warm.
195
+ if (Array.isArray(cached)) _setInMemoryCatalog(cached);
189
196
  }
190
197
  if (!Array.isArray(_inMemoryCatalog)) return null;
191
198
  const entry = _inMemoryCatalog.find(m => m?.id === model);
@@ -1094,7 +1094,7 @@ export class AnthropicOAuthProvider {
1094
1094
  continue;
1095
1095
  }
1096
1096
  const classifier = _classifyMidstreamError(err, midState);
1097
- // CC-parity stall recovery (2026-08-03 v3 postmortem): a
1097
+ // Stall recovery (2026-08-03 v3 postmortem): a
1098
1098
  // stalled stream that exposed NOTHING (no text/thinking
1099
1099
  // relayed, no tool emitted) is re-issued NON-STREAMING instead
1100
1100
  // of retrying the same streaming shape. Effort-mode models can
@@ -1103,8 +1103,7 @@ export class AnthropicOAuthProvider {
1103
1103
  // generation into the same timer (observed live: deterministic
1104
1104
  // 4×~138s beheading, ~552s per turn), while the non-streaming
1105
1105
  // transport simply waits for the full body (bounded by
1106
- // PROVIDER_NONSTREAM_TOTAL_TIMEOUT_MS). Claude Code does
1107
- // exactly this on its watchdog aborts. Replay is trivially
1106
+ // PROVIDER_NONSTREAM_TOTAL_TIMEOUT_MS). Replay is trivially
1108
1107
  // safe here — nothing was relayed or dispatched.
1109
1108
  if (classifier === 'stream_stalled'
1110
1109
  && _outcome?.replayUnsafe !== true
@@ -604,7 +604,7 @@ export class AnthropicProvider {
604
604
  continue;
605
605
  }
606
606
  const classifier = _classifyMidstreamError(err, midState);
607
- // CC-parity stall recovery (ported from anthropic-oauth,
607
+ // Stall recovery (shared with anthropic-oauth,
608
608
  // 2026-08-03): a stalled stream that exposed NOTHING is
609
609
  // re-issued non-streaming instead of retrying the same
610
610
  // streaming shape into the same idle window. Replay is
@@ -2,13 +2,11 @@
2
2
  * codex-client-meta.mjs — codex client identity headers for the OpenAI OAuth
3
3
  * transports.
4
4
  *
5
- * codex-rs sends two client-identity headers on EVERY request including the
6
- * WS handshake:
5
+ * The reference client sends two client-identity headers on EVERY request,
6
+ * including the WS handshake:
7
7
  * - `User-Agent: codex_cli_rs/<version> (<os> <ver>; <arch>) <terminal>`
8
- * (login/src/auth/default_client.rs get_codex_user_agent + default_headers)
9
8
  * - `version: <CARGO_PKG_VERSION>` via the built-in provider http_headers
10
- * (model-provider-info/src/lib.rs:339-343), merged into the WS handshake by
11
- * merge_request_headers (codex-api/src/endpoint/responses_websocket.rs).
9
+ * merged into the WS handshake.
12
10
  * The backend uses these for client gating (model catalog visibility measured
13
11
  * 2026-07-03) and plausibly for x-codex-turn-state issuance / sticky
14
12
  * cache-node routing, so mixdog mirrors both.
@@ -17,7 +15,7 @@ import os from 'node:os';
17
15
 
18
16
  // Offline fallback only; live value refreshes from npm (24h TTL, in-process).
19
17
  // The backend gates model exposure AND per-request model access on the client
20
- // version (gpt-5.6-* require >= 0.144.0 per codex models-manager/models.json,
18
+ // version (gpt-5.6-* require >= 0.144.0 per the published model catalog,
21
19
  // verified 2026-07-09), so keep this at the current release when bumping.
22
20
  const CODEX_CLIENT_VERSION_FLOOR = '0.144.1';
23
21
  const VERSION_TTL_MS = 24 * 60 * 60_000;
@@ -50,7 +50,7 @@
50
50
  * || toolCallsComplete > 0 || toolCallsDispatched > 0
51
51
  * sideEffectDispatched = toolCallsDispatched > 0
52
52
  * replayUnsafe = the ONE architecture-specific deny MixDog adds on top
53
- * of the Codex retry rules, because it streams UI text
53
+ * of the standard retry rules, because it streams UI text
54
54
  * and dispatches tools eagerly:
55
55
  * visibleOutput || sideEffectDispatched
56
56
  * || dispatchAmbiguous || unsafe marker
@@ -35,8 +35,8 @@ function _codexInstallationId(sendOpts) {
35
35
  || `mixdog-${_hashText(`${process.env.USERPROFILE || process.env.HOME || ''}:${process.cwd()}`, 32)}`;
36
36
  }
37
37
 
38
- // The identity block codex rebuilds per request (responses_metadata.rs
39
- // client_metadata()): never cached on the pooled socket, or a later turn would
38
+ // The identity block is rebuilt per request: never cached on the pooled
39
+ // socket, or a later turn would
40
40
  // replay the first turn's identity.
41
41
  function _codexMetadataBase(entry, { poolKey, cacheKey, sendOpts, handshake = false } = {}) {
42
42
  const sessionId = _cleanMetaString(sendOpts?.codexSessionId || sendOpts?.session?.codexSessionId || poolKey || cacheKey)
@@ -49,8 +49,8 @@ function _codexMetadataBase(entry, { poolKey, cacheKey, sendOpts, handshake = fa
49
49
  : _sessionStartedAtUnixMs(sessionId);
50
50
  const requestKind = _codexRequestKind(sendOpts, sessionId);
51
51
  const wireParity = process.env.MIXDOG_OAI_CODEX_WIRE_PARITY === '1';
52
- // codex opens the WS with a prewarm (empty turn_id) BEFORE the real turn
53
- // (client.rs). Under wire parity the handshake IS that prewarm, so its
52
+ // The reference client opens the WS with a prewarm (empty turn_id) BEFORE
53
+ // the real turn. Under wire parity the handshake IS that prewarm, so its
54
54
  // turn_id empties and its request_kind becomes 'prewarm' instead of
55
55
  // presenting the handshake as a live turn. Parity off is unchanged.
56
56
  const isPrewarm = requestKind === 'prewarm' || handshake === true;
@@ -66,7 +66,7 @@ function _codexMetadataBase(entry, { poolKey, cacheKey, sendOpts, handshake = fa
66
66
  turn_id: turnId,
67
67
  window_id: windowId,
68
68
  request_kind: effectiveRequestKind,
69
- // Richer codex turn-metadata (responses_metadata.rs:264-280). A/B
69
+ // Richer turn-metadata. A/B
70
70
  // 2026-07-04 showed no effect, so it stays behind a knob for future
71
71
  // probes; wire parity implies it.
72
72
  ...((process.env.MIXDOG_OAI_TURN_METADATA_RICH === '1' || wireParity) ? {
@@ -98,8 +98,8 @@ export function _metadataTrace(metadata) {
98
98
  };
99
99
  }
100
100
 
101
- // Handshake projection of the same identity. codex attaches these on every
102
- // request (client.rs:582-584, responses_metadata.rs:227-252); A/B 2026-07-04
101
+ // Handshake projection of the same identity. These ride on every
102
+ // request; A/B 2026-07-04
103
103
  // showed the turn-metadata blob alone lifts prefix-cache hits, so it is ON by
104
104
  // default. MIXDOG_OAI_TURN_METADATA overrides:
105
105
  // unset|1|turn-metadata : window-id + turn-metadata + installation-id
@@ -141,7 +141,7 @@ export function _withCodexWsClientMetadata(frame, entry, enabled, context = {})
141
141
  'x-codex-ws-stream-request-start-ms': String(Date.now()),
142
142
  };
143
143
  if (entry && typeof entry === 'object') {
144
- // codex scopes x-codex-turn-state to ONE turn (client.rs:263-279) while
144
+ // x-codex-turn-state is scoped to ONE turn while
145
145
  // pooled sockets span turns: attribute a captured token to the FIRST
146
146
  // turn that observes it, then drop it once turn_id moves on. An empty
147
147
  // turn_id (parity prewarm) is a valid owner, so the check is against
@@ -126,8 +126,8 @@ function _isMaxOutputIncompleteReason(reason) {
126
126
  return /^(?:max_output_tokens|max_tokens|length|output_token_limit)$/i.test(String(reason || '').trim());
127
127
  }
128
128
 
129
- // Wire-level `end_turn` on a terminal Responses frame (codex-rs
130
- // codex-api/src/sse/responses.rs ResponseCompleted.end_turn: Option<bool>).
129
+ // Wire-level `end_turn` on a terminal Responses frame: an optional boolean on
130
+ // the completed response.
131
131
  // Optional by contract: only a real boolean normalizes; a missing/non-boolean
132
132
  // field stays undefined so absence is never collapsed into false.
133
133
  export function _endTurnFromEvent(event) {
@@ -178,9 +178,9 @@ function _buildOpenAIHttpFallbackHeaders({ auth, cacheKey }) {
178
178
  };
179
179
  if (cacheKey) {
180
180
  const sid = String(cacheKey);
181
- // Codex-native anchors (see openai-ws-pool _buildHandshakeHeaders):
182
- // `session-id`/`thread-id` (hyphen) match codex-rs headers.rs; legacy
183
- // underscore `session_id` kept for backward compat.
181
+ // Backend-native anchors (see openai-ws-pool _buildHandshakeHeaders):
182
+ // the hyphenated `session-id`/`thread-id` pair; legacy underscore
183
+ // `session_id` kept for backward compat.
184
184
  headers.session_id = sid;
185
185
  headers['session-id'] = sid;
186
186
  headers['thread-id'] = sid;
@@ -244,9 +244,8 @@ export async function sendViaHttpSse({
244
244
  const responsesUrl = auth?.type === 'openai-direct'
245
245
  ? OPENAI_DIRECT_RESPONSES_URL
246
246
  : CODEX_RESPONSES_URL;
247
- // Request-body zstd (codex parity: core client.rs
248
- // responses_request_compression enables zstd for the codex backend on the
249
- // OpenAI provider — the server decompresses Content-Encoding: zstd).
247
+ // Request-body zstd: compression is enabled for the codex backend on the
248
+ // OpenAI provider — the server decompresses Content-Encoding: zstd.
250
249
  // openai-direct is excluded: only the codex backend is verified. Env
251
250
  // kill-switch plus a process-wide latch flipped on the first 400 seen on
252
251
  // a compressed request, which then replays that attempt uncompressed.
@@ -88,7 +88,7 @@ globalThis.__mixdogOpenaiWsRuntimeLoaded = true;
88
88
  // by connect/handshake and pre-output stream failures.
89
89
  const MIDSTREAM_WS_TRANSIENT_RETRY_LIMIT = MIDSTREAM_RETRY_POLICY.ws.transientCloseRetries;
90
90
  const MIDSTREAM_DEFAULT_RETRY_LIMIT = MIDSTREAM_RETRY_POLICY.ws.defaultRetries;
91
- // Codex core/src/util.rs uses a 200ms base, factor 2, and symmetric ±10%
91
+ // The reference client uses a 200ms base, factor 2, and symmetric ±10%
92
92
  // jitter for each of its five stream retries.
93
93
  const MIDSTREAM_BACKOFF_MS = Object.freeze([200, 400, 800, 1600, 3200]);
94
94
  const CODEX_RETRY_JITTER_RATIO = 0.1;
@@ -773,7 +773,7 @@ export async function sendViaWebSocket({
773
773
  let result;
774
774
  const streamTimeouts = null;
775
775
  try {
776
- // codex prewarm gate (client.rs:1686-1688): only when the session
776
+ // Prewarm gate: only when the session
777
777
  // has no prior request state. A reused pooled socket with a live
778
778
  // chain must go straight to the real request.
779
779
  if (warmupBody && typeof warmupBody === 'object' && !completedWarmup
@@ -203,7 +203,7 @@ export function toOpenAIResponsesTool(t) {
203
203
 
204
204
  export const _convertMessagesToResponsesInputForTest = convertMessagesToResponsesInput;
205
205
 
206
- // codex build_reasoning() (core/src/client.rs:785-805) only attaches the
206
+ // The reference client only attaches the
207
207
  // reasoning object when model_info.supports_reasoning_summaries; models
208
208
  // without summary support get NO reasoning field at all. Mirror that via the
209
209
  // cached codex catalog; unknown models default to true (gpt-5 family all
@@ -218,7 +218,7 @@ function _codexModelSupportsReasoningSummaries(id) {
218
218
  return true;
219
219
  }
220
220
 
221
- // codex reasoning_effort_for_request (core/src/client.rs): `ultra` collapses to
221
+ // Effort normalization: `ultra` collapses to
222
222
  // `max` on the wire — the openai-oauth backend does not accept `ultra`. Every
223
223
  // other effort passes through unchanged; empty/unknown falls back to medium.
224
224
  export function _normalizeReasoningEffort(effort) {
@@ -251,12 +251,12 @@ export function buildRequestBody(messages, model, tools, sendOpts) {
251
251
  const value = String(item || '').trim();
252
252
  if (value && !include.includes(value)) include.push(value);
253
253
  }
254
- // Field order MIRRORS codex-rs ResponsesApiRequest (common.rs struct order):
254
+ // Field order MIRRORS the reference request struct:
255
255
  // model, instructions, input, tools, tool_choice, parallel_tool_calls,
256
256
  // reasoning, store, stream, include, service_tier, prompt_cache_key, text.
257
257
  // JSON serialization order is load-bearing for the server prompt cache
258
- // (exact-prefix match): matching codex's byte layout keeps our requests on
259
- // the same cache-routing shape codex warms. tools/service_tier/
258
+ // (exact-prefix match): matching that byte layout keeps our requests on
259
+ // the same cache-routing shape the backend warms. tools/service_tier/
260
260
  // prompt_cache_key are appended below in the same relative order.
261
261
  const body = {
262
262
  model,
@@ -264,17 +264,15 @@ export function buildRequestBody(messages, model, tools, sendOpts) {
264
264
  input,
265
265
  tool_choice: opts.toolChoice || 'auto',
266
266
  parallel_tool_calls: true,
267
- // codex build_reasoning() sends { effort, summary } — summary defaults to
268
- // ReasoningSummary::Auto (protocol config_types.rs), serialized lowercase
269
- // as "auto". Matching this keeps our reasoning object byte-identical to
270
- // codex so the server prompt-cache prefix hash lines up. codex also
271
- // normalizes `ultra` -> `max` on the wire (reasoning_effort_for_request
272
- // in core/src/client.rs); the openai-oauth backend does not accept
273
- // `ultra` as a wire value, so mirror that mapping here.
274
- // WIRE-VERIFIED (codex desktop logs_2.sqlite, 40 response.create
275
- // captures, 2026-07-03): codex sends reasoning as {"effort":"..."}
276
- // with NO summary field on gpt-5.5, regardless of what the repo's
277
- // build_reasoning() suggests. Match the observed bytes.
267
+ // The reference client sends { effort, summary } — summary defaults
268
+ // to "auto" (lowercase on the wire). Matching this keeps our
269
+ // reasoning object byte-identical so the server prompt-cache prefix
270
+ // hash lines up. `ultra` is normalized to `max` on the wire too; the
271
+ // openai-oauth backend does not accept `ultra` as a wire value, so
272
+ // mirror that mapping here.
273
+ // WIRE-VERIFIED (40 response.create captures, 2026-07-03): the wire
274
+ // carries reasoning as {"effort":"..."} with NO summary field on
275
+ // gpt-5.5. Match the observed bytes.
278
276
  reasoning: { effort: _normalizeReasoningEffort(opts.effort) },
279
277
  store: process.env.MIXDOG_OAI_STORE === 'true' ? true : false,
280
278
  stream: true,
@@ -41,8 +41,8 @@ export function _envOn(name) {
41
41
  // (MIXDOG_OAI_CODEX_WIRE_PARITY=1 + ws-delta + underscore session_id) exactly
42
42
  // as-is. These add EXTRA parity dimensions for backend fingerprint probes.
43
43
 
44
- // codex sends dashed RFC-4122 UUIDs as session-id/thread-id (client.rs:1033-
45
- // 1057); we key those dashed handshake headers off the underscore cacheKey by
44
+ // The reference client sends dashed RFC-4122 UUIDs as session-id/thread-id;
45
+ // we key those dashed handshake headers off the underscore cacheKey by
46
46
  // default. Opt in with MIXDOG_OAI_CODEX_WIRE_PARITY_UUID_IDS to reshape ONLY
47
47
  // the dashed pair (session-id/thread-id/x-client-request-id) into codex's UUID
48
48
  // format. The value is derived deterministically from the id so it stays
@@ -108,11 +108,12 @@ export function _codexBetaFeatures() {
108
108
  return out.join(',');
109
109
  }
110
110
 
111
- // --- Opt-in raw WS capture for byte-diff against codex-rs -------------------
111
+ // --- Opt-in raw WS capture for wire byte-diff -------------------------------
112
112
  // Enabled ONLY when MIXDOG_OAI_WS_DUMP_DIR names a directory. Persists the
113
113
  // (redacted) handshake header metadata and the exact serialized
114
114
  // response.create frame bytes so our wire format can be byte-diffed against
115
- // codex. Secrets (Authorization / Cookie / account-id / routing tokens) are
115
+ // the reference client. Secrets (Authorization / Cookie / account-id /
116
+ // routing tokens) are
116
117
  // hashed, never written in clear. When the env is unset both helpers are
117
118
  // no-ops, so there is no default behavior change.
118
119
  const _WS_DUMP_SECRET_RE = /^(authorization|proxy-authorization|cookie|set-cookie|chatgpt-account-id|x-codex-turn-state|session_id|session-id|thread-id|x-codex-parent-thread-id|x-client-request-id|x-session-affinity)$/i;
@@ -110,16 +110,16 @@ function _selectIdleEntry(entries, compatibility) {
110
110
  }
111
111
 
112
112
  // --- Cache-route probe state (2026-07-04 hunt) -----------------------------
113
- // CF cookie stickiness (codex chatgpt_cloudflare_cookies.rs:22-55 persists
113
+ // CF cookie stickiness (the reference client persists
114
114
  // __cf_bm/_cfuvid across HTTP clients; our WS handshakes never echo them, so
115
115
  // Cloudflare may re-shard every fresh socket). Jar is per-process, keyed by
116
116
  // auth account. Env knobs (A/B):
117
117
  // MIXDOG_OAI_CF_COOKIES=1 capture Set-Cookie from the 101 upgrade and
118
118
  // send Cookie on subsequent handshakes
119
119
  // MIXDOG_OAI_SESSION_AFFINITY=1 send x-session-affinity: <cacheKey>
120
- // (opencode request.ts:187, ws-pool.ts:66)
121
- // MIXDOG_OAI_WS_URL_SESSION=0 drop the ?session_id= URL query (codex/pi/
122
- // opencode all use the bare WS URL)
120
+ // (a known cache-affinity hint)
121
+ // MIXDOG_OAI_WS_URL_SESSION=0 drop the ?session_id= URL query (reference
122
+ // clients all use the bare WS URL)
123
123
  function _getPoolArr(poolKey) {
124
124
  if (!poolKey) return null;
125
125
  let arr = _wsPool.get(poolKey);
@@ -340,15 +340,15 @@ function _buildHandshakeHeaders({ auth, sessionToken, turnState, cacheKey: _cach
340
340
  'x-codex-beta-features': _codexBetaFeatures(),
341
341
  };
342
342
  const isOpenAiOauth = auth.type !== 'xai' && auth.type !== 'openai-direct';
343
- // codex-rs sends only the dashed session-id/thread-id pair
344
- // (client.rs:1033-1057), but OUR backend measurements disagree with pure
345
- // codex parity here: 2026-04-19 probes showed the OAuth backend dedupes
343
+ // The reference client sends only the dashed session-id/thread-id pair,
344
+ // but OUR backend measurements disagree with pure
345
+ // wire parity here: 2026-04-19 probes showed the OAuth backend dedupes
346
346
  // its in-memory prefix state by the underscore session_id handshake
347
347
  // header, and the only 0.0%-miss full-frame rounds (R7/R8, R15 regressed
348
348
  // to 13% after this header was dropped) all had it present. Send both.
349
349
  // The underscore session_id is the backend prefix-dedupe key (2026-04-19
350
- // probes; R15 regressed to 13% miss when it was dropped). Codex parity
351
- // (client.rs:1033-1057) sends ONLY the dashed pair, but dropping this
350
+ // probes; R15 regressed to 13% miss when it was dropped). Strict wire
351
+ // parity sends ONLY the dashed pair, but dropping this
352
352
  // header is a KNOWN cache-unsafe change — so the general parity flag no
353
353
  // longer silently drops it. Keep it unless an operator EXPLICITLY opts
354
354
  // into the codex-exact dashed-only wire via
@@ -432,7 +432,7 @@ function _openSocket({ auth, sessionToken, turnState, externalSignal, cacheKey,
432
432
  if (process.env.MIXDOG_DEBUG_AGENT) {
433
433
  process.stderr.write(`[agent-trace] ws-open-start url=${baseUrl} tokenHash=${createHash('sha256').update(String(sessionToken)).digest('hex').slice(0, 8)} ts=${_wsOpenStart}\n`);
434
434
  }
435
- // Bare WS URL by default (codex/pi/opencode parity). Interleaved A/B
435
+ // Bare WS URL by default (reference-client parity). Interleaved A/B
436
436
  // (2026-07-04, ivA/ivB, 24 sessions each, alternating rounds to cancel
437
437
  // server-time noise): dropping the ?session_id= query improved it1
438
438
  // warmup-prefix hits 15/24 -> 22/24 and it2 full hits 11 -> 15 (miss
@@ -126,9 +126,8 @@ function _writeWsLifecycleTrace(lifecycle) {
126
126
  process.stderr.write(`[ws-trace] t=${new Date().toISOString()} lifecycle=${lifecycle}\n`);
127
127
  }
128
128
 
129
- // Wire-level `end_turn` on a terminal Responses frame (codex-rs
130
- // codex-api/src/sse/responses.rs ResponseCompleted.end_turn: Option<bool>,
131
- // consumed in core/src/session/turn.rs:2299 as "false ⇒ needs follow-up").
129
+ // Wire-level `end_turn` on a terminal Responses frame: an optional boolean on
130
+ // the completed response, where "false ⇒ needs follow-up".
132
131
  // The field is optional: only a real boolean is normalized; anything else —
133
132
  // including a missing field — stays undefined so absence is preserved and no
134
133
  // caller can mistake "server said nothing" for "server said true/false".