mixdog 0.9.92 → 0.9.93
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +147 -51
- package/package.json +6 -5
- package/scripts/code-graph-description-contract.mjs +6 -8
- package/scripts/tmp-cdp-errors.mjs +41 -0
- package/scripts/tmp-cdp-inspect.mjs +41 -0
- package/scripts/tui-transcript-jitter-harness.mjs +2 -18
- package/src/rules/agent/00-core.md +1 -2
- package/src/rules/lead/01-general.md +1 -0
- package/src/rules/shared/01-tool.md +19 -22
- package/src/runtime/agent/orchestrator/agent-runtime/cache-strategy.mjs +16 -3
- package/src/runtime/agent/orchestrator/agent-trace.mjs +17 -0
- package/src/runtime/agent/orchestrator/context/collect.mjs +2 -1
- package/src/runtime/agent/orchestrator/providers/anthropic-effort.mjs +9 -1
- package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +79 -21
- package/src/runtime/agent/orchestrator/providers/anthropic.mjs +40 -19
- package/src/runtime/agent/orchestrator/providers/lib/anthropic-request-utils.mjs +18 -1
- package/src/runtime/agent/orchestrator/providers/openai-oauth-http-sse.mjs +44 -8
- package/src/runtime/agent/orchestrator/providers/openai-ws-events.mjs +3 -0
- package/src/runtime/agent/orchestrator/providers/openai-ws-stream.mjs +2 -0
- package/src/runtime/agent/orchestrator/session/agent-loop.mjs +22 -5
- package/src/runtime/agent/orchestrator/session/eager-dispatch.mjs +35 -29
- package/src/runtime/agent/orchestrator/session/loop/stored-tool-args.mjs +11 -2
- package/src/runtime/agent/orchestrator/session/loop/tool-classify.mjs +5 -6
- package/src/runtime/agent/orchestrator/session/loop/tool-exec.mjs +31 -2
- package/src/runtime/agent/orchestrator/session/manager/compaction-runner.mjs +60 -0
- package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +60 -31
- package/src/runtime/agent/orchestrator/session/manager.mjs +1 -1
- package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +12 -3
- package/src/runtime/agent/orchestrator/session/store/listing.mjs +17 -0
- package/src/runtime/agent/orchestrator/session/store-summary-reader.mjs +101 -0
- package/src/runtime/agent/orchestrator/session/store.mjs +30 -0
- package/src/runtime/agent/orchestrator/session/tool-batch.mjs +19 -23
- package/src/runtime/agent/orchestrator/stall-policy.mjs +31 -21
- package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +4 -1
- package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +12 -6
- package/src/runtime/agent/orchestrator/tools/code-graph-tool-defs.mjs +4 -3
- package/src/runtime/agent/orchestrator/tools/lib/pwsh-standby-pool.mjs +47 -14
- package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +100 -0
- package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +6 -5
- package/src/runtime/agent/orchestrator/tools/shell-command.mjs +9 -0
- package/src/runtime/agent/orchestrator/tools/shell-state.mjs +32 -2
- package/src/runtime/channels/backends/discord-gateway.mjs +6 -32
- package/src/runtime/channels/lib/inbound-handler.mjs +19 -3
- package/src/runtime/channels/lib/scheduler.mjs +51 -3
- package/src/runtime/channels/lib/worker-main.mjs +4 -0
- package/src/runtime/channels/tool-defs.mjs +4 -2
- package/src/runtime/memory/lib/query-handlers.mjs +11 -3
- package/src/runtime/memory/tool-defs.mjs +5 -5
- package/src/runtime/shared/channel-notification-routing.mjs +8 -2
- package/src/runtime/shared/llm/http-agent.mjs +11 -0
- package/src/session-runtime/lifecycle-api.mjs +26 -1
- package/src/session-runtime/tool-catalog-data.mjs +5 -2
- package/src/session-runtime/workflow.mjs +7 -5
- package/src/standalone/agent-tool/tag-registry.mjs +5 -1
- package/src/standalone/explore-tool.mjs +1 -1
- package/src/tui/dist/index.mjs +68 -45
- package/src/tui/engine/agent-envelope.mjs +52 -3
- package/src/tui/engine/turn.mjs +8 -9
- package/src/tui/engine.mjs +18 -41
- package/src/workflows/solo/WORKFLOW.md +0 -6
- package/src/workflows/solo-bench/WORKFLOW.md +17 -0
|
@@ -38,6 +38,22 @@ function extractCachedTokens(usage) {
|
|
|
38
38
|
return 0;
|
|
39
39
|
}
|
|
40
40
|
|
|
41
|
+
function extractCacheWriteTokens(usage) {
|
|
42
|
+
const candidates = [
|
|
43
|
+
usage?.input_tokens_details?.cache_write_tokens,
|
|
44
|
+
usage?.prompt_tokens_details?.cache_write_tokens,
|
|
45
|
+
usage?.inputTokensDetails?.cacheWriteTokens,
|
|
46
|
+
usage?.promptTokensDetails?.cacheWriteTokens,
|
|
47
|
+
usage?.cache_write_tokens,
|
|
48
|
+
usage?.cacheWriteTokens,
|
|
49
|
+
];
|
|
50
|
+
for (const value of candidates) {
|
|
51
|
+
const n = Number(value);
|
|
52
|
+
if (Number.isFinite(n)) return n;
|
|
53
|
+
}
|
|
54
|
+
return 0;
|
|
55
|
+
}
|
|
56
|
+
|
|
41
57
|
// Lightweight fingerprint of the conversation prefix. Hashes the first 4096
|
|
42
58
|
// characters of JSON.stringify(messages) — enough to detect prefix mutation
|
|
43
59
|
// across iterations (which invalidates the provider prompt cache) without
|
|
@@ -267,6 +283,7 @@ export {
|
|
|
267
283
|
appendAgentTrace,
|
|
268
284
|
drainAgentTrace,
|
|
269
285
|
estimateProviderPayloadBytes,
|
|
286
|
+
extractCacheWriteTokens,
|
|
270
287
|
extractCachedTokens,
|
|
271
288
|
messagePrefixHash,
|
|
272
289
|
traceAgentFetch,
|
|
@@ -275,7 +275,7 @@ export function buildSkillToolEnvelope(name, content, skillDir) {
|
|
|
275
275
|
};
|
|
276
276
|
}
|
|
277
277
|
|
|
278
|
-
function compactSkillManifestText(value, max =
|
|
278
|
+
function compactSkillManifestText(value, max = 100) {
|
|
279
279
|
const text = String(value || '').replace(/\s+/g, ' ').trim();
|
|
280
280
|
return text.length > max ? `${text.slice(0, Math.max(1, max - 3))}...` : text;
|
|
281
281
|
}
|
|
@@ -528,6 +528,7 @@ export function buildSkillToolDefs(skills, { ownerIsAgentSession = false } = {})
|
|
|
528
528
|
name: { type: 'string', description: 'Skill name' },
|
|
529
529
|
},
|
|
530
530
|
required: ['name'],
|
|
531
|
+
additionalProperties: false,
|
|
531
532
|
},
|
|
532
533
|
},
|
|
533
534
|
];
|
|
@@ -208,7 +208,15 @@ export function applyAnthropicEffortToBody(
|
|
|
208
208
|
// modelSupportsEffort() allowlist so older models never receive it.
|
|
209
209
|
// Set unconditionally (independent of `normalized`) so effort-capable
|
|
210
210
|
// turns always carry adaptive thinking + round-trip signatures.
|
|
211
|
-
|
|
211
|
+
// MIXDOG_ANTHROPIC_THINKING_DISPLAY=omitted (operator/bench knob):
|
|
212
|
+
// CC-parity mode — no thinking blocks come back, so nothing is
|
|
213
|
+
// replayed into later requests (saves the 1h cache-write + re-read on
|
|
214
|
+
// accumulated thinking) at the cost of losing visible reasoning and
|
|
215
|
+
// cross-iteration thinking continuity. Default stays summarized.
|
|
216
|
+
const display = (process.env.MIXDOG_ANTHROPIC_THINKING_DISPLAY || '').trim() === 'omitted'
|
|
217
|
+
? 'omitted'
|
|
218
|
+
: 'summarized';
|
|
219
|
+
body.thinking = { type: 'adaptive', display };
|
|
212
220
|
// Adaptive/4.7+ models reject any non-default sampling param with a 400.
|
|
213
221
|
delete body.temperature;
|
|
214
222
|
delete body.top_p;
|
|
@@ -64,6 +64,18 @@ import {
|
|
|
64
64
|
stampAnthropicStreamOutcome,
|
|
65
65
|
} from './anthropic-sse.mjs';
|
|
66
66
|
import { buildAnthropicBetaHeaders, supportsAnthropicFastMode } from './anthropic-betas.mjs';
|
|
67
|
+
import { gzipSync } from 'node:zlib';
|
|
68
|
+
|
|
69
|
+
// Request-body gzip gate (see the fetch site below). Env kill-switch
|
|
70
|
+
// (MIXDOG_ANTHROPIC_REQ_GZIP=0) plus a process-wide latch flipped on the
|
|
71
|
+
// first 400 response to a compressed request. Small bodies skip compression:
|
|
72
|
+
// below ~8KB the CPU + header cost outweighs the upload saving.
|
|
73
|
+
const ANTHROPIC_REQ_GZIP_MIN_BYTES = 8 * 1024;
|
|
74
|
+
let _anthropicReqGzipLatch = false;
|
|
75
|
+
function _anthropicReqGzipDisabled() {
|
|
76
|
+
return _anthropicReqGzipLatch || process.env.MIXDOG_ANTHROPIC_REQ_GZIP === '0';
|
|
77
|
+
}
|
|
78
|
+
function _disableAnthropicReqGzip() { _anthropicReqGzipLatch = true; }
|
|
67
79
|
import {
|
|
68
80
|
applyAnthropicEffortToBody,
|
|
69
81
|
effortValuesForModel,
|
|
@@ -654,7 +666,16 @@ export class AnthropicOAuthProvider {
|
|
|
654
666
|
// provider-visible cache breakpoint off the cached one — the
|
|
655
667
|
// exact COLD-turn bug this change fixes. Order is fixed:
|
|
656
668
|
// build → sanitize (once) → mark → JSON.stringify.
|
|
657
|
-
|
|
669
|
+
// Request-body gzip (probe-verified 2026-08-04: /v1/messages
|
|
670
|
+
// returns 200 for Content-Encoding: gzip, 400 for zstd). Large
|
|
671
|
+
// turn bodies (system prompt + history, typically 50-100KB+)
|
|
672
|
+
// compress ~5-10x, trimming upload time off every call's
|
|
673
|
+
// header wait. Latch OFF process-wide on the first 400 seen on
|
|
674
|
+
// a compressed request and retry that attempt uncompressed, so
|
|
675
|
+
// a server-side behavior change can never wedge the session.
|
|
676
|
+
const rawBody = Buffer.from(JSON.stringify(requestBody));
|
|
677
|
+
const useGzip = !_anthropicReqGzipDisabled() && rawBody.length >= ANTHROPIC_REQ_GZIP_MIN_BYTES;
|
|
678
|
+
const sendAttempt = (gz) => fetch(API_URL, {
|
|
658
679
|
method: 'POST',
|
|
659
680
|
headers: {
|
|
660
681
|
'Authorization': `Bearer ${accessToken}`,
|
|
@@ -669,11 +690,18 @@ export class AnthropicOAuthProvider {
|
|
|
669
690
|
'user-agent': `claude-cli/${resolveCliVersion()} (external, sdk-cli)`,
|
|
670
691
|
'x-app': 'cli',
|
|
671
692
|
'Content-Type': 'application/json',
|
|
693
|
+
...(gz ? { 'Content-Encoding': 'gzip' } : {}),
|
|
672
694
|
},
|
|
673
|
-
body:
|
|
695
|
+
body: gz ? gzipSync(rawBody) : rawBody,
|
|
674
696
|
signal: controller.signal,
|
|
675
697
|
dispatcher: getLlmDispatcher(),
|
|
676
698
|
});
|
|
699
|
+
let response = await sendAttempt(useGzip);
|
|
700
|
+
if (useGzip && response.status === 400) {
|
|
701
|
+
_disableAnthropicReqGzip();
|
|
702
|
+
try { await response.arrayBuffer(); } catch { /* drain best-effort */ }
|
|
703
|
+
response = await sendAttempt(false);
|
|
704
|
+
}
|
|
677
705
|
|
|
678
706
|
traceAgentFetch({
|
|
679
707
|
sessionId,
|
|
@@ -780,25 +808,13 @@ export class AnthropicOAuthProvider {
|
|
|
780
808
|
// clock starting at the first stall (see createStallRetryBudget).
|
|
781
809
|
const stallRetryBudget = createStallRetryBudget();
|
|
782
810
|
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
}
|
|
790
|
-
let resetAccepted = false;
|
|
791
|
-
try {
|
|
792
|
-
resetAccepted = await onTextReset({
|
|
793
|
-
chars: exposedChars,
|
|
794
|
-
reason: 'anthropic-streaming-fallback',
|
|
795
|
-
}) === true;
|
|
796
|
-
} catch {}
|
|
797
|
-
if (!resetAccepted) {
|
|
798
|
-
try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
|
|
799
|
-
throw streamingError;
|
|
800
|
-
}
|
|
801
|
-
try { controller?.abort?.(streamingError); } catch {}
|
|
811
|
+
// Core non-streaming re-issue: abort the dead stream and repeat the
|
|
812
|
+
// SAME request with stream:false. Shared by the exposed-text recovery
|
|
813
|
+
// (which must first get the owner's onTextReset acknowledgement) and
|
|
814
|
+
// the no-exposure stall fallback below (trivially safe — nothing was
|
|
815
|
+
// relayed or dispatched, so there is nothing to withdraw or replay).
|
|
816
|
+
const issueNonStreamingFallback = async (controller, abortReason) => {
|
|
817
|
+
try { controller?.abort?.(abortReason); } catch {}
|
|
802
818
|
try { onStageChange?.('requesting', { transport: 'non-streaming-fallback' }); } catch {}
|
|
803
819
|
let fallback = await requestWithRetry(creds.accessToken, { ...body, stream: false });
|
|
804
820
|
if (fallback.response.status === 401) {
|
|
@@ -825,6 +841,27 @@ export class AnthropicOAuthProvider {
|
|
|
825
841
|
}
|
|
826
842
|
};
|
|
827
843
|
|
|
844
|
+
const recoverNonStreaming = async (midState, streamingError, controller) => {
|
|
845
|
+
const exposedChars = Number(midState?.emittedTextChars) || 0;
|
|
846
|
+
if (!onTextReset || exposedChars <= 0
|
|
847
|
+
|| midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
|
|
848
|
+
try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
|
|
849
|
+
throw streamingError;
|
|
850
|
+
}
|
|
851
|
+
let resetAccepted = false;
|
|
852
|
+
try {
|
|
853
|
+
resetAccepted = await onTextReset({
|
|
854
|
+
chars: exposedChars,
|
|
855
|
+
reason: 'anthropic-streaming-fallback',
|
|
856
|
+
}) === true;
|
|
857
|
+
} catch {}
|
|
858
|
+
if (!resetAccepted) {
|
|
859
|
+
try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
|
|
860
|
+
throw streamingError;
|
|
861
|
+
}
|
|
862
|
+
return issueNonStreamingFallback(controller, streamingError);
|
|
863
|
+
};
|
|
864
|
+
|
|
828
865
|
try {
|
|
829
866
|
for (let attemptIndex = 0; attemptIndex <= MAX_MIDSTREAM_RETRIES; attemptIndex++) {
|
|
830
867
|
let response, controller, cancelHandler;
|
|
@@ -1057,6 +1094,27 @@ export class AnthropicOAuthProvider {
|
|
|
1057
1094
|
continue;
|
|
1058
1095
|
}
|
|
1059
1096
|
const classifier = _classifyMidstreamError(err, midState);
|
|
1097
|
+
// CC-parity stall recovery (2026-08-03 v3 postmortem): a
|
|
1098
|
+
// stalled stream that exposed NOTHING (no text/thinking
|
|
1099
|
+
// relayed, no tool emitted) is re-issued NON-STREAMING instead
|
|
1100
|
+
// of retrying the same streaming shape. Effort-mode models can
|
|
1101
|
+
// legitimately think in silence past any streaming idle
|
|
1102
|
+
// window; an in-place streaming retry re-runs the same silent
|
|
1103
|
+
// generation into the same timer (observed live: deterministic
|
|
1104
|
+
// 4×~138s beheading, ~552s per turn), while the non-streaming
|
|
1105
|
+
// transport simply waits for the full body (bounded by
|
|
1106
|
+
// PROVIDER_NONSTREAM_TOTAL_TIMEOUT_MS). Claude Code does
|
|
1107
|
+
// exactly this on its watchdog aborts. Replay is trivially
|
|
1108
|
+
// safe here — nothing was relayed or dispatched.
|
|
1109
|
+
if (classifier === 'stream_stalled'
|
|
1110
|
+
&& _outcome?.replayUnsafe !== true
|
|
1111
|
+
&& !midState.emittedText
|
|
1112
|
+
&& !midState.emittedToolCall
|
|
1113
|
+
&& !midState.partialToolCall
|
|
1114
|
+
&& !midState.emittedThinking) {
|
|
1115
|
+
try { process.stderr.write('[anthropic-oauth] stream stalled with no exposure — retrying non-streaming\n'); } catch {}
|
|
1116
|
+
return await issueNonStreamingFallback(controller, err);
|
|
1117
|
+
}
|
|
1060
1118
|
if (classifier === 'stream_stalled' && !stallRetryBudget.allowStallRetry()) {
|
|
1061
1119
|
try { process.stderr.write(`[anthropic-oauth] stall retry budget exhausted (${STREAM_STALL_RETRY_BUDGET_MS}ms since first stall) — surfacing for fresh-request retry\n`); } catch {}
|
|
1062
1120
|
try { controller?.abort?.(err); } catch { /* best-effort teardown */ }
|
|
@@ -320,25 +320,11 @@ export class AnthropicProvider {
|
|
|
320
320
|
};
|
|
321
321
|
};
|
|
322
322
|
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
throw streamingError;
|
|
329
|
-
}
|
|
330
|
-
let resetAccepted = false;
|
|
331
|
-
try {
|
|
332
|
-
resetAccepted = await onTextReset({
|
|
333
|
-
chars: exposedChars,
|
|
334
|
-
reason: 'anthropic-streaming-fallback',
|
|
335
|
-
}) === true;
|
|
336
|
-
} catch {}
|
|
337
|
-
if (!resetAccepted) {
|
|
338
|
-
try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
|
|
339
|
-
throw streamingError;
|
|
340
|
-
}
|
|
341
|
-
try { streamController.abort?.(streamingError); } catch {}
|
|
323
|
+
// Core non-streaming re-issue shared by the exposed-text recovery
|
|
324
|
+
// (onTextReset-gated) and the no-exposure stall fallback (trivially
|
|
325
|
+
// safe — nothing was relayed or dispatched). Mirrors anthropic-oauth.
|
|
326
|
+
const issueNonStreamingFallback = async (streamController, abortReason) => {
|
|
327
|
+
try { streamController.abort?.(abortReason); } catch {}
|
|
342
328
|
try { onStageChange?.('requesting', { transport: 'non-streaming-fallback' }); } catch {}
|
|
343
329
|
const nonStreamingParams = { ...params, stream: false };
|
|
344
330
|
const message = await withRetry(
|
|
@@ -362,6 +348,27 @@ export class AnthropicProvider {
|
|
|
362
348
|
return buildReturnFromParse(normalizeAnthropicNonStreamingResponse(message, useModel));
|
|
363
349
|
};
|
|
364
350
|
|
|
351
|
+
const recoverNonStreaming = async (midState, streamingError, streamController) => {
|
|
352
|
+
const exposedChars = Number(midState?.emittedTextChars) || 0;
|
|
353
|
+
if (!onTextReset || exposedChars <= 0
|
|
354
|
+
|| midState.emittedToolCall || midState.partialToolCall || midState.emittedThinking) {
|
|
355
|
+
try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
|
|
356
|
+
throw streamingError;
|
|
357
|
+
}
|
|
358
|
+
let resetAccepted = false;
|
|
359
|
+
try {
|
|
360
|
+
resetAccepted = await onTextReset({
|
|
361
|
+
chars: exposedChars,
|
|
362
|
+
reason: 'anthropic-streaming-fallback',
|
|
363
|
+
}) === true;
|
|
364
|
+
} catch {}
|
|
365
|
+
if (!resetAccepted) {
|
|
366
|
+
try { streamingError.liveTextEmitted = true; streamingError.unsafeToRetry = true; } catch {}
|
|
367
|
+
throw streamingError;
|
|
368
|
+
}
|
|
369
|
+
return issueNonStreamingFallback(streamController, streamingError);
|
|
370
|
+
};
|
|
371
|
+
|
|
365
372
|
try {
|
|
366
373
|
for (let attemptIndex = 0; attemptIndex <= MAX_MIDSTREAM_RETRIES; attemptIndex++) {
|
|
367
374
|
const streamController = createAbortController();
|
|
@@ -597,6 +604,20 @@ export class AnthropicProvider {
|
|
|
597
604
|
continue;
|
|
598
605
|
}
|
|
599
606
|
const classifier = _classifyMidstreamError(err, midState);
|
|
607
|
+
// CC-parity stall recovery (ported from anthropic-oauth,
|
|
608
|
+
// 2026-08-03): a stalled stream that exposed NOTHING is
|
|
609
|
+
// re-issued non-streaming instead of retrying the same
|
|
610
|
+
// streaming shape into the same idle window. Replay is
|
|
611
|
+
// trivially safe — nothing was relayed or dispatched.
|
|
612
|
+
if (classifier === 'stream_stalled'
|
|
613
|
+
&& _outcome?.replayUnsafe !== true
|
|
614
|
+
&& !midState.emittedText
|
|
615
|
+
&& !midState.emittedToolCall
|
|
616
|
+
&& !midState.partialToolCall
|
|
617
|
+
&& !midState.emittedThinking) {
|
|
618
|
+
try { process.stderr.write(`[${this.name}] stream stalled with no exposure — retrying non-streaming\n`); } catch {}
|
|
619
|
+
return await issueNonStreamingFallback(streamController, err);
|
|
620
|
+
}
|
|
600
621
|
if (classifier === 'stream_stalled' && !stallRetryBudget.allowStallRetry()) {
|
|
601
622
|
try {
|
|
602
623
|
process.stderr.write(
|
|
@@ -268,10 +268,27 @@ export function applyAnthropicCacheMarkers(sanitizedMessages, {
|
|
|
268
268
|
const tailIdx = sanitizedMessages.length - 1;
|
|
269
269
|
return hasUserText(sanitizedMessages[tailIdx]) ? tailIdx : -1;
|
|
270
270
|
};
|
|
271
|
+
// True-tip anchor: when the request ends with a PERSISTED user text turn
|
|
272
|
+
// (the current prompt — it re-appears verbatim in every later request's
|
|
273
|
+
// prefix), mark it first. Without this, mid-session turn-first requests
|
|
274
|
+
// (multi-turn only; single-turn sessions are already covered by
|
|
275
|
+
// firstRequestUserPromptIdx) leave the fresh prompt unmarked: its tokens
|
|
276
|
+
// bill once at $5/M uncached, then again as a cache write when a later
|
|
277
|
+
// anchor advances past them — cost bounded by the prompt's size, so it
|
|
278
|
+
// matters for large pasted prompts. (2026-08-03 A/B note: session totals'
|
|
279
|
+
// totalUncachedInputTokens = input + cacheWrite by design, see
|
|
280
|
+
// uncachedInputTokensForProvider; billing-uncached input measured via
|
|
281
|
+
// usage.json was already ~0 on single-turn bench tasks before and after
|
|
282
|
+
// this change.) Synthetic system-reminder tails are excluded by
|
|
283
|
+
// hasUserText, so per-call volatile content still never keys the cache.
|
|
284
|
+
const currentTailUserIdx = () => {
|
|
285
|
+
const tailIdx = sanitizedMessages.length - 1;
|
|
286
|
+
return hasUserText(sanitizedMessages[tailIdx]) ? tailIdx : -1;
|
|
287
|
+
};
|
|
271
288
|
if (messageTtl !== null) {
|
|
272
289
|
const slots = Math.max(0, Math.min(4, Number(messageSlots) || 0));
|
|
273
290
|
const marked = new Set();
|
|
274
|
-
const candidates = [latestToolResultTailIdx(), previousUserTextAnchorIdx(), firstRequestUserPromptIdx()];
|
|
291
|
+
const candidates = [currentTailUserIdx(), latestToolResultTailIdx(), previousUserTextAnchorIdx(), firstRequestUserPromptIdx()];
|
|
275
292
|
for (const idx of candidates) {
|
|
276
293
|
if (slots <= 0) break;
|
|
277
294
|
if (idx < 0 || marked.has(idx)) continue;
|
|
@@ -7,7 +7,10 @@
|
|
|
7
7
|
* (scripts/openai-oauth-http-sse-toolcall-smoke.mjs) and fallback headers.
|
|
8
8
|
*/
|
|
9
9
|
import { randomBytes } from 'crypto';
|
|
10
|
+
import zlib from 'node:zlib';
|
|
10
11
|
import {
|
|
12
|
+
extractCacheWriteTokens,
|
|
13
|
+
extractCachedTokens,
|
|
11
14
|
traceAgentFetch,
|
|
12
15
|
traceAgentSse,
|
|
13
16
|
traceAgentUsage,
|
|
@@ -47,6 +50,21 @@ const CODEX_REQUEST_MAX_RETRIES = 4;
|
|
|
47
50
|
const CODEX_REQUEST_BACKOFF_MS = Object.freeze([200, 400, 800, 1600]);
|
|
48
51
|
const CODEX_RETRY_JITTER_RATIO = 0.1;
|
|
49
52
|
|
|
53
|
+
// Request-body zstd gate (see sendViaHttpSse). Namespace import: on Node
|
|
54
|
+
// runtimes without zlib zstd bindings the named export would fail at module
|
|
55
|
+
// load, so the call site typeof-guards zlib.zstdCompressSync instead.
|
|
56
|
+
const OPENAI_REQ_ZSTD_MIN_BYTES = 8 * 1024;
|
|
57
|
+
let _openaiReqZstdLatch = false;
|
|
58
|
+
function _openaiReqZstdDisabled() {
|
|
59
|
+
return _openaiReqZstdLatch || process.env.MIXDOG_OPENAI_REQ_ZSTD === '0';
|
|
60
|
+
}
|
|
61
|
+
function _disableOpenaiReqZstd() { _openaiReqZstdLatch = true; }
|
|
62
|
+
function _zstdHeaders(headers, bodyForSend) {
|
|
63
|
+
return bodyForSend.encoding
|
|
64
|
+
? { ...headers, 'Content-Encoding': bodyForSend.encoding }
|
|
65
|
+
: headers;
|
|
66
|
+
}
|
|
67
|
+
|
|
50
68
|
export function _envFlag(name, fallback = true) {
|
|
51
69
|
const raw = process.env[name];
|
|
52
70
|
if (raw == null || raw === '') return fallback;
|
|
@@ -72,11 +90,6 @@ function _parseJsonObject(value) {
|
|
|
72
90
|
}
|
|
73
91
|
}
|
|
74
92
|
|
|
75
|
-
function _extractCachedTokens(usage) {
|
|
76
|
-
const details = usage?.input_tokens_details || usage?.prompt_tokens_details || {};
|
|
77
|
-
return Number(details.cached_tokens ?? details.cached ?? usage?.cached_tokens ?? 0) || 0;
|
|
78
|
-
}
|
|
79
|
-
|
|
80
93
|
function _sseEventsFromBuffer(buffer) {
|
|
81
94
|
const frames = [];
|
|
82
95
|
let rest = buffer.replace(/\r\n/g, '\n');
|
|
@@ -231,6 +244,19 @@ export async function sendViaHttpSse({
|
|
|
231
244
|
const responsesUrl = auth?.type === 'openai-direct'
|
|
232
245
|
? OPENAI_DIRECT_RESPONSES_URL
|
|
233
246
|
: CODEX_RESPONSES_URL;
|
|
247
|
+
// Request-body zstd (codex parity: core client.rs
|
|
248
|
+
// responses_request_compression enables zstd for the codex backend on the
|
|
249
|
+
// OpenAI provider — the server decompresses Content-Encoding: zstd).
|
|
250
|
+
// openai-direct is excluded: only the codex backend is verified. Env
|
|
251
|
+
// kill-switch plus a process-wide latch flipped on the first 400 seen on
|
|
252
|
+
// a compressed request, which then replays that attempt uncompressed.
|
|
253
|
+
const _rawReqBytes = Buffer.from(JSON.stringify(body));
|
|
254
|
+
let _reqBodyForSend = auth?.type !== 'openai-direct'
|
|
255
|
+
&& !_openaiReqZstdDisabled()
|
|
256
|
+
&& typeof zlib.zstdCompressSync === 'function'
|
|
257
|
+
&& _rawReqBytes.length >= OPENAI_REQ_ZSTD_MIN_BYTES
|
|
258
|
+
? { bytes: zlib.zstdCompressSync(_rawReqBytes), encoding: 'zstd' }
|
|
259
|
+
: { bytes: _rawReqBytes, encoding: null };
|
|
234
260
|
let response;
|
|
235
261
|
for (let attempt = 0; attempt <= CODEX_REQUEST_MAX_RETRIES; attempt++) {
|
|
236
262
|
const headerTimeout = createTimeoutSignal(
|
|
@@ -253,8 +279,8 @@ export async function sendViaHttpSse({
|
|
|
253
279
|
} catch {}
|
|
254
280
|
response = await fetchFn(responsesUrl, {
|
|
255
281
|
method: 'POST',
|
|
256
|
-
headers,
|
|
257
|
-
body:
|
|
282
|
+
headers: _zstdHeaders(headers, _reqBodyForSend),
|
|
283
|
+
body: _reqBodyForSend.bytes,
|
|
258
284
|
signal: headerTimeout.signal,
|
|
259
285
|
dispatcher: getLlmDispatcher(),
|
|
260
286
|
});
|
|
@@ -267,6 +293,15 @@ export async function sendViaHttpSse({
|
|
|
267
293
|
headerTimeout.cleanup();
|
|
268
294
|
}
|
|
269
295
|
|
|
296
|
+
// zstd rejection fallback: a 400 on a compressed request latches
|
|
297
|
+
// compression OFF process-wide and replays this attempt uncompressed.
|
|
298
|
+
if (response && response.status === 400 && _reqBodyForSend.encoding) {
|
|
299
|
+
_disableOpenaiReqZstd();
|
|
300
|
+
_reqBodyForSend = { bytes: _rawReqBytes, encoding: null };
|
|
301
|
+
await response.arrayBuffer().catch(() => {});
|
|
302
|
+
response = undefined;
|
|
303
|
+
continue;
|
|
304
|
+
}
|
|
270
305
|
const retryableStatus = response && response.status >= 500 && response.status <= 599;
|
|
271
306
|
// Typed transient transport failures only (errno / SDK connection
|
|
272
307
|
// type). An unknown pre-response failure throws immediately instead of
|
|
@@ -783,7 +818,8 @@ export async function sendViaHttpSse({
|
|
|
783
818
|
usage = {
|
|
784
819
|
inputTokens: resp.usage.input_tokens || 0,
|
|
785
820
|
outputTokens: resp.usage.output_tokens || 0,
|
|
786
|
-
cachedTokens:
|
|
821
|
+
cachedTokens: extractCachedTokens(resp.usage),
|
|
822
|
+
cacheWriteTokens: extractCacheWriteTokens(resp.usage),
|
|
787
823
|
promptTokens: resp.usage.input_tokens || 0,
|
|
788
824
|
raw: serviceTier ? { ...resp.usage, service_tier: serviceTier } : resp.usage,
|
|
789
825
|
};
|
|
@@ -39,6 +39,9 @@ export function _combineUsageWithWarmup(actual, warmup, { separateMainContext =
|
|
|
39
39
|
inputTokens: _usageNum(actual.inputTokens) + _usageNum(warmup.inputTokens),
|
|
40
40
|
outputTokens: _usageNum(actual.outputTokens) + _usageNum(warmup.outputTokens),
|
|
41
41
|
cachedTokens: _usageNum(actual.cachedTokens) + _usageNum(warmup.cachedTokens),
|
|
42
|
+
...(actual.cacheWriteTokens != null || warmup.cacheWriteTokens != null
|
|
43
|
+
? { cacheWriteTokens: _usageNum(actual.cacheWriteTokens) + _usageNum(warmup.cacheWriteTokens) }
|
|
44
|
+
: {}),
|
|
42
45
|
promptTokens: _usageNum(actual.promptTokens) + _usageNum(warmup.promptTokens),
|
|
43
46
|
warmupInputTokens: _usageNum(warmup.inputTokens),
|
|
44
47
|
warmupCachedTokens: _usageNum(warmup.cachedTokens),
|
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
import { randomBytes } from 'crypto';
|
|
17
17
|
import { performance } from 'node:perf_hooks';
|
|
18
18
|
import {
|
|
19
|
+
extractCacheWriteTokens,
|
|
19
20
|
extractCachedTokens,
|
|
20
21
|
appendAgentTrace,
|
|
21
22
|
} from '../agent-trace.mjs';
|
|
@@ -1090,6 +1091,7 @@ export async function _streamResponse({
|
|
|
1090
1091
|
inputTokens: u.input_tokens || 0,
|
|
1091
1092
|
outputTokens: u.output_tokens || 0,
|
|
1092
1093
|
cachedTokens: extractCachedTokens(u),
|
|
1094
|
+
cacheWriteTokens: extractCacheWriteTokens(u),
|
|
1093
1095
|
// openai-oauth reports input_tokens as the total
|
|
1094
1096
|
// prompt volume (cached portion is a subset, not
|
|
1095
1097
|
// additive). Alias into the cross-provider
|
|
@@ -227,7 +227,17 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
|
|
|
227
227
|
// turn's push (deferBodies below) collapse to markers now — the model has
|
|
228
228
|
// already seen them on that turn's follow-up send. Failed bodies stay
|
|
229
229
|
// verbatim for retry.
|
|
230
|
-
|
|
230
|
+
// Out-of-loop transcript mutations (post-turn/manual compaction in
|
|
231
|
+
// manager/compaction-runner.mjs) run where no send opts exist; they park a
|
|
232
|
+
// one-shot intent on the session so the FIRST send of the next turn tags
|
|
233
|
+
// its expected cache break instead of an unexplained prefix mismatch.
|
|
234
|
+
if (!opts.cacheBreakIntent && typeof sessionRef?.pendingCacheBreakIntent === 'string') {
|
|
235
|
+
opts.cacheBreakIntent = sessionRef.pendingCacheBreakIntent;
|
|
236
|
+
delete sessionRef.pendingCacheBreakIntent;
|
|
237
|
+
}
|
|
238
|
+
if (compactSettledToolCallBodies(messages) && !opts.cacheBreakIntent) {
|
|
239
|
+
opts.cacheBreakIntent = 'deferred_body_compaction';
|
|
240
|
+
}
|
|
231
241
|
// ---- Codex turn stop hook (refs/codex core/src/session/turn.rs:372-404) --
|
|
232
242
|
// A no-tool assistant message is TERMINAL. Only a structured provider
|
|
233
243
|
// follow-up signal (end_turn=false / pause_turn), pending input, tool
|
|
@@ -526,7 +536,10 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
|
|
|
526
536
|
const _m = messages[_i];
|
|
527
537
|
if (_m && _m.role === 'tool' && typeof _m.content === 'string' && _m.content.includes('⚠')) {
|
|
528
538
|
const _stripped = stripSoftWarns(_m.content);
|
|
529
|
-
if (_stripped !== _m.content)
|
|
539
|
+
if (_stripped !== _m.content) {
|
|
540
|
+
_m.content = _stripped;
|
|
541
|
+
if (!opts.cacheBreakIntent) opts.cacheBreakIntent = 'soft_warn_strip';
|
|
542
|
+
}
|
|
530
543
|
}
|
|
531
544
|
}
|
|
532
545
|
sendTools = snapshotProviderRequestTools({
|
|
@@ -645,14 +658,16 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
|
|
|
645
658
|
);
|
|
646
659
|
const _sendEndedAt = Date.now();
|
|
647
660
|
if (_sendResult.action === 'retry') {
|
|
648
|
-
|
|
661
|
+
// Keep opts.cacheBreakIntent: the failed send never consumed the
|
|
662
|
+
// tag, and the reactive-compact retry that follows IS the tagged
|
|
663
|
+
// transition — deleting it here made retry-side cache_break rows
|
|
664
|
+
// log intentional_transition: null.
|
|
649
665
|
contextOverflowRetryUsed = true;
|
|
650
666
|
reactiveOverflowRetryPending = true;
|
|
651
667
|
continue;
|
|
652
668
|
}
|
|
653
669
|
if (_sendResult.action === 'retry_transport') {
|
|
654
670
|
_transportRetriesUsed += 1;
|
|
655
|
-
delete opts.cacheBreakIntent;
|
|
656
671
|
continue;
|
|
657
672
|
}
|
|
658
673
|
response = _sendResult.response;
|
|
@@ -1021,7 +1036,9 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
|
|
|
1021
1036
|
// Settle earlier deferred bodies before this turn's message lands:
|
|
1022
1037
|
// every previous call already has its result row, so successful bodies
|
|
1023
1038
|
// compact to markers while failed ones keep their full retry text.
|
|
1024
|
-
compactSettledToolCallBodies(messages)
|
|
1039
|
+
if (compactSettledToolCallBodies(messages) && !opts.cacheBreakIntent) {
|
|
1040
|
+
opts.cacheBreakIntent = 'deferred_body_compaction';
|
|
1041
|
+
}
|
|
1025
1042
|
messages.push(_assistantTurnMsg);
|
|
1026
1043
|
try { opts.onAssistantMessageCommitted?.(_assistantTurnMsg); } catch {}
|
|
1027
1044
|
const _callsToExecute = calls;
|
|
@@ -1,18 +1,18 @@
|
|
|
1
1
|
// Eager tool-dispatch controller, extracted from agent-loop.mjs. Owns the
|
|
2
2
|
// per-turn pending promise map, the intra-turn in-flight signature set, and
|
|
3
|
-
// the mutation epoch.
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
7
|
-
//
|
|
3
|
+
// the mutation epoch. Read-only calls may start while the provider is still
|
|
4
|
+
// streaming. Side-effect calls stream-start too until the first apply_patch
|
|
5
|
+
// appears in the stream (they are definitively in segment 0); after that they
|
|
6
|
+
// wait for their segment bounded by apply_patch barriers. Patches execute
|
|
7
|
+
// serially as barriers, so a failed patch can skip every later side effect.
|
|
8
8
|
import { normalizeToolEnvelope } from './tool-envelope.mjs';
|
|
9
9
|
import { isInvalidToolArgsMarker } from '../providers/openai-compat-stream.mjs';
|
|
10
|
-
import { _intraTurnSig, _isMutationTool, _isReadTool, _isScopedCacheableTool,
|
|
10
|
+
import { _intraTurnSig, _isMutationTool, _isOrderedGateSkippable, _isReadTool, _isScopedCacheableTool, _stripMcpPrefix } from './loop/tool-classify.mjs';
|
|
11
11
|
import { tryReadCached, tryScopedToolCached } from './read-dedup.mjs';
|
|
12
12
|
import { preDispatchDenyForSession } from './loop/pre-dispatch-deny.mjs';
|
|
13
13
|
import { executeTool } from './loop/tool-exec.mjs';
|
|
14
14
|
import { crossTurnSignature } from './loop/completion-guards.mjs';
|
|
15
|
-
import { getToolKind, isParallelDispatchable, isToolCallDedupEligible } from './loop/tool-helpers.mjs';
|
|
15
|
+
import { getToolKind, isEagerDispatchable, isParallelDispatchable, isToolCallDedupEligible } from './loop/tool-helpers.mjs';
|
|
16
16
|
|
|
17
17
|
export function createEagerDispatcher({
|
|
18
18
|
tools, cwd, sessionId, sessionRef, signal, opts,
|
|
@@ -40,18 +40,11 @@ export function createEagerDispatcher({
|
|
|
40
40
|
// resets at the turn boundary without leaking across getIterations().
|
|
41
41
|
const _eagerInFlightSigs = new Map();
|
|
42
42
|
const epoch = { mutation: 0 };
|
|
43
|
-
//
|
|
44
|
-
//
|
|
45
|
-
//
|
|
46
|
-
//
|
|
47
|
-
|
|
48
|
-
let _streamSawMutation = false;
|
|
49
|
-
const _hasEarlierMutation = (calls, index) => {
|
|
50
|
-
for (let k = 0; k < index; k += 1) {
|
|
51
|
-
if (_isMutationTool(calls[k]?.name)) return true;
|
|
52
|
-
}
|
|
53
|
-
return false;
|
|
54
|
-
};
|
|
43
|
+
// True once an apply_patch tool_use has been seen in THIS turn's
|
|
44
|
+
// stream. Before that, every side-effect call streamed so far sits
|
|
45
|
+
// before the first patch in call order (segment 0), so it may start
|
|
46
|
+
// immediately — a patch that arrives later cannot gate an EARLIER call.
|
|
47
|
+
let _streamSeenMutation = false;
|
|
55
48
|
const startEagerTool = (call) => {
|
|
56
49
|
if (!call?.id || pending.has(call.id) || !isParallelDispatchable(call.name)) return null;
|
|
57
50
|
// Never eager-execute a call whose arguments failed to parse
|
|
@@ -107,7 +100,7 @@ export function createEagerDispatcher({
|
|
|
107
100
|
if (_dedupEligible) _eagerInFlightSigs.set(_sig, call.id);
|
|
108
101
|
entry.promise = (async () => {
|
|
109
102
|
try {
|
|
110
|
-
return { ok: true, value: await executeToolFn(call.name, call.arguments, cwd, sessionId, sessionRef, { toolCallId: call.id, signal, notifyFn: opts.notifyFn, toolApprovalHook: opts.onToolApproval, iteration: getNextIteration() }) };
|
|
103
|
+
return { ok: true, value: await executeToolFn(call.name, call.arguments, cwd, sessionId, sessionRef, { toolCallId: call.id, signal, notifyFn: opts.notifyFn, toolApprovalHook: opts.onToolApproval, iteration: getNextIteration(), deferShellCwdCommit: true }) };
|
|
111
104
|
} catch (error) {
|
|
112
105
|
return { ok: false, error };
|
|
113
106
|
}
|
|
@@ -168,16 +161,21 @@ export function createEagerDispatcher({
|
|
|
168
161
|
return entry;
|
|
169
162
|
};
|
|
170
163
|
const startEagerRun = (calls, startIndex, dupSet) => {
|
|
164
|
+
const _nextOrderedMutationIndex = calls.findIndex(
|
|
165
|
+
(call, index) => index >= startIndex && _isMutationTool(call?.name),
|
|
166
|
+
);
|
|
171
167
|
for (let j = startIndex; j < calls.length; j += 1) {
|
|
172
168
|
const call = calls[j];
|
|
173
|
-
//
|
|
174
|
-
//
|
|
175
|
-
//
|
|
169
|
+
// Side effects may eager-start in the current segment but not
|
|
170
|
+
// across the next apply_patch barrier. Known read-only work
|
|
171
|
+
// may cross it and is protected by mutation-epoch re-execution.
|
|
176
172
|
if (!call?.id || !isParallelDispatchable(call.name)) continue;
|
|
177
173
|
if (dupSet && dupSet.has(call.id)) continue;
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
174
|
+
if (
|
|
175
|
+
_nextOrderedMutationIndex >= 0
|
|
176
|
+
&& j >= _nextOrderedMutationIndex
|
|
177
|
+
&& _isOrderedGateSkippable(call.name)
|
|
178
|
+
) continue;
|
|
181
179
|
// A null return here is NOT a state barrier. It means a
|
|
182
180
|
// non-barrier stub — intra-turn in-flight dup, repeat-failure /
|
|
183
181
|
// cross-turn dedup, pre-dispatch-deny, invalid-args, or a cache
|
|
@@ -188,9 +186,17 @@ export function createEagerDispatcher({
|
|
|
188
186
|
}
|
|
189
187
|
};
|
|
190
188
|
const onToolCall = (call) => {
|
|
191
|
-
if (_isMutationTool(call?.name)) {
|
|
192
|
-
|
|
193
|
-
|
|
189
|
+
if (_isMutationTool(call?.name)) {
|
|
190
|
+
_streamSeenMutation = true;
|
|
191
|
+
return;
|
|
192
|
+
}
|
|
193
|
+
// Declared read-only calls always overlap streaming (epoch guard
|
|
194
|
+
// re-executes them if a later patch lands). Side-effect calls may
|
|
195
|
+
// stream-start only while NO apply_patch has streamed yet: their
|
|
196
|
+
// call-order position is already fixed before the first barrier,
|
|
197
|
+
// matching exactly what startEagerRun would do post-batch. Once a
|
|
198
|
+
// patch has streamed, later side effects wait for their segment.
|
|
199
|
+
if (!isEagerDispatchable(call?.name, tools) && _streamSeenMutation) return;
|
|
194
200
|
startEagerTool(call);
|
|
195
201
|
};
|
|
196
202
|
return { pending, epoch, startEagerTool, startEagerRun, onToolCall };
|