maka-agent 0.2.0-dev.31.20260913 → 0.2.0-dev.32.20260914
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/pi-transcript.js +3 -4
- package/native/runtime-host-windows-task-launcher/prebuilds/win32-x64/maka-runtime-host-task-launcher.exe +0 -0
- package/node_modules/@maka/core/dist/agent-graph-schedule.js +15 -4
- package/node_modules/@maka/core/dist/executor-id.js +22 -0
- package/node_modules/@maka/core/dist/external-session.js +18 -0
- package/node_modules/@maka/core/dist/model-call-usage-projection.js +0 -13
- package/node_modules/@maka/core/dist/runtime-event.js +19 -3
- package/node_modules/@maka/core/dist/session-send-projection.js +6 -3
- package/node_modules/@maka/core/dist/session.js +8 -2
- package/node_modules/@maka/core/dist/shell-run-result.js +1 -0
- package/node_modules/@maka/core/dist/shell-run.js +4 -0
- package/node_modules/@maka/core/dist/work-board.js +94 -1
- package/node_modules/@maka/core/package.json +1 -0
- package/node_modules/@maka/eval/dist/fleet-simulation.js +270 -0
- package/node_modules/@maka/eval/dist/fleet-store.js +243 -0
- package/node_modules/@maka/eval/dist/fleet-worker.js +125 -0
- package/node_modules/@maka/eval/dist/fleet.js +329 -0
- package/node_modules/@maka/eval/dist/index.js +3 -0
- package/node_modules/@maka/runtime/dist/agent-run.js +2 -17
- package/node_modules/@maka/runtime/dist/ai-sdk-compaction.js +6 -2
- package/node_modules/@maka/runtime/dist/ai-sdk-message-projection.js +36 -24
- package/node_modules/@maka/runtime/dist/ai-sdk-turn.js +257 -330
- package/node_modules/@maka/runtime/dist/background-task-health-tool.js +104 -0
- package/node_modules/@maka/runtime/dist/history-compaction.js +3 -1
- package/node_modules/@maka/runtime/dist/local-web-fetch.js +1 -1
- package/node_modules/@maka/runtime/dist/model-adapter.js +62 -53
- package/node_modules/@maka/runtime/dist/model-history.js +1 -5
- package/node_modules/@maka/runtime/dist/plugin-executor-backend.js +311 -0
- package/node_modules/@maka/runtime/dist/plugin-executor-service.js +344 -0
- package/node_modules/@maka/runtime/dist/provider-error-classification.js +7 -1
- package/node_modules/@maka/runtime/dist/provider-request-telemetry.js +35 -8
- package/node_modules/@maka/runtime/dist/runtime-event-backfill.js +7 -1
- package/node_modules/@maka/runtime/dist/runtime-event-read-model.js +1 -0
- package/node_modules/@maka/runtime/dist/runtime-invocation-route.js +49 -0
- package/node_modules/@maka/runtime/dist/runtime-kernel.js +72 -150
- package/node_modules/@maka/runtime/dist/runtime-read-model.js +0 -2
- package/node_modules/@maka/runtime/dist/session-event-runtime-mapper.js +3 -0
- package/node_modules/@maka/runtime/dist/session-manager.js +100 -58
- package/node_modules/@maka/runtime/dist/shell-run-manager.js +9 -3
- package/node_modules/@maka/runtime/dist/shell-run-tool-result.js +1 -0
- package/node_modules/@maka/runtime/dist/stream-graph-schedule-reconcile.js +1 -0
- package/node_modules/@maka/runtime/dist/stream-graph-supervisor-tools.js +26 -4
- package/node_modules/@maka/runtime/dist/subagent-tools.js +7 -0
- package/node_modules/@maka/runtime/dist/tool-runtime.js +45 -466
- package/node_modules/@maka/runtime/package.json +3 -0
- package/node_modules/@maka/runtime-host/dist/adapter/session-projector.js +7 -2
- package/node_modules/@maka/runtime-host/dist/client/session-catalog-summary.js +1 -0
- package/node_modules/@maka/runtime-host/dist/protocol/external-session.js +24 -3
- package/node_modules/@maka/runtime-host/dist/protocol/index.js +4 -1
- package/node_modules/@maka/runtime-host/dist/protocol/plugin-platform.js +42 -5
- package/node_modules/@maka/runtime-host/dist/protocol/session-catalog.js +31 -3
- package/node_modules/@maka/runtime-host/dist/protocol/session-continuity.js +6 -0
- package/node_modules/@maka/runtime-host/dist/server/child-agent-composition.js +1 -0
- package/node_modules/@maka/runtime-host/dist/server/execution-artifacts.js +1 -22
- package/node_modules/@maka/runtime-host/dist/server/execution-composition.js +54 -8
- package/node_modules/@maka/runtime-host/dist/server/external-session-coordinator.js +14 -1
- package/node_modules/@maka/runtime-host/dist/server/host-session-availability.js +10 -2
- package/node_modules/@maka/runtime-host/dist/server/plugin-platform-coordinator.js +6 -0
- package/node_modules/@maka/runtime-host/dist/server/plugin-platform.js +6 -0
- package/node_modules/@maka/runtime-host/dist/server/session-catalog-coordinator.js +56 -16
- package/node_modules/@maka/runtime-host/dist/server/session-continuity-coordinator.js +4 -3
- package/node_modules/@maka/runtime-host/dist/server/session-revision-coordinator.js +4 -0
- package/node_modules/@maka/runtime-host/dist/server/web-fetch-tool.js +37 -1
- package/node_modules/@maka/storage/dist/claude-code-session-adapter.js +450 -234
- package/node_modules/@maka/storage/dist/claude-code-transcript-lineage.js +106 -77
- package/node_modules/@maka/storage/dist/legacy-run-header.js +2 -2
- package/node_modules/@maka/storage/dist/model-call-usage-sql.js +1 -1
- package/node_modules/@maka/storage/dist/session-store.js +12 -2
- package/node_modules/@maka/storage/dist/sqlite-long-term-memory-store.js +86 -12
- package/node_modules/@maka/storage/dist/work-board-store.js +40 -1
- package/package.json +1 -1
|
@@ -39,6 +39,7 @@ import { REQUEST_SANDBOX_BOUNDARY_TOOL_NAME, SANDBOX_BOUNDARY_DENIED_FOR_TURN, S
|
|
|
39
39
|
import { buildRuntimeEventModelReplayPlan, buildSteeringEnvelope, collectToolActivityTurnIds, compatibleProviderReasoningReplayEventIds, formatTextWithInlineRefs, steeringMessagesMissingFromBase, steeringModelMessage, } from './model-history.js';
|
|
40
40
|
import { toolSchemaCharsForDiagnostics, requestCompositionToolSchemas, stableHash, toolCatalogHash, } from './request-shape.js';
|
|
41
41
|
import { toolAvailabilityHash } from './tool-availability.js';
|
|
42
|
+
import { hasFinalizedReasoning } from './ai-sdk-message-projection.js';
|
|
42
43
|
import { renderSwarmModePrompt } from './swarm-mode.js';
|
|
43
44
|
import { renderGraphModePrompt } from './graph-mode.js';
|
|
44
45
|
import { modelUsesNativeOpenAiResponses } from './model-runtime.js';
|
|
@@ -345,13 +346,6 @@ function joinPromptFragments(fragments) {
|
|
|
345
346
|
}
|
|
346
347
|
const MAX_WAITING_CODE_MODE_CELLS = 1;
|
|
347
348
|
const MAX_PROVIDER_ATTEMPTS_PER_STEP = 10;
|
|
348
|
-
const MAX_IDLE_WATCHDOG_RETRIES_PER_STEP = 1;
|
|
349
|
-
const MAX_INCOMPLETE_STREAM_RETRIES_PER_STEP = 1;
|
|
350
|
-
// A mid-stream cut after partial thinking seals one transcript fragment per
|
|
351
|
-
// retry. A gateway that systematically kills long thinking streams (the
|
|
352
|
-
// 2026-08-28 incident shape) would otherwise spend the full attempt budget
|
|
353
|
-
// accumulating fragments before failing anyway, so fail fast after one.
|
|
354
|
-
const MAX_SEALED_THINKING_RETRIES_PER_STEP = 1;
|
|
355
349
|
const PROVIDER_RETRY_BASE_DELAY_MS = 1_000;
|
|
356
350
|
const PROVIDER_RETRY_MAX_DELAY_MS = 32_000;
|
|
357
351
|
const PROVIDER_RETRY_JITTER_FACTOR = 0.25;
|
|
@@ -375,9 +369,6 @@ function providerRetryReason(kind) {
|
|
|
375
369
|
return 'unknown';
|
|
376
370
|
}
|
|
377
371
|
}
|
|
378
|
-
function isIncompleteProviderFinishReason(reason) {
|
|
379
|
-
return reason === undefined || reason === 'other' || reason === 'unknown';
|
|
380
|
-
}
|
|
381
372
|
/**
|
|
382
373
|
* The mutable state of ONE `send()`.
|
|
383
374
|
*
|
|
@@ -611,11 +602,6 @@ export class AiSdkTurn {
|
|
|
611
602
|
let stepTextPartStartOffset = 0;
|
|
612
603
|
let stepThinkingParts = [];
|
|
613
604
|
let stepThinkingPartsById = new Map();
|
|
614
|
-
let stepContentOrder = [];
|
|
615
|
-
const recordStepContent = (kind) => {
|
|
616
|
-
if (!stepContentOrder.includes(kind))
|
|
617
|
-
stepContentOrder.push(kind);
|
|
618
|
-
};
|
|
619
605
|
// Flush the current step's AssistantMessage (text + thinking) and the paired
|
|
620
606
|
// terminal thinking/text events, then clear the per-step accumulators.
|
|
621
607
|
// Persist when the step produced text OR reasoning — a thinking-only step
|
|
@@ -631,17 +617,37 @@ export class AiSdkTurn {
|
|
|
631
617
|
stepTextPartStartOffset = 0;
|
|
632
618
|
stepThinkingParts = [];
|
|
633
619
|
stepThinkingPartsById = new Map();
|
|
634
|
-
stepContentOrder = [];
|
|
635
620
|
};
|
|
636
|
-
const flushStep = async () => {
|
|
621
|
+
const flushStep = async (interrupted = false) => {
|
|
637
622
|
const hasThinking = stepThinkingParts.length > 0;
|
|
638
623
|
if (stepText.length === 0 && !hasThinking) {
|
|
639
624
|
resetStep();
|
|
640
625
|
return;
|
|
641
626
|
}
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
627
|
+
// A provider may finalize one reasoning item before the next item fails.
|
|
628
|
+
// Keep completed and interrupted parts in separate transcript rows so
|
|
629
|
+
// missing-ledger recovery preserves the same model visibility.
|
|
630
|
+
const fragments = [];
|
|
631
|
+
for (const part of stepThinkingParts) {
|
|
632
|
+
const partInterrupted = interrupted && !hasFinalizedReasoning(part);
|
|
633
|
+
let fragment = fragments.at(-1);
|
|
634
|
+
if (!fragment || fragment.interrupted !== partInterrupted) {
|
|
635
|
+
fragment = { thinking: [], text: '', interrupted: partInterrupted };
|
|
636
|
+
fragments.push(fragment);
|
|
637
|
+
}
|
|
638
|
+
fragment.thinking.push(part);
|
|
639
|
+
}
|
|
640
|
+
if (stepText.length > 0) {
|
|
641
|
+
let fragment = fragments.at(-1);
|
|
642
|
+
if (!fragment || fragment.interrupted !== interrupted) {
|
|
643
|
+
fragment = { thinking: [], text: '', interrupted };
|
|
644
|
+
fragments.push(fragment);
|
|
645
|
+
}
|
|
646
|
+
fragment.text = stepText;
|
|
647
|
+
}
|
|
648
|
+
for (const [index, fragment] of fragments.entries()) {
|
|
649
|
+
const stepId = index === 0 ? currentStepMessageId : this.deps.newId();
|
|
650
|
+
for (const part of fragment.thinking) {
|
|
645
651
|
queue.push({
|
|
646
652
|
type: 'thinking_complete',
|
|
647
653
|
id: this.deps.newId(),
|
|
@@ -649,6 +655,7 @@ export class AiSdkTurn {
|
|
|
649
655
|
ts: this.deps.now(),
|
|
650
656
|
messageId: stepId,
|
|
651
657
|
text: part.text,
|
|
658
|
+
...(fragment.interrupted ? { interrupted: true } : {}),
|
|
652
659
|
...(part.signature !== undefined ? { signature: part.signature } : {}),
|
|
653
660
|
// No sanitiser here, unlike the tool call below: these options are
|
|
654
661
|
// not the provider's object. `translateChunk` rebuilds reasoning
|
|
@@ -661,24 +668,25 @@ export class AiSdkTurn {
|
|
|
661
668
|
: {}),
|
|
662
669
|
});
|
|
663
670
|
}
|
|
671
|
+
queue.push({
|
|
672
|
+
type: 'text_complete',
|
|
673
|
+
id: this.deps.newId(),
|
|
674
|
+
turnId,
|
|
675
|
+
ts: this.deps.now(),
|
|
676
|
+
messageId: stepId,
|
|
677
|
+
text: fragment.text,
|
|
678
|
+
...(fragment.interrupted ? { interrupted: true } : {}),
|
|
679
|
+
...(index === fragments.length - 1 && stepTextProviderOptions !== undefined
|
|
680
|
+
? { providerOptions: stepTextProviderOptions }
|
|
681
|
+
: {}),
|
|
682
|
+
});
|
|
664
683
|
}
|
|
665
|
-
queue.push({
|
|
666
|
-
type: 'text_complete',
|
|
667
|
-
id: this.deps.newId(),
|
|
668
|
-
turnId,
|
|
669
|
-
ts: this.deps.now(),
|
|
670
|
-
messageId: stepId,
|
|
671
|
-
text: stepText,
|
|
672
|
-
...(stepTextProviderOptions !== undefined
|
|
673
|
-
? { providerOptions: stepTextProviderOptions }
|
|
674
|
-
: {}),
|
|
675
|
-
});
|
|
676
684
|
this.finalAssistantText = stepText.length > 0 ? stepText : undefined;
|
|
677
685
|
resetStep();
|
|
678
686
|
};
|
|
679
687
|
let tokenUsage;
|
|
680
688
|
let tokenUsageCostUsd;
|
|
681
|
-
// Per-send sum of every
|
|
689
|
+
// Per-send sum of every completed step's usage, merged at settlement
|
|
682
690
|
// boundary. When the send aborts (mid-turn exhaust, user stop, stream
|
|
683
691
|
// error) the SDK's cumulative `usage` promise may not resolve, but this sum is
|
|
684
692
|
// real provider-reported evidence for the steps that did finish — IF every
|
|
@@ -945,7 +953,9 @@ export class AiSdkTurn {
|
|
|
945
953
|
idleTimeoutMs: this.deps.backend.streamIdleTimeoutMs,
|
|
946
954
|
...this.deps.backend.streamWatchdogTimer,
|
|
947
955
|
onTimeout: (timeout) => {
|
|
948
|
-
const error = new Error(formatStreamWatchdogError(timeout))
|
|
956
|
+
const error = Object.assign(new Error(formatStreamWatchdogError(timeout)), {
|
|
957
|
+
code: 'MODEL_STREAM_TIMEOUT',
|
|
958
|
+
});
|
|
949
959
|
watchdogTimeoutState.current = { phase: timeout.phase, error };
|
|
950
960
|
providerRequestAbortController.abort(error);
|
|
951
961
|
},
|
|
@@ -1020,7 +1030,9 @@ export class AiSdkTurn {
|
|
|
1020
1030
|
// folded through the same reducer before it becomes messages. Without
|
|
1021
1031
|
// this, a result archived at step N is rebuilt in full at step N+1 and
|
|
1022
1032
|
// the ledger's account of what the model sees stops being true.
|
|
1023
|
-
const foldedReplayEvents =
|
|
1033
|
+
const foldedReplayEvents = projectionCheckpoint
|
|
1034
|
+
? await this.deps.compaction.foldEffectiveModelHistory(replayEvents, pruned.projectionSnapshot)
|
|
1035
|
+
: pruned.events;
|
|
1024
1036
|
const replayPlan = buildRuntimeEventModelReplayPlan(foldedReplayEvents, {
|
|
1025
1037
|
toolActivityTurnIds: collectToolActivityTurnIds([
|
|
1026
1038
|
...(input.runtimeContext ?? []),
|
|
@@ -1068,11 +1080,6 @@ export class AiSdkTurn {
|
|
|
1068
1080
|
const requestProjection = shapedProjection;
|
|
1069
1081
|
const completedProviderSteps = [];
|
|
1070
1082
|
let requestMessages = messages;
|
|
1071
|
-
// The compaction module runs at most once per send. This tracks the
|
|
1072
|
-
// reactive entry; the proactive one sets the same flag on the mid-turn
|
|
1073
|
-
// state, and each consults the other, so a send that already folded
|
|
1074
|
-
// reports the oversized message instead of folding again (#4559).
|
|
1075
|
-
let overflowRetryUsed = false;
|
|
1076
1083
|
let result;
|
|
1077
1084
|
let providerOutcome;
|
|
1078
1085
|
let finishReason = 'stop';
|
|
@@ -1153,41 +1160,27 @@ export class AiSdkTurn {
|
|
|
1153
1160
|
: undefined;
|
|
1154
1161
|
providerRequestTracker?.setStep(runtimeSteps, requestCompositionId);
|
|
1155
1162
|
let attemptMessages = projectedMessages;
|
|
1156
|
-
let providerAttempt =
|
|
1157
|
-
let idleWatchdogRetryCount = 0;
|
|
1158
|
-
let incompleteStreamRetryCount = 0;
|
|
1159
|
-
let sealedThinkingRetryCount = 0;
|
|
1163
|
+
let providerAttempt = 0;
|
|
1160
1164
|
const returnedToolCalls = [];
|
|
1161
|
-
let providerToolActivityCount = 0;
|
|
1162
1165
|
const providerToolInputs = new Map();
|
|
1163
1166
|
let providerStepUsage;
|
|
1164
1167
|
for (;;) {
|
|
1168
|
+
providerAttempt += 1;
|
|
1169
|
+
// Local calls are only admitted after a successful provider outcome.
|
|
1170
|
+
// A new physical request must not inherit the failed one's intents.
|
|
1171
|
+
returnedToolCalls.length = 0;
|
|
1165
1172
|
providerRequestAbortController = new AbortController();
|
|
1166
1173
|
watchdogTimeoutState.current = null;
|
|
1167
1174
|
startWatchdog();
|
|
1168
1175
|
// Monotonic facts for this physical request. The step accumulators
|
|
1169
1176
|
// are cleared after flushStep(), so they cannot decide whether a
|
|
1170
1177
|
// later stream failure is safe to retry.
|
|
1171
|
-
let
|
|
1172
|
-
let attemptSawThinking = false;
|
|
1178
|
+
let attemptSawVisibleContent = false;
|
|
1173
1179
|
let attemptSawToolActivity = false;
|
|
1174
|
-
let
|
|
1175
|
-
let
|
|
1176
|
-
const attemptHasNoObservableOutput = () => !
|
|
1177
|
-
|
|
1178
|
-
!attemptSawToolActivity &&
|
|
1179
|
-
!attemptSawContinuationMetadata &&
|
|
1180
|
-
!attemptReachedStepBoundary;
|
|
1181
|
-
// Thinking is the only output that can be sealed into its own
|
|
1182
|
-
// message before a retry: flushStep() closes the fragment under
|
|
1183
|
-
// the current message id and the retry streams into a fresh one,
|
|
1184
|
-
// so the user never sees spliced or duplicated content. Text,
|
|
1185
|
-
// tool activity, continuation metadata, and step boundaries stay
|
|
1186
|
-
// non-recoverable for the reasons each of them is tracked.
|
|
1187
|
-
const attemptCanRecoverWithSealedThinking = () => !attemptSawText &&
|
|
1188
|
-
!attemptSawToolActivity &&
|
|
1189
|
-
!attemptSawContinuationMetadata &&
|
|
1190
|
-
!attemptReachedStepBoundary;
|
|
1180
|
+
let attemptSawToolInput = false;
|
|
1181
|
+
let attemptSawReplayBarrier = false;
|
|
1182
|
+
const attemptHasNoObservableOutput = () => !attemptSawVisibleContent && !attemptSawToolActivity && !attemptSawReplayBarrier;
|
|
1183
|
+
const attemptCanReplay = () => !attemptSawToolActivity && !attemptSawReplayBarrier;
|
|
1191
1184
|
this.memorySourceMessages = [...attemptMessages];
|
|
1192
1185
|
this.memorySourceEventMessagePositions =
|
|
1193
1186
|
this.deps.messageProjection.memoryEventMessagePositions(attemptMessages);
|
|
@@ -1240,185 +1233,18 @@ export class AiSdkTurn {
|
|
|
1240
1233
|
// trailer and consume the one authoritative outcome below.
|
|
1241
1234
|
break;
|
|
1242
1235
|
}
|
|
1243
|
-
const incompleteFinish = (event.kind === 'finish' || event.kind === 'step-finish') &&
|
|
1244
|
-
isIncompleteProviderFinishReason(event.finishReason);
|
|
1245
|
-
if ((event.kind === 'finish' || event.kind === 'step-finish') && !incompleteFinish) {
|
|
1246
|
-
attemptReachedStepBoundary = true;
|
|
1247
|
-
}
|
|
1248
|
-
if (event.kind === 'step-finish') {
|
|
1249
|
-
// AI SDK can synthesize `finish-step(other)` when the provider
|
|
1250
|
-
// stream reaches EOF without a terminal frame. That is not a
|
|
1251
|
-
// completed model step and must not consume the step budget or
|
|
1252
|
-
// checkpoint imaginary usage before the safe retry below.
|
|
1253
|
-
if (!incompleteFinish) {
|
|
1254
|
-
// Step boundary: AI SDK 7 delimits steps with `finish-step`
|
|
1255
|
-
// (and `step-finish` for legacy replay fixtures); the adapter
|
|
1256
|
-
// reduces both to this event. A duplicate boundary is harmless:
|
|
1257
|
-
// the second flush no-ops (accumulators already cleared) and one
|
|
1258
|
-
// extra id rotation just discards an unused id.
|
|
1259
|
-
runtimeSteps += 1;
|
|
1260
|
-
const stepUsage = event.usage;
|
|
1261
|
-
providerStepUsage = stepUsage;
|
|
1262
|
-
if (!stepUsage)
|
|
1263
|
-
sawUnusableStepUsage = true;
|
|
1264
|
-
// Silent eviction / rewrite check (#4559): this step only
|
|
1265
|
-
// appended (no fold, no prune, no image omission) yet the
|
|
1266
|
-
// provider counted no more input tokens than for the previous
|
|
1267
|
-
// request. Not-greater, not strictly-fewer: a provider that
|
|
1268
|
-
// truncates to a fixed window (Ollama's `num_ctx`) reports the
|
|
1269
|
-
// same total on every later request while Maka keeps
|
|
1270
|
-
// appending, so a plateau is the signal, and an equal count
|
|
1271
|
-
// after an append is already impossible without provider-side
|
|
1272
|
-
// eviction or rewriting. Input against input: the previous
|
|
1273
|
-
// reply's reasoning may not be resent, so input + output is
|
|
1274
|
-
// not the floor of the next input on every wire.
|
|
1275
|
-
const completedRequestIndex = runtimeSteps - 1;
|
|
1276
|
-
// A finalization step resolves an empty tool set, so its
|
|
1277
|
-
// request legitimately drops several thousand schema tokens
|
|
1278
|
-
// with no fold, prune or image omission. Maka shaped that
|
|
1279
|
-
// request; the provider did not drop anything.
|
|
1280
|
-
const toolSchemaShrank = lastStepActiveToolCount !== undefined &&
|
|
1281
|
-
activeToolsForRequest.length < lastStepActiveToolCount;
|
|
1282
|
-
// Across the send boundary the comparison is the same one,
|
|
1283
|
-
// against the last request a provider accepted before this
|
|
1284
|
-
// send. A provider that truncates to a fixed window reports
|
|
1285
|
-
// the same input on every later request while the user keeps
|
|
1286
|
-
// adding turns, and a send of one or two steps never sees
|
|
1287
|
-
// that from the inside: the live evidence plateaus at 3,716
|
|
1288
|
-
// input tokens across eight turns with nothing reported
|
|
1289
|
-
// (#4623). The first request of a send therefore compares
|
|
1290
|
-
// against the persisted anchor, which is route-validated
|
|
1291
|
-
// where it is read; a fold before that request would explain
|
|
1292
|
-
// a smaller input by itself, so it disables the comparison.
|
|
1293
|
-
const acrossSends = completedRequestIndex === 0;
|
|
1294
|
-
const priorInput = acrossSends
|
|
1295
|
-
? midTurnState?.compactionAppliedThisSend === true
|
|
1296
|
-
? undefined
|
|
1297
|
-
: midTurnState?.priorAcceptedInputTokens
|
|
1298
|
-
: lastStepInputTokens;
|
|
1299
|
-
if (!this.deps.session.contextProviderDroppingReported &&
|
|
1300
|
-
!toolSchemaShrank &&
|
|
1301
|
-
midTurnState &&
|
|
1302
|
-
priorInput !== undefined &&
|
|
1303
|
-
midTurnState.replacedStepNumber !== completedRequestIndex &&
|
|
1304
|
-
pruneAppliedAtStep !== completedRequestIndex &&
|
|
1305
|
-
midTurnState.omittedImageToolResults.size === 0 &&
|
|
1306
|
-
stepUsage !== undefined &&
|
|
1307
|
-
Number.isFinite(stepUsage.inputTokens) &&
|
|
1308
|
-
stepUsage.inputTokens > 0 &&
|
|
1309
|
-
// Across sends the test is equality, not "did not grow".
|
|
1310
|
-
// Inside a send Maka knows it only appended, so any
|
|
1311
|
-
// shortfall is the provider's. Across the boundary it does
|
|
1312
|
-
// not: a manual compaction leaves the pre-compaction anchor
|
|
1313
|
-
// behind, a turn can carry a smaller tool set, and a user
|
|
1314
|
-
// can edit or branch history. All three shrink the input
|
|
1315
|
-
// legitimately, and none of them lands on exactly the same
|
|
1316
|
-
// count. A provider truncating to a fixed window does, on
|
|
1317
|
-
// every later request.
|
|
1318
|
-
(acrossSends
|
|
1319
|
-
? stepUsage.inputTokens === priorInput
|
|
1320
|
-
: stepUsage.inputTokens <= priorInput)) {
|
|
1321
|
-
this.deps.session.contextProviderDroppingReported = true;
|
|
1322
|
-
await this.recordSystemNote('context_provider_dropping', turnId, {
|
|
1323
|
-
inputTokens: stepUsage.inputTokens,
|
|
1324
|
-
priorInputTokens: priorInput,
|
|
1325
|
-
});
|
|
1326
|
-
}
|
|
1327
|
-
// Fail closed: reset on every step boundary so a missing final
|
|
1328
|
-
// step's usage does not leave a stale value from an earlier step.
|
|
1329
|
-
// The reply needed more room than the declared window had
|
|
1330
|
-
// left after this request's own input. Both halves are the
|
|
1331
|
-
// provider's numbers, read after the fact: the reserve that
|
|
1332
|
-
// should have kept them apart was measured from a smaller
|
|
1333
|
-
// previous reply. Say so once per send; the next request
|
|
1334
|
-
// folds anyway because the baseline now exceeds the window.
|
|
1335
|
-
if (!contextWindowOverrunNoteWritten &&
|
|
1336
|
-
midTurnState?.capacity !== undefined &&
|
|
1337
|
-
stepUsage !== undefined &&
|
|
1338
|
-
Number.isFinite(stepUsage.inputTokens) &&
|
|
1339
|
-
stepUsage.inputTokens > 0 &&
|
|
1340
|
-
Number.isFinite(stepUsage.outputTokens) &&
|
|
1341
|
-
stepUsage.outputTokens > 0 &&
|
|
1342
|
-
stepUsage.inputTokens + stepUsage.outputTokens > midTurnState.capacity) {
|
|
1343
|
-
contextWindowOverrunNoteWritten = true;
|
|
1344
|
-
await this.recordSystemNote('context_window_overrun', turnId, {
|
|
1345
|
-
usedTokens: stepUsage.inputTokens + stepUsage.outputTokens,
|
|
1346
|
-
declaredContextWindow: midTurnState.capacity,
|
|
1347
|
-
});
|
|
1348
|
-
}
|
|
1349
|
-
// Nothing declared, and the provider accepted a request past
|
|
1350
|
-
// the window this model reports. Every other signal in this
|
|
1351
|
-
// design stays dark there: no rejection to recover from, no
|
|
1352
|
-
// plateau to read, and no declaration to arm the proactive
|
|
1353
|
-
// threshold, so the session degrades quietly and
|
|
1354
|
-
// indefinitely (#4634). Report the two real numbers and
|
|
1355
|
-
// leave the decision with the user: a reported window is a
|
|
1356
|
-
// hint, and Maka still declares nothing on their behalf.
|
|
1357
|
-
//
|
|
1358
|
-
// Once per crossing, not once per send. On these providers
|
|
1359
|
-
// usage keeps growing past the line (305K → 322K observed),
|
|
1360
|
-
// so the note fires on the transition: the previous accepted
|
|
1361
|
-
// total was still inside the reported window and this one is
|
|
1362
|
-
// not. The baseline carries that previous total across
|
|
1363
|
-
// sessions through the persisted anchor, so a resumed
|
|
1364
|
-
// session does not repeat a crossing it already reported.
|
|
1365
|
-
if (!contextReportedWindowNoteWritten &&
|
|
1366
|
-
midTurnState !== undefined &&
|
|
1367
|
-
midTurnState.capacity === undefined &&
|
|
1368
|
-
stepUsage !== undefined &&
|
|
1369
|
-
Number.isFinite(stepUsage.inputTokens) &&
|
|
1370
|
-
stepUsage.inputTokens > 0 &&
|
|
1371
|
-
Number.isFinite(stepUsage.outputTokens)) {
|
|
1372
|
-
const reported = resolveSelectedModelContextWindow(this.deps.backend.connection, this.deps.backend.modelId);
|
|
1373
|
-
const used = stepUsage.inputTokens + Math.max(0, stepUsage.outputTokens);
|
|
1374
|
-
// `baselineTokens` still describes the request before this
|
|
1375
|
-
// one: the capacity hook sets it from the previous step, or
|
|
1376
|
-
// from the persisted anchor on a send's first request.
|
|
1377
|
-
const previousTotal = midTurnState.baselineTokens;
|
|
1378
|
-
const crossedNow = reported !== undefined &&
|
|
1379
|
-
used > reported &&
|
|
1380
|
-
(previousTotal === undefined || previousTotal <= reported);
|
|
1381
|
-
if (reported !== undefined && crossedNow) {
|
|
1382
|
-
contextReportedWindowNoteWritten = true;
|
|
1383
|
-
await this.recordSystemNote('context_reported_window_exceeded', turnId, {
|
|
1384
|
-
usedTokens: used,
|
|
1385
|
-
reportedContextWindow: reported,
|
|
1386
|
-
});
|
|
1387
|
-
}
|
|
1388
|
-
}
|
|
1389
|
-
lastStepInputTokens = stepUsage?.inputTokens;
|
|
1390
|
-
lastStepOutputTokens = stepUsage?.outputTokens;
|
|
1391
|
-
lastStepActiveToolCount = activeToolsForRequest.length;
|
|
1392
|
-
// A `finishReason: length` is deliberately not a trigger. The
|
|
1393
|
-
// reply may have been cut because the provider ran out of
|
|
1394
|
-
// window room, or because the provider's own output cap is
|
|
1395
|
-
// lower than the one Maka sends. Those are indistinguishable
|
|
1396
|
-
// from outside, and an indistinguishable signal must not
|
|
1397
|
-
// drive an action; the cut reply is visible to the user
|
|
1398
|
-
// either way (#4559).
|
|
1399
|
-
if (stepUsage) {
|
|
1400
|
-
completedStepUsage = mergeNormalizedUsage(completedStepUsage, stepUsage);
|
|
1401
|
-
this.deps.session.cumulativeUsageCheckpoint = mergeNormalizedUsage(this.deps.session.cumulativeUsageCheckpoint, stepUsage);
|
|
1402
|
-
await this.deps.backend.recordUsageCheckpoint?.({
|
|
1403
|
-
...this.deps.session.cumulativeUsageCheckpoint,
|
|
1404
|
-
costUsd: this.deps.providerTelemetry.normalizedUsageCostUsd(this.deps.session.cumulativeUsageCheckpoint),
|
|
1405
|
-
});
|
|
1406
|
-
}
|
|
1407
|
-
}
|
|
1408
|
-
}
|
|
1409
1236
|
if (event.kind === 'text-start') {
|
|
1410
1237
|
if (stepText.length > 0 && event.providerItemBoundary === true) {
|
|
1238
|
+
attemptSawReplayBarrier = true;
|
|
1411
1239
|
await flushStep();
|
|
1412
1240
|
currentStepMessageId = this.deps.newId();
|
|
1413
1241
|
}
|
|
1414
1242
|
stepTextPartStartOffset = stepText.length;
|
|
1415
1243
|
}
|
|
1416
1244
|
else if (event.kind === 'text') {
|
|
1417
|
-
if (event.text.length > 0)
|
|
1418
|
-
recordStepContent('text');
|
|
1419
1245
|
stepText += event.text;
|
|
1420
1246
|
if (event.text.length > 0)
|
|
1421
|
-
|
|
1247
|
+
attemptSawVisibleContent = true;
|
|
1422
1248
|
queue.push({
|
|
1423
1249
|
type: 'text_delta',
|
|
1424
1250
|
id: this.deps.newId(),
|
|
@@ -1430,7 +1256,7 @@ export class AiSdkTurn {
|
|
|
1430
1256
|
}
|
|
1431
1257
|
else if (event.kind === 'text-end') {
|
|
1432
1258
|
if (event.providerOptions !== undefined) {
|
|
1433
|
-
|
|
1259
|
+
attemptSawReplayBarrier = true;
|
|
1434
1260
|
stepTextProviderOptions = mergeTextProviderOptions(stepTextProviderOptions, stripUndefinedDeep(event.providerOptions), stepTextPartStartOffset);
|
|
1435
1261
|
}
|
|
1436
1262
|
if (event.providerItemBoundary === true) {
|
|
@@ -1440,7 +1266,7 @@ export class AiSdkTurn {
|
|
|
1440
1266
|
}
|
|
1441
1267
|
else if (event.kind === 'thinking-start') {
|
|
1442
1268
|
if (event.providerOptions !== undefined) {
|
|
1443
|
-
|
|
1269
|
+
attemptSawReplayBarrier = true;
|
|
1444
1270
|
}
|
|
1445
1271
|
const part = {
|
|
1446
1272
|
text: '',
|
|
@@ -1455,12 +1281,10 @@ export class AiSdkTurn {
|
|
|
1455
1281
|
}
|
|
1456
1282
|
else if (event.kind === 'thinking') {
|
|
1457
1283
|
if (event.text.length > 0)
|
|
1458
|
-
|
|
1459
|
-
if (event.text.length > 0)
|
|
1460
|
-
attemptSawThinking = true;
|
|
1284
|
+
attemptSawVisibleContent = true;
|
|
1461
1285
|
if (event.providerOptions !== undefined) {
|
|
1462
1286
|
if (event.providerOptionsOrigin !== 'maka_transport') {
|
|
1463
|
-
|
|
1287
|
+
attemptSawReplayBarrier = true;
|
|
1464
1288
|
}
|
|
1465
1289
|
}
|
|
1466
1290
|
const partId = event.reasoningPartId ?? responsesReasoningItemId(event.providerOptions);
|
|
@@ -1516,7 +1340,7 @@ export class AiSdkTurn {
|
|
|
1516
1340
|
});
|
|
1517
1341
|
}
|
|
1518
1342
|
else if (event.kind === 'thinking-signature') {
|
|
1519
|
-
|
|
1343
|
+
attemptSawReplayBarrier = true;
|
|
1520
1344
|
let part = event.reasoningPartId
|
|
1521
1345
|
? stepThinkingPartsById.get(event.reasoningPartId)
|
|
1522
1346
|
: stepThinkingParts.at(-1);
|
|
@@ -1529,17 +1353,17 @@ export class AiSdkTurn {
|
|
|
1529
1353
|
}
|
|
1530
1354
|
part.signature = event.signature;
|
|
1531
1355
|
}
|
|
1532
|
-
else if (event.kind === '
|
|
1356
|
+
else if (event.kind === 'tool-input') {
|
|
1357
|
+
attemptSawToolInput = true;
|
|
1533
1358
|
// The provider has started its own tool. Even without a
|
|
1534
1359
|
// final tool-call/result event, retrying can repeat external
|
|
1535
1360
|
// work that the Runtime cannot observe or reconcile.
|
|
1536
|
-
|
|
1361
|
+
if (event.providerExecuted)
|
|
1362
|
+
attemptSawToolActivity = true;
|
|
1537
1363
|
}
|
|
1538
1364
|
else if (event.kind === 'tool-call') {
|
|
1539
|
-
attemptSawToolActivity = true;
|
|
1540
|
-
recordStepContent('tools');
|
|
1541
1365
|
if (event.toolCall.providerExecuted) {
|
|
1542
|
-
|
|
1366
|
+
attemptSawToolActivity = true;
|
|
1543
1367
|
providerToolInputs.set(event.toolCall.toolCallId, event.toolCall.input);
|
|
1544
1368
|
queue.push({
|
|
1545
1369
|
type: 'tool_start',
|
|
@@ -1566,7 +1390,6 @@ export class AiSdkTurn {
|
|
|
1566
1390
|
}
|
|
1567
1391
|
else if (event.kind === 'provider-tool-result') {
|
|
1568
1392
|
attemptSawToolActivity = true;
|
|
1569
|
-
providerToolActivityCount += 1;
|
|
1570
1393
|
const providerOutput = stripUndefinedDeep(event.output);
|
|
1571
1394
|
queue.push({
|
|
1572
1395
|
type: 'tool_result',
|
|
@@ -1581,44 +1404,185 @@ export class AiSdkTurn {
|
|
|
1581
1404
|
});
|
|
1582
1405
|
providerToolInputs.delete(event.toolCallId);
|
|
1583
1406
|
}
|
|
1584
|
-
else if (event.kind === 'step-finish' && !incompleteFinish) {
|
|
1585
|
-
// The step's text/thinking deltas are all in (the stream is
|
|
1586
|
-
// drained in order), so flush this step's AssistantMessage and
|
|
1587
|
-
// rotate to a fresh id for the next step. Tool settlement
|
|
1588
|
-
// below receives this step's pre-rotation id, so durable replay
|
|
1589
|
-
// can regroup calls with this reasoning/text.
|
|
1590
|
-
await flushStep();
|
|
1591
|
-
if (midTurnState) {
|
|
1592
|
-
// Durability clock: step N's thinking/text completion events
|
|
1593
|
-
// are enqueued by flushStep just above, so only after this
|
|
1594
|
-
// boundary can a seq-ack wait for step N mean anything. Wake
|
|
1595
|
-
// waiters AFTER the increment or they would re-check a stale
|
|
1596
|
-
// count and sleep.
|
|
1597
|
-
midTurnState.flushedSteps += 1;
|
|
1598
|
-
queue.wake();
|
|
1599
|
-
}
|
|
1600
|
-
}
|
|
1601
1407
|
}
|
|
1602
1408
|
watchdogState.current?.stop();
|
|
1603
1409
|
// This timeout belongs to the physical request that just settled.
|
|
1604
1410
|
// Consume it before recovery/flush work: a later persistence error
|
|
1605
1411
|
// must not be reported as the already-handled watchdog timeout.
|
|
1606
|
-
|
|
1412
|
+
consumeWatchdogTimeout();
|
|
1607
1413
|
providerOutcome = await result.outcome;
|
|
1608
|
-
|
|
1609
|
-
|
|
1610
|
-
|
|
1611
|
-
|
|
1612
|
-
!
|
|
1613
|
-
|
|
1614
|
-
|
|
1615
|
-
(
|
|
1616
|
-
|
|
1617
|
-
|
|
1618
|
-
|
|
1619
|
-
|
|
1414
|
+
if (providerOutcome.kind === 'completed') {
|
|
1415
|
+
runtimeSteps += 1;
|
|
1416
|
+
const stepUsage = providerOutcome.usage;
|
|
1417
|
+
providerStepUsage = stepUsage;
|
|
1418
|
+
if (!stepUsage)
|
|
1419
|
+
sawUnusableStepUsage = true;
|
|
1420
|
+
// Silent eviction / rewrite check (#4559): this step only
|
|
1421
|
+
// appended (no fold, no prune, no image omission) yet the
|
|
1422
|
+
// provider counted no more input tokens than for the previous
|
|
1423
|
+
// request. Not-greater, not strictly-fewer: a provider that
|
|
1424
|
+
// truncates to a fixed window (Ollama's `num_ctx`) reports the
|
|
1425
|
+
// same total on every later request while Maka keeps
|
|
1426
|
+
// appending, so a plateau is the signal, and an equal count
|
|
1427
|
+
// after an append is already impossible without provider-side
|
|
1428
|
+
// eviction or rewriting. Input against input: the previous
|
|
1429
|
+
// reply's reasoning may not be resent, so input + output is
|
|
1430
|
+
// not the floor of the next input on every wire.
|
|
1431
|
+
const completedRequestIndex = runtimeSteps - 1;
|
|
1432
|
+
// A finalization step resolves an empty tool set, so its
|
|
1433
|
+
// request legitimately drops several thousand schema tokens
|
|
1434
|
+
// with no fold, prune or image omission. Maka shaped that
|
|
1435
|
+
// request; the provider did not drop anything.
|
|
1436
|
+
const toolSchemaShrank = lastStepActiveToolCount !== undefined &&
|
|
1437
|
+
activeToolsForRequest.length < lastStepActiveToolCount;
|
|
1438
|
+
// Across the send boundary the comparison is the same one,
|
|
1439
|
+
// against the last request a provider accepted before this
|
|
1440
|
+
// send. A provider that truncates to a fixed window reports
|
|
1441
|
+
// the same input on every later request while the user keeps
|
|
1442
|
+
// adding turns, and a send of one or two steps never sees
|
|
1443
|
+
// that from the inside: the live evidence plateaus at 3,716
|
|
1444
|
+
// input tokens across eight turns with nothing reported
|
|
1445
|
+
// (#4623). The first request of a send therefore compares
|
|
1446
|
+
// against the persisted anchor, which is route-validated
|
|
1447
|
+
// where it is read; a fold before that request would explain
|
|
1448
|
+
// a smaller input by itself, so it disables the comparison.
|
|
1449
|
+
const acrossSends = completedRequestIndex === 0;
|
|
1450
|
+
const priorInput = acrossSends
|
|
1451
|
+
? midTurnState?.compactionAppliedThisSend === true
|
|
1452
|
+
? undefined
|
|
1453
|
+
: midTurnState?.priorAcceptedInputTokens
|
|
1454
|
+
: lastStepInputTokens;
|
|
1455
|
+
if (!this.deps.session.contextProviderDroppingReported &&
|
|
1456
|
+
!toolSchemaShrank &&
|
|
1457
|
+
midTurnState &&
|
|
1458
|
+
priorInput !== undefined &&
|
|
1459
|
+
midTurnState.replacedStepNumber !== completedRequestIndex &&
|
|
1460
|
+
pruneAppliedAtStep !== completedRequestIndex &&
|
|
1461
|
+
midTurnState.omittedImageToolResults.size === 0 &&
|
|
1462
|
+
stepUsage !== undefined &&
|
|
1463
|
+
Number.isFinite(stepUsage.inputTokens) &&
|
|
1464
|
+
stepUsage.inputTokens > 0 &&
|
|
1465
|
+
// Across sends the test is equality, not "did not grow".
|
|
1466
|
+
// Inside a send Maka knows it only appended, so any
|
|
1467
|
+
// shortfall is the provider's. Across the boundary it does
|
|
1468
|
+
// not: a manual compaction leaves the pre-compaction anchor
|
|
1469
|
+
// behind, a turn can carry a smaller tool set, and a user
|
|
1470
|
+
// can edit or branch history. All three shrink the input
|
|
1471
|
+
// legitimately, and none of them lands on exactly the same
|
|
1472
|
+
// count. A provider truncating to a fixed window does, on
|
|
1473
|
+
// every later request.
|
|
1474
|
+
(acrossSends
|
|
1475
|
+
? stepUsage.inputTokens === priorInput
|
|
1476
|
+
: stepUsage.inputTokens <= priorInput)) {
|
|
1477
|
+
this.deps.session.contextProviderDroppingReported = true;
|
|
1478
|
+
await this.recordSystemNote('context_provider_dropping', turnId, {
|
|
1479
|
+
inputTokens: stepUsage.inputTokens,
|
|
1480
|
+
priorInputTokens: priorInput,
|
|
1481
|
+
});
|
|
1482
|
+
}
|
|
1483
|
+
// Fail closed: reset on every step boundary so a missing final
|
|
1484
|
+
// step's usage does not leave a stale value from an earlier step.
|
|
1485
|
+
// The reply needed more room than the declared window had
|
|
1486
|
+
// left after this request's own input. Both halves are the
|
|
1487
|
+
// provider's numbers, read after the fact: the reserve that
|
|
1488
|
+
// should have kept them apart was measured from a smaller
|
|
1489
|
+
// previous reply. Say so once per send; the next request
|
|
1490
|
+
// folds anyway because the baseline now exceeds the window.
|
|
1491
|
+
if (!contextWindowOverrunNoteWritten &&
|
|
1492
|
+
midTurnState?.capacity !== undefined &&
|
|
1493
|
+
stepUsage !== undefined &&
|
|
1494
|
+
Number.isFinite(stepUsage.inputTokens) &&
|
|
1495
|
+
stepUsage.inputTokens > 0 &&
|
|
1496
|
+
Number.isFinite(stepUsage.outputTokens) &&
|
|
1497
|
+
stepUsage.outputTokens > 0 &&
|
|
1498
|
+
stepUsage.inputTokens + stepUsage.outputTokens > midTurnState.capacity) {
|
|
1499
|
+
contextWindowOverrunNoteWritten = true;
|
|
1500
|
+
await this.recordSystemNote('context_window_overrun', turnId, {
|
|
1501
|
+
usedTokens: stepUsage.inputTokens + stepUsage.outputTokens,
|
|
1502
|
+
declaredContextWindow: midTurnState.capacity,
|
|
1503
|
+
});
|
|
1504
|
+
}
|
|
1505
|
+
// Nothing declared, and the provider accepted a request past
|
|
1506
|
+
// the window this model reports. Every other signal in this
|
|
1507
|
+
// design stays dark there: no rejection to recover from, no
|
|
1508
|
+
// plateau to read, and no declaration to arm the proactive
|
|
1509
|
+
// threshold, so the session degrades quietly and
|
|
1510
|
+
// indefinitely (#4634). Report the two real numbers and
|
|
1511
|
+
// leave the decision with the user: a reported window is a
|
|
1512
|
+
// hint, and Maka still declares nothing on their behalf.
|
|
1513
|
+
//
|
|
1514
|
+
// Once per crossing, not once per send. On these providers
|
|
1515
|
+
// usage keeps growing past the line (305K → 322K observed),
|
|
1516
|
+
// so the note fires on the transition: the previous accepted
|
|
1517
|
+
// total was still inside the reported window and this one is
|
|
1518
|
+
// not. The baseline carries that previous total across
|
|
1519
|
+
// sessions through the persisted anchor, so a resumed
|
|
1520
|
+
// session does not repeat a crossing it already reported.
|
|
1521
|
+
if (!contextReportedWindowNoteWritten &&
|
|
1522
|
+
midTurnState !== undefined &&
|
|
1523
|
+
midTurnState.capacity === undefined &&
|
|
1524
|
+
stepUsage !== undefined &&
|
|
1525
|
+
Number.isFinite(stepUsage.inputTokens) &&
|
|
1526
|
+
stepUsage.inputTokens > 0 &&
|
|
1527
|
+
Number.isFinite(stepUsage.outputTokens)) {
|
|
1528
|
+
const reported = resolveSelectedModelContextWindow(this.deps.backend.connection, this.deps.backend.modelId);
|
|
1529
|
+
const used = stepUsage.inputTokens + Math.max(0, stepUsage.outputTokens);
|
|
1530
|
+
// `baselineTokens` still describes the request before this
|
|
1531
|
+
// one: the capacity hook sets it from the previous step, or
|
|
1532
|
+
// from the persisted anchor on a send's first request.
|
|
1533
|
+
const previousTotal = midTurnState.baselineTokens;
|
|
1534
|
+
const crossedNow = reported !== undefined &&
|
|
1535
|
+
used > reported &&
|
|
1536
|
+
(previousTotal === undefined || previousTotal <= reported);
|
|
1537
|
+
if (reported !== undefined && crossedNow) {
|
|
1538
|
+
contextReportedWindowNoteWritten = true;
|
|
1539
|
+
await this.recordSystemNote('context_reported_window_exceeded', turnId, {
|
|
1540
|
+
usedTokens: used,
|
|
1541
|
+
reportedContextWindow: reported,
|
|
1542
|
+
});
|
|
1543
|
+
}
|
|
1544
|
+
}
|
|
1545
|
+
lastStepInputTokens = stepUsage?.inputTokens;
|
|
1546
|
+
lastStepOutputTokens = stepUsage?.outputTokens;
|
|
1547
|
+
lastStepActiveToolCount = activeToolsForRequest.length;
|
|
1548
|
+
// A `finishReason: length` is deliberately not a trigger. The
|
|
1549
|
+
// reply may have been cut because the provider ran out of
|
|
1550
|
+
// window room, or because the provider's own output cap is
|
|
1551
|
+
// lower than the one Maka sends. Those are indistinguishable
|
|
1552
|
+
// from outside, and an indistinguishable signal must not
|
|
1553
|
+
// drive an action; the cut reply is visible to the user
|
|
1554
|
+
// either way (#4559).
|
|
1555
|
+
if (stepUsage) {
|
|
1556
|
+
completedStepUsage = mergeNormalizedUsage(completedStepUsage, stepUsage);
|
|
1557
|
+
this.deps.session.cumulativeUsageCheckpoint = mergeNormalizedUsage(this.deps.session.cumulativeUsageCheckpoint, stepUsage);
|
|
1558
|
+
await this.deps.backend.recordUsageCheckpoint?.({
|
|
1559
|
+
...this.deps.session.cumulativeUsageCheckpoint,
|
|
1560
|
+
costUsd: this.deps.providerTelemetry.normalizedUsageCostUsd(this.deps.session.cumulativeUsageCheckpoint),
|
|
1561
|
+
});
|
|
1562
|
+
}
|
|
1563
|
+
await flushStep();
|
|
1564
|
+
if (midTurnState) {
|
|
1565
|
+
// Durability clock: step N's thinking/text completion events
|
|
1566
|
+
// are enqueued by flushStep just above, so only after this
|
|
1567
|
+
// boundary can a seq-ack wait for step N mean anything. Wake
|
|
1568
|
+
// waiters AFTER the increment or they would re-check a stale
|
|
1569
|
+
// count and sleep.
|
|
1570
|
+
midTurnState.flushedSteps += 1;
|
|
1571
|
+
queue.wake();
|
|
1572
|
+
}
|
|
1573
|
+
}
|
|
1574
|
+
if (providerOutcome.kind === 'failed' && !this.aborted) {
|
|
1575
|
+
const failure = providerOutcome.failure;
|
|
1576
|
+
// An output-free context rejection did not sample a response.
|
|
1577
|
+
// Preserve metering for completed steps across compaction recovery.
|
|
1578
|
+
if (!providerOutcome.usage &&
|
|
1579
|
+
!(failure.kind === 'context_overflow' &&
|
|
1580
|
+
attemptHasNoObservableOutput() &&
|
|
1581
|
+
!attemptSawToolInput &&
|
|
1582
|
+
returnedToolCalls.length === 0))
|
|
1583
|
+
sawUnusableStepUsage = true;
|
|
1620
1584
|
if (this.loopStopRequested) {
|
|
1621
|
-
terminalProviderError =
|
|
1585
|
+
terminalProviderError = failure;
|
|
1622
1586
|
terminalProviderErrorReason =
|
|
1623
1587
|
lastCompletedStepHadToolResult && failure.kind === 'timeout'
|
|
1624
1588
|
? 'model_after_tool_timeout'
|
|
@@ -1629,10 +1593,11 @@ export class AiSdkTurn {
|
|
|
1629
1593
|
// more step; with the send-level budget already spent there is
|
|
1630
1594
|
// nothing left to grant it, so the error is terminal.
|
|
1631
1595
|
const stepBudgetRemains = maxSteps === undefined || runtimeSteps < maxSteps;
|
|
1632
|
-
const recovered = stepBudgetRemains &&
|
|
1596
|
+
const recovered = stepBudgetRemains &&
|
|
1597
|
+
providerAttempt < MAX_PROVIDER_ATTEMPTS_PER_STEP &&
|
|
1598
|
+
attemptHasNoObservableOutput()
|
|
1633
1599
|
? await this.deps.compaction.recoverFromOverflowError({
|
|
1634
|
-
error:
|
|
1635
|
-
retryAlreadyUsed: overflowRetryUsed || (midTurnState?.compactionAttemptedThisSend ?? false),
|
|
1600
|
+
error: failure,
|
|
1636
1601
|
midTurnState,
|
|
1637
1602
|
turnId,
|
|
1638
1603
|
stepNumber: runtimeSteps,
|
|
@@ -1651,7 +1616,6 @@ export class AiSdkTurn {
|
|
|
1651
1616
|
})
|
|
1652
1617
|
: undefined;
|
|
1653
1618
|
if (recovered) {
|
|
1654
|
-
overflowRetryUsed = true;
|
|
1655
1619
|
attemptMessages = recovered.messages;
|
|
1656
1620
|
continue;
|
|
1657
1621
|
}
|
|
@@ -1690,30 +1654,10 @@ export class AiSdkTurn {
|
|
|
1690
1654
|
contextOverflowAfterCompactionNoteWritten = true;
|
|
1691
1655
|
await this.recordSystemNote('context_overflow_after_compaction', turnId);
|
|
1692
1656
|
}
|
|
1693
|
-
const idleWatchdogRecovery = settledWatchdogTimeout?.phase === 'idle' &&
|
|
1694
|
-
idleWatchdogRetryCount < MAX_IDLE_WATCHDOG_RETRIES_PER_STEP &&
|
|
1695
|
-
attemptCanRecoverWithSealedThinking();
|
|
1696
|
-
const incompleteStreamRecovery = incompleteStreamTerminal &&
|
|
1697
|
-
incompleteStreamRetryCount < MAX_INCOMPLETE_STREAM_RETRIES_PER_STEP &&
|
|
1698
|
-
incompleteStreamHasNoObservableOutput;
|
|
1699
|
-
// Same seal-and-retry contract as the watchdog path, entered when
|
|
1700
|
-
// the failure arrives as a retryable provider/network error
|
|
1701
|
-
// instead of a local idle timeout. `!idleWatchdogRecovery` keeps
|
|
1702
|
-
// every watchdog-shaped outcome on its existing path, and
|
|
1703
|
-
// `!attemptHasNoObservableOutput()` keeps no-output retries on
|
|
1704
|
-
// the plain budget so this one is spent only on sealed fragments.
|
|
1705
|
-
const sealedThinkingRecovery = !idleWatchdogRecovery &&
|
|
1706
|
-
failure.retryable &&
|
|
1707
|
-
sealedThinkingRetryCount < MAX_SEALED_THINKING_RETRIES_PER_STEP &&
|
|
1708
|
-
attemptCanRecoverWithSealedThinking() &&
|
|
1709
|
-
!attemptHasNoObservableOutput();
|
|
1710
1657
|
// The stopping gate also supplies the durable reason. An absent
|
|
1711
1658
|
// decision means this attempt is allowed to retry.
|
|
1712
1659
|
let retry;
|
|
1713
|
-
if (!(
|
|
1714
|
-
idleWatchdogRecovery ||
|
|
1715
|
-
incompleteStreamRecovery ||
|
|
1716
|
-
sealedThinkingRecovery)) {
|
|
1660
|
+
if (!attemptCanReplay()) {
|
|
1717
1661
|
retry = {
|
|
1718
1662
|
decision: 'declined',
|
|
1719
1663
|
because: attemptSawToolActivity ? 'side_effects' : 'observable_output',
|
|
@@ -1728,33 +1672,20 @@ export class AiSdkTurn {
|
|
|
1728
1672
|
else if (failure.kind === 'context_overflow') {
|
|
1729
1673
|
retry = { decision: 'declined', because: 'policy' };
|
|
1730
1674
|
}
|
|
1731
|
-
else if (!
|
|
1732
|
-
retry =
|
|
1733
|
-
incompleteStreamTerminal &&
|
|
1734
|
-
incompleteStreamRetryCount >= MAX_INCOMPLETE_STREAM_RETRIES_PER_STEP
|
|
1735
|
-
? { decision: 'exhausted', attempts: providerAttempt }
|
|
1736
|
-
: { decision: 'declined', because: 'policy' };
|
|
1675
|
+
else if (!failure.retryable) {
|
|
1676
|
+
retry = { decision: 'declined', because: 'policy' };
|
|
1737
1677
|
}
|
|
1738
1678
|
if (!retry) {
|
|
1739
|
-
|
|
1740
|
-
|
|
1741
|
-
|
|
1742
|
-
|
|
1743
|
-
if (incompleteStreamRecovery)
|
|
1744
|
-
incompleteStreamRetryCount += 1;
|
|
1745
|
-
if ((idleWatchdogRecovery || sealedThinkingRecovery) &&
|
|
1746
|
-
stepThinkingParts.length > 0) {
|
|
1747
|
-
await flushStep();
|
|
1748
|
-
currentStepMessageId = this.deps.newId();
|
|
1749
|
-
}
|
|
1679
|
+
// Preserve the failed fragment for the transcript, but exclude
|
|
1680
|
+
// it from model history when a later tool step reloads the ledger.
|
|
1681
|
+
await flushStep(true);
|
|
1682
|
+
currentStepMessageId = this.deps.newId();
|
|
1750
1683
|
// The failed request did not return authoritative usage. Keep
|
|
1751
1684
|
// effectiveness recoverable, but fail final metering closed.
|
|
1752
1685
|
sawUnusableStepUsage = true;
|
|
1753
1686
|
const delayMs = providerRetryDelayMs(providerAttempt, failure.retryAfterMs);
|
|
1754
1687
|
const nextAttempt = providerAttempt + 1;
|
|
1755
|
-
const maxAttempts =
|
|
1756
|
-
? nextAttempt
|
|
1757
|
-
: MAX_PROVIDER_ATTEMPTS_PER_STEP;
|
|
1688
|
+
const maxAttempts = MAX_PROVIDER_ATTEMPTS_PER_STEP;
|
|
1758
1689
|
const reason = providerRetryReason(failure.kind);
|
|
1759
1690
|
queue.push({
|
|
1760
1691
|
type: 'provider_retry',
|
|
@@ -1769,14 +1700,13 @@ export class AiSdkTurn {
|
|
|
1769
1700
|
reason,
|
|
1770
1701
|
});
|
|
1771
1702
|
await this.deps.providerRetrySleep(delayMs, turnAbortController.signal);
|
|
1772
|
-
providerAttempt = nextAttempt;
|
|
1773
1703
|
queue.push({
|
|
1774
1704
|
type: 'provider_retry',
|
|
1775
1705
|
id: this.deps.newId(),
|
|
1776
1706
|
turnId,
|
|
1777
1707
|
ts: this.deps.now(),
|
|
1778
1708
|
phase: 'started',
|
|
1779
|
-
attempt:
|
|
1709
|
+
attempt: nextAttempt,
|
|
1780
1710
|
maxAttempts,
|
|
1781
1711
|
reason,
|
|
1782
1712
|
});
|
|
@@ -1786,7 +1716,7 @@ export class AiSdkTurn {
|
|
|
1786
1716
|
// safe fold): surface the real provider error via the terminal
|
|
1787
1717
|
// handler after settling any authoritative usage — never a
|
|
1788
1718
|
// fabricated success.
|
|
1789
|
-
terminalProviderError =
|
|
1719
|
+
terminalProviderError = failure;
|
|
1790
1720
|
terminalRetry = { error: terminalProviderError, retry };
|
|
1791
1721
|
terminalProviderErrorReason =
|
|
1792
1722
|
lastCompletedStepHadToolResult && failure.kind === 'timeout'
|
|
@@ -1803,10 +1733,7 @@ export class AiSdkTurn {
|
|
|
1803
1733
|
if (this.aborted) {
|
|
1804
1734
|
throw Object.assign(new Error('aborted'), { name: 'AbortError' });
|
|
1805
1735
|
}
|
|
1806
|
-
// Catch-all: flush any residual step content if the provider closed the
|
|
1807
|
-
// stream without a trailing `finish-step` for the last step.
|
|
1808
1736
|
const providerStepId = currentStepMessageId;
|
|
1809
|
-
await flushStep();
|
|
1810
1737
|
if (providerOutcome.kind !== 'completed')
|
|
1811
1738
|
throw providerOutcome.failure;
|
|
1812
1739
|
finishReason = providerOutcome.finishReason;
|
|
@@ -2109,11 +2036,11 @@ export class AiSdkTurn {
|
|
|
2109
2036
|
streamStatus = this.aborted ? 'aborted' : 'error';
|
|
2110
2037
|
streamErrorClass = this.deps.modelAdapter.classifyError(currentWatchdogTimeout()?.error ?? err);
|
|
2111
2038
|
// Flush the in-flight step's partial text/thinking before the terminal
|
|
2112
|
-
// abort/error events. Earlier steps already flushed at
|
|
2113
|
-
//
|
|
2039
|
+
// abort/error events. Earlier steps already flushed at settlement;
|
|
2040
|
+
// this keeps their and this step's streamed-out output on
|
|
2114
2041
|
// BOTH exits — user stop and provider error / watchdog timeout — so the
|
|
2115
2042
|
// transcript keeps what the user actually saw.
|
|
2116
|
-
await flushStep().catch(() => { });
|
|
2043
|
+
await flushStep(!this.aborted).catch(() => { });
|
|
2117
2044
|
if (this.aborted) {
|
|
2118
2045
|
queue.push({
|
|
2119
2046
|
type: 'abort',
|