maka-agent 0.2.0-dev.31.20260913 → 0.2.0-dev.32.20260914

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/dist/pi-transcript.js +3 -4
  2. package/native/runtime-host-windows-task-launcher/prebuilds/win32-x64/maka-runtime-host-task-launcher.exe +0 -0
  3. package/node_modules/@maka/core/dist/agent-graph-schedule.js +15 -4
  4. package/node_modules/@maka/core/dist/executor-id.js +22 -0
  5. package/node_modules/@maka/core/dist/external-session.js +18 -0
  6. package/node_modules/@maka/core/dist/model-call-usage-projection.js +0 -13
  7. package/node_modules/@maka/core/dist/runtime-event.js +19 -3
  8. package/node_modules/@maka/core/dist/session-send-projection.js +6 -3
  9. package/node_modules/@maka/core/dist/session.js +8 -2
  10. package/node_modules/@maka/core/dist/shell-run-result.js +1 -0
  11. package/node_modules/@maka/core/dist/shell-run.js +4 -0
  12. package/node_modules/@maka/core/dist/work-board.js +94 -1
  13. package/node_modules/@maka/core/package.json +1 -0
  14. package/node_modules/@maka/eval/dist/fleet-simulation.js +270 -0
  15. package/node_modules/@maka/eval/dist/fleet-store.js +243 -0
  16. package/node_modules/@maka/eval/dist/fleet-worker.js +125 -0
  17. package/node_modules/@maka/eval/dist/fleet.js +329 -0
  18. package/node_modules/@maka/eval/dist/index.js +3 -0
  19. package/node_modules/@maka/runtime/dist/agent-run.js +2 -17
  20. package/node_modules/@maka/runtime/dist/ai-sdk-compaction.js +6 -2
  21. package/node_modules/@maka/runtime/dist/ai-sdk-message-projection.js +36 -24
  22. package/node_modules/@maka/runtime/dist/ai-sdk-turn.js +257 -330
  23. package/node_modules/@maka/runtime/dist/background-task-health-tool.js +104 -0
  24. package/node_modules/@maka/runtime/dist/history-compaction.js +3 -1
  25. package/node_modules/@maka/runtime/dist/local-web-fetch.js +1 -1
  26. package/node_modules/@maka/runtime/dist/model-adapter.js +62 -53
  27. package/node_modules/@maka/runtime/dist/model-history.js +1 -5
  28. package/node_modules/@maka/runtime/dist/plugin-executor-backend.js +311 -0
  29. package/node_modules/@maka/runtime/dist/plugin-executor-service.js +344 -0
  30. package/node_modules/@maka/runtime/dist/provider-error-classification.js +7 -1
  31. package/node_modules/@maka/runtime/dist/provider-request-telemetry.js +35 -8
  32. package/node_modules/@maka/runtime/dist/runtime-event-backfill.js +7 -1
  33. package/node_modules/@maka/runtime/dist/runtime-event-read-model.js +1 -0
  34. package/node_modules/@maka/runtime/dist/runtime-invocation-route.js +49 -0
  35. package/node_modules/@maka/runtime/dist/runtime-kernel.js +72 -150
  36. package/node_modules/@maka/runtime/dist/runtime-read-model.js +0 -2
  37. package/node_modules/@maka/runtime/dist/session-event-runtime-mapper.js +3 -0
  38. package/node_modules/@maka/runtime/dist/session-manager.js +100 -58
  39. package/node_modules/@maka/runtime/dist/shell-run-manager.js +9 -3
  40. package/node_modules/@maka/runtime/dist/shell-run-tool-result.js +1 -0
  41. package/node_modules/@maka/runtime/dist/stream-graph-schedule-reconcile.js +1 -0
  42. package/node_modules/@maka/runtime/dist/stream-graph-supervisor-tools.js +26 -4
  43. package/node_modules/@maka/runtime/dist/subagent-tools.js +7 -0
  44. package/node_modules/@maka/runtime/dist/tool-runtime.js +45 -466
  45. package/node_modules/@maka/runtime/package.json +3 -0
  46. package/node_modules/@maka/runtime-host/dist/adapter/session-projector.js +7 -2
  47. package/node_modules/@maka/runtime-host/dist/client/session-catalog-summary.js +1 -0
  48. package/node_modules/@maka/runtime-host/dist/protocol/external-session.js +24 -3
  49. package/node_modules/@maka/runtime-host/dist/protocol/index.js +4 -1
  50. package/node_modules/@maka/runtime-host/dist/protocol/plugin-platform.js +42 -5
  51. package/node_modules/@maka/runtime-host/dist/protocol/session-catalog.js +31 -3
  52. package/node_modules/@maka/runtime-host/dist/protocol/session-continuity.js +6 -0
  53. package/node_modules/@maka/runtime-host/dist/server/child-agent-composition.js +1 -0
  54. package/node_modules/@maka/runtime-host/dist/server/execution-artifacts.js +1 -22
  55. package/node_modules/@maka/runtime-host/dist/server/execution-composition.js +54 -8
  56. package/node_modules/@maka/runtime-host/dist/server/external-session-coordinator.js +14 -1
  57. package/node_modules/@maka/runtime-host/dist/server/host-session-availability.js +10 -2
  58. package/node_modules/@maka/runtime-host/dist/server/plugin-platform-coordinator.js +6 -0
  59. package/node_modules/@maka/runtime-host/dist/server/plugin-platform.js +6 -0
  60. package/node_modules/@maka/runtime-host/dist/server/session-catalog-coordinator.js +56 -16
  61. package/node_modules/@maka/runtime-host/dist/server/session-continuity-coordinator.js +4 -3
  62. package/node_modules/@maka/runtime-host/dist/server/session-revision-coordinator.js +4 -0
  63. package/node_modules/@maka/runtime-host/dist/server/web-fetch-tool.js +37 -1
  64. package/node_modules/@maka/storage/dist/claude-code-session-adapter.js +450 -234
  65. package/node_modules/@maka/storage/dist/claude-code-transcript-lineage.js +106 -77
  66. package/node_modules/@maka/storage/dist/legacy-run-header.js +2 -2
  67. package/node_modules/@maka/storage/dist/model-call-usage-sql.js +1 -1
  68. package/node_modules/@maka/storage/dist/session-store.js +12 -2
  69. package/node_modules/@maka/storage/dist/sqlite-long-term-memory-store.js +86 -12
  70. package/node_modules/@maka/storage/dist/work-board-store.js +40 -1
  71. package/package.json +1 -1
@@ -39,6 +39,7 @@ import { REQUEST_SANDBOX_BOUNDARY_TOOL_NAME, SANDBOX_BOUNDARY_DENIED_FOR_TURN, S
39
39
  import { buildRuntimeEventModelReplayPlan, buildSteeringEnvelope, collectToolActivityTurnIds, compatibleProviderReasoningReplayEventIds, formatTextWithInlineRefs, steeringMessagesMissingFromBase, steeringModelMessage, } from './model-history.js';
40
40
  import { toolSchemaCharsForDiagnostics, requestCompositionToolSchemas, stableHash, toolCatalogHash, } from './request-shape.js';
41
41
  import { toolAvailabilityHash } from './tool-availability.js';
42
+ import { hasFinalizedReasoning } from './ai-sdk-message-projection.js';
42
43
  import { renderSwarmModePrompt } from './swarm-mode.js';
43
44
  import { renderGraphModePrompt } from './graph-mode.js';
44
45
  import { modelUsesNativeOpenAiResponses } from './model-runtime.js';
@@ -345,13 +346,6 @@ function joinPromptFragments(fragments) {
345
346
  }
346
347
  const MAX_WAITING_CODE_MODE_CELLS = 1;
347
348
  const MAX_PROVIDER_ATTEMPTS_PER_STEP = 10;
348
- const MAX_IDLE_WATCHDOG_RETRIES_PER_STEP = 1;
349
- const MAX_INCOMPLETE_STREAM_RETRIES_PER_STEP = 1;
350
- // A mid-stream cut after partial thinking seals one transcript fragment per
351
- // retry. A gateway that systematically kills long thinking streams (the
352
- // 2026-08-28 incident shape) would otherwise spend the full attempt budget
353
- // accumulating fragments before failing anyway, so fail fast after one.
354
- const MAX_SEALED_THINKING_RETRIES_PER_STEP = 1;
355
349
  const PROVIDER_RETRY_BASE_DELAY_MS = 1_000;
356
350
  const PROVIDER_RETRY_MAX_DELAY_MS = 32_000;
357
351
  const PROVIDER_RETRY_JITTER_FACTOR = 0.25;
@@ -375,9 +369,6 @@ function providerRetryReason(kind) {
375
369
  return 'unknown';
376
370
  }
377
371
  }
378
- function isIncompleteProviderFinishReason(reason) {
379
- return reason === undefined || reason === 'other' || reason === 'unknown';
380
- }
381
372
  /**
382
373
  * The mutable state of ONE `send()`.
383
374
  *
@@ -611,11 +602,6 @@ export class AiSdkTurn {
611
602
  let stepTextPartStartOffset = 0;
612
603
  let stepThinkingParts = [];
613
604
  let stepThinkingPartsById = new Map();
614
- let stepContentOrder = [];
615
- const recordStepContent = (kind) => {
616
- if (!stepContentOrder.includes(kind))
617
- stepContentOrder.push(kind);
618
- };
619
605
  // Flush the current step's AssistantMessage (text + thinking) and the paired
620
606
  // terminal thinking/text events, then clear the per-step accumulators.
621
607
  // Persist when the step produced text OR reasoning — a thinking-only step
@@ -631,17 +617,37 @@ export class AiSdkTurn {
631
617
  stepTextPartStartOffset = 0;
632
618
  stepThinkingParts = [];
633
619
  stepThinkingPartsById = new Map();
634
- stepContentOrder = [];
635
620
  };
636
- const flushStep = async () => {
621
+ const flushStep = async (interrupted = false) => {
637
622
  const hasThinking = stepThinkingParts.length > 0;
638
623
  if (stepText.length === 0 && !hasThinking) {
639
624
  resetStep();
640
625
  return;
641
626
  }
642
- const stepId = currentStepMessageId;
643
- if (hasThinking) {
644
- for (const part of stepThinkingParts) {
627
+ // A provider may finalize one reasoning item before the next item fails.
628
+ // Keep completed and interrupted parts in separate transcript rows so
629
+ // missing-ledger recovery preserves the same model visibility.
630
+ const fragments = [];
631
+ for (const part of stepThinkingParts) {
632
+ const partInterrupted = interrupted && !hasFinalizedReasoning(part);
633
+ let fragment = fragments.at(-1);
634
+ if (!fragment || fragment.interrupted !== partInterrupted) {
635
+ fragment = { thinking: [], text: '', interrupted: partInterrupted };
636
+ fragments.push(fragment);
637
+ }
638
+ fragment.thinking.push(part);
639
+ }
640
+ if (stepText.length > 0) {
641
+ let fragment = fragments.at(-1);
642
+ if (!fragment || fragment.interrupted !== interrupted) {
643
+ fragment = { thinking: [], text: '', interrupted };
644
+ fragments.push(fragment);
645
+ }
646
+ fragment.text = stepText;
647
+ }
648
+ for (const [index, fragment] of fragments.entries()) {
649
+ const stepId = index === 0 ? currentStepMessageId : this.deps.newId();
650
+ for (const part of fragment.thinking) {
645
651
  queue.push({
646
652
  type: 'thinking_complete',
647
653
  id: this.deps.newId(),
@@ -649,6 +655,7 @@ export class AiSdkTurn {
649
655
  ts: this.deps.now(),
650
656
  messageId: stepId,
651
657
  text: part.text,
658
+ ...(fragment.interrupted ? { interrupted: true } : {}),
652
659
  ...(part.signature !== undefined ? { signature: part.signature } : {}),
653
660
  // No sanitiser here, unlike the tool call below: these options are
654
661
  // not the provider's object. `translateChunk` rebuilds reasoning
@@ -661,24 +668,25 @@ export class AiSdkTurn {
661
668
  : {}),
662
669
  });
663
670
  }
671
+ queue.push({
672
+ type: 'text_complete',
673
+ id: this.deps.newId(),
674
+ turnId,
675
+ ts: this.deps.now(),
676
+ messageId: stepId,
677
+ text: fragment.text,
678
+ ...(fragment.interrupted ? { interrupted: true } : {}),
679
+ ...(index === fragments.length - 1 && stepTextProviderOptions !== undefined
680
+ ? { providerOptions: stepTextProviderOptions }
681
+ : {}),
682
+ });
664
683
  }
665
- queue.push({
666
- type: 'text_complete',
667
- id: this.deps.newId(),
668
- turnId,
669
- ts: this.deps.now(),
670
- messageId: stepId,
671
- text: stepText,
672
- ...(stepTextProviderOptions !== undefined
673
- ? { providerOptions: stepTextProviderOptions }
674
- : {}),
675
- });
676
684
  this.finalAssistantText = stepText.length > 0 ? stepText : undefined;
677
685
  resetStep();
678
686
  };
679
687
  let tokenUsage;
680
688
  let tokenUsageCostUsd;
681
- // Per-send sum of every COMPLETED step's usage, merged at each finish-step
689
+ // Per-send sum of every completed step's usage, merged at settlement
682
690
  // boundary. When the send aborts (mid-turn exhaust, user stop, stream
683
691
  // error) the SDK's cumulative `usage` promise may not resolve, but this sum is
684
692
  // real provider-reported evidence for the steps that did finish — IF every
@@ -945,7 +953,9 @@ export class AiSdkTurn {
945
953
  idleTimeoutMs: this.deps.backend.streamIdleTimeoutMs,
946
954
  ...this.deps.backend.streamWatchdogTimer,
947
955
  onTimeout: (timeout) => {
948
- const error = new Error(formatStreamWatchdogError(timeout));
956
+ const error = Object.assign(new Error(formatStreamWatchdogError(timeout)), {
957
+ code: 'MODEL_STREAM_TIMEOUT',
958
+ });
949
959
  watchdogTimeoutState.current = { phase: timeout.phase, error };
950
960
  providerRequestAbortController.abort(error);
951
961
  },
@@ -1020,7 +1030,9 @@ export class AiSdkTurn {
1020
1030
  // folded through the same reducer before it becomes messages. Without
1021
1031
  // this, a result archived at step N is rebuilt in full at step N+1 and
1022
1032
  // the ledger's account of what the model sees stops being true.
1023
- const foldedReplayEvents = await this.deps.compaction.foldEffectiveModelHistory(replayEvents, pruned.projectionSnapshot);
1033
+ const foldedReplayEvents = projectionCheckpoint
1034
+ ? await this.deps.compaction.foldEffectiveModelHistory(replayEvents, pruned.projectionSnapshot)
1035
+ : pruned.events;
1024
1036
  const replayPlan = buildRuntimeEventModelReplayPlan(foldedReplayEvents, {
1025
1037
  toolActivityTurnIds: collectToolActivityTurnIds([
1026
1038
  ...(input.runtimeContext ?? []),
@@ -1068,11 +1080,6 @@ export class AiSdkTurn {
1068
1080
  const requestProjection = shapedProjection;
1069
1081
  const completedProviderSteps = [];
1070
1082
  let requestMessages = messages;
1071
- // The compaction module runs at most once per send. This tracks the
1072
- // reactive entry; the proactive one sets the same flag on the mid-turn
1073
- // state, and each consults the other, so a send that already folded
1074
- // reports the oversized message instead of folding again (#4559).
1075
- let overflowRetryUsed = false;
1076
1083
  let result;
1077
1084
  let providerOutcome;
1078
1085
  let finishReason = 'stop';
@@ -1153,41 +1160,27 @@ export class AiSdkTurn {
1153
1160
  : undefined;
1154
1161
  providerRequestTracker?.setStep(runtimeSteps, requestCompositionId);
1155
1162
  let attemptMessages = projectedMessages;
1156
- let providerAttempt = 1;
1157
- let idleWatchdogRetryCount = 0;
1158
- let incompleteStreamRetryCount = 0;
1159
- let sealedThinkingRetryCount = 0;
1163
+ let providerAttempt = 0;
1160
1164
  const returnedToolCalls = [];
1161
- let providerToolActivityCount = 0;
1162
1165
  const providerToolInputs = new Map();
1163
1166
  let providerStepUsage;
1164
1167
  for (;;) {
1168
+ providerAttempt += 1;
1169
+ // Local calls are only admitted after a successful provider outcome.
1170
+ // A new physical request must not inherit the failed one's intents.
1171
+ returnedToolCalls.length = 0;
1165
1172
  providerRequestAbortController = new AbortController();
1166
1173
  watchdogTimeoutState.current = null;
1167
1174
  startWatchdog();
1168
1175
  // Monotonic facts for this physical request. The step accumulators
1169
1176
  // are cleared after flushStep(), so they cannot decide whether a
1170
1177
  // later stream failure is safe to retry.
1171
- let attemptSawText = false;
1172
- let attemptSawThinking = false;
1178
+ let attemptSawVisibleContent = false;
1173
1179
  let attemptSawToolActivity = false;
1174
- let attemptSawContinuationMetadata = false;
1175
- let attemptReachedStepBoundary = false;
1176
- const attemptHasNoObservableOutput = () => !attemptSawText &&
1177
- !attemptSawThinking &&
1178
- !attemptSawToolActivity &&
1179
- !attemptSawContinuationMetadata &&
1180
- !attemptReachedStepBoundary;
1181
- // Thinking is the only output that can be sealed into its own
1182
- // message before a retry: flushStep() closes the fragment under
1183
- // the current message id and the retry streams into a fresh one,
1184
- // so the user never sees spliced or duplicated content. Text,
1185
- // tool activity, continuation metadata, and step boundaries stay
1186
- // non-recoverable for the reasons each of them is tracked.
1187
- const attemptCanRecoverWithSealedThinking = () => !attemptSawText &&
1188
- !attemptSawToolActivity &&
1189
- !attemptSawContinuationMetadata &&
1190
- !attemptReachedStepBoundary;
1180
+ let attemptSawToolInput = false;
1181
+ let attemptSawReplayBarrier = false;
1182
+ const attemptHasNoObservableOutput = () => !attemptSawVisibleContent && !attemptSawToolActivity && !attemptSawReplayBarrier;
1183
+ const attemptCanReplay = () => !attemptSawToolActivity && !attemptSawReplayBarrier;
1191
1184
  this.memorySourceMessages = [...attemptMessages];
1192
1185
  this.memorySourceEventMessagePositions =
1193
1186
  this.deps.messageProjection.memoryEventMessagePositions(attemptMessages);
@@ -1240,185 +1233,18 @@ export class AiSdkTurn {
1240
1233
  // trailer and consume the one authoritative outcome below.
1241
1234
  break;
1242
1235
  }
1243
- const incompleteFinish = (event.kind === 'finish' || event.kind === 'step-finish') &&
1244
- isIncompleteProviderFinishReason(event.finishReason);
1245
- if ((event.kind === 'finish' || event.kind === 'step-finish') && !incompleteFinish) {
1246
- attemptReachedStepBoundary = true;
1247
- }
1248
- if (event.kind === 'step-finish') {
1249
- // AI SDK can synthesize `finish-step(other)` when the provider
1250
- // stream reaches EOF without a terminal frame. That is not a
1251
- // completed model step and must not consume the step budget or
1252
- // checkpoint imaginary usage before the safe retry below.
1253
- if (!incompleteFinish) {
1254
- // Step boundary: AI SDK 7 delimits steps with `finish-step`
1255
- // (and `step-finish` for legacy replay fixtures); the adapter
1256
- // reduces both to this event. A duplicate boundary is harmless:
1257
- // the second flush no-ops (accumulators already cleared) and one
1258
- // extra id rotation just discards an unused id.
1259
- runtimeSteps += 1;
1260
- const stepUsage = event.usage;
1261
- providerStepUsage = stepUsage;
1262
- if (!stepUsage)
1263
- sawUnusableStepUsage = true;
1264
- // Silent eviction / rewrite check (#4559): this step only
1265
- // appended (no fold, no prune, no image omission) yet the
1266
- // provider counted no more input tokens than for the previous
1267
- // request. Not-greater, not strictly-fewer: a provider that
1268
- // truncates to a fixed window (Ollama's `num_ctx`) reports the
1269
- // same total on every later request while Maka keeps
1270
- // appending, so a plateau is the signal, and an equal count
1271
- // after an append is already impossible without provider-side
1272
- // eviction or rewriting. Input against input: the previous
1273
- // reply's reasoning may not be resent, so input + output is
1274
- // not the floor of the next input on every wire.
1275
- const completedRequestIndex = runtimeSteps - 1;
1276
- // A finalization step resolves an empty tool set, so its
1277
- // request legitimately drops several thousand schema tokens
1278
- // with no fold, prune or image omission. Maka shaped that
1279
- // request; the provider did not drop anything.
1280
- const toolSchemaShrank = lastStepActiveToolCount !== undefined &&
1281
- activeToolsForRequest.length < lastStepActiveToolCount;
1282
- // Across the send boundary the comparison is the same one,
1283
- // against the last request a provider accepted before this
1284
- // send. A provider that truncates to a fixed window reports
1285
- // the same input on every later request while the user keeps
1286
- // adding turns, and a send of one or two steps never sees
1287
- // that from the inside: the live evidence plateaus at 3,716
1288
- // input tokens across eight turns with nothing reported
1289
- // (#4623). The first request of a send therefore compares
1290
- // against the persisted anchor, which is route-validated
1291
- // where it is read; a fold before that request would explain
1292
- // a smaller input by itself, so it disables the comparison.
1293
- const acrossSends = completedRequestIndex === 0;
1294
- const priorInput = acrossSends
1295
- ? midTurnState?.compactionAppliedThisSend === true
1296
- ? undefined
1297
- : midTurnState?.priorAcceptedInputTokens
1298
- : lastStepInputTokens;
1299
- if (!this.deps.session.contextProviderDroppingReported &&
1300
- !toolSchemaShrank &&
1301
- midTurnState &&
1302
- priorInput !== undefined &&
1303
- midTurnState.replacedStepNumber !== completedRequestIndex &&
1304
- pruneAppliedAtStep !== completedRequestIndex &&
1305
- midTurnState.omittedImageToolResults.size === 0 &&
1306
- stepUsage !== undefined &&
1307
- Number.isFinite(stepUsage.inputTokens) &&
1308
- stepUsage.inputTokens > 0 &&
1309
- // Across sends the test is equality, not "did not grow".
1310
- // Inside a send Maka knows it only appended, so any
1311
- // shortfall is the provider's. Across the boundary it does
1312
- // not: a manual compaction leaves the pre-compaction anchor
1313
- // behind, a turn can carry a smaller tool set, and a user
1314
- // can edit or branch history. All three shrink the input
1315
- // legitimately, and none of them lands on exactly the same
1316
- // count. A provider truncating to a fixed window does, on
1317
- // every later request.
1318
- (acrossSends
1319
- ? stepUsage.inputTokens === priorInput
1320
- : stepUsage.inputTokens <= priorInput)) {
1321
- this.deps.session.contextProviderDroppingReported = true;
1322
- await this.recordSystemNote('context_provider_dropping', turnId, {
1323
- inputTokens: stepUsage.inputTokens,
1324
- priorInputTokens: priorInput,
1325
- });
1326
- }
1327
- // Fail closed: reset on every step boundary so a missing final
1328
- // step's usage does not leave a stale value from an earlier step.
1329
- // The reply needed more room than the declared window had
1330
- // left after this request's own input. Both halves are the
1331
- // provider's numbers, read after the fact: the reserve that
1332
- // should have kept them apart was measured from a smaller
1333
- // previous reply. Say so once per send; the next request
1334
- // folds anyway because the baseline now exceeds the window.
1335
- if (!contextWindowOverrunNoteWritten &&
1336
- midTurnState?.capacity !== undefined &&
1337
- stepUsage !== undefined &&
1338
- Number.isFinite(stepUsage.inputTokens) &&
1339
- stepUsage.inputTokens > 0 &&
1340
- Number.isFinite(stepUsage.outputTokens) &&
1341
- stepUsage.outputTokens > 0 &&
1342
- stepUsage.inputTokens + stepUsage.outputTokens > midTurnState.capacity) {
1343
- contextWindowOverrunNoteWritten = true;
1344
- await this.recordSystemNote('context_window_overrun', turnId, {
1345
- usedTokens: stepUsage.inputTokens + stepUsage.outputTokens,
1346
- declaredContextWindow: midTurnState.capacity,
1347
- });
1348
- }
1349
- // Nothing declared, and the provider accepted a request past
1350
- // the window this model reports. Every other signal in this
1351
- // design stays dark there: no rejection to recover from, no
1352
- // plateau to read, and no declaration to arm the proactive
1353
- // threshold, so the session degrades quietly and
1354
- // indefinitely (#4634). Report the two real numbers and
1355
- // leave the decision with the user: a reported window is a
1356
- // hint, and Maka still declares nothing on their behalf.
1357
- //
1358
- // Once per crossing, not once per send. On these providers
1359
- // usage keeps growing past the line (305K → 322K observed),
1360
- // so the note fires on the transition: the previous accepted
1361
- // total was still inside the reported window and this one is
1362
- // not. The baseline carries that previous total across
1363
- // sessions through the persisted anchor, so a resumed
1364
- // session does not repeat a crossing it already reported.
1365
- if (!contextReportedWindowNoteWritten &&
1366
- midTurnState !== undefined &&
1367
- midTurnState.capacity === undefined &&
1368
- stepUsage !== undefined &&
1369
- Number.isFinite(stepUsage.inputTokens) &&
1370
- stepUsage.inputTokens > 0 &&
1371
- Number.isFinite(stepUsage.outputTokens)) {
1372
- const reported = resolveSelectedModelContextWindow(this.deps.backend.connection, this.deps.backend.modelId);
1373
- const used = stepUsage.inputTokens + Math.max(0, stepUsage.outputTokens);
1374
- // `baselineTokens` still describes the request before this
1375
- // one: the capacity hook sets it from the previous step, or
1376
- // from the persisted anchor on a send's first request.
1377
- const previousTotal = midTurnState.baselineTokens;
1378
- const crossedNow = reported !== undefined &&
1379
- used > reported &&
1380
- (previousTotal === undefined || previousTotal <= reported);
1381
- if (reported !== undefined && crossedNow) {
1382
- contextReportedWindowNoteWritten = true;
1383
- await this.recordSystemNote('context_reported_window_exceeded', turnId, {
1384
- usedTokens: used,
1385
- reportedContextWindow: reported,
1386
- });
1387
- }
1388
- }
1389
- lastStepInputTokens = stepUsage?.inputTokens;
1390
- lastStepOutputTokens = stepUsage?.outputTokens;
1391
- lastStepActiveToolCount = activeToolsForRequest.length;
1392
- // A `finishReason: length` is deliberately not a trigger. The
1393
- // reply may have been cut because the provider ran out of
1394
- // window room, or because the provider's own output cap is
1395
- // lower than the one Maka sends. Those are indistinguishable
1396
- // from outside, and an indistinguishable signal must not
1397
- // drive an action; the cut reply is visible to the user
1398
- // either way (#4559).
1399
- if (stepUsage) {
1400
- completedStepUsage = mergeNormalizedUsage(completedStepUsage, stepUsage);
1401
- this.deps.session.cumulativeUsageCheckpoint = mergeNormalizedUsage(this.deps.session.cumulativeUsageCheckpoint, stepUsage);
1402
- await this.deps.backend.recordUsageCheckpoint?.({
1403
- ...this.deps.session.cumulativeUsageCheckpoint,
1404
- costUsd: this.deps.providerTelemetry.normalizedUsageCostUsd(this.deps.session.cumulativeUsageCheckpoint),
1405
- });
1406
- }
1407
- }
1408
- }
1409
1236
  if (event.kind === 'text-start') {
1410
1237
  if (stepText.length > 0 && event.providerItemBoundary === true) {
1238
+ attemptSawReplayBarrier = true;
1411
1239
  await flushStep();
1412
1240
  currentStepMessageId = this.deps.newId();
1413
1241
  }
1414
1242
  stepTextPartStartOffset = stepText.length;
1415
1243
  }
1416
1244
  else if (event.kind === 'text') {
1417
- if (event.text.length > 0)
1418
- recordStepContent('text');
1419
1245
  stepText += event.text;
1420
1246
  if (event.text.length > 0)
1421
- attemptSawText = true;
1247
+ attemptSawVisibleContent = true;
1422
1248
  queue.push({
1423
1249
  type: 'text_delta',
1424
1250
  id: this.deps.newId(),
@@ -1430,7 +1256,7 @@ export class AiSdkTurn {
1430
1256
  }
1431
1257
  else if (event.kind === 'text-end') {
1432
1258
  if (event.providerOptions !== undefined) {
1433
- attemptSawContinuationMetadata = true;
1259
+ attemptSawReplayBarrier = true;
1434
1260
  stepTextProviderOptions = mergeTextProviderOptions(stepTextProviderOptions, stripUndefinedDeep(event.providerOptions), stepTextPartStartOffset);
1435
1261
  }
1436
1262
  if (event.providerItemBoundary === true) {
@@ -1440,7 +1266,7 @@ export class AiSdkTurn {
1440
1266
  }
1441
1267
  else if (event.kind === 'thinking-start') {
1442
1268
  if (event.providerOptions !== undefined) {
1443
- attemptSawContinuationMetadata = true;
1269
+ attemptSawReplayBarrier = true;
1444
1270
  }
1445
1271
  const part = {
1446
1272
  text: '',
@@ -1455,12 +1281,10 @@ export class AiSdkTurn {
1455
1281
  }
1456
1282
  else if (event.kind === 'thinking') {
1457
1283
  if (event.text.length > 0)
1458
- recordStepContent('thinking');
1459
- if (event.text.length > 0)
1460
- attemptSawThinking = true;
1284
+ attemptSawVisibleContent = true;
1461
1285
  if (event.providerOptions !== undefined) {
1462
1286
  if (event.providerOptionsOrigin !== 'maka_transport') {
1463
- attemptSawContinuationMetadata = true;
1287
+ attemptSawReplayBarrier = true;
1464
1288
  }
1465
1289
  }
1466
1290
  const partId = event.reasoningPartId ?? responsesReasoningItemId(event.providerOptions);
@@ -1516,7 +1340,7 @@ export class AiSdkTurn {
1516
1340
  });
1517
1341
  }
1518
1342
  else if (event.kind === 'thinking-signature') {
1519
- attemptSawContinuationMetadata = true;
1343
+ attemptSawReplayBarrier = true;
1520
1344
  let part = event.reasoningPartId
1521
1345
  ? stepThinkingPartsById.get(event.reasoningPartId)
1522
1346
  : stepThinkingParts.at(-1);
@@ -1529,17 +1353,17 @@ export class AiSdkTurn {
1529
1353
  }
1530
1354
  part.signature = event.signature;
1531
1355
  }
1532
- else if (event.kind === 'provider-tool-input') {
1356
+ else if (event.kind === 'tool-input') {
1357
+ attemptSawToolInput = true;
1533
1358
  // The provider has started its own tool. Even without a
1534
1359
  // final tool-call/result event, retrying can repeat external
1535
1360
  // work that the Runtime cannot observe or reconcile.
1536
- attemptSawToolActivity = true;
1361
+ if (event.providerExecuted)
1362
+ attemptSawToolActivity = true;
1537
1363
  }
1538
1364
  else if (event.kind === 'tool-call') {
1539
- attemptSawToolActivity = true;
1540
- recordStepContent('tools');
1541
1365
  if (event.toolCall.providerExecuted) {
1542
- providerToolActivityCount += 1;
1366
+ attemptSawToolActivity = true;
1543
1367
  providerToolInputs.set(event.toolCall.toolCallId, event.toolCall.input);
1544
1368
  queue.push({
1545
1369
  type: 'tool_start',
@@ -1566,7 +1390,6 @@ export class AiSdkTurn {
1566
1390
  }
1567
1391
  else if (event.kind === 'provider-tool-result') {
1568
1392
  attemptSawToolActivity = true;
1569
- providerToolActivityCount += 1;
1570
1393
  const providerOutput = stripUndefinedDeep(event.output);
1571
1394
  queue.push({
1572
1395
  type: 'tool_result',
@@ -1581,44 +1404,185 @@ export class AiSdkTurn {
1581
1404
  });
1582
1405
  providerToolInputs.delete(event.toolCallId);
1583
1406
  }
1584
- else if (event.kind === 'step-finish' && !incompleteFinish) {
1585
- // The step's text/thinking deltas are all in (the stream is
1586
- // drained in order), so flush this step's AssistantMessage and
1587
- // rotate to a fresh id for the next step. Tool settlement
1588
- // below receives this step's pre-rotation id, so durable replay
1589
- // can regroup calls with this reasoning/text.
1590
- await flushStep();
1591
- if (midTurnState) {
1592
- // Durability clock: step N's thinking/text completion events
1593
- // are enqueued by flushStep just above, so only after this
1594
- // boundary can a seq-ack wait for step N mean anything. Wake
1595
- // waiters AFTER the increment or they would re-check a stale
1596
- // count and sleep.
1597
- midTurnState.flushedSteps += 1;
1598
- queue.wake();
1599
- }
1600
- }
1601
1407
  }
1602
1408
  watchdogState.current?.stop();
1603
1409
  // This timeout belongs to the physical request that just settled.
1604
1410
  // Consume it before recovery/flush work: a later persistence error
1605
1411
  // must not be reported as the already-handled watchdog timeout.
1606
- const settledWatchdogTimeout = consumeWatchdogTimeout();
1412
+ consumeWatchdogTimeout();
1607
1413
  providerOutcome = await result.outcome;
1608
- const incompleteStreamTerminal = providerOutcome.kind === 'truncated';
1609
- const incompleteStreamHasNoObservableOutput = incompleteStreamTerminal &&
1610
- !attemptSawText &&
1611
- !attemptSawThinking &&
1612
- !attemptSawToolActivity &&
1613
- !attemptSawContinuationMetadata;
1614
- const attemptFailure = settledWatchdogTimeout?.error ??
1615
- (providerOutcome.kind === 'completed' ? undefined : providerOutcome.failure);
1616
- if (attemptFailure && !this.aborted) {
1617
- const failure = settledWatchdogTimeout || providerOutcome.kind === 'completed'
1618
- ? this.deps.modelAdapter.normalizeFailure(attemptFailure)
1619
- : providerOutcome.failure;
1414
+ if (providerOutcome.kind === 'completed') {
1415
+ runtimeSteps += 1;
1416
+ const stepUsage = providerOutcome.usage;
1417
+ providerStepUsage = stepUsage;
1418
+ if (!stepUsage)
1419
+ sawUnusableStepUsage = true;
1420
+ // Silent eviction / rewrite check (#4559): this step only
1421
+ // appended (no fold, no prune, no image omission) yet the
1422
+ // provider counted no more input tokens than for the previous
1423
+ // request. Not-greater, not strictly-fewer: a provider that
1424
+ // truncates to a fixed window (Ollama's `num_ctx`) reports the
1425
+ // same total on every later request while Maka keeps
1426
+ // appending, so a plateau is the signal, and an equal count
1427
+ // after an append is already impossible without provider-side
1428
+ // eviction or rewriting. Input against input: the previous
1429
+ // reply's reasoning may not be resent, so input + output is
1430
+ // not the floor of the next input on every wire.
1431
+ const completedRequestIndex = runtimeSteps - 1;
1432
+ // A finalization step resolves an empty tool set, so its
1433
+ // request legitimately drops several thousand schema tokens
1434
+ // with no fold, prune or image omission. Maka shaped that
1435
+ // request; the provider did not drop anything.
1436
+ const toolSchemaShrank = lastStepActiveToolCount !== undefined &&
1437
+ activeToolsForRequest.length < lastStepActiveToolCount;
1438
+ // Across the send boundary the comparison is the same one,
1439
+ // against the last request a provider accepted before this
1440
+ // send. A provider that truncates to a fixed window reports
1441
+ // the same input on every later request while the user keeps
1442
+ // adding turns, and a send of one or two steps never sees
1443
+ // that from the inside: the live evidence plateaus at 3,716
1444
+ // input tokens across eight turns with nothing reported
1445
+ // (#4623). The first request of a send therefore compares
1446
+ // against the persisted anchor, which is route-validated
1447
+ // where it is read; a fold before that request would explain
1448
+ // a smaller input by itself, so it disables the comparison.
1449
+ const acrossSends = completedRequestIndex === 0;
1450
+ const priorInput = acrossSends
1451
+ ? midTurnState?.compactionAppliedThisSend === true
1452
+ ? undefined
1453
+ : midTurnState?.priorAcceptedInputTokens
1454
+ : lastStepInputTokens;
1455
+ if (!this.deps.session.contextProviderDroppingReported &&
1456
+ !toolSchemaShrank &&
1457
+ midTurnState &&
1458
+ priorInput !== undefined &&
1459
+ midTurnState.replacedStepNumber !== completedRequestIndex &&
1460
+ pruneAppliedAtStep !== completedRequestIndex &&
1461
+ midTurnState.omittedImageToolResults.size === 0 &&
1462
+ stepUsage !== undefined &&
1463
+ Number.isFinite(stepUsage.inputTokens) &&
1464
+ stepUsage.inputTokens > 0 &&
1465
+ // Across sends the test is equality, not "did not grow".
1466
+ // Inside a send Maka knows it only appended, so any
1467
+ // shortfall is the provider's. Across the boundary it does
1468
+ // not: a manual compaction leaves the pre-compaction anchor
1469
+ // behind, a turn can carry a smaller tool set, and a user
1470
+ // can edit or branch history. All three shrink the input
1471
+ // legitimately, and none of them lands on exactly the same
1472
+ // count. A provider truncating to a fixed window does, on
1473
+ // every later request.
1474
+ (acrossSends
1475
+ ? stepUsage.inputTokens === priorInput
1476
+ : stepUsage.inputTokens <= priorInput)) {
1477
+ this.deps.session.contextProviderDroppingReported = true;
1478
+ await this.recordSystemNote('context_provider_dropping', turnId, {
1479
+ inputTokens: stepUsage.inputTokens,
1480
+ priorInputTokens: priorInput,
1481
+ });
1482
+ }
1483
+ // Fail closed: reset on every step boundary so a missing final
1484
+ // step's usage does not leave a stale value from an earlier step.
1485
+ // The reply needed more room than the declared window had
1486
+ // left after this request's own input. Both halves are the
1487
+ // provider's numbers, read after the fact: the reserve that
1488
+ // should have kept them apart was measured from a smaller
1489
+ // previous reply. Say so once per send; the next request
1490
+ // folds anyway because the baseline now exceeds the window.
1491
+ if (!contextWindowOverrunNoteWritten &&
1492
+ midTurnState?.capacity !== undefined &&
1493
+ stepUsage !== undefined &&
1494
+ Number.isFinite(stepUsage.inputTokens) &&
1495
+ stepUsage.inputTokens > 0 &&
1496
+ Number.isFinite(stepUsage.outputTokens) &&
1497
+ stepUsage.outputTokens > 0 &&
1498
+ stepUsage.inputTokens + stepUsage.outputTokens > midTurnState.capacity) {
1499
+ contextWindowOverrunNoteWritten = true;
1500
+ await this.recordSystemNote('context_window_overrun', turnId, {
1501
+ usedTokens: stepUsage.inputTokens + stepUsage.outputTokens,
1502
+ declaredContextWindow: midTurnState.capacity,
1503
+ });
1504
+ }
1505
+ // Nothing declared, and the provider accepted a request past
1506
+ // the window this model reports. Every other signal in this
1507
+ // design stays dark there: no rejection to recover from, no
1508
+ // plateau to read, and no declaration to arm the proactive
1509
+ // threshold, so the session degrades quietly and
1510
+ // indefinitely (#4634). Report the two real numbers and
1511
+ // leave the decision with the user: a reported window is a
1512
+ // hint, and Maka still declares nothing on their behalf.
1513
+ //
1514
+ // Once per crossing, not once per send. On these providers
1515
+ // usage keeps growing past the line (305K → 322K observed),
1516
+ // so the note fires on the transition: the previous accepted
1517
+ // total was still inside the reported window and this one is
1518
+ // not. The baseline carries that previous total across
1519
+ // sessions through the persisted anchor, so a resumed
1520
+ // session does not repeat a crossing it already reported.
1521
+ if (!contextReportedWindowNoteWritten &&
1522
+ midTurnState !== undefined &&
1523
+ midTurnState.capacity === undefined &&
1524
+ stepUsage !== undefined &&
1525
+ Number.isFinite(stepUsage.inputTokens) &&
1526
+ stepUsage.inputTokens > 0 &&
1527
+ Number.isFinite(stepUsage.outputTokens)) {
1528
+ const reported = resolveSelectedModelContextWindow(this.deps.backend.connection, this.deps.backend.modelId);
1529
+ const used = stepUsage.inputTokens + Math.max(0, stepUsage.outputTokens);
1530
+ // `baselineTokens` still describes the request before this
1531
+ // one: the capacity hook sets it from the previous step, or
1532
+ // from the persisted anchor on a send's first request.
1533
+ const previousTotal = midTurnState.baselineTokens;
1534
+ const crossedNow = reported !== undefined &&
1535
+ used > reported &&
1536
+ (previousTotal === undefined || previousTotal <= reported);
1537
+ if (reported !== undefined && crossedNow) {
1538
+ contextReportedWindowNoteWritten = true;
1539
+ await this.recordSystemNote('context_reported_window_exceeded', turnId, {
1540
+ usedTokens: used,
1541
+ reportedContextWindow: reported,
1542
+ });
1543
+ }
1544
+ }
1545
+ lastStepInputTokens = stepUsage?.inputTokens;
1546
+ lastStepOutputTokens = stepUsage?.outputTokens;
1547
+ lastStepActiveToolCount = activeToolsForRequest.length;
1548
+ // A `finishReason: length` is deliberately not a trigger. The
1549
+ // reply may have been cut because the provider ran out of
1550
+ // window room, or because the provider's own output cap is
1551
+ // lower than the one Maka sends. Those are indistinguishable
1552
+ // from outside, and an indistinguishable signal must not
1553
+ // drive an action; the cut reply is visible to the user
1554
+ // either way (#4559).
1555
+ if (stepUsage) {
1556
+ completedStepUsage = mergeNormalizedUsage(completedStepUsage, stepUsage);
1557
+ this.deps.session.cumulativeUsageCheckpoint = mergeNormalizedUsage(this.deps.session.cumulativeUsageCheckpoint, stepUsage);
1558
+ await this.deps.backend.recordUsageCheckpoint?.({
1559
+ ...this.deps.session.cumulativeUsageCheckpoint,
1560
+ costUsd: this.deps.providerTelemetry.normalizedUsageCostUsd(this.deps.session.cumulativeUsageCheckpoint),
1561
+ });
1562
+ }
1563
+ await flushStep();
1564
+ if (midTurnState) {
1565
+ // Durability clock: step N's thinking/text completion events
1566
+ // are enqueued by flushStep just above, so only after this
1567
+ // boundary can a seq-ack wait for step N mean anything. Wake
1568
+ // waiters AFTER the increment or they would re-check a stale
1569
+ // count and sleep.
1570
+ midTurnState.flushedSteps += 1;
1571
+ queue.wake();
1572
+ }
1573
+ }
1574
+ if (providerOutcome.kind === 'failed' && !this.aborted) {
1575
+ const failure = providerOutcome.failure;
1576
+ // An output-free context rejection did not sample a response.
1577
+ // Preserve metering for completed steps across compaction recovery.
1578
+ if (!providerOutcome.usage &&
1579
+ !(failure.kind === 'context_overflow' &&
1580
+ attemptHasNoObservableOutput() &&
1581
+ !attemptSawToolInput &&
1582
+ returnedToolCalls.length === 0))
1583
+ sawUnusableStepUsage = true;
1620
1584
  if (this.loopStopRequested) {
1621
- terminalProviderError = settledWatchdogTimeout?.error ?? failure;
1585
+ terminalProviderError = failure;
1622
1586
  terminalProviderErrorReason =
1623
1587
  lastCompletedStepHadToolResult && failure.kind === 'timeout'
1624
1588
  ? 'model_after_tool_timeout'
@@ -1629,10 +1593,11 @@ export class AiSdkTurn {
1629
1593
  // more step; with the send-level budget already spent there is
1630
1594
  // nothing left to grant it, so the error is terminal.
1631
1595
  const stepBudgetRemains = maxSteps === undefined || runtimeSteps < maxSteps;
1632
- const recovered = stepBudgetRemains && attemptHasNoObservableOutput()
1596
+ const recovered = stepBudgetRemains &&
1597
+ providerAttempt < MAX_PROVIDER_ATTEMPTS_PER_STEP &&
1598
+ attemptHasNoObservableOutput()
1633
1599
  ? await this.deps.compaction.recoverFromOverflowError({
1634
- error: attemptFailure,
1635
- retryAlreadyUsed: overflowRetryUsed || (midTurnState?.compactionAttemptedThisSend ?? false),
1600
+ error: failure,
1636
1601
  midTurnState,
1637
1602
  turnId,
1638
1603
  stepNumber: runtimeSteps,
@@ -1651,7 +1616,6 @@ export class AiSdkTurn {
1651
1616
  })
1652
1617
  : undefined;
1653
1618
  if (recovered) {
1654
- overflowRetryUsed = true;
1655
1619
  attemptMessages = recovered.messages;
1656
1620
  continue;
1657
1621
  }
@@ -1690,30 +1654,10 @@ export class AiSdkTurn {
1690
1654
  contextOverflowAfterCompactionNoteWritten = true;
1691
1655
  await this.recordSystemNote('context_overflow_after_compaction', turnId);
1692
1656
  }
1693
- const idleWatchdogRecovery = settledWatchdogTimeout?.phase === 'idle' &&
1694
- idleWatchdogRetryCount < MAX_IDLE_WATCHDOG_RETRIES_PER_STEP &&
1695
- attemptCanRecoverWithSealedThinking();
1696
- const incompleteStreamRecovery = incompleteStreamTerminal &&
1697
- incompleteStreamRetryCount < MAX_INCOMPLETE_STREAM_RETRIES_PER_STEP &&
1698
- incompleteStreamHasNoObservableOutput;
1699
- // Same seal-and-retry contract as the watchdog path, entered when
1700
- // the failure arrives as a retryable provider/network error
1701
- // instead of a local idle timeout. `!idleWatchdogRecovery` keeps
1702
- // every watchdog-shaped outcome on its existing path, and
1703
- // `!attemptHasNoObservableOutput()` keeps no-output retries on
1704
- // the plain budget so this one is spent only on sealed fragments.
1705
- const sealedThinkingRecovery = !idleWatchdogRecovery &&
1706
- failure.retryable &&
1707
- sealedThinkingRetryCount < MAX_SEALED_THINKING_RETRIES_PER_STEP &&
1708
- attemptCanRecoverWithSealedThinking() &&
1709
- !attemptHasNoObservableOutput();
1710
1657
  // The stopping gate also supplies the durable reason. An absent
1711
1658
  // decision means this attempt is allowed to retry.
1712
1659
  let retry;
1713
- if (!(attemptHasNoObservableOutput() ||
1714
- idleWatchdogRecovery ||
1715
- incompleteStreamRecovery ||
1716
- sealedThinkingRecovery)) {
1660
+ if (!attemptCanReplay()) {
1717
1661
  retry = {
1718
1662
  decision: 'declined',
1719
1663
  because: attemptSawToolActivity ? 'side_effects' : 'observable_output',
@@ -1728,33 +1672,20 @@ export class AiSdkTurn {
1728
1672
  else if (failure.kind === 'context_overflow') {
1729
1673
  retry = { decision: 'declined', because: 'policy' };
1730
1674
  }
1731
- else if (!(failure.retryable || idleWatchdogRecovery || incompleteStreamRecovery)) {
1732
- retry =
1733
- incompleteStreamTerminal &&
1734
- incompleteStreamRetryCount >= MAX_INCOMPLETE_STREAM_RETRIES_PER_STEP
1735
- ? { decision: 'exhausted', attempts: providerAttempt }
1736
- : { decision: 'declined', because: 'policy' };
1675
+ else if (!failure.retryable) {
1676
+ retry = { decision: 'declined', because: 'policy' };
1737
1677
  }
1738
1678
  if (!retry) {
1739
- if (idleWatchdogRecovery)
1740
- idleWatchdogRetryCount += 1;
1741
- if (sealedThinkingRecovery)
1742
- sealedThinkingRetryCount += 1;
1743
- if (incompleteStreamRecovery)
1744
- incompleteStreamRetryCount += 1;
1745
- if ((idleWatchdogRecovery || sealedThinkingRecovery) &&
1746
- stepThinkingParts.length > 0) {
1747
- await flushStep();
1748
- currentStepMessageId = this.deps.newId();
1749
- }
1679
+ // Preserve the failed fragment for the transcript, but exclude
1680
+ // it from model history when a later tool step reloads the ledger.
1681
+ await flushStep(true);
1682
+ currentStepMessageId = this.deps.newId();
1750
1683
  // The failed request did not return authoritative usage. Keep
1751
1684
  // effectiveness recoverable, but fail final metering closed.
1752
1685
  sawUnusableStepUsage = true;
1753
1686
  const delayMs = providerRetryDelayMs(providerAttempt, failure.retryAfterMs);
1754
1687
  const nextAttempt = providerAttempt + 1;
1755
- const maxAttempts = idleWatchdogRecovery || incompleteStreamRecovery || sealedThinkingRecovery
1756
- ? nextAttempt
1757
- : MAX_PROVIDER_ATTEMPTS_PER_STEP;
1688
+ const maxAttempts = MAX_PROVIDER_ATTEMPTS_PER_STEP;
1758
1689
  const reason = providerRetryReason(failure.kind);
1759
1690
  queue.push({
1760
1691
  type: 'provider_retry',
@@ -1769,14 +1700,13 @@ export class AiSdkTurn {
1769
1700
  reason,
1770
1701
  });
1771
1702
  await this.deps.providerRetrySleep(delayMs, turnAbortController.signal);
1772
- providerAttempt = nextAttempt;
1773
1703
  queue.push({
1774
1704
  type: 'provider_retry',
1775
1705
  id: this.deps.newId(),
1776
1706
  turnId,
1777
1707
  ts: this.deps.now(),
1778
1708
  phase: 'started',
1779
- attempt: providerAttempt,
1709
+ attempt: nextAttempt,
1780
1710
  maxAttempts,
1781
1711
  reason,
1782
1712
  });
@@ -1786,7 +1716,7 @@ export class AiSdkTurn {
1786
1716
  // safe fold): surface the real provider error via the terminal
1787
1717
  // handler after settling any authoritative usage — never a
1788
1718
  // fabricated success.
1789
- terminalProviderError = settledWatchdogTimeout?.error ?? failure;
1719
+ terminalProviderError = failure;
1790
1720
  terminalRetry = { error: terminalProviderError, retry };
1791
1721
  terminalProviderErrorReason =
1792
1722
  lastCompletedStepHadToolResult && failure.kind === 'timeout'
@@ -1803,10 +1733,7 @@ export class AiSdkTurn {
1803
1733
  if (this.aborted) {
1804
1734
  throw Object.assign(new Error('aborted'), { name: 'AbortError' });
1805
1735
  }
1806
- // Catch-all: flush any residual step content if the provider closed the
1807
- // stream without a trailing `finish-step` for the last step.
1808
1736
  const providerStepId = currentStepMessageId;
1809
- await flushStep();
1810
1737
  if (providerOutcome.kind !== 'completed')
1811
1738
  throw providerOutcome.failure;
1812
1739
  finishReason = providerOutcome.finishReason;
@@ -2109,11 +2036,11 @@ export class AiSdkTurn {
2109
2036
  streamStatus = this.aborted ? 'aborted' : 'error';
2110
2037
  streamErrorClass = this.deps.modelAdapter.classifyError(currentWatchdogTimeout()?.error ?? err);
2111
2038
  // Flush the in-flight step's partial text/thinking before the terminal
2112
- // abort/error events. Earlier steps already flushed at their
2113
- // `finish-step`; this keeps their and this step's streamed-out output on
2039
+ // abort/error events. Earlier steps already flushed at settlement;
2040
+ // this keeps their and this step's streamed-out output on
2114
2041
  // BOTH exits — user stop and provider error / watchdog timeout — so the
2115
2042
  // transcript keeps what the user actually saw.
2116
- await flushStep().catch(() => { });
2043
+ await flushStep(!this.aborted).catch(() => { });
2117
2044
  if (this.aborted) {
2118
2045
  queue.push({
2119
2046
  type: 'abort',