@librechat/agents 3.7.13 → 3.7.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/dist/cjs/agents/AgentContext.cjs +1 -1
  2. package/dist/cjs/common/constants.cjs +2 -0
  3. package/dist/cjs/common/constants.cjs.map +1 -1
  4. package/dist/cjs/graphs/Graph.cjs +46 -10
  5. package/dist/cjs/graphs/Graph.cjs.map +1 -1
  6. package/dist/cjs/hooks/index.cjs +2 -0
  7. package/dist/cjs/hooks/index.cjs.map +1 -1
  8. package/dist/cjs/langfuse.cjs +16 -0
  9. package/dist/cjs/langfuse.cjs.map +1 -1
  10. package/dist/cjs/llm/invoke.cjs +191 -81
  11. package/dist/cjs/llm/invoke.cjs.map +1 -1
  12. package/dist/cjs/llm/preempt.cjs +77 -1
  13. package/dist/cjs/llm/preempt.cjs.map +1 -1
  14. package/dist/cjs/llm/prepareProviderRequest.cjs +2 -2
  15. package/dist/cjs/main.cjs +7 -4
  16. package/dist/cjs/messages/core.cjs +6 -0
  17. package/dist/cjs/messages/core.cjs.map +1 -1
  18. package/dist/cjs/messages/index.cjs +1 -1
  19. package/dist/cjs/messages/prune.cjs +2 -2
  20. package/dist/cjs/run.cjs +3 -2
  21. package/dist/cjs/run.cjs.map +1 -1
  22. package/dist/cjs/stream.cjs +4 -6
  23. package/dist/cjs/stream.cjs.map +1 -1
  24. package/dist/cjs/summarization/node.cjs +1 -1
  25. package/dist/cjs/utils/tokens.cjs +1 -1
  26. package/dist/esm/agents/AgentContext.mjs +1 -1
  27. package/dist/esm/common/constants.mjs +2 -1
  28. package/dist/esm/common/constants.mjs.map +1 -1
  29. package/dist/esm/graphs/Graph.mjs +46 -10
  30. package/dist/esm/graphs/Graph.mjs.map +1 -1
  31. package/dist/esm/hooks/index.mjs +2 -1
  32. package/dist/esm/hooks/index.mjs.map +1 -1
  33. package/dist/esm/langfuse.mjs +16 -0
  34. package/dist/esm/langfuse.mjs.map +1 -1
  35. package/dist/esm/llm/invoke.mjs +191 -81
  36. package/dist/esm/llm/invoke.mjs.map +1 -1
  37. package/dist/esm/llm/preempt.mjs +72 -2
  38. package/dist/esm/llm/preempt.mjs.map +1 -1
  39. package/dist/esm/llm/prepareProviderRequest.mjs +2 -2
  40. package/dist/esm/main.mjs +7 -7
  41. package/dist/esm/messages/core.mjs +6 -1
  42. package/dist/esm/messages/core.mjs.map +1 -1
  43. package/dist/esm/messages/index.mjs +1 -1
  44. package/dist/esm/messages/prune.mjs +2 -2
  45. package/dist/esm/run.mjs +3 -2
  46. package/dist/esm/run.mjs.map +1 -1
  47. package/dist/esm/stream.mjs +4 -6
  48. package/dist/esm/stream.mjs.map +1 -1
  49. package/dist/esm/summarization/node.mjs +1 -1
  50. package/dist/esm/utils/tokens.mjs +1 -1
  51. package/dist/types/common/constants.d.ts +15 -0
  52. package/dist/types/graphs/Graph.d.ts +50 -0
  53. package/dist/types/hooks/index.d.ts +15 -0
  54. package/dist/types/llm/invoke.d.ts +17 -1
  55. package/dist/types/llm/preempt.d.ts +114 -0
  56. package/dist/types/messages/core.d.ts +19 -0
  57. package/dist/types/run.d.ts +5 -3
  58. package/dist/types/types/run.d.ts +58 -7
  59. package/package.json +1 -1
  60. package/src/common/constants.ts +15 -0
  61. package/src/graphs/Graph.ts +122 -6
  62. package/src/hooks/index.ts +15 -0
  63. package/src/langfuse.ts +53 -0
  64. package/src/llm/invoke.ts +614 -130
  65. package/src/llm/preempt.ts +320 -1
  66. package/src/messages/core.ts +28 -0
  67. package/src/run.ts +12 -4
  68. package/src/stream.ts +4 -12
  69. package/src/types/run.ts +58 -7
package/src/llm/invoke.ts CHANGED
@@ -16,6 +16,7 @@ import type { ToolOutputReferenceRegistry } from '@/tools/toolOutputReferences';
16
16
  import type { PreparedProviderRequest } from '@/llm/prepareProviderRequest';
17
17
  import type { ContextOverflowContext } from '@/utils/errors';
18
18
  import type { StreamLimitState } from '@/llm/streamLimits';
19
+ import type { PreemptAction } from '@/llm/preempt';
19
20
  import type * as t from '@/types';
20
21
  import {
21
22
  enforceStreamLimitsForWireChunk,
@@ -33,6 +34,13 @@ import {
33
34
  resolveProviderMessageProjectionInvariantMode,
34
35
  modifyDeltaProperties,
35
36
  } from '@/messages';
37
+ import {
38
+ canRestartPreempt,
39
+ notePreemptRestartedRun,
40
+ PREEMPT_RESTART_CONTROL_FLOW,
41
+ resolvePreemptAction,
42
+ resolveRestartGraceMs,
43
+ } from '@/llm/preempt';
36
44
  import {
37
45
  assertPreparedProviderRequestFor,
38
46
  prepareProviderRequest,
@@ -48,7 +56,7 @@ import { resolveClientOptionsModel } from '@/llm/request';
48
56
  import { safeDispatchCustomEvent } from '@/utils/events';
49
57
  import { getContextOverflowInfo } from '@/utils/errors';
50
58
  import { appendCallbacks } from '@/utils/callbacks';
51
- import { canSealPreempt } from '@/llm/preempt';
59
+ import { composeAbortSignals } from '@/utils/misc';
52
60
  import { initializeModel } from '@/llm/init';
53
61
 
54
62
  export {
@@ -110,6 +118,10 @@ export type OnChunk = (
110
118
  metadata?: Record<string, unknown>
111
119
  ) => void | Promise<void>;
112
120
 
121
+ /** Node coerces a `setTimeout` delay past this to 1ms, so a longer grace is
122
+ * reached by chaining rather than by one out-of-range timer. */
123
+ const MAX_TIMEOUT_MS = 2_147_483_647;
124
+
113
125
  /** Unique per-model-attempt sequence; see the stamp in `attemptInvoke`. */
114
126
  let streamLimitAttemptSeq = 0;
115
127
 
@@ -497,7 +509,21 @@ async function endSealedModelRun(
497
509
  prompt: BaseMessage[],
498
510
  llmRunId: string | undefined,
499
511
  config?: RunnableConfig,
500
- model?: t.ChatModel
512
+ model?: t.ChatModel,
513
+ /**
514
+ * Shapes the close for a DISCARDED turn. A seal's output is the partial
515
+ * answer it kept; a restart kept nothing, and marking it here is what keeps
516
+ * the two restart routes reading alike in a trace — the aborted one is
517
+ * relabelled through the tracing callback, and this is the manual twin.
518
+ */
519
+ llmOutput: Record<string, unknown> = {},
520
+ /**
521
+ * Set when the run was probably closed already, so a failed native close is
522
+ * the expected outcome rather than a fault. The synthetic fallback below is
523
+ * the real close in that case, and warning on every interrupt would bury the
524
+ * failures that do matter.
525
+ */
526
+ nativeCloseOptional = false
501
527
  ): Promise<void> {
502
528
  const metadata = config?.metadata as Record<string, unknown> | undefined;
503
529
  synthesizeSealedUsage(context, chunk, prompt, metadata);
@@ -537,7 +563,7 @@ async function endSealedModelRun(
537
563
  };
538
564
  await runManager.handleLLMEnd({
539
565
  generations: [[generation]],
540
- llmOutput: {},
566
+ llmOutput,
541
567
  });
542
568
  return;
543
569
  }
@@ -546,11 +572,13 @@ async function endSealedModelRun(
546
572
  * A sealed answer that reaches the user is worth more than a tidy
547
573
  * trace. Fall through to the custom event rather than failing the run.
548
574
  */
549
- // eslint-disable-next-line no-console
550
- console.warn(
551
- '[attemptInvoke] Native close of the sealed model run failed; falling back to a custom event:',
552
- e instanceof Error ? e.message : e
553
- );
575
+ if (!nativeCloseOptional) {
576
+ // eslint-disable-next-line no-console
577
+ console.warn(
578
+ '[attemptInvoke] Native close of the sealed model run failed; falling back to a custom event:',
579
+ e instanceof Error ? e.message : e
580
+ );
581
+ }
554
582
  }
555
583
  }
556
584
  await safeDispatchCustomEvent(
@@ -590,6 +618,17 @@ function appendStreamChunk({
590
618
  interface AttemptInvokeCommonParams {
591
619
  context?: InvokeContext;
592
620
  onChunk?: OnChunk;
621
+ /**
622
+ * The agent lane this attempt belongs to, as the model node names it. Only
623
+ * read to record a discarded turn, which the node cannot otherwise detect —
624
+ * a discard returns no message to carry `response_metadata.preempted`.
625
+ *
626
+ * Keyed rather than global for the same reason `pendingPreemptReturn` is:
627
+ * `MultiAgentGraph` routes every parallel agent through ONE graph instance,
628
+ * so a shared flag would let whichever lane finished first consume another
629
+ * lane's boundary — and inject that lane's words into the wrong turn.
630
+ */
631
+ preemptAgentId?: string;
593
632
  /** Accounting owner for callers that deliberately pass no `context`
594
633
  * (summarization) — used ONLY for the attempt's accounting lease, never
595
634
  * for charge claims. */
@@ -691,6 +730,7 @@ export async function attemptInvoke(
691
730
  request: resolveAttemptRequest(params, stampedConfig),
692
731
  context: params.context,
693
732
  onChunk: params.onChunk,
733
+ preemptAgentId: params.preemptAgentId,
694
734
  },
695
735
  stampedConfig
696
736
  );
@@ -706,7 +746,11 @@ async function attemptInvokeBody(
706
746
  request,
707
747
  context,
708
748
  onChunk,
709
- }: Pick<AttemptInvokeCommonParams, 'context' | 'onChunk'> & {
749
+ preemptAgentId,
750
+ }: Pick<
751
+ AttemptInvokeCommonParams,
752
+ 'context' | 'onChunk' | 'preemptAgentId'
753
+ > & {
710
754
  request: PreparedProviderRequest;
711
755
  },
712
756
  config: RunnableConfig
@@ -747,11 +791,238 @@ async function attemptInvokeBody(
747
791
  * survive the bound runnable. The same handler owns the opt-in projection
748
792
  * invariant so enabled diagnostics do not stack a second model callback.
749
793
  */
750
- const stream = await model.stream(messagesForProvider, invocationConfig);
751
794
  let finalChunk: AIMessageChunk | undefined;
752
- let preempted = false;
795
+ let preemptAction: PreemptAction = 'none';
753
796
  const registeredStreamHandler =
754
797
  getRegisteredDefaultChatStreamHandler(context);
798
+ /**
799
+ * The wake channel is armed for the seal-capable branch only. The other
800
+ * two consume through readers that lag the accumulation — the same reason
801
+ * the per-chunk poll lives in that branch alone — and an `onChunk`
802
+ * consumer owns the stream outright.
803
+ *
804
+ * An unnamed lane disarms it too. A caller that supplies a preemption
805
+ * source but no `preemptAgentId` cannot have its discard routed back to
806
+ * the right model node, so it keeps today's behavior — the request waits
807
+ * for a boundary — rather than ending a turn nothing will resume.
808
+ */
809
+ const restartLane =
810
+ onChunk == null &&
811
+ registeredStreamHandler == null &&
812
+ context?.preemption?.subscribe != null
813
+ ? preemptAgentId
814
+ : undefined;
815
+ const restartController =
816
+ restartLane == null ? undefined : new AbortController();
817
+ /**
818
+ * Composed, never replaced: the run's own signal must keep tearing this
819
+ * stream down. `composeAbortSignals` collapses back to a single signal
820
+ * when there is nothing to compose.
821
+ */
822
+ const streamConfig =
823
+ restartController == null
824
+ ? invocationConfig
825
+ : {
826
+ ...invocationConfig,
827
+ signal: composeAbortSignals(
828
+ invocationConfig.signal,
829
+ restartController.signal
830
+ ),
831
+ };
832
+ const restartGraceMs = resolveRestartGraceMs(
833
+ context?.preemption?.restartGraceMs
834
+ );
835
+ /**
836
+ * When THIS attempt first saw the request, which is what the grace window
837
+ * is measured against. Recorded on first observation rather than on arm:
838
+ * the host's flag is level-triggered and may already have been true for a
839
+ * previous attempt, and a retry inheriting an aged request would discard
840
+ * its very first chunk.
841
+ */
842
+ let preemptRequestedAt: number | undefined;
843
+ let restartGraceTimer: ReturnType<typeof setTimeout> | undefined;
844
+ /** True while a chunk is being handed to the stream handler and has not
845
+ * yet been folded into `finalChunk`. See the guard in
846
+ * {@link evaluatePreemptRestart}. */
847
+ let dispatchingChunk = false;
848
+ /** Read through a call so the check sees the value the wake handler
849
+ * writes: control-flow analysis at the loop head still holds the
850
+ * initializer, and would narrow the comparison away as unreachable. */
851
+ const restartDecided = (): boolean => preemptAction === 'restart';
852
+ /**
853
+ * How the stream ended when a restart was chosen, because each way leaves
854
+ * the model run in a different state and needs a different close:
855
+ * - `aborted`: LangChain closed it through its error path, where the
856
+ * marker relabels it and supplies the discarded attempt's usage.
857
+ * - `broke`: the iterator was closed early, which fires NEITHER end nor
858
+ * error callbacks, so this is the one route that must close natively.
859
+ * - `exhausted`: the stream finished on its own, so LangChain already
860
+ * emitted its end. Closing again would warn and change nothing — and
861
+ * the generation is honestly an ordinary completed one, since nothing
862
+ * was torn down; only the turn built on it is being discarded.
863
+ */
864
+ let restartRoute: 'aborted' | 'broke' | 'exhausted' | undefined;
865
+ /**
866
+ * Cleared when the attempt returns. A host may deliver its wake through a
867
+ * queue or a message bus, so one can already be in flight when the
868
+ * unsubscribe runs — and a late claim would take the seal slot, abort a
869
+ * finished controller, and never reach a boundary to release it, blocking
870
+ * every later preemption in the run.
871
+ */
872
+ let attemptActive = true;
873
+ /**
874
+ * Only an exhausted stream leaves nothing for this attempt to close: it
875
+ * emitted its own end. Both other routes attempt the native close, because
876
+ * whether the run is still open is NOT observable from here — an adapter
877
+ * that ignores the abort keeps it open, while one that honors it (or that
878
+ * simply hands over a chunk buffered before the abort landed) has already
879
+ * closed it through the error path. `endSealedModelRun` degrades to the
880
+ * synthetic event when the run is gone, which is what makes both right.
881
+ */
882
+ const restartNeedsNativeClose = (): boolean => restartRoute !== 'exhausted';
883
+ /**
884
+ * An aborted run is EXPECTED to be closed already, so a failed native
885
+ * close is the normal outcome there and must not warn on every interrupt.
886
+ * A broken iterator fired no callback at all, so a failure is worth
887
+ * hearing about.
888
+ */
889
+ const restartCloseMayHaveHappened = (): boolean =>
890
+ restartRoute === 'aborted';
891
+
892
+ /**
893
+ * The wake is a hint; this is where the request is actually read. The
894
+ * shape is judged at the instant we look at it, and the teardown that
895
+ * follows is what makes any later chunk irrelevant — so a wake arriving
896
+ * once an answer has started, or one that loses the shared seal slot,
897
+ * leaves the stream untouched and waits for the ordinary boundary.
898
+ *
899
+ * A `seal` verdict is deliberately ignored here. Sealing KEEPS the
900
+ * accumulated turn, and only the chunk loop can hand it over intact; this
901
+ * path exists solely to end turns that have nothing to hand over.
902
+ *
903
+ * The self-rescheduling timer is what covers the silent window. Nothing
904
+ * else will look again — a provider that has gone quiet produces no chunk
905
+ * to poll on — so without it a request arriving mid-grace would wait out
906
+ * the whole turn, which is the stall this path exists to remove.
907
+ */
908
+ const evaluatePreemptRestart = (): boolean => {
909
+ /** Reports whether a restart IS decided, not whether this call decided
910
+ * it. A synchronous wake during `subscribe` can settle the action
911
+ * before the pre-call read runs, and a caller that only learned "I did
912
+ * not convert" would go on to issue a provider request against an
913
+ * already-aborted signal. */
914
+ if (!attemptActive) {
915
+ return false;
916
+ }
917
+ if (restartController == null || preemptAction !== 'none') {
918
+ return preemptAction === 'restart';
919
+ }
920
+ if (context?.shouldPreemptStream() !== true) {
921
+ /** The request was withdrawn. Forget when it arrived, or a NEW request
922
+ * later in this same attempt would inherit the old clock and convert
923
+ * without ever getting its grace. */
924
+ preemptRequestedAt = undefined;
925
+ return false;
926
+ }
927
+ /** A chunk is mid-dispatch to the host, so `finalChunk` describes only
928
+ * the chunks BEFORE it. Judging the turn now could read a text chunk
929
+ * the host has already been shown as an empty turn and discard it. The
930
+ * per-chunk poll runs immediately after the append with the complete
931
+ * accumulation, so nothing is lost by declining here. */
932
+ if (dispatchingChunk) {
933
+ return false;
934
+ }
935
+ preemptRequestedAt ??= Date.now();
936
+ const requestAgeMs = Date.now() - preemptRequestedAt;
937
+ const action = resolvePreemptAction({
938
+ chunk: finalChunk,
939
+ requestAgeMs,
940
+ graceMs: restartGraceMs,
941
+ });
942
+ if (action !== 'restart') {
943
+ if (
944
+ action === 'none' &&
945
+ restartGraceTimer == null &&
946
+ canRestartPreempt(finalChunk)
947
+ ) {
948
+ /**
949
+ * Scheduled for what is LEFT of the window, not a fresh one: the
950
+ * clock starts when the request is first seen, so a wake arriving
951
+ * mid-window must not push the conversion further out than a wake
952
+ * arriving at its start.
953
+ *
954
+ * The handle is released as the timer fires so a look that changes
955
+ * nothing — the shape moved, the host disarmed and re-armed — can
956
+ * still schedule the next one.
957
+ */
958
+ restartGraceTimer = setTimeout(
959
+ () => {
960
+ restartGraceTimer = undefined;
961
+ /**
962
+ * Contained, because this frame has no invocation to reject
963
+ * into: a host predicate that throws here would surface as an
964
+ * uncaught exception and can take the process down, while the
965
+ * same throw from the per-chunk poll is held by the attempt's
966
+ * promise. The cost of swallowing is one missed look, and a
967
+ * predicate that throws has no request to honor anyway.
968
+ */
969
+ try {
970
+ evaluatePreemptRestart();
971
+ } catch {
972
+ /** empty */
973
+ }
974
+ },
975
+ /**
976
+ * Clamped, then CHAINED: a delay past Node's ceiling silently
977
+ * becomes 1ms, and this timer reschedules itself, so an
978
+ * out-of-range grace would spin at ~1ms for the life of the
979
+ * stream. Each firing re-reads the real age, so the requested
980
+ * deadline survives being reached in several hops.
981
+ */
982
+ Math.min(MAX_TIMEOUT_MS, Math.max(0, restartGraceMs - requestAgeMs))
983
+ );
984
+ /** The grace is a fallback, never a reason to hold the process
985
+ * open: a run that ends before the window elapses must not be kept
986
+ * alive by a timer whose only job is to look again. */
987
+ restartGraceTimer.unref();
988
+ }
989
+ return false;
990
+ }
991
+ if (!context.claimPreemptRestart()) {
992
+ return false;
993
+ }
994
+ preemptAction = 'restart';
995
+ /**
996
+ * Recorded BEFORE the abort: the adapter's cancellation error can reach
997
+ * the tracing callback synchronously, and a run marked afterwards would
998
+ * already have closed as a failure.
999
+ *
1000
+ * `sealedRunId` being set is also what proves the request went OUT, so
1001
+ * charging the prompt against it is honest. Usage is resolved here, onto
1002
+ * the same chunk the discard reports later — the run closes through the
1003
+ * error path, which carries no output, so a marker without it would make
1004
+ * every restart look free. `synthesizeSealedUsage` no-ops when the
1005
+ * provider already streamed usage, and again when the discard path runs,
1006
+ * so this is one computation, not two.
1007
+ */
1008
+ if (sealedRunId != null) {
1009
+ finalChunk ??= new AIMessageChunk({ content: '' });
1010
+ synthesizeSealedUsage(
1011
+ context,
1012
+ finalChunk,
1013
+ messagesForProvider,
1014
+ config.metadata as Record<string, unknown> | undefined
1015
+ );
1016
+ notePreemptRestartedRun(sealedRunId, finalChunk);
1017
+ }
1018
+ restartRoute = 'aborted';
1019
+ restartController.abort();
1020
+ return true;
1021
+ };
1022
+ const unsubscribeWake =
1023
+ restartController == null
1024
+ ? undefined
1025
+ : context?.preemption?.subscribe?.(evaluatePreemptRestart);
755
1026
  /** A sibling's trip aborts the composed signal, but an adapter that
756
1027
  * ignores cancellation keeps yielding — and text-only chunks with the
757
1028
  * event cap off never throw in enforcement, so nothing else would stop
@@ -767,137 +1038,245 @@ async function attemptInvokeBody(
767
1038
  }
768
1039
  };
769
1040
 
770
- if (onChunk) {
771
- const attemptMetadata = config.metadata as
772
- | Record<string, unknown>
773
- | undefined;
774
- for await (const chunk of stream) {
775
- throwIfBreakerTripped();
776
- /** An onChunk consumer replaces the stream handler entirely, so
777
- * stream limits are enforced here for every such caller — public
778
- * package consumers get no other accounting. The internal
779
- * summarization onChunk charges producer-side itself and passes no
780
- * context, precisely so this claim and its own never stack. */
781
- if (context != null) {
782
- enforceStreamLimitsForWireChunk({
783
- graph: context,
784
- metadata: attemptMetadata,
785
- chunk,
786
- });
787
- }
788
- await onChunk(chunk, attemptMetadata);
789
- finalChunk = appendStreamChunk({
790
- current: finalChunk,
791
- next: chunk,
792
- provider,
793
- });
794
- }
795
- } else if (registeredStreamHandler == null) {
796
- const metadata = config.metadata as Record<string, unknown> | undefined;
797
- const streamHandler = new ChatModelStreamHandler();
798
- for await (const chunk of stream) {
799
- throwIfBreakerTripped();
800
- const handlingChunk = getStreamHandlingChunk({
801
- current: finalChunk,
802
- next: chunk,
803
- provider,
804
- });
805
- if (handlingChunk != null) {
806
- await streamHandler.handle(
807
- GraphEvents.CHAT_MODEL_STREAM,
808
- { chunk: handlingChunk },
809
- metadata,
810
- context
811
- );
812
- } else if (context != null) {
1041
+ try {
1042
+ /**
1043
+ * Subscribing is not enough on its own. `shouldPreempt` is
1044
+ * level-triggered, so a request armed BEFORE this attempt began — during
1045
+ * setup, or on a previous attempt a fallback replaced — is already true
1046
+ * and will never produce another wake. Reading it once here is the
1047
+ * difference between honoring that request and waiting out the turn.
1048
+ *
1049
+ * Read before the provider request is issued so the grace clock starts
1050
+ * from the attempt's own beginning rather than from its first chunk. A
1051
+ * host that set `restartGraceMs` to zero skips the call entirely here,
1052
+ * having streamed nothing and opened no model run — which is why the
1053
+ * close below is conditional.
1054
+ */
1055
+ if (!evaluatePreemptRestart()) {
1056
+ const stream = await model.stream(messagesForProvider, streamConfig);
1057
+ if (onChunk) {
1058
+ const attemptMetadata = config.metadata as
1059
+ | Record<string, unknown>
1060
+ | undefined;
1061
+ for await (const chunk of stream) {
1062
+ throwIfBreakerTripped();
1063
+ /** An onChunk consumer replaces the stream handler entirely, so
1064
+ * stream limits are enforced here for every such caller — public
1065
+ * package consumers get no other accounting. The internal
1066
+ * summarization onChunk charges producer-side itself and passes no
1067
+ * context, precisely so this claim and its own never stack. */
1068
+ if (context != null) {
1069
+ enforceStreamLimitsForWireChunk({
1070
+ graph: context,
1071
+ metadata: attemptMetadata,
1072
+ chunk,
1073
+ });
1074
+ }
1075
+ await onChunk(chunk, attemptMetadata);
1076
+ finalChunk = appendStreamChunk({
1077
+ current: finalChunk,
1078
+ next: chunk,
1079
+ provider,
1080
+ });
1081
+ }
1082
+ } else if (registeredStreamHandler == null) {
1083
+ const metadata = config.metadata as Record<string, unknown> | undefined;
1084
+ const streamHandler = new ChatModelStreamHandler();
1085
+ for await (const chunk of stream) {
1086
+ throwIfBreakerTripped();
1087
+ /**
1088
+ * The decision is final, so stop consuming here rather than
1089
+ * trusting the adapter to honor the abort. An adapter that ignores
1090
+ * cancellation — or one whose iterator hands over a chunk buffered
1091
+ * before the abort landed — would otherwise keep dispatching text
1092
+ * the host is about to be told was discarded, and in the ignoring
1093
+ * case would hold the boundary until the whole response finished:
1094
+ * exactly the stall a restart exists to end.
1095
+ */
1096
+ if (restartDecided()) {
1097
+ break;
1098
+ }
1099
+ const handlingChunk = getStreamHandlingChunk({
1100
+ current: finalChunk,
1101
+ next: chunk,
1102
+ provider,
1103
+ });
1104
+ if (handlingChunk != null) {
1105
+ dispatchingChunk = true;
1106
+ try {
1107
+ await streamHandler.handle(
1108
+ GraphEvents.CHAT_MODEL_STREAM,
1109
+ { chunk: handlingChunk },
1110
+ metadata,
1111
+ context
1112
+ );
1113
+ } finally {
1114
+ dispatchingChunk = false;
1115
+ }
1116
+ } else if (context != null) {
1117
+ /**
1118
+ * A replay-skipped chunk yields no handling chunk, and in this
1119
+ * local branch no `streamEvents` consumer judges the wire event
1120
+ * either — yet a cumulative OpenRouter replay can still carry
1121
+ * `tool_call_chunks` or complete `tool_calls` that are appended
1122
+ * below. Charge the full limits (event budget and argument bytes)
1123
+ * directly so neither cap can be bypassed. Consumer side: the
1124
+ * local handler.handle call above claims as consumer, and one
1125
+ * reused chunk object can alternate between these two arms.
1126
+ */
1127
+ enforceStreamLimitsForWireChunk({
1128
+ graph: context,
1129
+ metadata,
1130
+ chunk,
1131
+ side: 'consumer',
1132
+ });
1133
+ }
1134
+ finalChunk = appendStreamChunk({
1135
+ current: finalChunk,
1136
+ next: chunk,
1137
+ provider,
1138
+ });
1139
+ /**
1140
+ * Only this loop may seal. The registered-handler branch below
1141
+ * dispatches through `run.ts`'s decoupled `streamEvents` consumer,
1142
+ * which can lag the accumulated chunk — sealing there would let the
1143
+ * host index a content part the user has not been shown yet.
1144
+ */
1145
+ /**
1146
+ * Cheap poll first, shape check second, budget claim last. The claim
1147
+ * is what makes this safe under a parallel `MultiAgentGraph`: several
1148
+ * agents share one graph and can each see the poll as true, but only
1149
+ * one can take the slot, and a chunk that cannot seal never spends it.
1150
+ *
1151
+ * `resolvePreemptAction` prefers a seal wherever one is available,
1152
+ * so a turn that already produced an answer keeps it. `restart` is
1153
+ * reached only for an accumulation holding nothing but reasoning —
1154
+ * the case a seal can never accept, and the reason an interrupt
1155
+ * armed during a long thinking stretch used to wait for the whole
1156
+ * turn.
1157
+ */
1158
+ if (context?.shouldPreemptStream() !== true) {
1159
+ /** Withdrawn: forget the clock, so a later request in this same
1160
+ * attempt still gets its own grace. */
1161
+ preemptRequestedAt = undefined;
1162
+ } else {
1163
+ preemptRequestedAt ??= Date.now();
1164
+ const action = resolvePreemptAction({
1165
+ chunk: finalChunk,
1166
+ requestAgeMs: Date.now() - preemptRequestedAt,
1167
+ graceMs: restartGraceMs,
1168
+ });
1169
+ /**
1170
+ * A restart needs the lane that names where its boundary is
1171
+ * owed. Without one — a host still on the seal-only contract,
1172
+ * which supplies no `subscribe` — the discard would return no
1173
+ * message AND record no lane, so the node would dispatch no
1174
+ * boundary, inject nothing, and end the turn empty with the
1175
+ * steer still queued. Such a host also never opted into having
1176
+ * its turns discarded, so its request waits for a seal exactly
1177
+ * as it does today.
1178
+ */
1179
+ const claimed =
1180
+ action === 'seal'
1181
+ ? context.claimPreemptSeal()
1182
+ : action === 'restart' &&
1183
+ restartLane != null &&
1184
+ context.claimPreemptRestart();
1185
+ if (claimed) {
1186
+ preemptAction = action;
1187
+ if (action === 'restart') {
1188
+ restartRoute = 'broke';
1189
+ }
1190
+ break;
1191
+ }
1192
+ }
1193
+ }
1194
+ } else {
1195
+ const metadata = config.metadata as Record<string, unknown> | undefined;
813
1196
  /**
814
- * A replay-skipped chunk yields no handling chunk, and in this
815
- * local branch no `streamEvents` consumer judges the wire event
816
- * either — yet a cumulative OpenRouter replay can still carry
817
- * `tool_call_chunks` or complete `tool_calls` that are appended
818
- * below. Charge the full limits (event budget and argument bytes)
819
- * directly so neither cap can be bypassed. Consumer side: the
820
- * local handler.handle call above claims as consumer, and one
821
- * reused chunk object can alternate between these two arms.
1197
+ * The original wire chunk still reaches the registered handler through
1198
+ * `streamEvents` (where the late-reasoning skip discards it AFTER the
1199
+ * event guard counts it), so this inline re-dispatch of the transformed
1200
+ * chunk is marked to not consume a second event-budget slot. Allocated
1201
+ * once per attempt, only when a transformation occurs.
822
1202
  */
823
- enforceStreamLimitsForWireChunk({
824
- graph: context,
825
- metadata,
826
- chunk,
827
- side: 'consumer',
828
- });
1203
+ let redispatchMetadata: Record<string, unknown> | undefined;
1204
+ for await (const chunk of stream) {
1205
+ throwIfBreakerTripped();
1206
+ /**
1207
+ * Charged synchronously, ahead of the decoupled `streamEvents`
1208
+ * reader that will echo this same chunk to the registered handler:
1209
+ * a lagging reader would otherwise let an oversized complete call
1210
+ * return to LangGraph and reach ToolNode before the queued handler
1211
+ * throws. The chunk is marked so the echo skips accounting.
1212
+ */
1213
+ if (context != null) {
1214
+ enforceStreamLimitsForWireChunk({ graph: context, metadata, chunk });
1215
+ }
1216
+ const handlingChunk = getStreamHandlingChunk({
1217
+ current: finalChunk,
1218
+ next: chunk,
1219
+ provider,
1220
+ });
1221
+ if (handlingChunk != null && handlingChunk !== chunk) {
1222
+ redispatchMetadata ??= {
1223
+ ...(metadata ?? {}),
1224
+ [STREAM_LIMIT_REDISPATCH_KEY]: true,
1225
+ };
1226
+ await registeredStreamHandler.handle(
1227
+ GraphEvents.CHAT_MODEL_STREAM,
1228
+ { chunk: handlingChunk },
1229
+ redispatchMetadata,
1230
+ context
1231
+ );
1232
+ }
1233
+ finalChunk = appendStreamChunk({
1234
+ current: finalChunk,
1235
+ next: chunk,
1236
+ provider,
1237
+ });
1238
+ }
829
1239
  }
830
- finalChunk = appendStreamChunk({
831
- current: finalChunk,
832
- next: chunk,
833
- provider,
834
- });
835
1240
  /**
836
- * Only this loop may seal. The registered-handler branch below
837
- * dispatches through `run.ts`'s decoupled `streamEvents` consumer,
838
- * which can lag the accumulated chunk — sealing there would let the
839
- * host index a content part the user has not been shown yet.
840
- */
841
- /**
842
- * Cheap poll first, shape check second, budget claim last. The claim
843
- * is what makes this safe under a parallel `MultiAgentGraph`: several
844
- * agents share one graph and can each see the poll as true, but only
845
- * one can take the slot, and a chunk that cannot seal never spends it.
1241
+ * The stream reached its natural end while a request was still armed
1242
+ * and the turn still holds nothing worth keeping. The grace exists to
1243
+ * avoid stealing a seal that was about to become possible — and the
1244
+ * final shape is proof none was: no text ever arrived. Converting here
1245
+ * spends no extra provider call (the stream is over) and spares the
1246
+ * host a reasoning-only turn followed by its own steer, which is the
1247
+ * shape the seal gate refuses to build in the first place.
846
1248
  */
847
1249
  if (
1250
+ preemptAction === 'none' &&
1251
+ restartLane != null &&
848
1252
  context?.shouldPreemptStream() === true &&
849
- canSealPreempt(finalChunk) &&
850
- context.claimPreemptSeal()
1253
+ canRestartPreempt(finalChunk) &&
1254
+ context.claimPreemptRestart()
851
1255
  ) {
852
- preempted = true;
853
- break;
1256
+ preemptAction = 'restart';
1257
+ restartRoute = 'exhausted';
854
1258
  }
855
1259
  }
856
- } else {
857
- const metadata = config.metadata as Record<string, unknown> | undefined;
1260
+ } catch (error) {
858
1261
  /**
859
- * The original wire chunk still reaches the registered handler through
860
- * `streamEvents` (where the late-reasoning skip discards it AFTER the
861
- * event guard counts it), so this inline re-dispatch of the transformed
862
- * chunk is marked to not consume a second event-budget slot. Allocated
863
- * once per attempt, only when a transformation occurs.
1262
+ * A restart tears the provider stream down mid-flight, which surfaces
1263
+ * as the composed signal's abort — from the iteration, or from stream
1264
+ * creation when the request was already outstanding. Swallowed only when
1265
+ * THIS attempt asked for it: `preemptAction` is set immediately before
1266
+ * the abort and by nothing else, so a run-level abort, a tripped stream
1267
+ * limit and every provider error still propagate.
864
1268
  */
865
- let redispatchMetadata: Record<string, unknown> | undefined;
866
- for await (const chunk of stream) {
867
- throwIfBreakerTripped();
868
- /**
869
- * Charged synchronously, ahead of the decoupled `streamEvents`
870
- * reader that will echo this same chunk to the registered handler:
871
- * a lagging reader would otherwise let an oversized complete call
872
- * return to LangGraph and reach ToolNode before the queued handler
873
- * throws. The chunk is marked so the echo skips accounting.
874
- */
875
- if (context != null) {
876
- enforceStreamLimitsForWireChunk({ graph: context, metadata, chunk });
877
- }
878
- const handlingChunk = getStreamHandlingChunk({
879
- current: finalChunk,
880
- next: chunk,
881
- provider,
882
- });
883
- if (handlingChunk != null && handlingChunk !== chunk) {
884
- redispatchMetadata ??= {
885
- ...(metadata ?? {}),
886
- [STREAM_LIMIT_REDISPATCH_KEY]: true,
887
- };
888
- await registeredStreamHandler.handle(
889
- GraphEvents.CHAT_MODEL_STREAM,
890
- { chunk: handlingChunk },
891
- redispatchMetadata,
892
- context
893
- );
894
- }
895
- finalChunk = appendStreamChunk({
896
- current: finalChunk,
897
- next: chunk,
898
- provider,
899
- });
1269
+ const ownAbort =
1270
+ restartRoute === 'aborted' &&
1271
+ !(error instanceof StreamLimitExceededError) &&
1272
+ config.signal?.aborted !== true;
1273
+ if (!ownAbort) {
1274
+ throw error;
900
1275
  }
1276
+ } finally {
1277
+ attemptActive = false;
1278
+ unsubscribeWake?.();
1279
+ clearTimeout(restartGraceTimer);
901
1280
  }
902
1281
 
903
1282
  if (providerUsesManualToolStream(provider)) {
@@ -907,7 +1286,104 @@ async function attemptInvokeBody(
907
1286
  );
908
1287
  }
909
1288
 
910
- if (preempted && finalChunk != null) {
1289
+ if (preemptAction === 'restart') {
1290
+ /**
1291
+ * The turn is thrown away, but a model run that OPENED still has to be
1292
+ * closed:
1293
+ * its callback span is open, and a host that renders from the stream
1294
+ * has already drawn the reasoning that is about to vanish. The synthetic
1295
+ * end carries `preemptDiscarded` so both can tell this apart from a seal
1296
+ * — a seal's content survives into the next prompt, a discard's does
1297
+ * not, and a host that keeps rendering it would show the user words the
1298
+ * model no longer has.
1299
+ *
1300
+ * Usage is still synthesized from whatever accumulated. Those reasoning
1301
+ * tokens were spent and billed; dropping them from the report would make
1302
+ * an interrupted turn look free.
1303
+ *
1304
+ * Skipped entirely when the request never went out — a preempt already
1305
+ * outstanding when the attempt began short-circuits above the provider
1306
+ * call, so there is no open span and no stream for a host to unwind, and
1307
+ * a synthetic end would announce a run that never started.
1308
+ */
1309
+ if (restartRoute !== 'exhausted') {
1310
+ /**
1311
+ * The step really was cut short, so it closes `cancelled` before the
1312
+ * model-end below can close it `completed`. An exhausted turn is left
1313
+ * alone: its stream reached its own end, and the step describing it is
1314
+ * already closed and honest — only the turn built on it is discarded.
1315
+ */
1316
+ await context?.cancelOpenMessageStep(
1317
+ config.metadata as Record<string, unknown> | undefined
1318
+ );
1319
+ }
1320
+ if (finalChunk != null || sealedRunId != null) {
1321
+ const discardedChunk = finalChunk ?? new AIMessageChunk({ content: '' });
1322
+ const responseMetadata = {
1323
+ ...discardedChunk.response_metadata,
1324
+ preempted: true,
1325
+ preemptDiscarded: true,
1326
+ };
1327
+ discardedChunk.response_metadata = responseMetadata;
1328
+ discardedChunk.lc_kwargs.response_metadata = responseMetadata;
1329
+ if (restartRoute === 'exhausted') {
1330
+ /**
1331
+ * The stream finished on its own, so the host has ALREADY had this
1332
+ * turn's model-end, with its usage. Replaying it through
1333
+ * `endSealedModelRun` would charge the same tokens twice in any host
1334
+ * accumulating usage from that event. What the host still needs is
1335
+ * the one thing the natural end could not say — that the turn it
1336
+ * just completed is being thrown away — so that notification carries
1337
+ * the marker and nothing else.
1338
+ */
1339
+ await safeDispatchCustomEvent(
1340
+ GraphEvents.CHAT_MODEL_END,
1341
+ {
1342
+ output: new AIMessageChunk({
1343
+ content: '',
1344
+ response_metadata: responseMetadata,
1345
+ }),
1346
+ },
1347
+ config
1348
+ );
1349
+ } else {
1350
+ /**
1351
+ * Both cut-short routes come here and both pass the run id: see
1352
+ * {@link restartNeedsNativeClose} for why the run's state is not
1353
+ * observable and the close has to work either way. An aborted run
1354
+ * that WAS already closed falls through to the synthetic event,
1355
+ * where the marker relabels it and supplies this usage.
1356
+ */
1357
+ await endSealedModelRun(
1358
+ context,
1359
+ discardedChunk,
1360
+ messagesForProvider,
1361
+ restartNeedsNativeClose() ? sealedRunId : undefined,
1362
+ config,
1363
+ model,
1364
+ PREEMPT_RESTART_CONTROL_FLOW,
1365
+ restartCloseMayHaveHappened()
1366
+ );
1367
+ }
1368
+ }
1369
+ /** Non-null on every path that can reach here: it arms the controller
1370
+ * the wake path needs, and the per-chunk path refuses a restart
1371
+ * without it. */
1372
+ if (restartLane != null) {
1373
+ context?.notePreemptRestart(restartLane);
1374
+ }
1375
+ /**
1376
+ * No message reaches graph state, so the injected user turn lands
1377
+ * directly after the previous one — which is what makes this safe on
1378
+ * every provider: adjacent user turns are native on Anthropic, OpenAI
1379
+ * and Gemini, and normalized by `coalesceAdjacentUserTurns` for the
1380
+ * strict-alternation providers. A seal needs a non-empty assistant turn
1381
+ * precisely because it leaves one behind; a discard leaves none.
1382
+ */
1383
+ return { messages: [] };
1384
+ }
1385
+
1386
+ if (preemptAction === 'seal' && finalChunk != null) {
911
1387
  const responseMetadata = {
912
1388
  ...finalChunk.response_metadata,
913
1389
  preempted: true,
@@ -1018,6 +1494,7 @@ export async function tryFallbackProviders({
1018
1494
  context,
1019
1495
  onChunk,
1020
1496
  streamLimitState,
1497
+ preemptAgentId,
1021
1498
  overflowContext,
1022
1499
  prepareProviderRequest: prepareFallbackRequest,
1023
1500
  prepareProviderMessages,
@@ -1032,6 +1509,11 @@ export async function tryFallbackProviders({
1032
1509
  /** Accounting-lease owner forwarded to each fallback attempt (see
1033
1510
  * `AttemptInvokeParams.streamLimitState`). */
1034
1511
  streamLimitState?: StreamLimitState;
1512
+ /** Forwarded so a fallback-served attempt records a discarded turn in the
1513
+ * SAME lane the primary would have (see
1514
+ * `AttemptInvokeParams.preemptAgentId`). Dropping it here would leave the
1515
+ * node's boundary undispatched and the lane holding an empty turn. */
1516
+ preemptAgentId?: string;
1035
1517
  /**
1036
1518
  * Prompt-size corroboration for signatures that are not self-describing.
1037
1519
  * Vertex AI's overflow is a bare `400` with no reason, so without this a
@@ -1147,6 +1629,7 @@ export async function tryFallbackProviders({
1147
1629
  context,
1148
1630
  onChunk,
1149
1631
  streamLimitState,
1632
+ preemptAgentId,
1150
1633
  },
1151
1634
  fbConfig
1152
1635
  );
@@ -1159,6 +1642,7 @@ export async function tryFallbackProviders({
1159
1642
  context,
1160
1643
  onChunk,
1161
1644
  streamLimitState,
1645
+ preemptAgentId,
1162
1646
  },
1163
1647
  fbConfig
1164
1648
  );