@librechat/agents 3.7.13 → 3.7.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/dist/cjs/agents/AgentContext.cjs +1 -1
  2. package/dist/cjs/common/constants.cjs +2 -0
  3. package/dist/cjs/common/constants.cjs.map +1 -1
  4. package/dist/cjs/graphs/Graph.cjs +46 -10
  5. package/dist/cjs/graphs/Graph.cjs.map +1 -1
  6. package/dist/cjs/hooks/index.cjs +2 -0
  7. package/dist/cjs/hooks/index.cjs.map +1 -1
  8. package/dist/cjs/langfuse.cjs +16 -0
  9. package/dist/cjs/langfuse.cjs.map +1 -1
  10. package/dist/cjs/llm/google/utils/common.cjs +13 -6
  11. package/dist/cjs/llm/google/utils/common.cjs.map +1 -1
  12. package/dist/cjs/llm/invoke.cjs +191 -81
  13. package/dist/cjs/llm/invoke.cjs.map +1 -1
  14. package/dist/cjs/llm/preempt.cjs +77 -1
  15. package/dist/cjs/llm/preempt.cjs.map +1 -1
  16. package/dist/cjs/llm/prepareProviderRequest.cjs +2 -2
  17. package/dist/cjs/main.cjs +7 -4
  18. package/dist/cjs/messages/core.cjs +6 -0
  19. package/dist/cjs/messages/core.cjs.map +1 -1
  20. package/dist/cjs/messages/index.cjs +1 -1
  21. package/dist/cjs/messages/prune.cjs +2 -2
  22. package/dist/cjs/run.cjs +3 -2
  23. package/dist/cjs/run.cjs.map +1 -1
  24. package/dist/cjs/stream.cjs +4 -6
  25. package/dist/cjs/stream.cjs.map +1 -1
  26. package/dist/cjs/summarization/node.cjs +1 -1
  27. package/dist/cjs/utils/tokens.cjs +1 -1
  28. package/dist/esm/agents/AgentContext.mjs +1 -1
  29. package/dist/esm/common/constants.mjs +2 -1
  30. package/dist/esm/common/constants.mjs.map +1 -1
  31. package/dist/esm/graphs/Graph.mjs +46 -10
  32. package/dist/esm/graphs/Graph.mjs.map +1 -1
  33. package/dist/esm/hooks/index.mjs +2 -1
  34. package/dist/esm/hooks/index.mjs.map +1 -1
  35. package/dist/esm/langfuse.mjs +16 -0
  36. package/dist/esm/langfuse.mjs.map +1 -1
  37. package/dist/esm/llm/google/utils/common.mjs +13 -6
  38. package/dist/esm/llm/google/utils/common.mjs.map +1 -1
  39. package/dist/esm/llm/invoke.mjs +191 -81
  40. package/dist/esm/llm/invoke.mjs.map +1 -1
  41. package/dist/esm/llm/preempt.mjs +72 -2
  42. package/dist/esm/llm/preempt.mjs.map +1 -1
  43. package/dist/esm/llm/prepareProviderRequest.mjs +2 -2
  44. package/dist/esm/main.mjs +7 -7
  45. package/dist/esm/messages/core.mjs +6 -1
  46. package/dist/esm/messages/core.mjs.map +1 -1
  47. package/dist/esm/messages/index.mjs +1 -1
  48. package/dist/esm/messages/prune.mjs +2 -2
  49. package/dist/esm/run.mjs +3 -2
  50. package/dist/esm/run.mjs.map +1 -1
  51. package/dist/esm/stream.mjs +4 -6
  52. package/dist/esm/stream.mjs.map +1 -1
  53. package/dist/esm/summarization/node.mjs +1 -1
  54. package/dist/esm/utils/tokens.mjs +1 -1
  55. package/dist/types/common/constants.d.ts +15 -0
  56. package/dist/types/graphs/Graph.d.ts +50 -0
  57. package/dist/types/hooks/index.d.ts +15 -0
  58. package/dist/types/llm/invoke.d.ts +17 -1
  59. package/dist/types/llm/preempt.d.ts +114 -0
  60. package/dist/types/messages/core.d.ts +19 -0
  61. package/dist/types/run.d.ts +5 -3
  62. package/dist/types/types/run.d.ts +58 -7
  63. package/package.json +1 -1
  64. package/src/common/constants.ts +15 -0
  65. package/src/graphs/Graph.ts +122 -6
  66. package/src/hooks/index.ts +15 -0
  67. package/src/langfuse.ts +53 -0
  68. package/src/llm/google/utils/common.ts +40 -12
  69. package/src/llm/invoke.ts +614 -130
  70. package/src/llm/preempt.ts +320 -1
  71. package/src/messages/core.ts +28 -0
  72. package/src/run.ts +12 -4
  73. package/src/stream.ts +4 -12
  74. package/src/types/run.ts +58 -7
@@ -1416,6 +1416,9 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1416
1416
  * cleanup runs.
1417
1417
  */
1418
1418
  preemptSealCount = 0;
1419
+ /** Discards honored, counted apart from seals — see
1420
+ * {@link claimPreemptRestart}. */
1421
+ preemptRestartCount = 0;
1419
1422
  /** Boundaries that produced nothing to inject, so the turn stopped early. */
1420
1423
  preemptEmptyBoundaries = 0;
1421
1424
  /**
@@ -1463,6 +1466,21 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1463
1466
  * another's turn.
1464
1467
  */
1465
1468
  pendingPreemptReturn = new Set<string>();
1469
+ /**
1470
+ * Agent IDs whose model turn a preempt DISCARDED rather than sealed. A
1471
+ * discard returns no message, so the node cannot read
1472
+ * `response_metadata.preempted` to learn a boundary is owed — this set is
1473
+ * the only carrier.
1474
+ *
1475
+ * Keyed by agent for the same reason {@link pendingPreemptReturn} is: one
1476
+ * graph instance serves every parallel agent, and a single flag would let
1477
+ * whichever lane finished first consume another lane's boundary.
1478
+ *
1479
+ * Not derivable from `preemptSealInFlight`: an ordinary seal holds that slot
1480
+ * too, and a claim that never reached its boundary would be indistinguishable
1481
+ * from a discard.
1482
+ */
1483
+ private preemptRestartPending = new Set<string>();
1466
1484
 
1467
1485
  constructor(
1468
1486
  {
@@ -1712,6 +1730,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1712
1730
  private resetPreemptTurnState(): void {
1713
1731
  this.preemptSealBudgetUsed = 0;
1714
1732
  this.preemptSealInFlight = false;
1733
+ this.preemptRestartPending.clear();
1715
1734
  this.pendingPreemptReturn.clear();
1716
1735
  }
1717
1736
 
@@ -1724,6 +1743,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1724
1743
  */
1725
1744
  private resetPreemptTotals(): void {
1726
1745
  this.preemptSealCount = 0;
1746
+ this.preemptRestartCount = 0;
1727
1747
  this.preemptEmptyBoundaries = 0;
1728
1748
  this.preemptIncomplete = false;
1729
1749
  this.preemptHaltReason = undefined;
@@ -1796,12 +1816,32 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1796
1816
  * rather than sealing for a message it would never receive.
1797
1817
  */
1798
1818
  claimPreemptSeal(): boolean {
1819
+ return this.claimPreemptSlot('seal');
1820
+ }
1821
+
1822
+ /**
1823
+ * The restart twin. It takes the SAME slot and spends the SAME budget —
1824
+ * both moves end one model turn and cost one extra superstep, and the
1825
+ * mutual exclusion argument above applies identically — but it is counted
1826
+ * apart: a restart preserved no partial assistant message, so reporting it
1827
+ * as a seal would tell every consumer of `getPreemptStats().seals` and the
1828
+ * boundary hook's `sealCount` that a turn was kept when it was discarded.
1829
+ */
1830
+ claimPreemptRestart(): boolean {
1831
+ return this.claimPreemptSlot('restart');
1832
+ }
1833
+
1834
+ private claimPreemptSlot(kind: 'seal' | 'restart'): boolean {
1799
1835
  if (!this.canClaimPreemptSeal()) {
1800
1836
  return false;
1801
1837
  }
1802
1838
  this.preemptSealInFlight = true;
1803
1839
  this.preemptSealBudgetUsed += 1;
1804
- this.preemptSealCount += 1;
1840
+ if (kind === 'seal') {
1841
+ this.preemptSealCount += 1;
1842
+ return true;
1843
+ }
1844
+ this.preemptRestartCount += 1;
1805
1845
  return true;
1806
1846
  }
1807
1847
 
@@ -1810,9 +1850,24 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1810
1850
  this.preemptSealInFlight = false;
1811
1851
  }
1812
1852
 
1853
+ /** See {@link preemptRestartPending}. Called from the model attempt. */
1854
+ notePreemptRestart(agentId: string): void {
1855
+ this.preemptRestartPending.add(agentId);
1856
+ }
1857
+
1858
+ /**
1859
+ * Reads and clears the lane's mark in one step, so a boundary is dispatched
1860
+ * exactly once per discard even though the model node runs again immediately
1861
+ * after.
1862
+ */
1863
+ private consumePreemptRestart(agentId: string): boolean {
1864
+ return this.preemptRestartPending.delete(agentId);
1865
+ }
1866
+
1813
1867
  getPreemptStats(): t.PreemptStats {
1814
1868
  return {
1815
1869
  seals: this.preemptSealCount,
1870
+ restarts: this.preemptRestartCount,
1816
1871
  emptyBoundaries: this.preemptEmptyBoundaries,
1817
1872
  };
1818
1873
  }
@@ -2047,6 +2102,35 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
2047
2102
  * unreachable — each step registers its calls before any completion can
2048
2103
  * reference it — and the mechanism was removed.)
2049
2104
  */
2105
+ /**
2106
+ * Closes the lane's open message step as `cancelled` when a preempt discards
2107
+ * the turn that step belongs to.
2108
+ *
2109
+ * Without it the step is closed `completed` by the discard's own model-end,
2110
+ * so `getRunSteps()` and every run-step subscriber keep reasoning the
2111
+ * replacement prompt does not contain — the graph state forgets the attempt
2112
+ * while the step view still shows it as a finished one.
2113
+ *
2114
+ * Only for turns that were CUT SHORT. A turn whose stream reached its own
2115
+ * end really did complete, and its step is already closed by then; the
2116
+ * terminal-status guard in {@link closeRunStep} leaves that alone.
2117
+ */
2118
+ async cancelOpenMessageStep(
2119
+ metadata?: Record<string, unknown>
2120
+ ): Promise<void> {
2121
+ const stepId = this.openMessageStepByAgent.get(
2122
+ this.getStepAgentKey(metadata)
2123
+ );
2124
+ if (stepId == null) {
2125
+ return;
2126
+ }
2127
+ try {
2128
+ await this.closeRunStep(stepId, 'cancelled', { metadata });
2129
+ } catch (_e) {
2130
+ /** A step-view detail must never take down the run it describes. */
2131
+ }
2132
+ }
2133
+
2050
2134
  async closeRunStep(
2051
2135
  stepId: string,
2052
2136
  status: Exclude<t.RunStepStatus, 'in_progress'>,
@@ -3907,6 +3991,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3907
3991
  {
3908
3992
  request: preparedRequest,
3909
3993
  context: this,
3994
+ preemptAgentId: agentId,
3910
3995
  },
3911
3996
  invokeConfig
3912
3997
  )
@@ -4105,6 +4190,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
4105
4190
  config: invokeConfig,
4106
4191
  primaryError,
4107
4192
  context: this,
4193
+ preemptAgentId: agentId,
4108
4194
  /**
4109
4195
  * Lets the chain recognise a fallback overflow whose signature
4110
4196
  * carries no reason of its own (Vertex AI's bare 400) and
@@ -4392,7 +4478,9 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
4392
4478
  { force: true }
4393
4479
  );
4394
4480
  }
4481
+ const preemptRestarted = this.consumePreemptRestart(agentId);
4395
4482
  if (
4483
+ preemptRestarted ||
4396
4484
  (responseMessage as AIMessageChunk | undefined)?.response_metadata
4397
4485
  .preempted === true
4398
4486
  ) {
@@ -4420,7 +4508,13 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
4420
4508
  * hosts use the counter for truncated-seal telemetry, and both
4421
4509
  * paths end the turn with nothing to resume from.
4422
4510
  */
4423
- if (injected.length === 0) {
4511
+ /**
4512
+ * Seals only. `emptyBoundaries` means a kept assistant turn that was
4513
+ * truncated with nothing to resume from; a restart preserved no turn
4514
+ * at all, so a halted one is not that — `preemptIncomplete` above
4515
+ * already records that the answer never arrived.
4516
+ */
4517
+ if (injected.length === 0 && !preemptRestarted) {
4424
4518
  this.preemptEmptyBoundaries += 1;
4425
4519
  }
4426
4520
  this.cleanupSignalListener();
@@ -4434,11 +4528,33 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
4434
4528
  return { messages: [...(result.messages ?? []), ...injected] };
4435
4529
  }
4436
4530
  /**
4437
- * Nothing to inject — the host cancelled or already drained. Do NOT
4438
- * self-loop: a trailing model turn with no new input is dropped by
4439
- * some Gemini models and read as prefill by Anthropic. Do NOT pretend
4440
- * the turn completed either; the answer really was cut short.
4531
+ * Nothing to inject — the host cancelled or already drained.
4532
+ *
4533
+ * After a SEAL, do not self-loop: a trailing model turn with no new
4534
+ * input is dropped by some Gemini models and read as prefill by
4535
+ * Anthropic. Do not pretend the turn completed either; the answer
4536
+ * really was cut short.
4537
+ *
4538
+ * After a RESTART there is no such trailing turn. The discard left
4539
+ * graph state exactly as the model node found it, so returning to the
4540
+ * node re-issues the same call — the only way the run can still
4541
+ * produce an answer, and the honest outcome of an interrupt whose
4542
+ * words were withdrawn before they landed. Bounded by the same seal
4543
+ * budget, so a host stuck arming and cancelling cannot loop forever.
4441
4544
  */
4545
+ if (preemptRestarted) {
4546
+ /**
4547
+ * NOT counted as an empty boundary. That counter means a seal that
4548
+ * ended the turn early with nothing to resume from — hosts read it
4549
+ * as truncated-answer telemetry — and this branch is the opposite:
4550
+ * the call is reissued and the run goes on to produce a complete
4551
+ * answer. Counting it here would make a successful restart look like
4552
+ * a truncated one in every host that persists on that signal.
4553
+ */
4554
+ this.pendingPreemptReturn.add(agentId);
4555
+ this.cleanupSignalListener();
4556
+ return result;
4557
+ }
4442
4558
  this.preemptEmptyBoundaries += 1;
4443
4559
  this.preemptIncomplete = true;
4444
4560
  }
@@ -37,6 +37,21 @@ export const HOOK_PREEMPT_BOUNDARY_CAPABLE = true;
37
37
  * and continue within the same `Run.processStream` lifecycle.
38
38
  */
39
39
  export const HOOK_STOP_CONTINUATION_CAPABLE = true;
40
+ /**
41
+ * Feature probe for hosts: a preempt request can also be honored BEFORE the
42
+ * turn has produced anything to keep — the in-flight model call is discarded
43
+ * and re-issued with the boundary's injection appended, and
44
+ * `StreamPreemption.subscribe` wakes the SDK during the silent window where
45
+ * the per-chunk poll cannot reach.
46
+ *
47
+ * Separate from {@link HOOK_PREEMPT_BOUNDARY_CAPABLE} because the two answer
48
+ * different questions for the user-facing control. An SDK with only the
49
+ * boundary can seal a turn that is already writing an answer, but an interrupt
50
+ * armed while the model is still thinking waits for the whole turn — so a host
51
+ * that probed the wrong flag would promise an interrupt it cannot deliver in
52
+ * exactly the window users reach for it most.
53
+ */
54
+ export const HOOK_PREEMPT_RESTART_CAPABLE = true;
40
55
  export {
41
56
  matchesQuery,
42
57
  hasNestedQuantifier,
package/src/langfuse.ts CHANGED
@@ -39,6 +39,10 @@ import {
39
39
  registerLangfuseManagedSpan,
40
40
  resolveLangfuseDestinationKey,
41
41
  } from '@/langfuseSpanRegistry';
42
+ import {
43
+ readPreemptRestartedRun,
44
+ PREEMPT_RESTART_CONTROL_FLOW,
45
+ } from '@/llm/preempt';
42
46
  import { isPresent, parseBooleanEnv } from '@/utils/misc';
43
47
 
44
48
  export {
@@ -672,6 +676,55 @@ class ScopedLangfuseCallbackHandler extends CallbackHandler {
672
676
  );
673
677
  }
674
678
 
679
+ /**
680
+ * A cooperative restart tears the provider stream down mid-flight, so the
681
+ * adapter reports cancellation and LangChain closes the generation through
682
+ * this path. That is control flow, not a failure — the graph catches it,
683
+ * injects, and calls the model again — so it closes as a successful
684
+ * generation rather than polluting an otherwise healthy trace with a failed
685
+ * one.
686
+ *
687
+ * Keyed on the run id the tear-down recorded, not on the error's shape:
688
+ * every provider adapter raises its own cancellation error, and matching on
689
+ * those would make the invariant depend on which provider served the call.
690
+ */
691
+ override handleLLMError(
692
+ ...args: Parameters<CallbackHandler['handleLLMError']>
693
+ ): ReturnType<CallbackHandler['handleLLMError']> {
694
+ const [, runId, parentRunId] = args;
695
+ const restarted = readPreemptRestartedRun(runId);
696
+ if (restarted != null) {
697
+ /**
698
+ * Closed WITH the discarded turn, not as an empty end. The provider
699
+ * consumed the whole prompt and may have billed reasoning tokens before
700
+ * the tear-down, and this is the only close the generation will get —
701
+ * the synthetic `CHAT_MODEL_END` that follows is a host stream event and
702
+ * cannot reopen it. Routed through the override so Bedrock usage is
703
+ * normalized exactly as it is for an ordinary end.
704
+ *
705
+ * The generation text stays empty on purpose: a discarded turn produced
706
+ * no answer, and replaying its reasoning as output would read as one.
707
+ *
708
+ * The lookup does not consume the record: a run can be closed through
709
+ * several handlers at once, and each must reach this branch rather than
710
+ * the first one leaving the others to export an error.
711
+ */
712
+ const generation: ChatGeneration = {
713
+ text: '',
714
+ message: restarted.message,
715
+ };
716
+ return this.handleLLMEnd(
717
+ {
718
+ generations: [[generation]],
719
+ llmOutput: PREEMPT_RESTART_CONTROL_FLOW,
720
+ },
721
+ runId,
722
+ parentRunId
723
+ );
724
+ }
725
+ return super.handleLLMError(...args);
726
+ }
727
+
675
728
  override handleToolStart(
676
729
  ...args: Parameters<CallbackHandler['handleToolStart']>
677
730
  ): ReturnType<CallbackHandler['handleToolStart']> {
@@ -659,28 +659,56 @@ export function convertBaseMessagesToContent(
659
659
  }
660
660
 
661
661
  /**
662
- * Gemini models that reject a request whose `contents` end with a `model`-role
663
- * turn (a "prefill"). Google enforces this on newer generations (Gemini 3.7
664
- * Flash, Gemini 3.6 Flash, Gemini 3.5 Flash-Lite) while older/sibling models
665
- * still accept a trailing model turn, so the rule is model-scoped rather than
666
- * version-wide. Extend this list as Google applies the restriction to further
667
- * models.
662
+ * Gemini Flash generation from which Google rejects a request whose `contents`
663
+ * end with a `model`-role turn (a "prefill"). Every Flash release from 3.6
664
+ * onward enforces it, so the cutoff is derived from the model id rather than
665
+ * enumerated - a new Flash model is covered on release with no change here.
666
+ *
667
+ * Scoped to Flash deliberately: dropping the turn silently degrades a working
668
+ * prefill into a fresh generation, so the rule only widens where Google
669
+ * documents the restriction. Lines that reject it without matching the cutoff
670
+ * are listed in {@link NO_PREFILL_GEMINI_MODELS}.
668
671
  * @see https://ai.google.dev/gemini-api/docs/latest-model#api-changes-and-parameter-updates
669
672
  */
670
- const NO_PREFILL_GEMINI_MODELS = [
671
- 'gemini-3.7-flash',
672
- 'gemini-3.6-flash',
673
- 'gemini-3.5-flash-lite',
674
- ] as const;
673
+ const NO_PREFILL_FLASH_MIN_VERSION = { major: 3, minor: 6 } as const;
674
+
675
+ /**
676
+ * `gemini-<major>[.<minor>]-flash`, with optional suffixes (`-latest`, `-lite`).
677
+ * Google ships both forms — `gemini-3.7-flash` and the major-only
678
+ * `gemini-3-flash-preview` — so the minor component is optional and an omitted
679
+ * one reads as `.0`.
680
+ */
681
+ const GEMINI_FLASH_VERSION_PATTERN = /^gemini-(\d+)(?:\.(\d+))?-flash(?:$|-)/;
682
+
683
+ /**
684
+ * Models that reject prefill despite predating
685
+ * {@link NO_PREFILL_FLASH_MIN_VERSION}. Gemini 3.5 Flash-Lite enforces the
686
+ * restriction while its sibling Gemini 3.5 Flash still accepts a trailing model
687
+ * turn, so the 3.5 generation cannot be expressed as a version cutoff.
688
+ */
689
+ const NO_PREFILL_GEMINI_MODELS = ['gemini-3.5-flash-lite'] as const;
675
690
 
676
691
  export function rejectsModelTurnPrefill(model?: string): boolean {
677
692
  if (model == null || model === '') {
678
693
  return false;
679
694
  }
680
695
  const modelId = model.toLowerCase().split('/').pop() ?? '';
681
- return NO_PREFILL_GEMINI_MODELS.some(
696
+ const listed = NO_PREFILL_GEMINI_MODELS.some(
682
697
  (id) => modelId === id || modelId.startsWith(`${id}-`)
683
698
  );
699
+ if (listed) {
700
+ return true;
701
+ }
702
+ const match = GEMINI_FLASH_VERSION_PATTERN.exec(modelId);
703
+ if (!match) {
704
+ return false;
705
+ }
706
+ const major = Number(match[1]);
707
+ const minor = Number(match[2] || '0');
708
+ if (major !== NO_PREFILL_FLASH_MIN_VERSION.major) {
709
+ return major > NO_PREFILL_FLASH_MIN_VERSION.major;
710
+ }
711
+ return minor >= NO_PREFILL_FLASH_MIN_VERSION.minor;
684
712
  }
685
713
 
686
714
  /**