@librechat/agents 3.7.13 → 3.7.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cjs/agents/AgentContext.cjs +1 -1
- package/dist/cjs/common/constants.cjs +2 -0
- package/dist/cjs/common/constants.cjs.map +1 -1
- package/dist/cjs/graphs/Graph.cjs +46 -10
- package/dist/cjs/graphs/Graph.cjs.map +1 -1
- package/dist/cjs/hooks/index.cjs +2 -0
- package/dist/cjs/hooks/index.cjs.map +1 -1
- package/dist/cjs/langfuse.cjs +16 -0
- package/dist/cjs/langfuse.cjs.map +1 -1
- package/dist/cjs/llm/google/utils/common.cjs +13 -6
- package/dist/cjs/llm/google/utils/common.cjs.map +1 -1
- package/dist/cjs/llm/invoke.cjs +191 -81
- package/dist/cjs/llm/invoke.cjs.map +1 -1
- package/dist/cjs/llm/preempt.cjs +77 -1
- package/dist/cjs/llm/preempt.cjs.map +1 -1
- package/dist/cjs/llm/prepareProviderRequest.cjs +2 -2
- package/dist/cjs/main.cjs +7 -4
- package/dist/cjs/messages/core.cjs +6 -0
- package/dist/cjs/messages/core.cjs.map +1 -1
- package/dist/cjs/messages/index.cjs +1 -1
- package/dist/cjs/messages/prune.cjs +2 -2
- package/dist/cjs/run.cjs +3 -2
- package/dist/cjs/run.cjs.map +1 -1
- package/dist/cjs/stream.cjs +4 -6
- package/dist/cjs/stream.cjs.map +1 -1
- package/dist/cjs/summarization/node.cjs +1 -1
- package/dist/cjs/utils/tokens.cjs +1 -1
- package/dist/esm/agents/AgentContext.mjs +1 -1
- package/dist/esm/common/constants.mjs +2 -1
- package/dist/esm/common/constants.mjs.map +1 -1
- package/dist/esm/graphs/Graph.mjs +46 -10
- package/dist/esm/graphs/Graph.mjs.map +1 -1
- package/dist/esm/hooks/index.mjs +2 -1
- package/dist/esm/hooks/index.mjs.map +1 -1
- package/dist/esm/langfuse.mjs +16 -0
- package/dist/esm/langfuse.mjs.map +1 -1
- package/dist/esm/llm/google/utils/common.mjs +13 -6
- package/dist/esm/llm/google/utils/common.mjs.map +1 -1
- package/dist/esm/llm/invoke.mjs +191 -81
- package/dist/esm/llm/invoke.mjs.map +1 -1
- package/dist/esm/llm/preempt.mjs +72 -2
- package/dist/esm/llm/preempt.mjs.map +1 -1
- package/dist/esm/llm/prepareProviderRequest.mjs +2 -2
- package/dist/esm/main.mjs +7 -7
- package/dist/esm/messages/core.mjs +6 -1
- package/dist/esm/messages/core.mjs.map +1 -1
- package/dist/esm/messages/index.mjs +1 -1
- package/dist/esm/messages/prune.mjs +2 -2
- package/dist/esm/run.mjs +3 -2
- package/dist/esm/run.mjs.map +1 -1
- package/dist/esm/stream.mjs +4 -6
- package/dist/esm/stream.mjs.map +1 -1
- package/dist/esm/summarization/node.mjs +1 -1
- package/dist/esm/utils/tokens.mjs +1 -1
- package/dist/types/common/constants.d.ts +15 -0
- package/dist/types/graphs/Graph.d.ts +50 -0
- package/dist/types/hooks/index.d.ts +15 -0
- package/dist/types/llm/invoke.d.ts +17 -1
- package/dist/types/llm/preempt.d.ts +114 -0
- package/dist/types/messages/core.d.ts +19 -0
- package/dist/types/run.d.ts +5 -3
- package/dist/types/types/run.d.ts +58 -7
- package/package.json +1 -1
- package/src/common/constants.ts +15 -0
- package/src/graphs/Graph.ts +122 -6
- package/src/hooks/index.ts +15 -0
- package/src/langfuse.ts +53 -0
- package/src/llm/google/utils/common.ts +40 -12
- package/src/llm/invoke.ts +614 -130
- package/src/llm/preempt.ts +320 -1
- package/src/messages/core.ts +28 -0
- package/src/run.ts +12 -4
- package/src/stream.ts +4 -12
- package/src/types/run.ts +58 -7
package/src/graphs/Graph.ts
CHANGED
|
@@ -1416,6 +1416,9 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
1416
1416
|
* cleanup runs.
|
|
1417
1417
|
*/
|
|
1418
1418
|
preemptSealCount = 0;
|
|
1419
|
+
/** Discards honored, counted apart from seals — see
|
|
1420
|
+
* {@link claimPreemptRestart}. */
|
|
1421
|
+
preemptRestartCount = 0;
|
|
1419
1422
|
/** Boundaries that produced nothing to inject, so the turn stopped early. */
|
|
1420
1423
|
preemptEmptyBoundaries = 0;
|
|
1421
1424
|
/**
|
|
@@ -1463,6 +1466,21 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
1463
1466
|
* another's turn.
|
|
1464
1467
|
*/
|
|
1465
1468
|
pendingPreemptReturn = new Set<string>();
|
|
1469
|
+
/**
|
|
1470
|
+
* Agent IDs whose model turn a preempt DISCARDED rather than sealed. A
|
|
1471
|
+
* discard returns no message, so the node cannot read
|
|
1472
|
+
* `response_metadata.preempted` to learn a boundary is owed — this set is
|
|
1473
|
+
* the only carrier.
|
|
1474
|
+
*
|
|
1475
|
+
* Keyed by agent for the same reason {@link pendingPreemptReturn} is: one
|
|
1476
|
+
* graph instance serves every parallel agent, and a single flag would let
|
|
1477
|
+
* whichever lane finished first consume another lane's boundary.
|
|
1478
|
+
*
|
|
1479
|
+
* Not derivable from `preemptSealInFlight`: an ordinary seal holds that slot
|
|
1480
|
+
* too, and a claim that never reached its boundary would be indistinguishable
|
|
1481
|
+
* from a discard.
|
|
1482
|
+
*/
|
|
1483
|
+
private preemptRestartPending = new Set<string>();
|
|
1466
1484
|
|
|
1467
1485
|
constructor(
|
|
1468
1486
|
{
|
|
@@ -1712,6 +1730,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
1712
1730
|
private resetPreemptTurnState(): void {
|
|
1713
1731
|
this.preemptSealBudgetUsed = 0;
|
|
1714
1732
|
this.preemptSealInFlight = false;
|
|
1733
|
+
this.preemptRestartPending.clear();
|
|
1715
1734
|
this.pendingPreemptReturn.clear();
|
|
1716
1735
|
}
|
|
1717
1736
|
|
|
@@ -1724,6 +1743,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
1724
1743
|
*/
|
|
1725
1744
|
private resetPreemptTotals(): void {
|
|
1726
1745
|
this.preemptSealCount = 0;
|
|
1746
|
+
this.preemptRestartCount = 0;
|
|
1727
1747
|
this.preemptEmptyBoundaries = 0;
|
|
1728
1748
|
this.preemptIncomplete = false;
|
|
1729
1749
|
this.preemptHaltReason = undefined;
|
|
@@ -1796,12 +1816,32 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
1796
1816
|
* rather than sealing for a message it would never receive.
|
|
1797
1817
|
*/
|
|
1798
1818
|
claimPreemptSeal(): boolean {
|
|
1819
|
+
return this.claimPreemptSlot('seal');
|
|
1820
|
+
}
|
|
1821
|
+
|
|
1822
|
+
/**
|
|
1823
|
+
* The restart twin. It takes the SAME slot and spends the SAME budget —
|
|
1824
|
+
* both moves end one model turn and cost one extra superstep, and the
|
|
1825
|
+
* mutual exclusion argument above applies identically — but it is counted
|
|
1826
|
+
* apart: a restart preserved no partial assistant message, so reporting it
|
|
1827
|
+
* as a seal would tell every consumer of `getPreemptStats().seals` and the
|
|
1828
|
+
* boundary hook's `sealCount` that a turn was kept when it was discarded.
|
|
1829
|
+
*/
|
|
1830
|
+
claimPreemptRestart(): boolean {
|
|
1831
|
+
return this.claimPreemptSlot('restart');
|
|
1832
|
+
}
|
|
1833
|
+
|
|
1834
|
+
private claimPreemptSlot(kind: 'seal' | 'restart'): boolean {
|
|
1799
1835
|
if (!this.canClaimPreemptSeal()) {
|
|
1800
1836
|
return false;
|
|
1801
1837
|
}
|
|
1802
1838
|
this.preemptSealInFlight = true;
|
|
1803
1839
|
this.preemptSealBudgetUsed += 1;
|
|
1804
|
-
|
|
1840
|
+
if (kind === 'seal') {
|
|
1841
|
+
this.preemptSealCount += 1;
|
|
1842
|
+
return true;
|
|
1843
|
+
}
|
|
1844
|
+
this.preemptRestartCount += 1;
|
|
1805
1845
|
return true;
|
|
1806
1846
|
}
|
|
1807
1847
|
|
|
@@ -1810,9 +1850,24 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
1810
1850
|
this.preemptSealInFlight = false;
|
|
1811
1851
|
}
|
|
1812
1852
|
|
|
1853
|
+
/** See {@link preemptRestartPending}. Called from the model attempt. */
|
|
1854
|
+
notePreemptRestart(agentId: string): void {
|
|
1855
|
+
this.preemptRestartPending.add(agentId);
|
|
1856
|
+
}
|
|
1857
|
+
|
|
1858
|
+
/**
|
|
1859
|
+
* Reads and clears the lane's mark in one step, so a boundary is dispatched
|
|
1860
|
+
* exactly once per discard even though the model node runs again immediately
|
|
1861
|
+
* after.
|
|
1862
|
+
*/
|
|
1863
|
+
private consumePreemptRestart(agentId: string): boolean {
|
|
1864
|
+
return this.preemptRestartPending.delete(agentId);
|
|
1865
|
+
}
|
|
1866
|
+
|
|
1813
1867
|
getPreemptStats(): t.PreemptStats {
|
|
1814
1868
|
return {
|
|
1815
1869
|
seals: this.preemptSealCount,
|
|
1870
|
+
restarts: this.preemptRestartCount,
|
|
1816
1871
|
emptyBoundaries: this.preemptEmptyBoundaries,
|
|
1817
1872
|
};
|
|
1818
1873
|
}
|
|
@@ -2047,6 +2102,35 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
2047
2102
|
* unreachable — each step registers its calls before any completion can
|
|
2048
2103
|
* reference it — and the mechanism was removed.)
|
|
2049
2104
|
*/
|
|
2105
|
+
/**
|
|
2106
|
+
* Closes the lane's open message step as `cancelled` when a preempt discards
|
|
2107
|
+
* the turn that step belongs to.
|
|
2108
|
+
*
|
|
2109
|
+
* Without it the step is closed `completed` by the discard's own model-end,
|
|
2110
|
+
* so `getRunSteps()` and every run-step subscriber keep reasoning the
|
|
2111
|
+
* replacement prompt does not contain — the graph state forgets the attempt
|
|
2112
|
+
* while the step view still shows it as a finished one.
|
|
2113
|
+
*
|
|
2114
|
+
* Only for turns that were CUT SHORT. A turn whose stream reached its own
|
|
2115
|
+
* end really did complete, and its step is already closed by then; the
|
|
2116
|
+
* terminal-status guard in {@link closeRunStep} leaves that alone.
|
|
2117
|
+
*/
|
|
2118
|
+
async cancelOpenMessageStep(
|
|
2119
|
+
metadata?: Record<string, unknown>
|
|
2120
|
+
): Promise<void> {
|
|
2121
|
+
const stepId = this.openMessageStepByAgent.get(
|
|
2122
|
+
this.getStepAgentKey(metadata)
|
|
2123
|
+
);
|
|
2124
|
+
if (stepId == null) {
|
|
2125
|
+
return;
|
|
2126
|
+
}
|
|
2127
|
+
try {
|
|
2128
|
+
await this.closeRunStep(stepId, 'cancelled', { metadata });
|
|
2129
|
+
} catch (_e) {
|
|
2130
|
+
/** A step-view detail must never take down the run it describes. */
|
|
2131
|
+
}
|
|
2132
|
+
}
|
|
2133
|
+
|
|
2050
2134
|
async closeRunStep(
|
|
2051
2135
|
stepId: string,
|
|
2052
2136
|
status: Exclude<t.RunStepStatus, 'in_progress'>,
|
|
@@ -3907,6 +3991,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
3907
3991
|
{
|
|
3908
3992
|
request: preparedRequest,
|
|
3909
3993
|
context: this,
|
|
3994
|
+
preemptAgentId: agentId,
|
|
3910
3995
|
},
|
|
3911
3996
|
invokeConfig
|
|
3912
3997
|
)
|
|
@@ -4105,6 +4190,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
4105
4190
|
config: invokeConfig,
|
|
4106
4191
|
primaryError,
|
|
4107
4192
|
context: this,
|
|
4193
|
+
preemptAgentId: agentId,
|
|
4108
4194
|
/**
|
|
4109
4195
|
* Lets the chain recognise a fallback overflow whose signature
|
|
4110
4196
|
* carries no reason of its own (Vertex AI's bare 400) and
|
|
@@ -4392,7 +4478,9 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
4392
4478
|
{ force: true }
|
|
4393
4479
|
);
|
|
4394
4480
|
}
|
|
4481
|
+
const preemptRestarted = this.consumePreemptRestart(agentId);
|
|
4395
4482
|
if (
|
|
4483
|
+
preemptRestarted ||
|
|
4396
4484
|
(responseMessage as AIMessageChunk | undefined)?.response_metadata
|
|
4397
4485
|
.preempted === true
|
|
4398
4486
|
) {
|
|
@@ -4420,7 +4508,13 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
4420
4508
|
* hosts use the counter for truncated-seal telemetry, and both
|
|
4421
4509
|
* paths end the turn with nothing to resume from.
|
|
4422
4510
|
*/
|
|
4423
|
-
|
|
4511
|
+
/**
|
|
4512
|
+
* Seals only. `emptyBoundaries` means a kept assistant turn that was
|
|
4513
|
+
* truncated with nothing to resume from; a restart preserved no turn
|
|
4514
|
+
* at all, so a halted one is not that — `preemptIncomplete` above
|
|
4515
|
+
* already records that the answer never arrived.
|
|
4516
|
+
*/
|
|
4517
|
+
if (injected.length === 0 && !preemptRestarted) {
|
|
4424
4518
|
this.preemptEmptyBoundaries += 1;
|
|
4425
4519
|
}
|
|
4426
4520
|
this.cleanupSignalListener();
|
|
@@ -4434,11 +4528,33 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
|
|
|
4434
4528
|
return { messages: [...(result.messages ?? []), ...injected] };
|
|
4435
4529
|
}
|
|
4436
4530
|
/**
|
|
4437
|
-
* Nothing to inject — the host cancelled or already drained.
|
|
4438
|
-
*
|
|
4439
|
-
*
|
|
4440
|
-
*
|
|
4531
|
+
* Nothing to inject — the host cancelled or already drained.
|
|
4532
|
+
*
|
|
4533
|
+
* After a SEAL, do not self-loop: a trailing model turn with no new
|
|
4534
|
+
* input is dropped by some Gemini models and read as prefill by
|
|
4535
|
+
* Anthropic. Do not pretend the turn completed either; the answer
|
|
4536
|
+
* really was cut short.
|
|
4537
|
+
*
|
|
4538
|
+
* After a RESTART there is no such trailing turn. The discard left
|
|
4539
|
+
* graph state exactly as the model node found it, so returning to the
|
|
4540
|
+
* node re-issues the same call — the only way the run can still
|
|
4541
|
+
* produce an answer, and the honest outcome of an interrupt whose
|
|
4542
|
+
* words were withdrawn before they landed. Bounded by the same seal
|
|
4543
|
+
* budget, so a host stuck arming and cancelling cannot loop forever.
|
|
4441
4544
|
*/
|
|
4545
|
+
if (preemptRestarted) {
|
|
4546
|
+
/**
|
|
4547
|
+
* NOT counted as an empty boundary. That counter means a seal that
|
|
4548
|
+
* ended the turn early with nothing to resume from — hosts read it
|
|
4549
|
+
* as truncated-answer telemetry — and this branch is the opposite:
|
|
4550
|
+
* the call is reissued and the run goes on to produce a complete
|
|
4551
|
+
* answer. Counting it here would make a successful restart look like
|
|
4552
|
+
* a truncated one in every host that persists on that signal.
|
|
4553
|
+
*/
|
|
4554
|
+
this.pendingPreemptReturn.add(agentId);
|
|
4555
|
+
this.cleanupSignalListener();
|
|
4556
|
+
return result;
|
|
4557
|
+
}
|
|
4442
4558
|
this.preemptEmptyBoundaries += 1;
|
|
4443
4559
|
this.preemptIncomplete = true;
|
|
4444
4560
|
}
|
package/src/hooks/index.ts
CHANGED
|
@@ -37,6 +37,21 @@ export const HOOK_PREEMPT_BOUNDARY_CAPABLE = true;
|
|
|
37
37
|
* and continue within the same `Run.processStream` lifecycle.
|
|
38
38
|
*/
|
|
39
39
|
export const HOOK_STOP_CONTINUATION_CAPABLE = true;
|
|
40
|
+
/**
|
|
41
|
+
* Feature probe for hosts: a preempt request can also be honored BEFORE the
|
|
42
|
+
* turn has produced anything to keep — the in-flight model call is discarded
|
|
43
|
+
* and re-issued with the boundary's injection appended, and
|
|
44
|
+
* `StreamPreemption.subscribe` wakes the SDK during the silent window where
|
|
45
|
+
* the per-chunk poll cannot reach.
|
|
46
|
+
*
|
|
47
|
+
* Separate from {@link HOOK_PREEMPT_BOUNDARY_CAPABLE} because the two answer
|
|
48
|
+
* different questions for the user-facing control. An SDK with only the
|
|
49
|
+
* boundary can seal a turn that is already writing an answer, but an interrupt
|
|
50
|
+
* armed while the model is still thinking waits for the whole turn — so a host
|
|
51
|
+
* that probed the wrong flag would promise an interrupt it cannot deliver in
|
|
52
|
+
* exactly the window users reach for it most.
|
|
53
|
+
*/
|
|
54
|
+
export const HOOK_PREEMPT_RESTART_CAPABLE = true;
|
|
40
55
|
export {
|
|
41
56
|
matchesQuery,
|
|
42
57
|
hasNestedQuantifier,
|
package/src/langfuse.ts
CHANGED
|
@@ -39,6 +39,10 @@ import {
|
|
|
39
39
|
registerLangfuseManagedSpan,
|
|
40
40
|
resolveLangfuseDestinationKey,
|
|
41
41
|
} from '@/langfuseSpanRegistry';
|
|
42
|
+
import {
|
|
43
|
+
readPreemptRestartedRun,
|
|
44
|
+
PREEMPT_RESTART_CONTROL_FLOW,
|
|
45
|
+
} from '@/llm/preempt';
|
|
42
46
|
import { isPresent, parseBooleanEnv } from '@/utils/misc';
|
|
43
47
|
|
|
44
48
|
export {
|
|
@@ -672,6 +676,55 @@ class ScopedLangfuseCallbackHandler extends CallbackHandler {
|
|
|
672
676
|
);
|
|
673
677
|
}
|
|
674
678
|
|
|
679
|
+
/**
|
|
680
|
+
* A cooperative restart tears the provider stream down mid-flight, so the
|
|
681
|
+
* adapter reports cancellation and LangChain closes the generation through
|
|
682
|
+
* this path. That is control flow, not a failure — the graph catches it,
|
|
683
|
+
* injects, and calls the model again — so it closes as a successful
|
|
684
|
+
* generation rather than polluting an otherwise healthy trace with a failed
|
|
685
|
+
* one.
|
|
686
|
+
*
|
|
687
|
+
* Keyed on the run id the tear-down recorded, not on the error's shape:
|
|
688
|
+
* every provider adapter raises its own cancellation error, and matching on
|
|
689
|
+
* those would make the invariant depend on which provider served the call.
|
|
690
|
+
*/
|
|
691
|
+
override handleLLMError(
|
|
692
|
+
...args: Parameters<CallbackHandler['handleLLMError']>
|
|
693
|
+
): ReturnType<CallbackHandler['handleLLMError']> {
|
|
694
|
+
const [, runId, parentRunId] = args;
|
|
695
|
+
const restarted = readPreemptRestartedRun(runId);
|
|
696
|
+
if (restarted != null) {
|
|
697
|
+
/**
|
|
698
|
+
* Closed WITH the discarded turn, not as an empty end. The provider
|
|
699
|
+
* consumed the whole prompt and may have billed reasoning tokens before
|
|
700
|
+
* the tear-down, and this is the only close the generation will get —
|
|
701
|
+
* the synthetic `CHAT_MODEL_END` that follows is a host stream event and
|
|
702
|
+
* cannot reopen it. Routed through the override so Bedrock usage is
|
|
703
|
+
* normalized exactly as it is for an ordinary end.
|
|
704
|
+
*
|
|
705
|
+
* The generation text stays empty on purpose: a discarded turn produced
|
|
706
|
+
* no answer, and replaying its reasoning as output would read as one.
|
|
707
|
+
*
|
|
708
|
+
* The lookup does not consume the record: a run can be closed through
|
|
709
|
+
* several handlers at once, and each must reach this branch rather than
|
|
710
|
+
* the first one leaving the others to export an error.
|
|
711
|
+
*/
|
|
712
|
+
const generation: ChatGeneration = {
|
|
713
|
+
text: '',
|
|
714
|
+
message: restarted.message,
|
|
715
|
+
};
|
|
716
|
+
return this.handleLLMEnd(
|
|
717
|
+
{
|
|
718
|
+
generations: [[generation]],
|
|
719
|
+
llmOutput: PREEMPT_RESTART_CONTROL_FLOW,
|
|
720
|
+
},
|
|
721
|
+
runId,
|
|
722
|
+
parentRunId
|
|
723
|
+
);
|
|
724
|
+
}
|
|
725
|
+
return super.handleLLMError(...args);
|
|
726
|
+
}
|
|
727
|
+
|
|
675
728
|
override handleToolStart(
|
|
676
729
|
...args: Parameters<CallbackHandler['handleToolStart']>
|
|
677
730
|
): ReturnType<CallbackHandler['handleToolStart']> {
|
|
@@ -659,28 +659,56 @@ export function convertBaseMessagesToContent(
|
|
|
659
659
|
}
|
|
660
660
|
|
|
661
661
|
/**
|
|
662
|
-
* Gemini
|
|
663
|
-
* turn (a "prefill").
|
|
664
|
-
*
|
|
665
|
-
*
|
|
666
|
-
*
|
|
667
|
-
*
|
|
662
|
+
* Gemini Flash generation from which Google rejects a request whose `contents`
|
|
663
|
+
* end with a `model`-role turn (a "prefill"). Every Flash release from 3.6
|
|
664
|
+
* onward enforces it, so the cutoff is derived from the model id rather than
|
|
665
|
+
* enumerated - a new Flash model is covered on release with no change here.
|
|
666
|
+
*
|
|
667
|
+
* Scoped to Flash deliberately: dropping the turn silently degrades a working
|
|
668
|
+
* prefill into a fresh generation, so the rule only widens where Google
|
|
669
|
+
* documents the restriction. Lines that reject it without matching the cutoff
|
|
670
|
+
* are listed in {@link NO_PREFILL_GEMINI_MODELS}.
|
|
668
671
|
* @see https://ai.google.dev/gemini-api/docs/latest-model#api-changes-and-parameter-updates
|
|
669
672
|
*/
|
|
670
|
-
const
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
673
|
+
const NO_PREFILL_FLASH_MIN_VERSION = { major: 3, minor: 6 } as const;
|
|
674
|
+
|
|
675
|
+
/**
|
|
676
|
+
* `gemini-<major>[.<minor>]-flash`, with optional suffixes (`-latest`, `-lite`).
|
|
677
|
+
* Google ships both forms — `gemini-3.7-flash` and the major-only
|
|
678
|
+
* `gemini-3-flash-preview` — so the minor component is optional and an omitted
|
|
679
|
+
* one reads as `.0`.
|
|
680
|
+
*/
|
|
681
|
+
const GEMINI_FLASH_VERSION_PATTERN = /^gemini-(\d+)(?:\.(\d+))?-flash(?:$|-)/;
|
|
682
|
+
|
|
683
|
+
/**
|
|
684
|
+
* Models that reject prefill despite predating
|
|
685
|
+
* {@link NO_PREFILL_FLASH_MIN_VERSION}. Gemini 3.5 Flash-Lite enforces the
|
|
686
|
+
* restriction while its sibling Gemini 3.5 Flash still accepts a trailing model
|
|
687
|
+
* turn, so the 3.5 generation cannot be expressed as a version cutoff.
|
|
688
|
+
*/
|
|
689
|
+
const NO_PREFILL_GEMINI_MODELS = ['gemini-3.5-flash-lite'] as const;
|
|
675
690
|
|
|
676
691
|
export function rejectsModelTurnPrefill(model?: string): boolean {
|
|
677
692
|
if (model == null || model === '') {
|
|
678
693
|
return false;
|
|
679
694
|
}
|
|
680
695
|
const modelId = model.toLowerCase().split('/').pop() ?? '';
|
|
681
|
-
|
|
696
|
+
const listed = NO_PREFILL_GEMINI_MODELS.some(
|
|
682
697
|
(id) => modelId === id || modelId.startsWith(`${id}-`)
|
|
683
698
|
);
|
|
699
|
+
if (listed) {
|
|
700
|
+
return true;
|
|
701
|
+
}
|
|
702
|
+
const match = GEMINI_FLASH_VERSION_PATTERN.exec(modelId);
|
|
703
|
+
if (!match) {
|
|
704
|
+
return false;
|
|
705
|
+
}
|
|
706
|
+
const major = Number(match[1]);
|
|
707
|
+
const minor = Number(match[2] || '0');
|
|
708
|
+
if (major !== NO_PREFILL_FLASH_MIN_VERSION.major) {
|
|
709
|
+
return major > NO_PREFILL_FLASH_MIN_VERSION.major;
|
|
710
|
+
}
|
|
711
|
+
return minor >= NO_PREFILL_FLASH_MIN_VERSION.minor;
|
|
684
712
|
}
|
|
685
713
|
|
|
686
714
|
/**
|