@librechat/agents 3.7.13 → 3.7.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cjs/agents/AgentContext.cjs +1 -1
- package/dist/cjs/common/constants.cjs +2 -0
- package/dist/cjs/common/constants.cjs.map +1 -1
- package/dist/cjs/graphs/Graph.cjs +46 -10
- package/dist/cjs/graphs/Graph.cjs.map +1 -1
- package/dist/cjs/hooks/index.cjs +2 -0
- package/dist/cjs/hooks/index.cjs.map +1 -1
- package/dist/cjs/langfuse.cjs +16 -0
- package/dist/cjs/langfuse.cjs.map +1 -1
- package/dist/cjs/llm/google/utils/common.cjs +13 -6
- package/dist/cjs/llm/google/utils/common.cjs.map +1 -1
- package/dist/cjs/llm/invoke.cjs +191 -81
- package/dist/cjs/llm/invoke.cjs.map +1 -1
- package/dist/cjs/llm/preempt.cjs +77 -1
- package/dist/cjs/llm/preempt.cjs.map +1 -1
- package/dist/cjs/llm/prepareProviderRequest.cjs +2 -2
- package/dist/cjs/main.cjs +7 -4
- package/dist/cjs/messages/core.cjs +6 -0
- package/dist/cjs/messages/core.cjs.map +1 -1
- package/dist/cjs/messages/index.cjs +1 -1
- package/dist/cjs/messages/prune.cjs +2 -2
- package/dist/cjs/run.cjs +3 -2
- package/dist/cjs/run.cjs.map +1 -1
- package/dist/cjs/stream.cjs +4 -6
- package/dist/cjs/stream.cjs.map +1 -1
- package/dist/cjs/summarization/node.cjs +1 -1
- package/dist/cjs/utils/tokens.cjs +1 -1
- package/dist/esm/agents/AgentContext.mjs +1 -1
- package/dist/esm/common/constants.mjs +2 -1
- package/dist/esm/common/constants.mjs.map +1 -1
- package/dist/esm/graphs/Graph.mjs +46 -10
- package/dist/esm/graphs/Graph.mjs.map +1 -1
- package/dist/esm/hooks/index.mjs +2 -1
- package/dist/esm/hooks/index.mjs.map +1 -1
- package/dist/esm/langfuse.mjs +16 -0
- package/dist/esm/langfuse.mjs.map +1 -1
- package/dist/esm/llm/google/utils/common.mjs +13 -6
- package/dist/esm/llm/google/utils/common.mjs.map +1 -1
- package/dist/esm/llm/invoke.mjs +191 -81
- package/dist/esm/llm/invoke.mjs.map +1 -1
- package/dist/esm/llm/preempt.mjs +72 -2
- package/dist/esm/llm/preempt.mjs.map +1 -1
- package/dist/esm/llm/prepareProviderRequest.mjs +2 -2
- package/dist/esm/main.mjs +7 -7
- package/dist/esm/messages/core.mjs +6 -1
- package/dist/esm/messages/core.mjs.map +1 -1
- package/dist/esm/messages/index.mjs +1 -1
- package/dist/esm/messages/prune.mjs +2 -2
- package/dist/esm/run.mjs +3 -2
- package/dist/esm/run.mjs.map +1 -1
- package/dist/esm/stream.mjs +4 -6
- package/dist/esm/stream.mjs.map +1 -1
- package/dist/esm/summarization/node.mjs +1 -1
- package/dist/esm/utils/tokens.mjs +1 -1
- package/dist/types/common/constants.d.ts +15 -0
- package/dist/types/graphs/Graph.d.ts +50 -0
- package/dist/types/hooks/index.d.ts +15 -0
- package/dist/types/llm/invoke.d.ts +17 -1
- package/dist/types/llm/preempt.d.ts +114 -0
- package/dist/types/messages/core.d.ts +19 -0
- package/dist/types/run.d.ts +5 -3
- package/dist/types/types/run.d.ts +58 -7
- package/package.json +1 -1
- package/src/common/constants.ts +15 -0
- package/src/graphs/Graph.ts +122 -6
- package/src/hooks/index.ts +15 -0
- package/src/langfuse.ts +53 -0
- package/src/llm/google/utils/common.ts +40 -12
- package/src/llm/invoke.ts +614 -130
- package/src/llm/preempt.ts +320 -1
- package/src/messages/core.ts +28 -0
- package/src/run.ts +12 -4
- package/src/stream.ts +4 -12
- package/src/types/run.ts +58 -7
package/src/llm/preempt.ts
CHANGED
|
@@ -1,6 +1,11 @@
|
|
|
1
1
|
// src/llm/preempt.ts
|
|
2
2
|
import type { AIMessageChunk } from '@langchain/core/messages';
|
|
3
|
-
import {
|
|
3
|
+
import {
|
|
4
|
+
ContentTypes,
|
|
5
|
+
DEFAULT_MAX_SEALS,
|
|
6
|
+
DEFAULT_PREEMPT_RESTART_GRACE_MS,
|
|
7
|
+
} from '@/common';
|
|
8
|
+
import { isReasoningContentBlock } from '@/messages/core';
|
|
4
9
|
|
|
5
10
|
/**
|
|
6
11
|
* Normalizes a host-supplied seal budget.
|
|
@@ -119,6 +124,61 @@ function countOpenToolCalls(
|
|
|
119
124
|
return open;
|
|
120
125
|
}
|
|
121
126
|
|
|
127
|
+
/**
|
|
128
|
+
* True for a text block a provider emitted before it had anything to say.
|
|
129
|
+
*
|
|
130
|
+
* Whitespace-only string content is already accepted above, and the block form
|
|
131
|
+
* of the same thing must agree: several providers open their message with an
|
|
132
|
+
* empty text part, and rejecting it would leave a turn neither sealable nor
|
|
133
|
+
* discardable — an armed interrupt would then wait out the whole turn for the
|
|
134
|
+
* sake of a block carrying nothing.
|
|
135
|
+
*/
|
|
136
|
+
function isBlankTextBlock(block: { type?: string; text?: unknown }): boolean {
|
|
137
|
+
if (block.type !== ContentTypes.TEXT) {
|
|
138
|
+
return false;
|
|
139
|
+
}
|
|
140
|
+
return typeof block.text === 'string' && block.text.trim() === '';
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* True when an OpenAI Responses turn already carries provider-side tool output.
|
|
145
|
+
*
|
|
146
|
+
* Responses reports its built-in tools differently from every other shape this
|
|
147
|
+
* module checks: a completed web search, code interpreter run, or apply-patch
|
|
148
|
+
* lands in `additional_kwargs.tool_outputs` or the authoritative
|
|
149
|
+
* `response_metadata.output`, while `content` stays empty and the `tool_calls`
|
|
150
|
+
* arrays are never populated. The ordinary gates therefore see an empty turn
|
|
151
|
+
* and would call it disposable — and reissuing the prompt would run that tool
|
|
152
|
+
* a SECOND time, rebilling a search and repeating whatever a shell or
|
|
153
|
+
* apply-patch call already did to the world.
|
|
154
|
+
*
|
|
155
|
+
* `output` is inspected item by item rather than treated as fatal on sight,
|
|
156
|
+
* because a reasoning-only Responses turn populates it too at natural
|
|
157
|
+
* completion, and that turn is precisely the one a restart should be able to
|
|
158
|
+
* discard.
|
|
159
|
+
*/
|
|
160
|
+
function hasResponsesProviderOutput(chunk: AIMessageChunk): boolean {
|
|
161
|
+
if (chunk.additional_kwargs.tool_outputs != null) {
|
|
162
|
+
return true;
|
|
163
|
+
}
|
|
164
|
+
const metadata = chunk.response_metadata as {
|
|
165
|
+
tool_outputs?: unknown;
|
|
166
|
+
output?: unknown;
|
|
167
|
+
};
|
|
168
|
+
if (metadata.tool_outputs != null) {
|
|
169
|
+
return true;
|
|
170
|
+
}
|
|
171
|
+
if (!Array.isArray(metadata.output)) {
|
|
172
|
+
return false;
|
|
173
|
+
}
|
|
174
|
+
for (const item of metadata.output) {
|
|
175
|
+
if (!isReasoningContentBlock(item as { type?: string })) {
|
|
176
|
+
return true;
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
return false;
|
|
180
|
+
}
|
|
181
|
+
|
|
122
182
|
/**
|
|
123
183
|
* Cooperative mid-generation seal gate. Returns true ONLY when sealing here
|
|
124
184
|
* yields a message sequence valid on EVERY supported provider:
|
|
@@ -176,3 +236,262 @@ export function canSealPreempt(chunk: AIMessageChunk | undefined): boolean {
|
|
|
176
236
|
}
|
|
177
237
|
return hasNonEmptyTextContent(chunk.content);
|
|
178
238
|
}
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* What a host's armed preempt request may do to the turn in flight.
|
|
242
|
+
*
|
|
243
|
+
* `seal` keeps the partial assistant turn and injects after it; `restart`
|
|
244
|
+
* throws the partial away and re-issues the model call with the injected turn
|
|
245
|
+
* appended to the prompt. They are not two flavors of the same move — one
|
|
246
|
+
* preserves work, the other deliberately discards it — so the decision is
|
|
247
|
+
* resolved once, here, rather than re-derived at each trigger.
|
|
248
|
+
*/
|
|
249
|
+
export type PreemptAction = 'none' | 'seal' | 'restart';
|
|
250
|
+
|
|
251
|
+
/**
|
|
252
|
+
* True when the accumulated turn holds nothing a seal would have to preserve,
|
|
253
|
+
* so the whole model call can be thrown away and re-issued instead.
|
|
254
|
+
*
|
|
255
|
+
* The membership test is a WHITELIST: every content block must be a known
|
|
256
|
+
* reasoning block, and an unrecognized block refuses the restart. Blacklisting
|
|
257
|
+
* would silently discard whatever a future provider adds, and the failure is
|
|
258
|
+
* asymmetric — refusing costs the user a slower interrupt (exactly today's
|
|
259
|
+
* behavior), while wrongly discarding destroys work the model already did and
|
|
260
|
+
* the host may already have rendered.
|
|
261
|
+
*
|
|
262
|
+
* Tool machinery of any kind refuses, and deliberately without the settled-id
|
|
263
|
+
* allowance {@link canSealPreempt} makes: a settled `web_search_tool_result`
|
|
264
|
+
* means the provider already ran and billed a search, and re-issuing the call
|
|
265
|
+
* would run it again. A seal there is free; a restart is not. The same
|
|
266
|
+
* argument covers Responses' sidecar shapes — see
|
|
267
|
+
* {@link hasResponsesProviderOutput}, which the ordinary tool-call gates
|
|
268
|
+
* cannot see at all.
|
|
269
|
+
*
|
|
270
|
+
* An accumulation that carries visible text is not handled here at all — the
|
|
271
|
+
* caller resolves `seal` first, because keeping the user's answer always beats
|
|
272
|
+
* discarding it.
|
|
273
|
+
*
|
|
274
|
+
* `undefined` — the provider has sent nothing at all — is the case this whole
|
|
275
|
+
* path exists for: it is the silent window between the request and the first
|
|
276
|
+
* chunk, where there is by definition nothing to lose.
|
|
277
|
+
*/
|
|
278
|
+
export function canRestartPreempt(chunk: AIMessageChunk | undefined): boolean {
|
|
279
|
+
if (chunk == null) {
|
|
280
|
+
return true;
|
|
281
|
+
}
|
|
282
|
+
if ((chunk.tool_calls?.length ?? 0) > 0) {
|
|
283
|
+
return false;
|
|
284
|
+
}
|
|
285
|
+
if ((chunk.tool_call_chunks?.length ?? 0) > 0) {
|
|
286
|
+
return false;
|
|
287
|
+
}
|
|
288
|
+
if ((chunk.invalid_tool_calls?.length ?? 0) > 0) {
|
|
289
|
+
return false;
|
|
290
|
+
}
|
|
291
|
+
if (hasResponsesProviderOutput(chunk)) {
|
|
292
|
+
return false;
|
|
293
|
+
}
|
|
294
|
+
const { content } = chunk;
|
|
295
|
+
if (typeof content === 'string') {
|
|
296
|
+
return content.trim() === '';
|
|
297
|
+
}
|
|
298
|
+
for (const block of content) {
|
|
299
|
+
if (isReasoningContentBlock(block) || isBlankTextBlock(block)) {
|
|
300
|
+
continue;
|
|
301
|
+
}
|
|
302
|
+
return false;
|
|
303
|
+
}
|
|
304
|
+
return true;
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
/**
|
|
308
|
+
* Normalizes a host-supplied restart grace, on the same rules as
|
|
309
|
+
* {@link resolveMaxSeals}: `0` is honored as "never wait", anything not finite
|
|
310
|
+
* falls back to the default. Negative values collapse to `0` rather than
|
|
311
|
+
* inverting the comparison in {@link resolvePreemptAction}.
|
|
312
|
+
*/
|
|
313
|
+
export function resolveRestartGraceMs(graceMs: number | undefined): number {
|
|
314
|
+
if (graceMs == null || !Number.isFinite(graceMs)) {
|
|
315
|
+
return DEFAULT_PREEMPT_RESTART_GRACE_MS;
|
|
316
|
+
}
|
|
317
|
+
return graceMs > 0 ? graceMs : 0;
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
/**
|
|
321
|
+
* The single decision point both preempt triggers share: the per-chunk poll in
|
|
322
|
+
* the stream loop, and the host wake-up that fires while the provider is
|
|
323
|
+
* silent.
|
|
324
|
+
*
|
|
325
|
+
* A SEAL is preferred wherever one is available, so a turn that has already
|
|
326
|
+
* produced an answer never loses it to a restart. A RESTART converts only once
|
|
327
|
+
* the request has outlived `graceMs`.
|
|
328
|
+
*
|
|
329
|
+
* The window is not politeness, it is correctness, and it applies to a silent
|
|
330
|
+
* provider exactly as it does to a reasoning one:
|
|
331
|
+
* - reasoning usually precedes text by moments, and discarding a turn that
|
|
332
|
+
* was about to become sealable trades a kept answer for a re-issued
|
|
333
|
+
* request. Only a genuinely long thinking stretch — the one an interrupt
|
|
334
|
+
* can otherwise wait out entirely — should convert.
|
|
335
|
+
* - `chunk` is what the CONSUMER has accumulated, and the provider stream
|
|
336
|
+
* buffers a chunk ahead of it. An empty accumulation therefore does not
|
|
337
|
+
* prove the provider produced nothing; it can also mean the first chunk is
|
|
338
|
+
* in flight. Converting on emptiness alone would discard that chunk — text
|
|
339
|
+
* included — on a race no caller can see. A provider that is still silent a
|
|
340
|
+
* window later has no such chunk outstanding.
|
|
341
|
+
*
|
|
342
|
+
* Non-mutating and allocation-free, like the poll that guards it: the seal
|
|
343
|
+
* budget is only spent once the caller acts on a non-`none` result.
|
|
344
|
+
*/
|
|
345
|
+
export function resolvePreemptAction({
|
|
346
|
+
chunk,
|
|
347
|
+
requestAgeMs,
|
|
348
|
+
graceMs,
|
|
349
|
+
}: {
|
|
350
|
+
chunk: AIMessageChunk | undefined;
|
|
351
|
+
requestAgeMs: number;
|
|
352
|
+
graceMs: number;
|
|
353
|
+
}): PreemptAction {
|
|
354
|
+
if (canSealPreempt(chunk)) {
|
|
355
|
+
return 'seal';
|
|
356
|
+
}
|
|
357
|
+
if (!canRestartPreempt(chunk)) {
|
|
358
|
+
return 'none';
|
|
359
|
+
}
|
|
360
|
+
return requestAgeMs >= graceMs ? 'restart' : 'none';
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
/**
|
|
364
|
+
* How long a restarted model run stays recorded when nothing consumes it.
|
|
365
|
+
*
|
|
366
|
+
* The consumer is LangChain's error callback, dispatched on a non-awaited
|
|
367
|
+
* queue, so a marker may legitimately sit unread for a while. A COUNT cap
|
|
368
|
+
* cannot tell that apart from a leak: a burst of concurrent restarts would
|
|
369
|
+
* evict markers whose callbacks were still queued, and each eviction turns an
|
|
370
|
+
* expected restart back into an error in the host's traces — the very thing
|
|
371
|
+
* the marker exists to prevent. Age can tell them apart, so the bound is time.
|
|
372
|
+
*
|
|
373
|
+
* It is a LEAK bound, not a deadline: nothing here can observe when the
|
|
374
|
+
* callback queue drains, so the window is set far past any latency that queue
|
|
375
|
+
* exhibits rather than tuned to it. A stall long enough to outlast this has
|
|
376
|
+
* already broken every other time-based assumption in the run.
|
|
377
|
+
*/
|
|
378
|
+
const PREEMPT_RESTARTED_RUN_TTL_MS = 300_000;
|
|
379
|
+
|
|
380
|
+
/**
|
|
381
|
+
* Model runs whose provider stream was torn down for a restart, with the
|
|
382
|
+
* accumulated turn so the close can carry its usage.
|
|
383
|
+
*
|
|
384
|
+
* Recorded rather than inferred from the thrown error: aborting makes the
|
|
385
|
+
* provider adapter raise whatever IT raises for cancellation, and matching on
|
|
386
|
+
* that shape would bind the tracing layer to per-provider error identity. The
|
|
387
|
+
* run id is unambiguous and already captured for the seal path.
|
|
388
|
+
*
|
|
389
|
+
* The message rides along because the run closes through the error path, which
|
|
390
|
+
* carries no output: without it a discarded attempt would report no usage at
|
|
391
|
+
* all, and the reasoning tokens the provider already billed would vanish from
|
|
392
|
+
* cost accounting. The later synthetic `CHAT_MODEL_END` cannot repair that —
|
|
393
|
+
* by then the generation is closed.
|
|
394
|
+
*
|
|
395
|
+
* Insertion-ordered, so expiry sweeps from the front and stops at the first
|
|
396
|
+
* live entry.
|
|
397
|
+
*/
|
|
398
|
+
const preemptRestartedRuns = new Map<string, PreemptRestartedRun>();
|
|
399
|
+
|
|
400
|
+
/**
|
|
401
|
+
* The `llmOutput` a discarded attempt closes with, so a restart reads as
|
|
402
|
+
* control flow rather than as an ordinary generation — `AGENTS.md` states that
|
|
403
|
+
* control flow is not an error, and a discarded turn is not an answer either.
|
|
404
|
+
*
|
|
405
|
+
* Shared so every close route emits it: which trigger fired must not change
|
|
406
|
+
* how the run looks in a trace.
|
|
407
|
+
*/
|
|
408
|
+
export const PREEMPT_RESTART_CONTROL_FLOW = {
|
|
409
|
+
controlFlow: 'PreemptRestart',
|
|
410
|
+
} as const;
|
|
411
|
+
|
|
412
|
+
/** A torn-down model run awaiting its tracing close. */
|
|
413
|
+
export interface PreemptRestartedRun {
|
|
414
|
+
/** The accumulated turn, carrying whatever usage was resolved for it. */
|
|
415
|
+
message: AIMessageChunk;
|
|
416
|
+
recordedAt: number;
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
/**
|
|
420
|
+
* Armed while anything is recorded, so expiry does not depend on another
|
|
421
|
+
* restart arriving to trigger it. Without this the final burst before a quiet
|
|
422
|
+
* period would hold its messages for the life of the process — and a host that
|
|
423
|
+
* enables preemption without tracing never consumes a single record, so that
|
|
424
|
+
* burst is the common case, not the corner one.
|
|
425
|
+
*
|
|
426
|
+
* `unref`'d: reclaiming a few messages is never a reason to keep a process
|
|
427
|
+
* alive.
|
|
428
|
+
*/
|
|
429
|
+
let sweepTimer: ReturnType<typeof setTimeout> | undefined;
|
|
430
|
+
|
|
431
|
+
function sweepExpiredRestartedRuns(now: number): void {
|
|
432
|
+
for (const [runId, record] of preemptRestartedRuns) {
|
|
433
|
+
if (now - record.recordedAt < PREEMPT_RESTARTED_RUN_TTL_MS) {
|
|
434
|
+
break;
|
|
435
|
+
}
|
|
436
|
+
preemptRestartedRuns.delete(runId);
|
|
437
|
+
}
|
|
438
|
+
if (preemptRestartedRuns.size === 0) {
|
|
439
|
+
if (sweepTimer != null) {
|
|
440
|
+
clearTimeout(sweepTimer);
|
|
441
|
+
sweepTimer = undefined;
|
|
442
|
+
}
|
|
443
|
+
return;
|
|
444
|
+
}
|
|
445
|
+
if (sweepTimer != null) {
|
|
446
|
+
return;
|
|
447
|
+
}
|
|
448
|
+
sweepTimer = setTimeout(() => {
|
|
449
|
+
sweepTimer = undefined;
|
|
450
|
+
sweepExpiredRestartedRuns(Date.now());
|
|
451
|
+
}, PREEMPT_RESTARTED_RUN_TTL_MS);
|
|
452
|
+
sweepTimer.unref();
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
/** Records a model run whose stream was torn down for a restart. */
|
|
456
|
+
export function notePreemptRestartedRun(
|
|
457
|
+
runId: string,
|
|
458
|
+
message: AIMessageChunk
|
|
459
|
+
): void {
|
|
460
|
+
const now = Date.now();
|
|
461
|
+
preemptRestartedRuns.set(runId, { message, recordedAt: now });
|
|
462
|
+
sweepExpiredRestartedRuns(now);
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
/**
|
|
466
|
+
* The record for a run that ended because a preempt discarded it, or
|
|
467
|
+
* `undefined`.
|
|
468
|
+
*
|
|
469
|
+
* Deliberately NON-consuming. A run can be closed through more than one
|
|
470
|
+
* tracing handler — a run-level one plus another supplied through the model's
|
|
471
|
+
* own `clientOptions.callbacks`, which `endSealedModelRun` already composes
|
|
472
|
+
* for exactly that reason — and each receives the same `handleLLMError`. A
|
|
473
|
+
* read that deleted the record would let only the first handler recognize the
|
|
474
|
+
* restart, and every other destination would export the same expected abort as
|
|
475
|
+
* an error. Reuse is not a concern the way it would be for a counter: run ids
|
|
476
|
+
* are UUIDs, and the sweep is what reclaims them.
|
|
477
|
+
*/
|
|
478
|
+
export function readPreemptRestartedRun(
|
|
479
|
+
runId: string
|
|
480
|
+
): PreemptRestartedRun | undefined {
|
|
481
|
+
return preemptRestartedRuns.get(runId);
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
/**
|
|
485
|
+
* Drops a record whose run turned out not to need it — the adapter ignored the
|
|
486
|
+
* abort, so the tear-down never reached LangChain's error path and the close
|
|
487
|
+
* happens manually instead.
|
|
488
|
+
*/
|
|
489
|
+
export function forgetPreemptRestartedRun(runId: string): void {
|
|
490
|
+
if (!preemptRestartedRuns.delete(runId)) {
|
|
491
|
+
return;
|
|
492
|
+
}
|
|
493
|
+
if (preemptRestartedRuns.size === 0 && sweepTimer != null) {
|
|
494
|
+
clearTimeout(sweepTimer);
|
|
495
|
+
sweepTimer = undefined;
|
|
496
|
+
}
|
|
497
|
+
}
|
package/src/messages/core.ts
CHANGED
|
@@ -224,6 +224,34 @@ function getAdditionalReasoningContent(
|
|
|
224
224
|
return getReasoningDetailsText(additionalKwargs.reasoning_details);
|
|
225
225
|
}
|
|
226
226
|
|
|
227
|
+
/**
|
|
228
|
+
* True when a content block is a provider reasoning block: Anthropic
|
|
229
|
+
* `thinking` / `redacted_thinking`, Google `reasoning`, Bedrock
|
|
230
|
+
* `reasoning_content`, and the streamed delta variants that prefix each of
|
|
231
|
+
* those names.
|
|
232
|
+
*
|
|
233
|
+
* Shared rather than re-derived because two very different readers must agree
|
|
234
|
+
* on the same answer: the stream classifier, which decides what the host has
|
|
235
|
+
* been shown, and the preempt gate, which decides whether an accumulated turn
|
|
236
|
+
* holds anything worth keeping. A turn the host renders as thinking-only must
|
|
237
|
+
* never look like a visible answer to the gate.
|
|
238
|
+
*
|
|
239
|
+
* Prefix matching, not equality, because providers stream these as suffixed
|
|
240
|
+
* delta types. `reasoning_content` needs no clause of its own — it already
|
|
241
|
+
* carries the `reasoning` prefix.
|
|
242
|
+
*/
|
|
243
|
+
export function isReasoningContentBlock(block: { type?: string }): boolean {
|
|
244
|
+
const type = block.type;
|
|
245
|
+
if (type == null) {
|
|
246
|
+
return false;
|
|
247
|
+
}
|
|
248
|
+
return (
|
|
249
|
+
type.startsWith(ContentTypes.THINKING) ||
|
|
250
|
+
type.startsWith(ContentTypes.REASONING) ||
|
|
251
|
+
type === 'redacted_thinking'
|
|
252
|
+
);
|
|
253
|
+
}
|
|
254
|
+
|
|
227
255
|
function hasReasoningContent(content: BaseMessage['content']): boolean {
|
|
228
256
|
if (!Array.isArray(content)) {
|
|
229
257
|
return false;
|
package/src/run.ts
CHANGED
|
@@ -942,12 +942,20 @@ export class Run<_T extends t.BaseGraphState> {
|
|
|
942
942
|
}
|
|
943
943
|
|
|
944
944
|
/**
|
|
945
|
-
* Cooperative-
|
|
946
|
-
* watch: it counts
|
|
947
|
-
* inject, which ends the turn early and leaves the answer unfinished.
|
|
945
|
+
* Cooperative-preemption counters for this run. `emptyBoundaries` is the one
|
|
946
|
+
* to watch: it counts SEALS whose `PreemptBoundary` produced nothing to
|
|
947
|
+
* inject, which ends the turn early and leaves the answer unfinished. A
|
|
948
|
+
* restart that injects nothing simply reissues the call, so it is not
|
|
949
|
+
* counted there.
|
|
948
950
|
*/
|
|
949
951
|
getPreemptStats(): t.PreemptStats {
|
|
950
|
-
return
|
|
952
|
+
return (
|
|
953
|
+
this.Graph?.getPreemptStats() ?? {
|
|
954
|
+
seals: 0,
|
|
955
|
+
restarts: 0,
|
|
956
|
+
emptyBoundaries: 0,
|
|
957
|
+
}
|
|
958
|
+
);
|
|
951
959
|
}
|
|
952
960
|
|
|
953
961
|
getToolCount(): number {
|
package/src/stream.ts
CHANGED
|
@@ -56,6 +56,7 @@ import {
|
|
|
56
56
|
} from '@/utils/truncation';
|
|
57
57
|
import { resolveToolOutcome, outcomeFieldsFromResult } from '@/tools/intentArg';
|
|
58
58
|
import { TOOL_OUTPUT_REF_PATTERN } from '@/tools/toolOutputReferences';
|
|
59
|
+
import { isReasoningContentBlock } from '@/messages/core';
|
|
59
60
|
import { safeDispatchCustomEvent } from '@/utils/events';
|
|
60
61
|
import { composeAbortSignals } from '@/utils/misc';
|
|
61
62
|
import { isGoogleLike } from '@/utils/llm';
|
|
@@ -454,15 +455,6 @@ function isTextContentPart(contentPart: t.MessageContentComplex): boolean {
|
|
|
454
455
|
return contentPart.type?.startsWith(ContentTypes.TEXT) ?? false;
|
|
455
456
|
}
|
|
456
457
|
|
|
457
|
-
function isReasoningContentPart(contentPart: t.MessageContentComplex): boolean {
|
|
458
|
-
return (
|
|
459
|
-
(contentPart.type?.startsWith(ContentTypes.THINKING) ?? false) ||
|
|
460
|
-
(contentPart.type?.startsWith(ContentTypes.REASONING) ?? false) ||
|
|
461
|
-
(contentPart.type?.startsWith(ContentTypes.REASONING_CONTENT) ?? false) ||
|
|
462
|
-
contentPart.type === 'redacted_thinking'
|
|
463
|
-
);
|
|
464
|
-
}
|
|
465
|
-
|
|
466
458
|
function getReasoningTextFromContentPart(
|
|
467
459
|
contentPart: t.MessageContentComplex
|
|
468
460
|
): string {
|
|
@@ -532,7 +524,7 @@ function shouldStartFreshMessageStepAfterGoogleServerSideTool({
|
|
|
532
524
|
}
|
|
533
525
|
return (
|
|
534
526
|
content.every((c) => isTextContentPart(c)) ||
|
|
535
|
-
content.every((c) =>
|
|
527
|
+
content.every((c) => isReasoningContentBlock(c))
|
|
536
528
|
);
|
|
537
529
|
}
|
|
538
530
|
|
|
@@ -649,7 +641,7 @@ async function dispatchGoogleServerSideToolStreamContent({
|
|
|
649
641
|
}
|
|
650
642
|
reasoningContent.push(
|
|
651
643
|
...content
|
|
652
|
-
.filter((contentPart) =>
|
|
644
|
+
.filter((contentPart) => isReasoningContentBlock(contentPart))
|
|
653
645
|
.map((contentPart) => ({
|
|
654
646
|
type: ContentTypes.THINK,
|
|
655
647
|
think: getReasoningTextFromContentPart(contentPart),
|
|
@@ -2034,7 +2026,7 @@ hasToolCallChunks: ${hasToolCallChunks}
|
|
|
2034
2026
|
},
|
|
2035
2027
|
metadata
|
|
2036
2028
|
);
|
|
2037
|
-
} else if (content.every((c) =>
|
|
2029
|
+
} else if (content.every((c) => isReasoningContentBlock(c))) {
|
|
2038
2030
|
await graph.dispatchReasoningDelta(
|
|
2039
2031
|
stepId,
|
|
2040
2032
|
{
|
package/src/types/run.ts
CHANGED
|
@@ -123,15 +123,23 @@ export type StandardGraphConfig = Omit<
|
|
|
123
123
|
> & { type?: 'standard'; signal?: AbortSignal };
|
|
124
124
|
|
|
125
125
|
/**
|
|
126
|
-
* Cooperative mid-generation preemption. Lets a host
|
|
127
|
-
*
|
|
128
|
-
*
|
|
129
|
-
*
|
|
130
|
-
*
|
|
126
|
+
* Cooperative mid-generation preemption. Lets a host end the live model stream
|
|
127
|
+
* early and inject whatever it queued, in one of two ways:
|
|
128
|
+
*
|
|
129
|
+
* - a SEAL, at the next provider-safe token boundary: the run is never
|
|
130
|
+
* aborted, the partial assistant turn is kept, and the graph self-loops into
|
|
131
|
+
* a fresh model call once the `PreemptBoundary` hook has injected.
|
|
132
|
+
* - a RESTART, when the turn holds nothing worth keeping — no text, no tool
|
|
133
|
+
* call, at most reasoning. The provider request IS torn down, its output is
|
|
134
|
+
* discarded, and the model is called again with the injection appended to
|
|
135
|
+
* the same prompt. Requires `subscribe`; see it for why the per-chunk poll
|
|
136
|
+
* alone cannot reach this case.
|
|
131
137
|
*
|
|
132
138
|
* Preconditions the host MUST satisfy:
|
|
133
139
|
* - `shouldPreempt` is polled once per streamed chunk on the top-level
|
|
134
|
-
* graph
|
|
140
|
+
* graph, and once more per model attempt if `subscribe` is supplied — at
|
|
141
|
+
* the attempt's start and on each wake. It must be synchronous,
|
|
142
|
+
* allocation-free and O(1) — never I/O.
|
|
135
143
|
* It must also be LEVEL-TRIGGERED (non-consuming): the SDK never clears
|
|
136
144
|
* the host's request, and a true result is only honored once the
|
|
137
145
|
* accumulated chunk is provider-safe, so the predicate may be polled
|
|
@@ -168,6 +176,40 @@ export interface StreamPreemption {
|
|
|
168
176
|
* consumes the request; a self-clearing read loses it on an unsafe chunk.
|
|
169
177
|
*/
|
|
170
178
|
shouldPreempt: () => boolean;
|
|
179
|
+
/**
|
|
180
|
+
* Optional wake-up channel for the window where the poll cannot reach: a
|
|
181
|
+
* provider that has sent nothing yet, or that is streaming reasoning the
|
|
182
|
+
* seal gate will never accept. `shouldPreempt` is only read per chunk, so a
|
|
183
|
+
* request armed during a long silent stretch would otherwise wait for the
|
|
184
|
+
* whole turn.
|
|
185
|
+
*
|
|
186
|
+
* The SDK subscribes once per model attempt and calls the returned
|
|
187
|
+
* unsubscribe when the attempt ends. `wake` is a HINT, never the request
|
|
188
|
+
* itself: it wakes the SDK, which then re-reads `shouldPreempt` as the sole
|
|
189
|
+
* authority. That keeps the level-triggered contract above intact and makes
|
|
190
|
+
* spurious wakes free — a host may call it on every arm without tracking
|
|
191
|
+
* which ones the SDK already saw.
|
|
192
|
+
*
|
|
193
|
+
* A wake is only honored when the accumulated turn holds nothing worth
|
|
194
|
+
* keeping, and it is honored by DISCARDING that turn: the in-flight provider
|
|
195
|
+
* request is torn down, its partial output is dropped, the `PreemptBoundary`
|
|
196
|
+
* hook injects, and the model is called again with the injection appended.
|
|
197
|
+
* Reasoning tokens already spent are billed and lost, which is the trade the
|
|
198
|
+
* host is asking for. A turn that has produced visible text seals instead,
|
|
199
|
+
* through the ordinary per-chunk path, and a wake never discards it.
|
|
200
|
+
*/
|
|
201
|
+
subscribe?: (wake: () => void) => () => void;
|
|
202
|
+
/**
|
|
203
|
+
* How long a request waits for the turn to become sealable before it may
|
|
204
|
+
* discard it instead. Defaults to `DEFAULT_PREEMPT_RESTART_GRACE_MS`; `0`
|
|
205
|
+
* converts as soon as the shape allows.
|
|
206
|
+
*
|
|
207
|
+
* The window keeps a restart from stealing a seal that was moments away, and
|
|
208
|
+
* from discarding a first chunk still in flight between the provider and the
|
|
209
|
+
* consumer. Both are why it applies to a silent turn as well as a thinking
|
|
210
|
+
* one.
|
|
211
|
+
*/
|
|
212
|
+
restartGraceMs?: number;
|
|
171
213
|
/**
|
|
172
214
|
* Max cooperative seals per run. Each seal costs one extra superstep, so
|
|
173
215
|
* this also bounds the recursion-limit headroom the run reserves.
|
|
@@ -175,9 +217,18 @@ export interface StreamPreemption {
|
|
|
175
217
|
maxSeals?: number;
|
|
176
218
|
}
|
|
177
219
|
|
|
178
|
-
/**
|
|
220
|
+
/**
|
|
221
|
+
* Per run: seals honored, discards honored, and boundaries that had nothing to
|
|
222
|
+
* inject.
|
|
223
|
+
*
|
|
224
|
+
* `seals` counts only turns that were KEPT — a partial assistant message
|
|
225
|
+
* survived into the next prompt. `restarts` counts turns that were discarded
|
|
226
|
+
* and reissued, which preserve nothing, so a consumer classifying a run as
|
|
227
|
+
* "sealed" must not read them together.
|
|
228
|
+
*/
|
|
179
229
|
export type PreemptStats = {
|
|
180
230
|
seals: number;
|
|
231
|
+
restarts: number;
|
|
181
232
|
emptyBoundaries: number;
|
|
182
233
|
};
|
|
183
234
|
|