@dudousxd/nestjs-agent-core 0.31.0 → 0.33.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -263,102 +263,6 @@ interface AgentIntake {
263
263
  }
264
264
  declare const DEFAULT_INTAKE_PREAMBLE = "A few questions before I start. I have pre-picked what I would choose, so confirming is enough.";
265
265
 
266
- /**
267
- * The ceiling on how much of a thread rides into a turn. Without one, `runAgentLoop` maps EVERY
268
- * message the store returns into the model's messages, so a long-lived thread grows until the
269
- * provider rejects the request — and every turn before that one pays for the whole transcript.
270
- * `@dudousxd/nestjs-agent-core` ships `windowHistory` as the built-in; anything satisfying this SPI
271
- * works. Wire it as `AgentLoopDeps.historyPolicy`, or via `AgentModule.forRoot({ history })` /
272
- * `@Agent({ history })`.
273
- */
274
-
275
- /** Whose history this is — enough for a policy to size the window per agent or per actor. */
276
- interface HistoryPolicyContext {
277
- threadId: string;
278
- actor: Actor;
279
- /** The agent running this turn. Undefined → the default agent. */
280
- agentName?: string;
281
- }
282
- /** How a policy split the thread: what rides into the turn, and what the ceiling left out. */
283
- interface HistorySelection {
284
- /** Sent to the model, oldest-first. */
285
- keep: ModelMessage[];
286
- /** Left out, oldest-first. Folded into a leading summary when the policy implements `summarize`. */
287
- drop: ModelMessage[];
288
- }
289
- /** A stand-in for the messages a window left out, plus what producing it cost. */
290
- interface HistorySummary {
291
- /** Prose the loop folds into the window as a leading `system` message. */
292
- text: string;
293
- /**
294
- * What the summarizer spent, when it called a model. Recorded as a `history_summary` usage row, so
295
- * a ceiling on context cost cannot itself become spend nothing accounts for. Omit for a summarizer
296
- * that calls no model (a rollup of tool names, a digest the app already stored).
297
- */
298
- usage?: MessageUsage;
299
- /** Accounting label for the model that produced it; falls back to `AgentLoopDeps.modelId`. */
300
- modelId?: string;
301
- }
302
- interface HistoryPolicy {
303
- /**
304
- * The most messages {@link select} can ever keep, where the ceiling can be stated as a row count.
305
- *
306
- * A hint the loop hands to the store, which then reads that many of the thread's newest rows
307
- * rather than its whole transcript (see `ThreadTurnReader`). Declaring it is a PROMISE about
308
- * `select`: that it keeps at most this many messages, and that they are the NEWEST ones — so a
309
- * window of this size is indistinguishable, to `select`, from the full transcript. A policy that
310
- * can keep more than this, or can keep something older than the newest `maxMessages`, must omit it
311
- * rather than select over rows the store was never asked for.
312
- *
313
- * Omit where the ceiling is not a row count at all — a token budget alone cannot name one, since
314
- * one message can be four tokens or forty thousand. Omitting costs only the bound on the READ; the
315
- * prompt is identical either way.
316
- */
317
- readonly maxMessages?: number;
318
- /**
319
- * Decide what the model sees.
320
- *
321
- * MUST be a pure function of `messages`. The loop calls it INSIDE the `load:thread` checkpoint and
322
- * records its result there, so the ceiling bounds the journal as well as the prompt: what the
323
- * checkpoint holds is the selection rather than the store's whole `ThreadDetail`, on a payload
324
- * every replay re-reads. It still adds no position of its own — a policy cannot change the name or
325
- * position of a single existing checkpoint.
326
- *
327
- * Read a clock, a feature flag or a database here and the resumed run windows differently from the
328
- * one that suspended: the model gets a different prompt, and any step whose existence depends on
329
- * the split lands at a position the history has no room for. Anything non-deterministic belongs in
330
- * {@link summarize}, which has a checkpoint of its own.
331
- *
332
- * The newest message must always be in `keep` — dropping it leaves the turn with nothing to answer.
333
- */
334
- select(messages: ModelMessage[], ctx: HistoryPolicyContext): HistorySelection;
335
- /**
336
- * Fold the dropped messages into prose the model reads in their place, prepended to the window as
337
- * a `system` message. Optional — without it, dropped messages are simply gone, and `load:thread`
338
- * does not record them either: a summarizer is the only thing that ever reads them back.
339
- *
340
- * Runs inside the loop's `history:summarize` checkpoint, so it may call a model or hit the
341
- * network: the first attempt's result is journaled and every replay reads it back instead of
342
- * re-summarizing. It runs once per RUN (not per model step), and only when `select` actually
343
- * dropped something.
344
- */
345
- summarize?(dropped: ModelMessage[], ctx: HistoryPolicyContext): Promise<HistorySummary>;
346
- }
347
- /**
348
- * The window an agent asks for declaratively — `AgentModule.forRoot({ history })` and
349
- * `@Agent({ history })`. Plain data, so it can live in a decorator's metadata; the NestJS layer
350
- * turns it into a `windowHistory` policy. A consumer needing anything the window can't express
351
- * supplies a {@link HistoryPolicy} instead.
352
- */
353
- interface AgentHistoryWindow {
354
- /** Keep at most this many of the newest messages. */
355
- maxMessages?: number;
356
- /** Keep the newest messages whose estimated tokens fit this budget. */
357
- maxTokens?: number;
358
- /** Fold what the window left out into a leading summary — one extra model call per run. */
359
- summarize?: boolean;
360
- }
361
-
362
266
  /**
363
267
  * The structured live-stream vocabulary carried over the {@link SinkWriter} byte channel.
364
268
  *
@@ -578,6 +482,22 @@ type AgentStreamEvent = {
578
482
  */
579
483
  | {
580
484
  kind: 'cancelled';
485
+ }
486
+ /**
487
+ * The thread's message queue changed — a snapshot of the whole queue, never a delta, so a client
488
+ * that missed one frame is corrected by the next. Written into the stream of the run that is
489
+ * holding the thread: when someone queues, edits, reorders or removes a waiting message, and, just
490
+ * before this run's own terminal frame, with what happens next — `started` names the queued
491
+ * message that became the next turn and that turn's run id (attach to it with
492
+ * `GET <base>/chat/:runId/stream`), `queue.paused` says why nothing starts.
493
+ */
494
+ | {
495
+ kind: 'queue';
496
+ queue: ChatQueueState;
497
+ started?: {
498
+ messageId: string;
499
+ runId: string;
500
+ };
581
501
  };
582
502
  /** Encode one event as an NDJSON line (`{...}\n`) for {@link SinkWriter.write}. */
583
503
  declare function encodeStreamEvent(event: AgentStreamEvent): Uint8Array;
@@ -591,6 +511,567 @@ declare function encodeStreamEvent(event: AgentStreamEvent): Uint8Array;
591
511
  */
592
512
  declare function decodeStreamEvent(line: string): AgentStreamEvent | null;
593
513
 
514
+ interface CreateThreadInput {
515
+ actor: Actor;
516
+ transient?: boolean;
517
+ title?: string;
518
+ }
519
+ interface AppendMessageInput {
520
+ threadId: string;
521
+ role: StoredMessage['role'];
522
+ content: string;
523
+ /** Which agent produced this message (assistant messages) — provenance. */
524
+ agentName?: string;
525
+ toolCalls?: ToolCallRequest[];
526
+ toolResults?: ToolResult[];
527
+ /** Files the user attached to this message (image/PDF). Persisted verbatim. */
528
+ attachments?: MessageAttachment[];
529
+ followUps?: string[];
530
+ usage?: MessageUsage;
531
+ /**
532
+ * The run (turn) that produced this message. Without it a consumer can only guess which turn a
533
+ * message belongs to by comparing timestamps against the run's `startedAt`, and that guess breaks
534
+ * the moment a turn is regenerated — the replaced answer is truncated away, so the times no longer
535
+ * line up 1:1. Optional so a caller predating this (and a host that appends messages outside a
536
+ * run) can omit it; the store persists it as `null` when absent.
537
+ */
538
+ runId?: string;
539
+ /** The step's streamed thinking. See {@link StoredMessage.reasoning}. */
540
+ reasoning?: string;
541
+ /** Time spent thinking in this step, in ms. See {@link StoredMessage.reasoningMs}. */
542
+ reasoningMs?: number;
543
+ /** Components pushed during this step. See {@link StoredMessage.ui}. */
544
+ ui?: AgentUiComponent[];
545
+ }
546
+ interface RecordToolCallInput {
547
+ toolCallId: string;
548
+ messageId: string;
549
+ toolName: string;
550
+ toolType: 'read' | 'action';
551
+ input: unknown;
552
+ status: ToolCallStatus;
553
+ /**
554
+ * The run (turn) this tool call belongs to — enables a governance surface to deep-link a tool
555
+ * call out to its trace waterfall. Optional so a caller predating this (or a store's own
556
+ * synthetic tool calls) can omit it; the store persists it as `null` when absent.
557
+ */
558
+ runId?: string;
559
+ /**
560
+ * Who has to approve this call (`'requester'` or a role), from the turn's `ApprovalPolicy`. Set on
561
+ * an action call that was put to a person — or approved by a remembered decision — and never
562
+ * otherwise; persisted as `null` when absent.
563
+ */
564
+ approver?: string;
565
+ /** ISO-8601 instant the approval request lapses. Absent → it never does. */
566
+ expiresAt?: string;
567
+ }
568
+ interface UpdateToolCallInput {
569
+ toolCallId: string;
570
+ status: ToolCallStatus;
571
+ output?: unknown;
572
+ error?: string;
573
+ executionMs?: number;
574
+ executedByRef?: string;
575
+ /** The approval asked for later calls of this tool in this thread to run without asking. */
576
+ remember?: boolean;
577
+ /** The surface the decision came through. See {@link import('../types.js').Decision.decidedVia}. */
578
+ decidedVia?: string;
579
+ }
580
+ /** What the approve/reject routes need to know about a call before they signal a decision on it. */
581
+ interface ToolCallApprovalState {
582
+ status: ToolCallStatus;
583
+ /** `null` for a call no policy put to anyone (an `ask`, or a row written before approvers existed). */
584
+ approver: string | null;
585
+ expiresAt: string | null;
586
+ }
587
+ /** Patch applied by {@link AgentStore.updateThread}. An omitted key leaves that field untouched. */
588
+ interface UpdateThreadInput {
589
+ title?: string;
590
+ /** `null` clears the thread's default agent (falls back to the module default). */
591
+ defaultAgent?: string | null;
592
+ /** `null` unpins the thread's model (turns run on the provider default). */
593
+ model?: string | null;
594
+ }
595
+ interface RecordUsageInput {
596
+ threadId: string;
597
+ actorRef: string;
598
+ messageId?: string;
599
+ modelId: string;
600
+ purpose: UsagePurpose;
601
+ usage: MessageUsage;
602
+ /** Provider-reported actual USD cost for this turn, when known (gateways report it). */
603
+ costUsd?: number;
604
+ }
605
+ interface RecordRunStartInput {
606
+ runId: string;
607
+ threadId: string;
608
+ actorRef: string;
609
+ agentName?: string;
610
+ /**
611
+ * The run that started this one, for a delegation's child run. The parent->child edge exists in
612
+ * the durable runtime's own journal, but only there: a governance surface reading run rows alone
613
+ * cannot roll a delegation's cost up to the turn that asked for it, and a DETACHED child outlives
614
+ * its parent's turn entirely, so nothing in the transcript pairs them either.
615
+ *
616
+ * Optional, and a store that persists nothing for it still works — it loses the tree, not the run.
617
+ */
618
+ parentRunId?: string;
619
+ /** sha256 hex of the run's resolved (pre-RAG) system prompt — identifies the prompt VERSION. */
620
+ promptHash?: string;
621
+ }
622
+ interface RecordRunEndInput {
623
+ runId: string;
624
+ /**
625
+ * `cancelled` is a THIRD terminal, not a flavour of `failed`: someone asked the run to stop and it
626
+ * did, which is the control working. A consumer computing a failure rate over these rows has to be
627
+ * able to leave it out — counting a user pressing Stop as an error pages whoever is on call for
628
+ * model failures. It carries no `errorCode`/`errorMessage`, since there is nothing to diagnose.
629
+ */
630
+ status: 'completed' | 'failed' | 'cancelled';
631
+ durationMs?: number;
632
+ errorCode?: string;
633
+ errorMessage?: string;
634
+ }
635
+ /** Which thread, and how many of its newest messages, {@link ThreadTurnReader.loadThreadForTurn} reads. */
636
+ interface ThreadTurnQuery {
637
+ threadId: string;
638
+ /** Omitted reads every message; `0` reads none. */
639
+ messageLimit?: number;
640
+ }
641
+ /** What a turn reads off a thread — a bounded window, not the transcript. */
642
+ interface ThreadTurnPage {
643
+ title: string;
644
+ defaultAgent: string | null;
645
+ /** Whether the THREAD has ever been answered, not whether {@link messages} holds an answer. */
646
+ hasAssistantMessage: boolean;
647
+ /** Oldest first, carrying only the fields a model turn reads. */
648
+ messages: StoredMessage[];
649
+ }
650
+ /**
651
+ * A store that can hand a turn the WINDOW it is about to send, instead of the thread's transcript.
652
+ *
653
+ * {@link AgentStore.getThread} materializes every message row, every attachment and every tool
654
+ * output a thread ever recorded, and the run then journals what it loaded — so a long thread pays
655
+ * for its whole history on every turn and again on every replay, to send a prompt bounded to its
656
+ * last few messages. This read is bounded by the database (`order by created_at desc limit ?`),
657
+ * projected to the columns a model turn actually reads.
658
+ *
659
+ * `hasAssistantMessage` is answered over the WHOLE thread, never the page: it answers "has this
660
+ * conversation been answered before?" — what a `thread-start` intake asks — and a thread whose window
661
+ * happens to hold only the user's last questions has still been answered. `null` for a thread that is
662
+ * unknown or soft-deleted, matching `getThread`.
663
+ *
664
+ * Probed STRUCTURALLY rather than declared on {@link AgentStore}, the same way `defaultAgentForThread`
665
+ * is: it is an optimization a store either offers or does not, and one that predates it still answers
666
+ * correctly through the full read.
667
+ */
668
+ interface ThreadTurnReader {
669
+ loadThreadForTurn(query: ThreadTurnQuery): Promise<ThreadTurnPage | null>;
670
+ }
671
+ /** ORM-agnostic persistence. Refs are string ids; adapters may add real relations. */
672
+ interface AgentStore {
673
+ createThread(input: CreateThreadInput): Promise<ThreadSummary>;
674
+ getThread(threadId: string): Promise<ThreadDetail | null>;
675
+ listThreads(actorRef: string, limit?: number): Promise<ThreadSummary[]>;
676
+ softDeleteThread(threadId: string): Promise<void>;
677
+ forkThread(threadId: string, fromMessageId: string): Promise<ThreadSummary>;
678
+ setTitle(threadId: string, title: string): Promise<void>;
679
+ /**
680
+ * Promote a transient thread to a persistent one so it shows up in {@link listThreads}. A
681
+ * transient thread is a scratch conversation the caller has not chosen to keep; "saving" it
682
+ * clears the flag. Idempotent — promoting an already-persistent thread is a no-op.
683
+ */
684
+ promoteThread(threadId: string): Promise<void>;
685
+ setActiveStream(threadId: string, runId: string | null): Promise<void>;
686
+ /**
687
+ * OPTIONAL: rename a thread and/or set its default agent in one write. Absent on a store that
688
+ * predates this — `setTitle` still covers title-only edits, so nothing else in the lib requires
689
+ * this method; the REST `PATCH /threads/:id` endpoint responds 501 for a `defaultAgent` change
690
+ * against a store that lacks it.
691
+ */
692
+ updateThread?(threadId: string, patch: UpdateThreadInput): Promise<void>;
693
+ /**
694
+ * OPTIONAL: the runId of a currently-running turn on this thread, or `null` if none is running.
695
+ * Lets a client that reconnects (page refresh) discover a run to reattach to via the existing
696
+ * `GET /chat/:runId/stream`, instead of only being told about a run right after starting it.
697
+ * Absent on a store that predates this — thread read/list payloads report `activeRunId: null`.
698
+ */
699
+ activeRunForThread?(threadId: string): Promise<string | null>;
700
+ /**
701
+ * OPTIONAL: persist the start of a run (turn). Replay-safe: called under a durable localStep.
702
+ * Absent on a store that predates run recording — reliability metrics degrade to zeros/empty.
703
+ */
704
+ recordRunStart?(run: RecordRunStartInput): Promise<void>;
705
+ /** OPTIONAL: settle a run's outcome. `errorCode`/`errorMessage` only when status is 'failed'. */
706
+ recordRunEnd?(end: RecordRunEndInput): Promise<void>;
707
+ /** OPTIONAL: bump the run's llm-step retry counter (dispatched-step attempt > 1). */
708
+ bumpRunRetries?(runId: string): Promise<void>;
709
+ /**
710
+ * The `actorRef` that owns a thread, or `null` if no such thread exists. The authorization seam
711
+ * for thread-scoped endpoints (detail / delete / fork): the service compares this against the
712
+ * resolved caller before acting, so one actor can never read or mutate another's thread.
713
+ */
714
+ ownerOfThread(threadId: string): Promise<string | null>;
715
+ /**
716
+ * The `actorRef` that owns the thread a tool call belongs to, or `null` if the call is unknown.
717
+ * The authorization seam for HITL approve / reject: the caller must own the run they approve.
718
+ */
719
+ ownerOfToolCall(toolCallId: string): Promise<string | null>;
720
+ /**
721
+ * The run awaiting a decision on `toolCallId`: the call's OWN `runId` when the row carries one,
722
+ * else the thread's `activeStreamId`. Both HITL approve/reject and an elicitation answer route
723
+ * through this, derived server-side from the tool call alone — so a decision reaches the exact run
724
+ * awaiting it, including a sub-agent's own child run, which the client never sees and could not
725
+ * name. No client-supplied runId is trusted (or needed).
726
+ *
727
+ * The row's own runId comes FIRST because `activeStreamId` names whichever run is streaming the
728
+ * thread right now, and that is only the same run while a thread holds exactly one. The fallback
729
+ * is for rows written before tool calls recorded a runId, which have nothing else to answer with.
730
+ */
731
+ runForToolCall(toolCallId: string): Promise<string | null>;
732
+ /**
733
+ * The `actorRef` that owns the thread currently streaming `runId` (its `activeStreamId`), or
734
+ * `null` if no thread is streaming it. The authorization seam for `cancel`: the caller must own
735
+ * the run they abort. Resolvable during the live window (a run cancel only matters while active).
736
+ */
737
+ ownerOfActiveStream(runId: string): Promise<string | null>;
738
+ appendMessage(input: AppendMessageInput): Promise<StoredMessage>;
739
+ /**
740
+ * Attach a turn's settled tool RESULTS to a message that was already appended, replacing whatever
741
+ * it held. A message's tool calls are known when it is written and their outputs are not, but a
742
+ * thread reader pairs the two off THAT MESSAGE — so an output that only ever reaches the tool-call
743
+ * table leaves every call on a reopened thread looking like a tool still running.
744
+ *
745
+ * Required rather than optional: a store that silently declines this renders a finished turn as a
746
+ * permanently in-flight one, with nothing logged and nothing to notice. A missing method should
747
+ * fail to compile instead.
748
+ */
749
+ setMessageToolResults(messageId: string, results: ToolResult[]): Promise<void>;
750
+ /**
751
+ * Replace the components persisted on an already-appended message (see {@link StoredMessage.ui}).
752
+ * The loop calls it once per step, after the step's tools ran, with every component the step
753
+ * showed — the model turn's own `ui` frames first, then what its tools pushed through
754
+ * `ctx.emitUi`, in call order, deduplicated by `id`. A full replacement, never an append, so a
755
+ * repeated call writes the same value.
756
+ *
757
+ * OPTIONAL: a store without it still streams tool-pushed components live; a reload then shows
758
+ * only the ones the model turn itself produced.
759
+ */
760
+ setMessageUi?(messageId: string, ui: AgentUiComponent[]): Promise<void>;
761
+ truncateFrom(threadId: string, messageId: string): Promise<void>;
762
+ /**
763
+ * OPTIONAL: the thread a message belongs to, or `null` when there is no such message. The
764
+ * authorization seam for message-scoped routes (feedback): the service resolves the thread's owner
765
+ * from it. Implement it together with {@link setMessageFeedback}.
766
+ */
767
+ threadOfMessage?(messageId: string): Promise<string | null>;
768
+ /**
769
+ * OPTIONAL: set (or, with `null`, clear) the rating on a message — see
770
+ * {@link import('../types.js').StoredMessage.feedback}. Absent → `POST /messages/:id/feedback`
771
+ * answers `501`.
772
+ */
773
+ setMessageFeedback?(messageId: string, feedback: MessageFeedback | null): Promise<void>;
774
+ recordToolCall(input: RecordToolCallInput): Promise<void>;
775
+ updateToolCall(input: UpdateToolCallInput): Promise<void>;
776
+ /**
777
+ * OPTIONAL: the names of the tools whose approval someone asked to REMEMBER in this thread — an
778
+ * approved call persisted with `remember: true`. The loop approves a later call of one of them
779
+ * without asking, inside that call's own `persist:toolcall` checkpoint. Absent → nothing is ever
780
+ * remembered, and every call asks.
781
+ */
782
+ rememberedApprovals?(threadId: string): Promise<string[]>;
783
+ /**
784
+ * OPTIONAL: a call's approval state, or `null` when the call is unknown. Read by the approve/reject
785
+ * routes to enforce the recorded approver and refuse a decision on a request that already
786
+ * expired. Absent → every call is treated as the requester's, with no expiry (the old behaviour).
787
+ */
788
+ toolCallApproval?(toolCallId: string): Promise<ToolCallApprovalState | null>;
789
+ /**
790
+ * OPTIONAL: the input a call was recorded with, or `null` when the call is unknown. Read by the
791
+ * answer route to check a reply against the questions it answers (a typed question's rules, a
792
+ * `required` one left empty) before it is signalled. Absent → answers are checked for shape only,
793
+ * and the loop drops what it cannot settle.
794
+ */
795
+ toolCallInput?(toolCallId: string): Promise<unknown>;
796
+ /**
797
+ * OPTIONAL: of `mediaIds`, the ones a message that still exists — in a thread owned by
798
+ * `actorRef` — still carries as an attachment. The inverse of
799
+ * {@link import('./attachment-staging.js').AttachmentStagingStore.list}: the host can enumerate
800
+ * the media it staged but cannot see a transcript, and this side sees every transcript but never
801
+ * holds the bytes, so neither can decide alone what is safe to delete.
802
+ *
803
+ * DERIVED, not tracked. A reference is not permanent: `truncateFrom` deletes messages — which is
804
+ * exactly what regenerating a turn does — so media that was referenced becomes unreferenced
805
+ * again. A flag set when a message is sent would never be unset by that delete, and the bytes
806
+ * would be pinned for ever with nothing pointing at them. Answering from the surviving message
807
+ * rows on every call is the only form of this that stays true after a truncation.
808
+ *
809
+ * Scoped to one actor, like every other read on this surface: media referenced only by ANOTHER
810
+ * actor's thread is reported unreferenced here, so this can never be turned into a probe for what
811
+ * exists in someone else's conversation. A host pairs it with its own per-actor inventory, so the
812
+ * candidate ids are already the caller's own.
813
+ *
814
+ * Returns each id at most once, in the order asked. Absent on a store that predates this — a
815
+ * caller must treat the absence as "cannot answer" and collect NOTHING, never as "nothing is
816
+ * referenced", which would delete every attachment the actor ever sent.
817
+ */
818
+ referencedMediaIds?(actorRef: string, mediaIds: readonly string[]): Promise<string[]>;
819
+ recordUsage(input: RecordUsageInput): Promise<void>;
820
+ /**
821
+ * The actor's spend for `day` (UTC): total tokens plus the summed provider-reported USD cost.
822
+ * `costUsd` is `0` when no turn on that day reported a cost (token-only providers). Feeds both
823
+ * quota enforcement (via {@link QuotaStore}) and the quota-today view.
824
+ */
825
+ quotaToday(actorRef: string, day: string): Promise<{
826
+ usedTokens: number;
827
+ costUsd: number;
828
+ }>;
829
+ /**
830
+ * OPTIONAL: the actor's spend over the UTC days `fromDay`..`toDay` (`YYYY-MM-DD`, inclusive) —
831
+ * {@link quotaToday} over a range. Feeds the monthly window of `GET <base>/quota`; absent → that
832
+ * window is left out.
833
+ */
834
+ usageBetween?(actorRef: string, fromDay: string, toDay: string): Promise<{
835
+ usedTokens: number;
836
+ costUsd: number;
837
+ }>;
838
+ }
839
+
840
+ /**
841
+ * Why a thread's queue stopped draining. A paused queue keeps its messages; nothing starts until
842
+ * someone resumes it (`POST <base>/threads/:id/queue/resume`, `chat.queue.resume()`).
843
+ *
844
+ * - `run_failed` the turn before it failed — the next message would likely fail the same way,
845
+ * and the person should see the error before more of their messages are spent.
846
+ * - `cancelled` someone pressed Stop. Stop means stop: the queue does not start behind it.
847
+ * An interrupt (`mode: 'interrupt'`) is the exception — its message starts.
848
+ * - `quota_exceeded` the next message's actor is over budget; resuming later retries the check.
849
+ * - `start_failed` the next message could not be started (the runner refused it).
850
+ */
851
+ type QueuePauseReason = 'run_failed' | 'cancelled' | 'quota_exceeded' | 'start_failed';
852
+ interface QueuePause {
853
+ reason: QueuePauseReason;
854
+ /** Human-facing detail (the failure, the quota window), when there is one. */
855
+ message?: string;
856
+ /** ISO-8601 instant the queue paused. */
857
+ at: string;
858
+ }
859
+ /**
860
+ * A message a person sent while a turn was still running on the thread, waiting its turn. It is not
861
+ * part of the transcript until it starts: the loop appends it as the turn's user message, exactly as
862
+ * if it had been sent then. Everything a send carries is captured at enqueue time, resolved and
863
+ * checked there (agent, model, attachments), so a queued message starts with no request around it.
864
+ */
865
+ interface QueuedMessage {
866
+ id: string;
867
+ threadId: string;
868
+ /** Who queued it. The turn it starts runs as this actor. */
869
+ actor: Actor;
870
+ content: string;
871
+ attachments?: MessageAttachment[];
872
+ agentName?: string;
873
+ model?: string;
874
+ pageContext?: PageContext;
875
+ /**
876
+ * Queued by an interrupt (`POST chat { mode: 'interrupt' }`): the running turn was cancelled to
877
+ * make room for it, so the cancel starts it instead of pausing the queue.
878
+ */
879
+ interrupt?: boolean;
880
+ createdAt: string;
881
+ updatedAt: string;
882
+ }
883
+ /** A queued message as the wire shows it — {@link QueuedMessage} without the actor or thread. */
884
+ interface QueuedMessageView {
885
+ id: string;
886
+ content: string;
887
+ attachments?: MessageAttachment[];
888
+ agentName?: string;
889
+ model?: string;
890
+ interrupt?: boolean;
891
+ createdAt: string;
892
+ updatedAt: string;
893
+ }
894
+ /**
895
+ * A thread's queue: its waiting messages in the order they will run, and whether it is draining.
896
+ * What `GET <base>/threads/:id/queue` answers, what `ThreadDetail.queue` carries, and what every
897
+ * `queue` stream frame snapshots.
898
+ */
899
+ interface ChatQueueState {
900
+ items: QueuedMessageView[];
901
+ paused: QueuePause | null;
902
+ }
903
+ interface EnqueueMessageInput {
904
+ threadId: string;
905
+ actor: Actor;
906
+ content: string;
907
+ attachments?: MessageAttachment[];
908
+ agentName?: string;
909
+ model?: string;
910
+ pageContext?: PageContext;
911
+ interrupt?: boolean;
912
+ /** `'tail'` (default) runs it after everything already waiting; `'head'` runs it next. */
913
+ at?: 'tail' | 'head';
914
+ }
915
+ /**
916
+ * What a store may change on a waiting message (`PATCH <base>/queue/:messageId` sets the text and
917
+ * attachments). An omitted key leaves the field as it is.
918
+ */
919
+ interface QueuedMessagePatch {
920
+ content?: string;
921
+ /** `null` drops every attachment. */
922
+ attachments?: MessageAttachment[] | null;
923
+ /**
924
+ * Mark (or unmark) the message as an interrupt — what `POST <base>/queue/:messageId/interrupt`
925
+ * does to a message that is already waiting, before it cancels the running turn for it.
926
+ */
927
+ interrupt?: boolean;
928
+ }
929
+ /**
930
+ * A store that can hold a thread's queue of waiting messages, and admit ONE run per thread at a time.
931
+ *
932
+ * Probed STRUCTURALLY ({@link isChatQueueStore}), like `ThreadTurnReader`: a store that predates it
933
+ * keeps the old behaviour (a send on a busy thread starts a second, concurrent run). All members or
934
+ * none — a queue without the admission primitives cannot be drained safely across processes.
935
+ *
936
+ * Admission is the thread's `activeStreamId`, compare-and-set:
937
+ * - {@link claimActiveStream} sets it to `runId` only when it is free, already `runId` (a retried
938
+ * claim is idempotent), or held by `replacing` (a handoff from the run that is settling, or a
939
+ * takeover from a holder the runner reports dead). Exactly one of two racing claims wins.
940
+ * - {@link releaseActiveStream} clears it only when `runId` still holds it — so a run that settles
941
+ * after it handed the thread to the next one cannot clear the next one's claim.
942
+ */
943
+ interface ChatQueueStore {
944
+ enqueueMessage(input: EnqueueMessageInput): Promise<QueuedMessage>;
945
+ /** The thread's waiting messages, in run order (the head first). */
946
+ listQueue(threadId: string): Promise<QueuedMessage[]>;
947
+ getQueuedMessage(id: string): Promise<QueuedMessage | null>;
948
+ /** `null` when there is no such message (it started, or was removed). */
949
+ updateQueuedMessage(id: string, patch: QueuedMessagePatch): Promise<QueuedMessage | null>;
950
+ /** Move a message to `index` (clamped) in its thread's run order. `false` when it is gone. */
951
+ moveQueuedMessage(id: string, index: number): Promise<boolean>;
952
+ /**
953
+ * Remove a message. `false` when it was already gone — which is also how a drain learns that the
954
+ * head it was about to start was deleted under it, so it must be a real conditional delete.
955
+ */
956
+ removeQueuedMessage(id: string): Promise<boolean>;
957
+ /** Remove every waiting message on the thread; answers how many there were. */
958
+ clearQueue(threadId: string): Promise<number>;
959
+ queuePause(threadId: string): Promise<QueuePause | null>;
960
+ setQueuePause(threadId: string, pause: QueuePause | null): Promise<void>;
961
+ /** The run currently holding the thread, or `null`. */
962
+ activeRunForThread(threadId: string): Promise<string | null>;
963
+ claimActiveStream(threadId: string, runId: string, options?: {
964
+ replacing?: string;
965
+ }): Promise<boolean>;
966
+ releaseActiveStream(threadId: string, runId: string): Promise<boolean>;
967
+ }
968
+ /** Whether `store` implements every {@link ChatQueueStore} member. */
969
+ declare function isChatQueueStore(store: AgentStore): store is AgentStore & ChatQueueStore;
970
+ /**
971
+ * Release the thread `runId` was streaming — conditionally when the store can compare, so a run
972
+ * that already handed the thread to the next queued turn does not clear that turn's claim; the old
973
+ * unconditional clear on a store that cannot.
974
+ */
975
+ declare function releaseThreadRun(store: AgentStore, threadId: string, runId: string): Promise<void>;
976
+ /** The wire view of a queued message. */
977
+ declare function queuedMessageView(message: QueuedMessage): QueuedMessageView;
978
+
979
+ /**
980
+ * The ceiling on how much of a thread rides into a turn. Without one, `runAgentLoop` maps EVERY
981
+ * message the store returns into the model's messages, so a long-lived thread grows until the
982
+ * provider rejects the request — and every turn before that one pays for the whole transcript.
983
+ * `@dudousxd/nestjs-agent-core` ships `windowHistory` as the built-in; anything satisfying this SPI
984
+ * works. Wire it as `AgentLoopDeps.historyPolicy`, or via `AgentModule.forRoot({ history })` /
985
+ * `@Agent({ history })`.
986
+ */
987
+
988
+ /** Whose history this is — enough for a policy to size the window per agent or per actor. */
989
+ interface HistoryPolicyContext {
990
+ threadId: string;
991
+ actor: Actor;
992
+ /** The agent running this turn. Undefined → the default agent. */
993
+ agentName?: string;
994
+ }
995
+ /** How a policy split the thread: what rides into the turn, and what the ceiling left out. */
996
+ interface HistorySelection {
997
+ /** Sent to the model, oldest-first. */
998
+ keep: ModelMessage[];
999
+ /** Left out, oldest-first. Folded into a leading summary when the policy implements `summarize`. */
1000
+ drop: ModelMessage[];
1001
+ }
1002
+ /** A stand-in for the messages a window left out, plus what producing it cost. */
1003
+ interface HistorySummary {
1004
+ /** Prose the loop folds into the window as a leading `system` message. */
1005
+ text: string;
1006
+ /**
1007
+ * What the summarizer spent, when it called a model. Recorded as a `history_summary` usage row, so
1008
+ * a ceiling on context cost cannot itself become spend nothing accounts for. Omit for a summarizer
1009
+ * that calls no model (a rollup of tool names, a digest the app already stored).
1010
+ */
1011
+ usage?: MessageUsage;
1012
+ /** Accounting label for the model that produced it; falls back to `AgentLoopDeps.modelId`. */
1013
+ modelId?: string;
1014
+ }
1015
+ interface HistoryPolicy {
1016
+ /**
1017
+ * The most messages {@link select} can ever keep, where the ceiling can be stated as a row count.
1018
+ *
1019
+ * A hint the loop hands to the store, which then reads that many of the thread's newest rows
1020
+ * rather than its whole transcript (see `ThreadTurnReader`). Declaring it is a PROMISE about
1021
+ * `select`: that it keeps at most this many messages, and that they are the NEWEST ones — so a
1022
+ * window of this size is indistinguishable, to `select`, from the full transcript. A policy that
1023
+ * can keep more than this, or can keep something older than the newest `maxMessages`, must omit it
1024
+ * rather than select over rows the store was never asked for.
1025
+ *
1026
+ * Omit where the ceiling is not a row count at all — a token budget alone cannot name one, since
1027
+ * one message can be four tokens or forty thousand. Omitting costs only the bound on the READ; the
1028
+ * prompt is identical either way.
1029
+ */
1030
+ readonly maxMessages?: number;
1031
+ /**
1032
+ * Decide what the model sees.
1033
+ *
1034
+ * MUST be a pure function of `messages`. The loop calls it INSIDE the `load:thread` checkpoint and
1035
+ * records its result there, so the ceiling bounds the journal as well as the prompt: what the
1036
+ * checkpoint holds is the selection rather than the store's whole `ThreadDetail`, on a payload
1037
+ * every replay re-reads. It still adds no position of its own — a policy cannot change the name or
1038
+ * position of a single existing checkpoint.
1039
+ *
1040
+ * Read a clock, a feature flag or a database here and the resumed run windows differently from the
1041
+ * one that suspended: the model gets a different prompt, and any step whose existence depends on
1042
+ * the split lands at a position the history has no room for. Anything non-deterministic belongs in
1043
+ * {@link summarize}, which has a checkpoint of its own.
1044
+ *
1045
+ * The newest message must always be in `keep` — dropping it leaves the turn with nothing to answer.
1046
+ */
1047
+ select(messages: ModelMessage[], ctx: HistoryPolicyContext): HistorySelection;
1048
+ /**
1049
+ * Fold the dropped messages into prose the model reads in their place, prepended to the window as
1050
+ * a `system` message. Optional — without it, dropped messages are simply gone, and `load:thread`
1051
+ * does not record them either: a summarizer is the only thing that ever reads them back.
1052
+ *
1053
+ * Runs inside the loop's `history:summarize` checkpoint, so it may call a model or hit the
1054
+ * network: the first attempt's result is journaled and every replay reads it back instead of
1055
+ * re-summarizing. It runs once per RUN (not per model step), and only when `select` actually
1056
+ * dropped something.
1057
+ */
1058
+ summarize?(dropped: ModelMessage[], ctx: HistoryPolicyContext): Promise<HistorySummary>;
1059
+ }
1060
+ /**
1061
+ * The window an agent asks for declaratively — `AgentModule.forRoot({ history })` and
1062
+ * `@Agent({ history })`. Plain data, so it can live in a decorator's metadata; the NestJS layer
1063
+ * turns it into a `windowHistory` policy. A consumer needing anything the window can't express
1064
+ * supplies a {@link HistoryPolicy} instead.
1065
+ */
1066
+ interface AgentHistoryWindow {
1067
+ /** Keep at most this many of the newest messages. */
1068
+ maxMessages?: number;
1069
+ /** Keep the newest messages whose estimated tokens fit this budget. */
1070
+ maxTokens?: number;
1071
+ /** Fold what the window left out into a leading summary — one extra model call per run. */
1072
+ summarize?: boolean;
1073
+ }
1074
+
594
1075
  /**
595
1076
  * How a person-facing surface talks about a tool WITHOUT ever naming it. Declared on the server,
596
1077
  * beside the tool's input schema, because whoever changes the input is the one who has to re-word
@@ -1295,6 +1776,12 @@ interface ToolCallApproval {
1295
1776
  }
1296
1777
  interface ThreadDetail extends ThreadSummary {
1297
1778
  messages: StoredMessage[];
1779
+ /**
1780
+ * Messages sent while a turn was running, waiting to run after it, and whether the queue is
1781
+ * draining. Present when the store supports a queue (`ChatQueueStore`); the REST read-model
1782
+ * omits it otherwise.
1783
+ */
1784
+ queue?: ChatQueueState;
1298
1785
  }
1299
1786
  type ToolCallStatus = 'auto_executed' | 'pending_approval' | 'executed' | 'rejected' | 'failed'
1300
1787
  /** An approval request lapsed before anyone decided; the tool never ran. */
@@ -1485,4 +1972,4 @@ interface ToolDescription {
1485
1972
  inputSchema?: StandardSchemaV1;
1486
1973
  }
1487
1974
 
1488
- export { type ElicitationInputType as $, type Actor as A, type ToolStepEnvelope as B, ALL_AGENTS as C, type DetachedDelivery as D, type ElicitationRequest as E, ASK_TOOL_DESCRIPTION as F, ASK_TOOL_NAME as G, type HumanReply as H, type AgentApprovalRequest as I, type AgentApprovalSettlement as J, type AgentAttachmentConfig as K, type LlmStepEnvelope as L, type ModelMessage as M, type AgentCatalogEntry as N, type AgentClientConfig as O, type PageContext as P, type QuotaState as Q, type AgentHistoryWindow as R, type StoredMessage as S, type ToolSpec as T, type UsagePurpose as U, type AskToolInput as V, DEFAULT_INTAKE_PREAMBLE as W, DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS as X, DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS as Y, ELICITATION_INPUT_TYPES as Z, type ElicitationInput as _, type ToolHandler as a, type ElicitationOption as a0, type ElicitationOutcome as a1, type ElicitationQuestion as a2, type ElicitationReply as a3, type ElicitationResult as a4, type HistoryPolicyContext as a5, type HistorySelection as a6, type HistorySummary as a7, type InvokeWithTransientRetryOptions as a8, MAX_ASK_QUESTIONS as a9, resolveElicitation as aA, resolveToolTransientRetryNumbers as aB, settleElicitation as aC, validateElicitationAnswer as aD, validateElicitationValue as aE, type MessageFeedbackValue as aa, type MessageRole as ab, type PromptContext as ac, type QuotaView as ad, type ToolCallApprovalStatus as ae, type ToolCatalogEntry as af, type ToolConfirmation as ag, type ToolDescription as ah, type ToolPresentationTone as ai, type ToolResultField as aj, type ToolResultView as ak, type ToolStepCtx as al, type ToolTransientRetryNumbers as am, type ToolTransientRetryOptions as an, askInputSchema as ao, askToolDefinition as ap, decodeStreamEvent as aq, encodeStreamEvent as ar, invokeWithTransientRetry as as, isTransientToolError as at, isTypedQuestion as au, normalizeElicitationReply as av, questionOptions as aw, readElicitationInput as ax, readElicitationQuestions as ay, renderElicitationAnswers as az, type ToolPresentation as b, type ToolDefinition as c, type ToolCallRequest as d, type MessageUsage as e, type AgentUiComponent as f, type AiToolCtx as g, type AgentStreamEvent as h, type ThreadSummary as i, type ThreadDetail as j, type ToolResult as k, type MessageAttachment as l, type MessageFeedback as m, type ToolCallStatus as n, type AgentRunInput as o, type ToolKind as p, type ToolCallApproval as q, type HistoryPolicy as r, type AgentDefinition as s, type AgentDelegation as t, type ToolDescribeScope as u, type PromptBuilder as v, type PromptContributor as w, type ToolTransientRetrySetting as x, type AgentIntake as y, type Decision as z };
1975
+ export { ASK_TOOL_NAME as $, type Actor as A, type ThreadDetail as B, type ChatQueueStore as C, type DetachedDelivery as D, type ElicitationRequest as E, type ToolCallApprovalState as F, type EnqueueMessageInput as G, type HumanReply as H, type QueuedMessage as I, type QueuedMessagePatch as J, type QueuePause as K, type LlmStepEnvelope as L, type ModelMessage as M, type AppendMessageInput as N, type ToolResult as O, type PageContext as P, type QuotaState as Q, type RecordRunStartInput as R, type StoredMessage as S, type ToolSpec as T, type UpdateThreadInput as U, type MessageFeedback as V, type RecordToolCallInput as W, type UpdateToolCallInput as X, type RecordUsageInput as Y, ALL_AGENTS as Z, ASK_TOOL_DESCRIPTION as _, type ToolHandler as a, validateElicitationAnswer as a$, type AgentApprovalRequest as a0, type AgentApprovalSettlement as a1, type AgentAttachmentConfig as a2, type AgentCatalogEntry as a3, type AgentClientConfig as a4, type AgentHistoryWindow as a5, type AskToolInput as a6, type ChatQueueState as a7, DEFAULT_INTAKE_PREAMBLE as a8, DEFAULT_TOOL_TRANSIENT_RETRY_ATTEMPTS as a9, type ToolConfirmation as aA, type ToolDescription as aB, type ToolPresentationTone as aC, type ToolResultField as aD, type ToolResultView as aE, type ToolStepCtx as aF, type ToolTransientRetryNumbers as aG, type ToolTransientRetryOptions as aH, type UsagePurpose as aI, askInputSchema as aJ, askToolDefinition as aK, decodeStreamEvent as aL, encodeStreamEvent as aM, invokeWithTransientRetry as aN, isChatQueueStore as aO, isTransientToolError as aP, isTypedQuestion as aQ, normalizeElicitationReply as aR, questionOptions as aS, queuedMessageView as aT, readElicitationInput as aU, readElicitationQuestions as aV, releaseThreadRun as aW, renderElicitationAnswers as aX, resolveElicitation as aY, resolveToolTransientRetryNumbers as aZ, settleElicitation as a_, DEFAULT_TOOL_TRANSIENT_RETRY_BACKOFF_MS as aa, ELICITATION_INPUT_TYPES as ab, type ElicitationInput as ac, type ElicitationInputType as ad, type ElicitationOption as ae, type ElicitationOutcome as af, type ElicitationQuestion as ag, type ElicitationReply as ah, type ElicitationResult as ai, type HistoryPolicyContext as aj, type HistorySelection as ak, type HistorySummary as al, type InvokeWithTransientRetryOptions as am, MAX_ASK_QUESTIONS as an, type MessageFeedbackValue as ao, type MessageRole as ap, type PromptContext as aq, type QueuePauseReason as ar, type QueuedMessageView as as, type QuotaView as at, type RecordRunEndInput as au, type ThreadTurnPage as av, type ThreadTurnQuery as aw, type ThreadTurnReader as ax, type ToolCallApprovalStatus as ay, type ToolCatalogEntry as az, type ToolPresentation as b, validateElicitationValue as b0, type ToolDefinition as c, type ToolCallRequest as d, type MessageUsage as e, type AgentUiComponent as f, type AiToolCtx as g, type AgentStreamEvent as h, type AgentRunInput as i, type MessageAttachment as j, type ToolKind as k, type ToolCallStatus as l, type ToolCallApproval as m, type HistoryPolicy as n, type AgentDefinition as o, type AgentDelegation as p, type AgentStore as q, type ToolDescribeScope as r, type PromptBuilder as s, type PromptContributor as t, type ToolTransientRetrySetting as u, type AgentIntake as v, type Decision as w, type ToolStepEnvelope as x, type CreateThreadInput as y, type ThreadSummary as z };