experimental-a2 0.12.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +47 -0
- package/dist/{actor-DJi3RsNu.d.ts → actor-BfQSE0KC.d.ts} +4 -4
- package/dist/{actor-DJi3RsNu.d.ts.map → actor-BfQSE0KC.d.ts.map} +1 -1
- package/dist/actor-client.d.ts +1 -1
- package/dist/actor-client.js +1 -1
- package/dist/actor-react.d.ts +3 -3
- package/dist/actor-react.js +2 -2
- package/dist/{actor-shared-DI7J5upy.js → actor-shared-B5tJfzt-.js} +2 -2
- package/dist/{actor-shared-DI7J5upy.js.map → actor-shared-B5tJfzt-.js.map} +1 -1
- package/dist/actor.d.ts +1 -1
- package/dist/actor.js +3 -3
- package/dist/ai-Cai-lCbj.d.ts +580 -0
- package/dist/ai-Cai-lCbj.d.ts.map +1 -0
- package/dist/ai-control-CcD4hh3y.js +119 -0
- package/dist/ai-control-CcD4hh3y.js.map +1 -0
- package/dist/ai-server.d.ts +16 -8
- package/dist/ai-server.d.ts.map +1 -1
- package/dist/ai-server.js +1322 -475
- package/dist/ai-server.js.map +1 -1
- package/dist/ai.d.ts +2 -334
- package/dist/ai.js +838 -85
- package/dist/ai.js.map +1 -1
- package/dist/{client-P_NNNRM-.d.ts → client-BAEABRZB.d.ts} +2 -2
- package/dist/{client-P_NNNRM-.d.ts.map → client-BAEABRZB.d.ts.map} +1 -1
- package/dist/{client-Bf6uSEAk.js → client-BYzHjkwU.js} +21 -6
- package/dist/client-BYzHjkwU.js.map +1 -0
- package/dist/client.d.ts +1 -1
- package/dist/client.js +1 -1
- package/dist/{contract-48bUMgcL.js → contract-CKRg_E4q.js} +3 -26
- package/dist/contract-CKRg_E4q.js.map +1 -0
- package/dist/index.d.ts +2 -2
- package/dist/index.js +1 -1
- package/dist/react.d.ts +2 -2
- package/dist/react.js +1 -1
- package/dist/{reducer-DJKWm3cp.d.ts → reducer-BcS9VDKC.d.ts} +4 -1
- package/dist/{reducer-DJKWm3cp.d.ts.map → reducer-BcS9VDKC.d.ts.map} +1 -1
- package/dist/reducer-DEMjEY_O.js +29 -0
- package/dist/reducer-DEMjEY_O.js.map +1 -0
- package/dist/scheduler-qstash.d.ts +2 -2
- package/dist/scheduler-qstash.js +1 -1
- package/dist/scheduler-vercel.d.ts +2 -2
- package/dist/scheduler-vercel.js +1 -1
- package/dist/{server-DjZZa1wr.d.ts → server-Bp5Nd1pF.d.ts} +3 -3
- package/dist/{server-DjZZa1wr.d.ts.map → server-Bp5Nd1pF.d.ts.map} +1 -1
- package/dist/{server-BeNADlCI.js → server-CjJSGcF7.js} +4 -3
- package/dist/server-CjJSGcF7.js.map +1 -0
- package/dist/server.d.ts +3 -3
- package/dist/server.js +1 -1
- package/dist/{store-DtDOWLSn.d.ts → store-D_yhNdPz.d.ts} +7 -2
- package/dist/{store-DtDOWLSn.d.ts.map → store-D_yhNdPz.d.ts.map} +1 -1
- package/dist/store-N8PXxDAS.js.map +1 -1
- package/dist/store-memory.d.ts +1 -1
- package/dist/store-postgres.d.ts +1 -1
- package/dist/store-postgres.js +19 -0
- package/dist/store-postgres.js.map +1 -1
- package/dist/store-redis-http.d.ts +1 -1
- package/dist/store-redis-http.js +1 -1
- package/dist/{store-redis-notify-BUCyXOn0.js → store-redis-notify-D2EI6gwX.js} +27 -2
- package/dist/store-redis-notify-D2EI6gwX.js.map +1 -0
- package/dist/store-redis.d.ts +1 -1
- package/dist/store-redis.js +1 -1
- package/dist/store-sqlite.d.ts +1 -1
- package/docs/guides/06-ai-agents.mdx +361 -76
- package/docs/reference/01-api.mdx +159 -27
- package/examples/playground/app/agent/[agentId]/agent-client.tsx +145 -33
- package/examples/playground/app/agent/[agentId]/agent-queue.tsx +289 -0
- package/examples/playground/app/agent/[agentId]/compaction/route.ts +14 -0
- package/examples/playground/app/agent/[agentId]/compaction-event.tsx +38 -0
- package/examples/playground/app/agent/[agentId]/compaction-panel.tsx +294 -0
- package/examples/playground/app/agent/compaction-settings.test.ts +144 -0
- package/examples/playground/app/agent/compaction-settings.ts +49 -0
- package/examples/playground/app/agent/compaction-timeline.test.ts +337 -0
- package/examples/playground/app/agent/compaction-timeline.ts +198 -0
- package/examples/playground/app/agent/model.ts +56 -1
- package/examples/playground/app/agent/server.ts +9 -2
- package/examples/playground/app/chat/[chatId]/chat-client.tsx +3 -13
- package/examples/playground/app/chat/model.ts +2 -2
- package/examples/playground/app/chat/server.ts +24 -17
- package/examples/playground/app/globals.css +333 -0
- package/examples/playground/package.json +1 -1
- package/package.json +1 -1
- package/src/ai-client-state.ts +185 -0
- package/src/ai-control-server.ts +829 -0
- package/src/ai-control-state.ts +152 -0
- package/src/ai-control.ts +139 -0
- package/src/ai-coordinator.ts +99 -32
- package/src/ai-model-metadata.ts +108 -0
- package/src/ai-progress-batches.ts +68 -0
- package/src/ai-projector.ts +76 -15
- package/src/ai-sdk-step.ts +5 -2
- package/src/ai-server.ts +920 -638
- package/src/ai.ts +650 -110
- package/src/client.ts +31 -9
- package/src/licenses/Apache-2.0.txt +55 -0
- package/src/parse-partial-json.ts +441 -0
- package/src/reducer.ts +6 -0
- package/src/server.ts +8 -4
- package/src/store-postgres.ts +27 -0
- package/src/store-redis-core.ts +53 -1
- package/src/store-redis-notify.ts +1 -0
- package/src/store.ts +6 -0
- package/dist/ai.d.ts.map +0 -1
- package/dist/client-Bf6uSEAk.js.map +0 -1
- package/dist/contract-48bUMgcL.js.map +0 -1
- package/dist/server-BeNADlCI.js.map +0 -1
- package/dist/store-redis-notify-BUCyXOn0.js.map +0 -1
|
@@ -135,10 +135,10 @@ export const GET = assistantServer.fetch
|
|
|
135
135
|
export const POST = assistantServer.fetch
|
|
136
136
|
```
|
|
137
137
|
|
|
138
|
-
The route never calls the model directly. The browser appends
|
|
139
|
-
|
|
140
|
-
`ai.generation.requested
|
|
141
|
-
log.
|
|
138
|
+
The route never calls the model directly. The browser appends
|
|
139
|
+
`ai.control.requested` commands. The controller admits inputs and schedules
|
|
140
|
+
`ai.generation.requested`; workers run the AI SDK and append progress to the
|
|
141
|
+
same log.
|
|
142
142
|
|
|
143
143
|
## Bind the reducer to React
|
|
144
144
|
|
|
@@ -309,11 +309,9 @@ export function AgentClient() {
|
|
|
309
309
|
}
|
|
310
310
|
```
|
|
311
311
|
|
|
312
|
-
|
|
313
|
-
`
|
|
314
|
-
|
|
315
|
-
event and the component restores the draft. The server owns the corresponding
|
|
316
|
-
generation request.
|
|
312
|
+
`push()` persists the command. Its receipt confirms admission to the inbox.
|
|
313
|
+
`state.inbox.items` contains pending inputs; `state.messages` contains the
|
|
314
|
+
admitted conversation. If persistence fails, the component restores the draft.
|
|
317
315
|
|
|
318
316
|
The form gives Enter its normal submit behavior. The thinking row is also
|
|
319
317
|
derived from `AIState`: it appears with the Agent label when the server's
|
|
@@ -323,7 +321,7 @@ it. Empty stream-start messages never create a blank conversation row.
|
|
|
323
321
|
Users can send another message while a response is active. The message enters
|
|
324
322
|
the durable log immediately, but its turn waits until the active assistant
|
|
325
323
|
response, including every tool step and approval, reaches a terminal event.
|
|
326
|
-
|
|
324
|
+
Inbox order defaults to arrival order and can be edited explicitly.
|
|
327
325
|
|
|
328
326
|
Pass `{ generate: false }` when a user message should update the conversation
|
|
329
327
|
without starting a model turn:
|
|
@@ -343,7 +341,7 @@ export async function recordPassiveMessage(
|
|
|
343
341
|
}
|
|
344
342
|
```
|
|
345
343
|
|
|
346
|
-
The passive user message
|
|
344
|
+
The passive user message appears in `AIState.messages` when admitted. A later user
|
|
347
345
|
message with the default generation behavior includes passive user messages
|
|
348
346
|
before it in model context. Passive user messages after that trigger remain
|
|
349
347
|
outside its prompt and wait for the next generating message. This ordering
|
|
@@ -355,7 +353,10 @@ already queued turn.
|
|
|
355
353
|
One generating user interaction becomes a durable sequence:
|
|
356
354
|
|
|
357
355
|
```text
|
|
358
|
-
browser ai.
|
|
356
|
+
browser ai.control.requested durable input command
|
|
357
|
+
server ai.message.created input admitted
|
|
358
|
+
server ai.control.committed accepted state changes
|
|
359
|
+
server ai.control.decided command receipt
|
|
359
360
|
server ai.generation.requested scheduled from durable facts
|
|
360
361
|
server ai.generation.started
|
|
361
362
|
server ai.generation.progress batched UIMessageChunk[]
|
|
@@ -414,15 +415,26 @@ continuation that has not been appended yet.
|
|
|
414
415
|
|
|
415
416
|
| Input | Events |
|
|
416
417
|
| --- | --- |
|
|
417
|
-
| `inputs.message(message, { generate? })` |
|
|
418
|
+
| `inputs.message(message, { generate? })` | queues input; `false` queues passive context |
|
|
418
419
|
| `inputs.seed(message)` | records `generate: false`; non-user roles require a trusted append |
|
|
419
|
-
| `inputs.approval(response)` |
|
|
420
|
+
| `inputs.approval(response)` | requests a decision on a pending approval |
|
|
420
421
|
| `inputs.input(response)` | records an application input response fact |
|
|
421
422
|
| `inputs.requestInput(request)` | records a trusted server request for application input |
|
|
422
|
-
| `inputs.retry(options)` |
|
|
423
|
-
| `inputs.
|
|
424
|
-
|
|
425
|
-
|
|
423
|
+
| `inputs.retry(options)` | requests a fresh attempt for a failed response |
|
|
424
|
+
| `inputs.stop({ turnId })` | ends the observed turn and advances |
|
|
425
|
+
| `inputs.steer({ turnId, message })` | prioritizes input after the observed turn's current step finishes |
|
|
426
|
+
| `inputs.queue.edit({ message, expectedRevision })` | edits a pending input |
|
|
427
|
+
| `inputs.queue.move({ inputId, beforeId })` | reorders pending input; `null` means the end |
|
|
428
|
+
| `inputs.queue.remove({ inputId })` | removes pending input from the inbox |
|
|
429
|
+
| `inputs.queue.sendNow({ inputId, turnId })` | atomically prioritizes queued input and interrupts the observed turn |
|
|
430
|
+
| `inputs.pause({ when })` | pauses now or after the active turn |
|
|
431
|
+
| `inputs.resume()` | resumes retained work and inbox admission |
|
|
432
|
+
| `inputs.toolResult(result)` | records a trusted deferred provider-tool result |
|
|
433
|
+
| `inputs.interrupt(options)` | addresses a specific generation or request |
|
|
434
|
+
|
|
435
|
+
Builders return `ai.control.requested` commands. Message, approval, input, and
|
|
436
|
+
retry builders identify a stable interaction. Other controls use the event ID
|
|
437
|
+
assigned by `push()` or `append()`; transport retries preserve that ID.
|
|
426
438
|
Generation scheduling considers only user-role `ai.message.created` facts.
|
|
427
439
|
Assistant and system message facts remain context regardless of the `generate`
|
|
428
440
|
field, and model output is recorded as generation progress rather than a new
|
|
@@ -482,10 +494,9 @@ The AI SDK infers the tool input and output from its schemas and
|
|
|
482
494
|
scope carrying the current durable handler attempt. Read it with
|
|
483
495
|
`handlerContext(agent)` from `experimental-a2/ai`: it returns the same
|
|
484
496
|
typed context bag an event handler receives, with `event`, `attempt`,
|
|
485
|
-
`session`, and `signal`.
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
event. The agent argument carries the types; A2 verifies it against the
|
|
497
|
+
`session`, and `signal`. Tools execute an `ai.tool.execution.requested`
|
|
498
|
+
event after authorization. Its payload contains the admitted `call`, its
|
|
499
|
+
`generation`, and its turn and execution version. The agent argument carries the types; A2 verifies it against the
|
|
489
500
|
server executing the tool and throws when a tool written for one agent
|
|
490
501
|
runs under another, or when `handlerContext()` is called outside a tool
|
|
491
502
|
execution. The scope survives awaited helpers and async iteration, so
|
|
@@ -547,7 +558,7 @@ and plan.
|
|
|
547
558
|
The schedule name uses `toolCallId`, and the message id uses the durable
|
|
548
559
|
triggering event id. A retry therefore converges on the same timer and message,
|
|
549
560
|
even if a provider reuses tool-call ids in a later generation. When the timer
|
|
550
|
-
arrives, `inputs.message()`
|
|
561
|
+
arrives, `inputs.message()` queues a durable input command. The
|
|
551
562
|
queued-turn policy starts a fresh response after the active response completes
|
|
552
563
|
or is interrupted. A failed response must be retried or interrupted first.
|
|
553
564
|
If the scheduler send fails ambiguously or transiently, A2 retries the same
|
|
@@ -664,40 +675,15 @@ import { useSession } from '../session'
|
|
|
664
675
|
|
|
665
676
|
export function StopButton() {
|
|
666
677
|
const { state, push, index } = useSession()
|
|
667
|
-
const active = state.
|
|
668
|
-
|
|
669
|
-
state.activeRequestId && state.activeResponseMessageId
|
|
670
|
-
? {
|
|
671
|
-
messageId: state.activeResponseMessageId,
|
|
672
|
-
requestId: state.activeRequestId,
|
|
673
|
-
}
|
|
674
|
-
: null
|
|
675
|
-
const waiting =
|
|
676
|
-
state.pendingApprovals[0] ??
|
|
677
|
-
state.pendingInputs[0] ??
|
|
678
|
-
state.tools.find((tool) => tool.status === 'running')
|
|
679
|
-
const target = active
|
|
680
|
-
? {
|
|
681
|
-
messageId: active.responseMessageId,
|
|
682
|
-
generationId: active.generationId,
|
|
683
|
-
}
|
|
684
|
-
: requested ??
|
|
685
|
-
(waiting
|
|
686
|
-
? {
|
|
687
|
-
messageId: waiting.messageId,
|
|
688
|
-
generationId: waiting.generationId,
|
|
689
|
-
}
|
|
690
|
-
: null)
|
|
691
|
-
|
|
692
|
-
if (!target) return null
|
|
678
|
+
const active = state.active
|
|
679
|
+
if (!active) return null
|
|
693
680
|
|
|
694
681
|
return (
|
|
695
682
|
<button
|
|
696
683
|
onClick={() =>
|
|
697
684
|
void push(
|
|
698
|
-
...inputs.
|
|
699
|
-
|
|
700
|
-
reason: 'Stopped by the user',
|
|
685
|
+
...inputs.stop({
|
|
686
|
+
turnId: active.turnId,
|
|
701
687
|
lastSeenIndex: index,
|
|
702
688
|
}),
|
|
703
689
|
)
|
|
@@ -709,17 +695,20 @@ export function StopButton() {
|
|
|
709
695
|
}
|
|
710
696
|
```
|
|
711
697
|
|
|
712
|
-
The
|
|
713
|
-
channel. `lastSeenIndex` is the exact confirmed log frontier visible when the
|
|
698
|
+
The accepted command ends the turn and reaches A2's cancellation channel. `lastSeenIndex` is the exact confirmed log frontier visible when the
|
|
714
699
|
user clicked. The reducer rewinds the indexed generation projection to that
|
|
715
|
-
frontier
|
|
716
|
-
|
|
700
|
+
frontier and keeps the partial text the user actually saw. Tool parts whose
|
|
701
|
+
arguments are still streaming are drafts: interruption removes them from the
|
|
702
|
+
conversation and subsequent model prompts, even if their current JSON parses.
|
|
703
|
+
Their progress remains in the raw log. Accepted tools still awaiting completion
|
|
704
|
+
become `output-error`. An accepted interruption
|
|
717
705
|
terminally fences that request or generation. Later generation, tool, approval,
|
|
718
706
|
input, and compaction events remain in raw history but cannot reactivate it or
|
|
719
707
|
alter the projection.
|
|
720
708
|
Completed tool results at or before the visible frontier remain completed.
|
|
721
709
|
|
|
722
|
-
|
|
710
|
+
For the lower-level `inputs.interrupt()` builder, between
|
|
711
|
+
`ai.generation.requested` and `ai.generation.started`, copy
|
|
723
712
|
`activeRequestId` as `requestId` and `activeResponseMessageId` as `messageId`.
|
|
724
713
|
Once `activeGeneration` exists, send its `generationId` and
|
|
725
714
|
`responseMessageId` instead. A request-owned interruption remains valid if
|
|
@@ -762,35 +751,112 @@ interaction that began elsewhere.
|
|
|
762
751
|
|
|
763
752
|
## Compact long conversations
|
|
764
753
|
|
|
765
|
-
Compaction is
|
|
766
|
-
|
|
754
|
+
Compaction is automatic for supported Gateway models. A2 uses the active model
|
|
755
|
+
with the same instructions, tool definitions, and provider settings, and appends
|
|
756
|
+
an internal user message asking for a summary. This preserves the request prefix
|
|
757
|
+
for prompt caching when the provider has a matching cache entry.
|
|
767
758
|
|
|
768
759
|
```ts server/compacted.ts
|
|
769
|
-
import type { UIMessage } from 'ai'
|
|
770
760
|
import { createAgentServer } from 'experimental-a2/ai/server'
|
|
771
761
|
import { assistant } from '../assistant'
|
|
772
762
|
|
|
773
763
|
export const compactedAssistantServer = createAgentServer({
|
|
774
764
|
agent: assistant,
|
|
775
765
|
model: 'openai/gpt-5.6-terra',
|
|
776
|
-
compaction: {
|
|
777
|
-
shouldCompact: ({ messages }) => messages.length > 40,
|
|
778
|
-
compact: async ({ messages }) => {
|
|
779
|
-
const summary = {
|
|
780
|
-
id: crypto.randomUUID(),
|
|
781
|
-
role: 'user',
|
|
782
|
-
parts: [{ type: 'text', text: 'Summary of the earlier conversation.' }],
|
|
783
|
-
} satisfies UIMessage
|
|
784
|
-
|
|
785
|
-
return [summary, ...messages.slice(-10)]
|
|
786
|
-
},
|
|
787
|
-
},
|
|
788
766
|
})
|
|
789
767
|
```
|
|
790
768
|
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
769
|
+
On first use of each Gateway model in a session, A2 includes
|
|
770
|
+
`ai.model.metadata.requested` in the existing generation-start append. A separate
|
|
771
|
+
handler fetches the public Gateway model catalog and records the selected limits
|
|
772
|
+
in `ai.model.metadata.resolved`. It runs outside the AI turn lane. The first model
|
|
773
|
+
request proceeds immediately; metadata that arrives afterward is available to
|
|
774
|
+
later steps. Returning to a previously used model reuses its durable metadata.
|
|
775
|
+
The catalog is also cached in memory for one hour across agent servers, with one
|
|
776
|
+
shared in-flight request and a five-second timeout. A failed lookup follows A2's
|
|
777
|
+
handler retry policy without failing the model generation. A model absent from
|
|
778
|
+
the catalog is recorded as unavailable. A pending entry means no result has been
|
|
779
|
+
recorded, including when the metadata handler exhausts its retry budget. A2 does
|
|
780
|
+
not start a fresh lookup on every turn after exhaustion. Inspect the handler
|
|
781
|
+
failure or use an explicit threshold in that case.
|
|
782
|
+
|
|
783
|
+
The default threshold is 75% of the context window. A larger configured
|
|
784
|
+
`generation.maxOutputTokens` lowers the threshold further to reserve that output
|
|
785
|
+
allowance. This is an input-context policy, not a new output limit. Limits and
|
|
786
|
+
metadata status are available in `state.modelMetadata`.
|
|
787
|
+
|
|
788
|
+
`compaction: { thresholdTokens: 80_000 }` overrides discovery with an explicit
|
|
789
|
+
positive safe integer. `compaction: { instructions: 'Preserve exact IDs.' }`
|
|
790
|
+
adds guidance to the built-in summary prompt while keeping the derived threshold.
|
|
791
|
+
`compaction: false` disables both compaction and metadata lookup. The `compaction`
|
|
792
|
+
option can also be a resolver receiving the same
|
|
793
|
+
`{ event, state, session, signal }` context as `model` and `instructions`. It
|
|
794
|
+
returns `false`, automatic options, or a custom policy once per generation and
|
|
795
|
+
can await `session.state(settingsReducer)` to read application settings.
|
|
796
|
+
Use this to select a threshold from the checkpointed AI state.
|
|
797
|
+
Changes after resolution apply to a later model step. Disabling compaction
|
|
798
|
+
preserves summaries that were already recorded. Custom provider
|
|
799
|
+
models and custom global SDK providers need an explicit threshold. A custom
|
|
800
|
+
`generate` callback keeps compaction disabled unless explicitly configured. Gateway
|
|
801
|
+
fallback model lists also need an explicit threshold appropriate for every model
|
|
802
|
+
they may use.
|
|
803
|
+
|
|
804
|
+
The input estimate starts from serialized UTF-8 bytes divided by four, including
|
|
805
|
+
model messages, instructions, and tool schemas. When a preceding model call has
|
|
806
|
+
reported token usage for the same model and context, A2 uses that measurement as
|
|
807
|
+
an anchor and estimates subsequent growth. It is not a provider tokenizer and
|
|
808
|
+
does not precisely measure new image or audio tokens. A2 performs no remote token
|
|
809
|
+
counting.
|
|
810
|
+
|
|
811
|
+
On a cold session without recorded metadata or an explicit threshold, automatic
|
|
812
|
+
compaction waits for discovery. An oversized first prompt is sent normally and
|
|
813
|
+
its provider error becomes a generation failure. A2 does not silently truncate
|
|
814
|
+
it or attempt automatic recovery from a context-limit error. Use an explicit threshold for
|
|
815
|
+
applications that import a large initial conversation. The estimate is a
|
|
816
|
+
compaction trigger, not a definitive overflow check. Provider context-limit
|
|
817
|
+
errors during generation or summarization remain generation failures.
|
|
818
|
+
|
|
819
|
+
Crossing the threshold adds one summary model request. Application generation
|
|
820
|
+
callbacks, stream transforms, and tool hooks do not run for the summary, and it
|
|
821
|
+
never executes local tools. Provider-executed tools, structured output, and forced
|
|
822
|
+
tool choices disable implicit automatic compaction; explicitly configuring
|
|
823
|
+
automatic compaction for those requests throws. Use a custom policy for them.
|
|
824
|
+
An empty summary or a summary tool call fails the generation instead of replacing
|
|
825
|
+
its context.
|
|
826
|
+
|
|
827
|
+
A2 records `ai.compaction.requested` before the call and
|
|
828
|
+
`ai.compaction.completed` after a successful summary. The completed event stores
|
|
829
|
+
the summary, its usage, and the exact event frontier it covers. Later model
|
|
830
|
+
requests receive the summary plus everything produced after that frontier,
|
|
831
|
+
including later tool steps in the same assistant message. Queued user messages
|
|
832
|
+
remain queued. Compaction waits until earlier tool calls have terminal results.
|
|
833
|
+
The raw log and `state.messages` remain available in full.
|
|
834
|
+
|
|
835
|
+
Custom `shouldCompact(context)` and `compact(context)` callbacks remain
|
|
836
|
+
available for deterministic replacement messages or application-specific
|
|
837
|
+
retention. Both receive the typed UI messages and `modelMessages`, the model-ready
|
|
838
|
+
context including any existing summary. Custom generation callbacks also receive
|
|
839
|
+
`modelMessages`; use it when forwarding context to a provider. A2 stores built-in
|
|
840
|
+
summaries separately from application-typed UI messages, so a summary does not
|
|
841
|
+
need your message metadata or schema.
|
|
842
|
+
|
|
843
|
+
Render compaction as activity alongside the conversation:
|
|
844
|
+
|
|
845
|
+
```tsx app/agent/compaction-status.tsx
|
|
846
|
+
'use client'
|
|
847
|
+
import { useSession } from './session'
|
|
848
|
+
|
|
849
|
+
export function CompactionStatus() {
|
|
850
|
+
const { state } = useSession()
|
|
851
|
+
return state.compaction?.status === 'running' ? (
|
|
852
|
+
<p role="status">Compacting…</p>
|
|
853
|
+
) : null
|
|
854
|
+
}
|
|
855
|
+
```
|
|
856
|
+
|
|
857
|
+
A failed or interrupted generation restores the prior compaction state. The
|
|
858
|
+
synthetic summary request is not a user turn or an assistant reply in
|
|
859
|
+
`state.messages`.
|
|
794
860
|
|
|
795
861
|
## Extend the assembly
|
|
796
862
|
|
|
@@ -883,7 +949,7 @@ export const customGenerationServer = createAgentServer({
|
|
|
883
949
|
```
|
|
884
950
|
|
|
885
951
|
`generate` receives compacted messages, resolved model and instructions, tools,
|
|
886
|
-
generation settings, durable request identity, current state
|
|
952
|
+
generation settings, durable request identity, current state, and
|
|
887
953
|
the abort signal. It returns exactly one model step as a
|
|
888
954
|
`ReadableStream<UIMessageChunk>`. A tool-aware replacement sends definitions to
|
|
889
955
|
the model without running local `execute` functions. A2 still owns durable
|
|
@@ -892,6 +958,33 @@ progress, tool execution, approval, continuation, interruption, and failure.
|
|
|
892
958
|
function owns its metadata and can pass a callback to `toUIMessageStream()` or
|
|
893
959
|
emit typed metadata chunks itself.
|
|
894
960
|
|
|
961
|
+
## Checkpoints and recovery reads
|
|
962
|
+
|
|
963
|
+
Model steps read the AI reducer and coordinator at a fixed log boundary. Model,
|
|
964
|
+
instruction, compaction, and custom-generation callbacks receive derived state.
|
|
965
|
+
They do not receive a raw history array. Their `session.state(reducer)` reads
|
|
966
|
+
application reducers at the captured prompt boundary by default; explicit
|
|
967
|
+
`through` options follow the ordinary state-read contract. Compaction selection
|
|
968
|
+
can be asynchronous when it needs a settings-state read.
|
|
969
|
+
|
|
970
|
+
Ordinary generation uses checkpointed state. A recovered attempt reads from its
|
|
971
|
+
saved prompt boundary to the captured current boundary, excluding abandoned
|
|
972
|
+
attempts. Tools recovering in a fresh process reconstruct their original prompt
|
|
973
|
+
from that checkpoint and the owning generation's start and compaction events.
|
|
974
|
+
Compacted context uses its checkpoint and subsequent events. These AI history
|
|
975
|
+
reads always have explicit lower and upper bounds. A2 core can rebuild a missing
|
|
976
|
+
or invalid reducer snapshot from the log.
|
|
977
|
+
|
|
978
|
+
Streaming progress updates the current message incrementally. Snapshots preserve
|
|
979
|
+
open text and reasoning parts and partial tool inputs. Indexed progress remains
|
|
980
|
+
available for exact interruption cutoffs; reordered events rebuild the projection.
|
|
981
|
+
|
|
982
|
+
While tool arguments arrive, `input-streaming` parts expose a best-effort partial
|
|
983
|
+
JSON value in `input`. A command string can appear and grow before its closing
|
|
984
|
+
quote arrives. This is a display draft, not validated execution input.
|
|
985
|
+
`tool-input-available` supplies the completed input. Interruption still discards
|
|
986
|
+
streaming drafts even when their partial JSON is readable.
|
|
987
|
+
|
|
895
988
|
## Delivery semantics
|
|
896
989
|
|
|
897
990
|
Model steps and tools run inside at-least-once A2 handlers. The returned-event
|
|
@@ -900,6 +993,12 @@ one atomic durable operation. Deterministic ids make scheduling and lifecycle
|
|
|
900
993
|
appends effectively once, and A2 marks an incomplete model attempt
|
|
901
994
|
`superseded` before starting its replacement.
|
|
902
995
|
|
|
996
|
+
An expired claim or superseded attempt preserves its abort reason through model,
|
|
997
|
+
resolver, compaction-policy, and tool calls. A2 retries that attempt without
|
|
998
|
+
spending its failure budget. Explicit user interruption remains terminal.
|
|
999
|
+
Streaming tool progress belongs to an execution attempt, so a retry can produce
|
|
1000
|
+
different preliminary output while preserving the tool-call id for idempotency.
|
|
1001
|
+
|
|
903
1002
|
Provider calls and external tool side effects remain outside that transaction.
|
|
904
1003
|
A process can die after an external effect succeeds and before its handler
|
|
905
1004
|
completion commits, so recovery may run the call again. Give side-effecting
|
|
@@ -911,3 +1010,189 @@ continuation, but it cannot roll back that external effect.
|
|
|
911
1010
|
Configure a [production scheduler](/guides/production) exactly as for any other A2
|
|
912
1011
|
server. It recovers an incomplete generation attempt after the original
|
|
913
1012
|
serverless invocation disappears.
|
|
1013
|
+
|
|
1014
|
+
## Inbox and turn controls
|
|
1015
|
+
|
|
1016
|
+
This input protocol requires matching client and server versions. Start a new
|
|
1017
|
+
agent session when adopting it. Existing event logs remain readable; finish
|
|
1018
|
+
pending work with the runtime that created it before upgrading.
|
|
1019
|
+
|
|
1020
|
+
`inputs.message(message)` places a message in the durable inbox. The controller
|
|
1021
|
+
admits one input at a time, freezes its revision, and adds it to `state.messages`.
|
|
1022
|
+
Pending inputs appear in `state.inbox.items`; they do not enter the model prompt.
|
|
1023
|
+
Each item has an `id`, `revision`, `message`, and `generate` flag.
|
|
1024
|
+
|
|
1025
|
+
`state.active` identifies the logical turn, its frozen `input`, and its `phase`.
|
|
1026
|
+
The turn ID stays the same across model steps, tools, and a pause. Stop ends that
|
|
1027
|
+
turn and admits the next queued input when admission is running.
|
|
1028
|
+
|
|
1029
|
+
```ts client/turn-controls.ts
|
|
1030
|
+
import { inputs, type AIState } from 'experimental-a2/ai'
|
|
1031
|
+
import type { UIMessage } from 'ai'
|
|
1032
|
+
|
|
1033
|
+
export function editPending(message: UIMessage, expectedRevision: number) {
|
|
1034
|
+
return inputs.queue.edit({ message, expectedRevision })
|
|
1035
|
+
}
|
|
1036
|
+
|
|
1037
|
+
export function movePending(inputId: string, beforeId: string | null) {
|
|
1038
|
+
return inputs.queue.move({ inputId, beforeId })
|
|
1039
|
+
}
|
|
1040
|
+
|
|
1041
|
+
export function removePending(inputId: string) {
|
|
1042
|
+
return inputs.queue.remove({ inputId })
|
|
1043
|
+
}
|
|
1044
|
+
|
|
1045
|
+
export function sendPendingNow(state: AIState, inputId: string) {
|
|
1046
|
+
return inputs.queue.sendNow({ inputId, turnId: state.active?.turnId ?? null })
|
|
1047
|
+
}
|
|
1048
|
+
|
|
1049
|
+
export function stopCurrent(state: AIState) {
|
|
1050
|
+
return state.active ? inputs.stop({ turnId: state.active.turnId }) : []
|
|
1051
|
+
}
|
|
1052
|
+
|
|
1053
|
+
export function steerCurrent(state: AIState, message: UIMessage) {
|
|
1054
|
+
return state.active
|
|
1055
|
+
? inputs.steer({ turnId: state.active.turnId, message })
|
|
1056
|
+
: inputs.message(message)
|
|
1057
|
+
}
|
|
1058
|
+
|
|
1059
|
+
export const pauseAfterTurn = () => inputs.pause({ when: 'after-turn' })
|
|
1060
|
+
export const pauseNow = () => inputs.pause({ when: 'now' })
|
|
1061
|
+
export const resume = () => inputs.resume()
|
|
1062
|
+
```
|
|
1063
|
+
|
|
1064
|
+
Pass these arrays to `push(...events)` from `useSession()`, or to a server
|
|
1065
|
+
session's `append(...events)`. `beforeId: null` moves an item to the end.
|
|
1066
|
+
Removal only applies to pending input. It preserves the immutable log and keeps
|
|
1067
|
+
the input identity reserved, so replay cannot recreate a removed prompt.
|
|
1068
|
+
Removing an admitted input returns `already-active`; removing an absent or
|
|
1069
|
+
already removed input returns `not-found`.
|
|
1070
|
+
|
|
1071
|
+
An edit that loses to admission returns `already-active`. Concurrent edits use
|
|
1072
|
+
`expectedRevision`; an outdated revision returns `revision-conflict`. A Stop
|
|
1073
|
+
aimed at a completed turn returns `stale-turn`.
|
|
1074
|
+
|
|
1075
|
+
Steer validates the new input and places it ahead of the inbox in one decision.
|
|
1076
|
+
If the observed turn is still active, it finishes its current model response and
|
|
1077
|
+
the tools produced by that response before handing off. Parallel tools all
|
|
1078
|
+
settle, and steering runs before the old turn starts another model step.
|
|
1079
|
+
Text and tool results produced after the click remain in the conversation.
|
|
1080
|
+
`lastSeenIndex` is not used to truncate output for steering or Send now.
|
|
1081
|
+
|
|
1082
|
+
When running work settles, unanswered approval or application-input requests
|
|
1083
|
+
are closed so the steering message can take over. The unapproved tool never
|
|
1084
|
+
executes. Provider-executed deferred results still count as running work when
|
|
1085
|
+
execution is authorized; a denied approval cannot block handoff.
|
|
1086
|
+
If the targeted turn has already finished, the input starts when idle or waits
|
|
1087
|
+
first behind a different active turn. Stop remains an immediate interruption.
|
|
1088
|
+
|
|
1089
|
+
Steering preserves the admission gate: steering while paused does not resume
|
|
1090
|
+
execution. Duplicate input identities and closed sessions still reject the command.
|
|
1091
|
+
|
|
1092
|
+
Send now uses the selected queued message's existing identity and latest
|
|
1093
|
+
accepted revision. It validates the selected input and the observed `turnId`,
|
|
1094
|
+
then moves the input first for the same handoff after the current step. Use
|
|
1095
|
+
`turnId: null` when no turn is active. If that expectation is stale, the command
|
|
1096
|
+
returns `stale-turn` without changing the queue.
|
|
1097
|
+
|
|
1098
|
+
Send now preserves pause just like steering. A paused queue keeps the selected
|
|
1099
|
+
input first until Resume. It accepts user-role inputs, including passive user
|
|
1100
|
+
input, which it marks for generation. Non-user context returns `not-user-input`;
|
|
1101
|
+
a conflicting response identity returns `duplicate-input`. Unsaved editor text
|
|
1102
|
+
is separate from the accepted queue revision; save it before sending.
|
|
1103
|
+
|
|
1104
|
+
A queued item's optional `afterStepOf` records the observed turn for steering.
|
|
1105
|
+
A matching active turn hands off after its current step.
|
|
1106
|
+
Editing preserves this intent; removing the item cancels it. Reordering the
|
|
1107
|
+
queue controls which input is admitted at that boundary.
|
|
1108
|
+
|
|
1109
|
+
### Optimistic conversation display
|
|
1110
|
+
|
|
1111
|
+
With `agent().reducer` or `createReducer()`, the client state is already optimistic.
|
|
1112
|
+
An idle, unpaused send appears in `state.messages` immediately, with
|
|
1113
|
+
`state.starting: true`. Another send waits in `state.inbox.items` while that
|
|
1114
|
+
optimistic turn or a confirmed turn is active. Paused sessions keep new sends in
|
|
1115
|
+
the inbox. Render the conversation and queue from `state`; `state.active` and
|
|
1116
|
+
execution controls remain server-confirmed.
|
|
1117
|
+
Steering appears immediately in `state.messages`, after the current response,
|
|
1118
|
+
and stays there while that step finishes. Send now has the same display when
|
|
1119
|
+
an active turn is targeted. These inputs are omitted from the client inbox,
|
|
1120
|
+
including while paused, but remain pending on the server until admission.
|
|
1121
|
+
Accepted steering survives receipt and hydration in this display. Multiple
|
|
1122
|
+
steering inputs appear in admission order, with the most recent first.
|
|
1123
|
+
No additional request, subscription, or component-owned queue is needed.
|
|
1124
|
+
|
|
1125
|
+
```tsx app/agent/[sessionId]/conversation.tsx
|
|
1126
|
+
'use client'
|
|
1127
|
+
import { useSession } from '../session'
|
|
1128
|
+
|
|
1129
|
+
export function Conversation() {
|
|
1130
|
+
const { state } = useSession()
|
|
1131
|
+
const { messages, inbox, starting } = state
|
|
1132
|
+
const text = (message: (typeof messages)[number]) =>
|
|
1133
|
+
message.parts.filter((part) => part.type === 'text').map((part) => part.text).join(' ')
|
|
1134
|
+
return (
|
|
1135
|
+
<section>
|
|
1136
|
+
<div aria-label="Conversation">
|
|
1137
|
+
{messages.map((message) => <p key={message.id}>{text(message)}</p>)}
|
|
1138
|
+
{starting ? <p role="status">Starting…</p> : null}
|
|
1139
|
+
</div>
|
|
1140
|
+
<ol aria-label="Queued messages">
|
|
1141
|
+
{inbox.items.map((item) => <li key={item.id}>{text(item.message)}</li>)}
|
|
1142
|
+
</ol>
|
|
1143
|
+
</section>
|
|
1144
|
+
)
|
|
1145
|
+
}
|
|
1146
|
+
```
|
|
1147
|
+
|
|
1148
|
+
The request echo keeps the optimistic display until `ai.control.decided` arrives. Accepted
|
|
1149
|
+
commits replace the confirmed base; rejected receipts remove the pending intent.
|
|
1150
|
+
A failed POST rolls back through the ordinary client optimistic overlay.
|
|
1151
|
+
Another client's accepted turn can move a speculative ordinary send back into
|
|
1152
|
+
the queue. Steering stays visible as a message while waiting behind that turn. `state.starting` describes an optimistic turn awaiting acceptance; it does
|
|
1153
|
+
not indicate a running model.
|
|
1154
|
+
|
|
1155
|
+
Server `session.state(reducer)`, reducer folds, checkpoints, and model prompts
|
|
1156
|
+
contain confirmed messages and inbox entries, with `starting: false`. The client
|
|
1157
|
+
applies pending intent and displays accepted steering only when returning its
|
|
1158
|
+
state, after folding events.
|
|
1159
|
+
A custom reducer that calls `reduceAIState()` or nests `agent.reducer.fold()`
|
|
1160
|
+
uses that confirmed fold; it does not inherit the built-in reducer's client display.
|
|
1161
|
+
`state.pendingQueueCommands` tracks unresolved queue commands incrementally, so
|
|
1162
|
+
the client does not scan raw history. Revisions remain server-confirmed; keep
|
|
1163
|
+
repeat editing and Send now disabled while an edit for that item awaits its
|
|
1164
|
+
receipt. Moving and removing an item do not require a revision.
|
|
1165
|
+
|
|
1166
|
+
### Persistence and application are separate
|
|
1167
|
+
|
|
1168
|
+
`push()` confirms that the command is persisted. The `ai.control.decided` event
|
|
1169
|
+
confirms its application, using the submitted event's ID as `commandId`. Its
|
|
1170
|
+
`outcome` is `applied` or `rejected`; a rejection includes a `reason`.
|
|
1171
|
+
`state.receipt` exposes the most recent user command receipt. Applications that
|
|
1172
|
+
have several commands in flight correlate receipts through the existing event
|
|
1173
|
+
stream. A rejection is a normal result and does not spend a handler retry.
|
|
1174
|
+
|
|
1175
|
+
### Pause and resume
|
|
1176
|
+
|
|
1177
|
+
Pause after the turn closes inbox admission and lets the active turn finish.
|
|
1178
|
+
Pause now also cancels current model and tool execution. `state.active.phase`
|
|
1179
|
+
becomes `pausing` while old work settles, then `paused`. Resume reopens admission
|
|
1180
|
+
and resumes retained work before admitting another input.
|
|
1181
|
+
|
|
1182
|
+
Resume starts a fresh model request from durable state. It does not resume the
|
|
1183
|
+
provider's old stream or undo completed tool effects. The controller retains
|
|
1184
|
+
accepted tool calls and records their outcomes while paused. A cancelled tool
|
|
1185
|
+
without a terminal result can execute again after resume; use the stable SDK
|
|
1186
|
+
`toolCallId` for external idempotency. No model or tool claim is held merely to
|
|
1187
|
+
keep a turn paused.
|
|
1188
|
+
|
|
1189
|
+
The controller runs on a short lane. Each decision commits accepted facts,
|
|
1190
|
+
compact state changes, work requests, and its receipt atomically. Workers use
|
|
1191
|
+
separate claims, report lifecycle batches, and wait for acceptance at model-start
|
|
1192
|
+
and compaction boundaries. Token progress appends directly. Explicit AI history
|
|
1193
|
+
reads remain bounded to a relevant generation suffix; core checkpoint rebuilding
|
|
1194
|
+
can replay older events when a snapshot is missing.
|
|
1195
|
+
|
|
1196
|
+
For a deferred provider-executed tool, trusted server code submits
|
|
1197
|
+
`inputs.toolResult(result)`. The controller checks its admitted call and records
|
|
1198
|
+
the result before continuing. Browser ingress rejects this server-only input.
|