experimental-a2 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/CHANGELOG.md +47 -0
  2. package/dist/{actor-DJi3RsNu.d.ts → actor-BfQSE0KC.d.ts} +4 -4
  3. package/dist/{actor-DJi3RsNu.d.ts.map → actor-BfQSE0KC.d.ts.map} +1 -1
  4. package/dist/actor-client.d.ts +1 -1
  5. package/dist/actor-client.js +1 -1
  6. package/dist/actor-react.d.ts +3 -3
  7. package/dist/actor-react.js +2 -2
  8. package/dist/{actor-shared-DI7J5upy.js → actor-shared-B5tJfzt-.js} +2 -2
  9. package/dist/{actor-shared-DI7J5upy.js.map → actor-shared-B5tJfzt-.js.map} +1 -1
  10. package/dist/actor.d.ts +1 -1
  11. package/dist/actor.js +3 -3
  12. package/dist/ai-Cai-lCbj.d.ts +580 -0
  13. package/dist/ai-Cai-lCbj.d.ts.map +1 -0
  14. package/dist/ai-control-CcD4hh3y.js +119 -0
  15. package/dist/ai-control-CcD4hh3y.js.map +1 -0
  16. package/dist/ai-server.d.ts +16 -8
  17. package/dist/ai-server.d.ts.map +1 -1
  18. package/dist/ai-server.js +1322 -475
  19. package/dist/ai-server.js.map +1 -1
  20. package/dist/ai.d.ts +2 -334
  21. package/dist/ai.js +838 -85
  22. package/dist/ai.js.map +1 -1
  23. package/dist/{client-P_NNNRM-.d.ts → client-BAEABRZB.d.ts} +2 -2
  24. package/dist/{client-P_NNNRM-.d.ts.map → client-BAEABRZB.d.ts.map} +1 -1
  25. package/dist/{client-Bf6uSEAk.js → client-BYzHjkwU.js} +21 -6
  26. package/dist/client-BYzHjkwU.js.map +1 -0
  27. package/dist/client.d.ts +1 -1
  28. package/dist/client.js +1 -1
  29. package/dist/{contract-48bUMgcL.js → contract-CKRg_E4q.js} +3 -26
  30. package/dist/contract-CKRg_E4q.js.map +1 -0
  31. package/dist/index.d.ts +2 -2
  32. package/dist/index.js +1 -1
  33. package/dist/react.d.ts +2 -2
  34. package/dist/react.js +1 -1
  35. package/dist/{reducer-DJKWm3cp.d.ts → reducer-BcS9VDKC.d.ts} +4 -1
  36. package/dist/{reducer-DJKWm3cp.d.ts.map → reducer-BcS9VDKC.d.ts.map} +1 -1
  37. package/dist/reducer-DEMjEY_O.js +29 -0
  38. package/dist/reducer-DEMjEY_O.js.map +1 -0
  39. package/dist/scheduler-qstash.d.ts +2 -2
  40. package/dist/scheduler-qstash.js +1 -1
  41. package/dist/scheduler-vercel.d.ts +2 -2
  42. package/dist/scheduler-vercel.js +1 -1
  43. package/dist/{server-DjZZa1wr.d.ts → server-Bp5Nd1pF.d.ts} +3 -3
  44. package/dist/{server-DjZZa1wr.d.ts.map → server-Bp5Nd1pF.d.ts.map} +1 -1
  45. package/dist/{server-BeNADlCI.js → server-CjJSGcF7.js} +4 -3
  46. package/dist/server-CjJSGcF7.js.map +1 -0
  47. package/dist/server.d.ts +3 -3
  48. package/dist/server.js +1 -1
  49. package/dist/{store-DtDOWLSn.d.ts → store-D_yhNdPz.d.ts} +7 -2
  50. package/dist/{store-DtDOWLSn.d.ts.map → store-D_yhNdPz.d.ts.map} +1 -1
  51. package/dist/store-N8PXxDAS.js.map +1 -1
  52. package/dist/store-memory.d.ts +1 -1
  53. package/dist/store-postgres.d.ts +1 -1
  54. package/dist/store-postgres.js +19 -0
  55. package/dist/store-postgres.js.map +1 -1
  56. package/dist/store-redis-http.d.ts +1 -1
  57. package/dist/store-redis-http.js +1 -1
  58. package/dist/{store-redis-notify-BUCyXOn0.js → store-redis-notify-D2EI6gwX.js} +27 -2
  59. package/dist/store-redis-notify-D2EI6gwX.js.map +1 -0
  60. package/dist/store-redis.d.ts +1 -1
  61. package/dist/store-redis.js +1 -1
  62. package/dist/store-sqlite.d.ts +1 -1
  63. package/docs/guides/06-ai-agents.mdx +361 -76
  64. package/docs/reference/01-api.mdx +159 -27
  65. package/examples/playground/app/agent/[agentId]/agent-client.tsx +145 -33
  66. package/examples/playground/app/agent/[agentId]/agent-queue.tsx +289 -0
  67. package/examples/playground/app/agent/[agentId]/compaction/route.ts +14 -0
  68. package/examples/playground/app/agent/[agentId]/compaction-event.tsx +38 -0
  69. package/examples/playground/app/agent/[agentId]/compaction-panel.tsx +294 -0
  70. package/examples/playground/app/agent/compaction-settings.test.ts +144 -0
  71. package/examples/playground/app/agent/compaction-settings.ts +49 -0
  72. package/examples/playground/app/agent/compaction-timeline.test.ts +337 -0
  73. package/examples/playground/app/agent/compaction-timeline.ts +198 -0
  74. package/examples/playground/app/agent/model.ts +56 -1
  75. package/examples/playground/app/agent/server.ts +9 -2
  76. package/examples/playground/app/chat/[chatId]/chat-client.tsx +3 -13
  77. package/examples/playground/app/chat/model.ts +2 -2
  78. package/examples/playground/app/chat/server.ts +24 -17
  79. package/examples/playground/app/globals.css +333 -0
  80. package/examples/playground/package.json +1 -1
  81. package/package.json +1 -1
  82. package/src/ai-client-state.ts +185 -0
  83. package/src/ai-control-server.ts +829 -0
  84. package/src/ai-control-state.ts +152 -0
  85. package/src/ai-control.ts +139 -0
  86. package/src/ai-coordinator.ts +99 -32
  87. package/src/ai-model-metadata.ts +108 -0
  88. package/src/ai-progress-batches.ts +68 -0
  89. package/src/ai-projector.ts +76 -15
  90. package/src/ai-sdk-step.ts +5 -2
  91. package/src/ai-server.ts +920 -638
  92. package/src/ai.ts +650 -110
  93. package/src/client.ts +31 -9
  94. package/src/licenses/Apache-2.0.txt +55 -0
  95. package/src/parse-partial-json.ts +441 -0
  96. package/src/reducer.ts +6 -0
  97. package/src/server.ts +8 -4
  98. package/src/store-postgres.ts +27 -0
  99. package/src/store-redis-core.ts +53 -1
  100. package/src/store-redis-notify.ts +1 -0
  101. package/src/store.ts +6 -0
  102. package/dist/ai.d.ts.map +0 -1
  103. package/dist/client-Bf6uSEAk.js.map +0 -1
  104. package/dist/contract-48bUMgcL.js.map +0 -1
  105. package/dist/server-BeNADlCI.js.map +0 -1
  106. package/dist/store-redis-notify-BUCyXOn0.js.map +0 -1
@@ -135,10 +135,10 @@ export const GET = assistantServer.fetch
135
135
  export const POST = assistantServer.fetch
136
136
  ```
137
137
 
138
- The route never calls the model directly. The browser appends user facts such
139
- as `ai.message.created`. Built-in server handlers schedule
140
- `ai.generation.requested`, run the AI SDK, and append progress back to the same
141
- log.
138
+ The route never calls the model directly. The browser appends
139
+ `ai.control.requested` commands. The controller admits inputs and schedules
140
+ `ai.generation.requested`; workers run the AI SDK and append progress to the
141
+ same log.
142
142
 
143
143
  ## Bind the reducer to React
144
144
 
@@ -309,11 +309,9 @@ export function AgentClient() {
309
309
  }
310
310
  ```
311
311
 
312
- There is no separate message state to reconcile. `push()` folds the
313
- `ai.message.created` fact locally before starting the request, so the user
314
- message appears immediately. If the request fails, A2 removes that optimistic
315
- event and the component restores the draft. The server owns the corresponding
316
- generation request.
312
+ `push()` persists the command. Its receipt confirms admission to the inbox.
313
+ `state.inbox.items` contains pending inputs; `state.messages` contains the
314
+ admitted conversation. If persistence fails, the component restores the draft.
317
315
 
318
316
  The form gives Enter its normal submit behavior. The thinking row is also
319
317
  derived from `AIState`: it appears with the Agent label when the server's
@@ -323,7 +321,7 @@ it. Empty stream-start messages never create a blank conversation row.
323
321
  Users can send another message while a response is active. The message enters
324
322
  the durable log immediately, but its turn waits until the active assistant
325
323
  response, including every tool step and approval, reaches a terminal event.
326
- Queued messages keep log order.
324
+ Inbox order defaults to arrival order and can be edited explicitly.
327
325
 
328
326
  Pass `{ generate: false }` when a user message should update the conversation
329
327
  without starting a model turn:
@@ -343,7 +341,7 @@ export async function recordPassiveMessage(
343
341
  }
344
342
  ```
345
343
 
346
- The passive user message still appears in `AIState.messages`. A later user
344
+ The passive user message appears in `AIState.messages` when admitted. A later user
347
345
  message with the default generation behavior includes passive user messages
348
346
  before it in model context. Passive user messages after that trigger remain
349
347
  outside its prompt and wait for the next generating message. This ordering
@@ -355,7 +353,10 @@ already queued turn.
355
353
  One generating user interaction becomes a durable sequence:
356
354
 
357
355
  ```text
358
- browser ai.message.created optimistic, then durable
356
+ browser ai.control.requested durable input command
357
+ server ai.message.created input admitted
358
+ server ai.control.committed accepted state changes
359
+ server ai.control.decided command receipt
359
360
  server ai.generation.requested scheduled from durable facts
360
361
  server ai.generation.started
361
362
  server ai.generation.progress batched UIMessageChunk[]
@@ -414,15 +415,26 @@ continuation that has not been appended yet.
414
415
 
415
416
  | Input | Events |
416
417
  | --- | --- |
417
- | `inputs.message(message, { generate? })` | records a message fact; `false` keeps context without scheduling |
418
+ | `inputs.message(message, { generate? })` | queues input; `false` queues passive context |
418
419
  | `inputs.seed(message)` | records `generate: false`; non-user roles require a trusted append |
419
- | `inputs.approval(response)` | records an approval decision fact |
420
+ | `inputs.approval(response)` | requests a decision on a pending approval |
420
421
  | `inputs.input(response)` | records an application input response fact |
421
422
  | `inputs.requestInput(request)` | records a trusted server request for application input |
422
- | `inputs.retry(options)` | records `ai.retry.requested` for a failed response |
423
- | `inputs.interrupt(options)` | interrupts an active response |
424
-
425
- The builders hide stable event ids, so the same interaction is safe to resend.
423
+ | `inputs.retry(options)` | requests a fresh attempt for a failed response |
424
+ | `inputs.stop({ turnId })` | ends the observed turn and advances |
425
+ | `inputs.steer({ turnId, message })` | prioritizes input after the observed turn's current step finishes |
426
+ | `inputs.queue.edit({ message, expectedRevision })` | edits a pending input |
427
+ | `inputs.queue.move({ inputId, beforeId })` | reorders pending input; `null` means the end |
428
+ | `inputs.queue.remove({ inputId })` | removes pending input from the inbox |
429
+ | `inputs.queue.sendNow({ inputId, turnId })` | atomically prioritizes queued input and interrupts the observed turn |
430
+ | `inputs.pause({ when })` | pauses now or after the active turn |
431
+ | `inputs.resume()` | resumes retained work and inbox admission |
432
+ | `inputs.toolResult(result)` | records a trusted deferred provider-tool result |
433
+ | `inputs.interrupt(options)` | addresses a specific generation or request |
434
+
435
+ Builders return `ai.control.requested` commands. Message, approval, input, and
436
+ retry builders identify a stable interaction. Other controls use the event ID
437
+ assigned by `push()` or `append()`; transport retries preserve that ID.
426
438
  Generation scheduling considers only user-role `ai.message.created` facts.
427
439
  Assistant and system message facts remain context regardless of the `generate`
428
440
  field, and model output is recorded as generation progress rather than a new
@@ -482,10 +494,9 @@ The AI SDK infers the tool input and output from its schemas and
482
494
  scope carrying the current durable handler attempt. Read it with
483
495
  `handlerContext(agent)` from `experimental-a2/ai`: it returns the same
484
496
  typed context bag an event handler receives, with `event`, `attempt`,
485
- `session`, and `signal`. Automatic tools see an `ai.tool.called` event.
486
- Approved tools see the `ai.approval.responded` event that authorized
487
- them. Narrow `ctx.event.type` if you need fields specific to either
488
- event. The agent argument carries the types; A2 verifies it against the
497
+ `session`, and `signal`. Tools execute an `ai.tool.execution.requested`
498
+ event after authorization. Its payload contains the admitted `call`, its
499
+ `generation`, and its turn and execution version. The agent argument carries the types; A2 verifies it against the
489
500
  server executing the tool and throws when a tool written for one agent
490
501
  runs under another, or when `handlerContext()` is called outside a tool
491
502
  execution. The scope survives awaited helpers and async iteration, so
@@ -547,7 +558,7 @@ and plan.
547
558
  The schedule name uses `toolCallId`, and the message id uses the durable
548
559
  triggering event id. A retry therefore converges on the same timer and message,
549
560
  even if a provider reuses tool-call ids in a later generation. When the timer
550
- arrives, `inputs.message()` records an ordinary `ai.message.created` fact. The
561
+ arrives, `inputs.message()` queues a durable input command. The
551
562
  queued-turn policy starts a fresh response after the active response completes
552
563
  or is interrupted. A failed response must be retried or interrupted first.
553
564
  If the scheduler send fails ambiguously or transiently, A2 retries the same
@@ -664,40 +675,15 @@ import { useSession } from '../session'
664
675
 
665
676
  export function StopButton() {
666
677
  const { state, push, index } = useSession()
667
- const active = state.activeGeneration
668
- const requested =
669
- state.activeRequestId && state.activeResponseMessageId
670
- ? {
671
- messageId: state.activeResponseMessageId,
672
- requestId: state.activeRequestId,
673
- }
674
- : null
675
- const waiting =
676
- state.pendingApprovals[0] ??
677
- state.pendingInputs[0] ??
678
- state.tools.find((tool) => tool.status === 'running')
679
- const target = active
680
- ? {
681
- messageId: active.responseMessageId,
682
- generationId: active.generationId,
683
- }
684
- : requested ??
685
- (waiting
686
- ? {
687
- messageId: waiting.messageId,
688
- generationId: waiting.generationId,
689
- }
690
- : null)
691
-
692
- if (!target) return null
678
+ const active = state.active
679
+ if (!active) return null
693
680
 
694
681
  return (
695
682
  <button
696
683
  onClick={() =>
697
684
  void push(
698
- ...inputs.interrupt({
699
- ...target,
700
- reason: 'Stopped by the user',
685
+ ...inputs.stop({
686
+ turnId: active.turnId,
701
687
  lastSeenIndex: index,
702
688
  }),
703
689
  )
@@ -709,17 +695,20 @@ export function StopButton() {
709
695
  }
710
696
  ```
711
697
 
712
- The optimistic event updates the UI immediately and reaches A2's cancellation
713
- channel. `lastSeenIndex` is the exact confirmed log frontier visible when the
698
+ The accepted command ends the turn and reaches A2's cancellation channel. `lastSeenIndex` is the exact confirmed log frontier visible when the
714
699
  user clicked. The reducer rewinds the indexed generation projection to that
715
- frontier, keeps the partial response the user actually saw, and turns
716
- incomplete visible tools into `output-error`. An accepted interruption
700
+ frontier and keeps the partial text the user actually saw. Tool parts whose
701
+ arguments are still streaming are drafts: interruption removes them from the
702
+ conversation and subsequent model prompts, even if their current JSON parses.
703
+ Their progress remains in the raw log. Accepted tools still awaiting completion
704
+ become `output-error`. An accepted interruption
717
705
  terminally fences that request or generation. Later generation, tool, approval,
718
706
  input, and compaction events remain in raw history but cannot reactivate it or
719
707
  alter the projection.
720
708
  Completed tool results at or before the visible frontier remain completed.
721
709
 
722
- Between `ai.generation.requested` and `ai.generation.started`, copy
710
+ For the lower-level `inputs.interrupt()` builder, between
711
+ `ai.generation.requested` and `ai.generation.started`, copy
723
712
  `activeRequestId` as `requestId` and `activeResponseMessageId` as `messageId`.
724
713
  Once `activeGeneration` exists, send its `generationId` and
725
714
  `responseMessageId` instead. A request-owned interruption remains valid if
@@ -762,35 +751,112 @@ interaction that began elsewhere.
762
751
 
763
752
  ## Compact long conversations
764
753
 
765
- Compaction is optional application policy. A2 records both the decision and
766
- replacement messages, so later prompts remain explainable from history:
754
+ Compaction is automatic for supported Gateway models. A2 uses the active model
755
+ with the same instructions, tool definitions, and provider settings, and appends
756
+ an internal user message asking for a summary. This preserves the request prefix
757
+ for prompt caching when the provider has a matching cache entry.
767
758
 
768
759
  ```ts server/compacted.ts
769
- import type { UIMessage } from 'ai'
770
760
  import { createAgentServer } from 'experimental-a2/ai/server'
771
761
  import { assistant } from '../assistant'
772
762
 
773
763
  export const compactedAssistantServer = createAgentServer({
774
764
  agent: assistant,
775
765
  model: 'openai/gpt-5.6-terra',
776
- compaction: {
777
- shouldCompact: ({ messages }) => messages.length > 40,
778
- compact: async ({ messages }) => {
779
- const summary = {
780
- id: crypto.randomUUID(),
781
- role: 'user',
782
- parts: [{ type: 'text', text: 'Summary of the earlier conversation.' }],
783
- } satisfies UIMessage
784
-
785
- return [summary, ...messages.slice(-10)]
786
- },
787
- },
788
766
  })
789
767
  ```
790
768
 
791
- The callback can call another model, create a deterministic summary, or retain
792
- selected messages. A2 owns only when the result enters the log and how later
793
- generations consume it.
769
+ On first use of each Gateway model in a session, A2 includes
770
+ `ai.model.metadata.requested` in the existing generation-start append. A separate
771
+ handler fetches the public Gateway model catalog and records the selected limits
772
+ in `ai.model.metadata.resolved`. It runs outside the AI turn lane. The first model
773
+ request proceeds immediately; metadata that arrives afterward is available to
774
+ later steps. Returning to a previously used model reuses its durable metadata.
775
+ The catalog is also cached in memory for one hour across agent servers, with one
776
+ shared in-flight request and a five-second timeout. A failed lookup follows A2's
777
+ handler retry policy without failing the model generation. A model absent from
778
+ the catalog is recorded as unavailable. A pending entry means no result has been
779
+ recorded, including when the metadata handler exhausts its retry budget. A2 does
780
+ not start a fresh lookup on every turn after exhaustion. Inspect the handler
781
+ failure or use an explicit threshold in that case.
782
+
783
+ The default threshold is 75% of the context window. A larger configured
784
+ `generation.maxOutputTokens` lowers the threshold further to reserve that output
785
+ allowance. This is an input-context policy, not a new output limit. Limits and
786
+ metadata status are available in `state.modelMetadata`.
787
+
788
+ `compaction: { thresholdTokens: 80_000 }` overrides discovery with an explicit
789
+ positive safe integer. `compaction: { instructions: 'Preserve exact IDs.' }`
790
+ adds guidance to the built-in summary prompt while keeping the derived threshold.
791
+ `compaction: false` disables both compaction and metadata lookup. The `compaction`
792
+ option can also be a resolver receiving the same
793
+ `{ event, state, session, signal }` context as `model` and `instructions`. It
794
+ returns `false`, automatic options, or a custom policy once per generation and
795
+ can await `session.state(settingsReducer)` to read application settings.
796
+ Use this to select a threshold from the checkpointed AI state.
797
+ Changes after resolution apply to a later model step. Disabling compaction
798
+ preserves summaries that were already recorded. Custom provider
799
+ models and custom global SDK providers need an explicit threshold. A custom
800
+ `generate` callback keeps compaction disabled unless explicitly configured. Gateway
801
+ fallback model lists also need an explicit threshold appropriate for every model
802
+ they may use.
803
+
804
+ The input estimate starts from serialized UTF-8 bytes divided by four, including
805
+ model messages, instructions, and tool schemas. When a preceding model call has
806
+ reported token usage for the same model and context, A2 uses that measurement as
807
+ an anchor and estimates subsequent growth. It is not a provider tokenizer and
808
+ does not precisely measure new image or audio tokens. A2 performs no remote token
809
+ counting.
810
+
811
+ On a cold session without recorded metadata or an explicit threshold, automatic
812
+ compaction waits for discovery. An oversized first prompt is sent normally and
813
+ its provider error becomes a generation failure. A2 does not silently truncate
814
+ it or attempt automatic recovery from a context-limit error. Use an explicit threshold for
815
+ applications that import a large initial conversation. The estimate is a
816
+ compaction trigger, not a definitive overflow check. Provider context-limit
817
+ errors during generation or summarization remain generation failures.
818
+
819
+ Crossing the threshold adds one summary model request. Application generation
820
+ callbacks, stream transforms, and tool hooks do not run for the summary, and it
821
+ never executes local tools. Provider-executed tools, structured output, and forced
822
+ tool choices disable implicit automatic compaction; explicitly configuring
823
+ automatic compaction for those requests throws. Use a custom policy for them.
824
+ An empty summary or a summary tool call fails the generation instead of replacing
825
+ its context.
826
+
827
+ A2 records `ai.compaction.requested` before the call and
828
+ `ai.compaction.completed` after a successful summary. The completed event stores
829
+ the summary, its usage, and the exact event frontier it covers. Later model
830
+ requests receive the summary plus everything produced after that frontier,
831
+ including later tool steps in the same assistant message. Queued user messages
832
+ remain queued. Compaction waits until earlier tool calls have terminal results.
833
+ The raw log and `state.messages` remain available in full.
834
+
835
+ Custom `shouldCompact(context)` and `compact(context)` callbacks remain
836
+ available for deterministic replacement messages or application-specific
837
+ retention. Both receive the typed UI messages and `modelMessages`, the model-ready
838
+ context including any existing summary. Custom generation callbacks also receive
839
+ `modelMessages`; use it when forwarding context to a provider. A2 stores built-in
840
+ summaries separately from application-typed UI messages, so a summary does not
841
+ need your message metadata or schema.
842
+
843
+ Render compaction as activity alongside the conversation:
844
+
845
+ ```tsx app/agent/compaction-status.tsx
846
+ 'use client'
847
+ import { useSession } from './session'
848
+
849
+ export function CompactionStatus() {
850
+ const { state } = useSession()
851
+ return state.compaction?.status === 'running' ? (
852
+ <p role="status">Compacting…</p>
853
+ ) : null
854
+ }
855
+ ```
856
+
857
+ A failed or interrupted generation restores the prior compaction state. The
858
+ synthetic summary request is not a user turn or an assistant reply in
859
+ `state.messages`.
794
860
 
795
861
  ## Extend the assembly
796
862
 
@@ -883,7 +949,7 @@ export const customGenerationServer = createAgentServer({
883
949
  ```
884
950
 
885
951
  `generate` receives compacted messages, resolved model and instructions, tools,
886
- generation settings, durable request identity, current state and history, and
952
+ generation settings, durable request identity, current state, and
887
953
  the abort signal. It returns exactly one model step as a
888
954
  `ReadableStream<UIMessageChunk>`. A tool-aware replacement sends definitions to
889
955
  the model without running local `execute` functions. A2 still owns durable
@@ -892,6 +958,33 @@ progress, tool execution, approval, continuation, interruption, and failure.
892
958
  function owns its metadata and can pass a callback to `toUIMessageStream()` or
893
959
  emit typed metadata chunks itself.
894
960
 
961
+ ## Checkpoints and recovery reads
962
+
963
+ Model steps read the AI reducer and coordinator at a fixed log boundary. Model,
964
+ instruction, compaction, and custom-generation callbacks receive derived state.
965
+ They do not receive a raw history array. Their `session.state(reducer)` reads
966
+ application reducers at the captured prompt boundary by default; explicit
967
+ `through` options follow the ordinary state-read contract. Compaction selection
968
+ can be asynchronous when it needs a settings-state read.
969
+
970
+ Ordinary generation uses checkpointed state. A recovered attempt reads from its
971
+ saved prompt boundary to the captured current boundary, excluding abandoned
972
+ attempts. Tools recovering in a fresh process reconstruct their original prompt
973
+ from that checkpoint and the owning generation's start and compaction events.
974
+ Compacted context uses its checkpoint and subsequent events. These AI history
975
+ reads always have explicit lower and upper bounds. A2 core can rebuild a missing
976
+ or invalid reducer snapshot from the log.
977
+
978
+ Streaming progress updates the current message incrementally. Snapshots preserve
979
+ open text and reasoning parts and partial tool inputs. Indexed progress remains
980
+ available for exact interruption cutoffs; reordered events rebuild the projection.
981
+
982
+ While tool arguments arrive, `input-streaming` parts expose a best-effort partial
983
+ JSON value in `input`. A command string can appear and grow before its closing
984
+ quote arrives. This is a display draft, not validated execution input.
985
+ `tool-input-available` supplies the completed input. Interruption still discards
986
+ streaming drafts even when their partial JSON is readable.
987
+
895
988
  ## Delivery semantics
896
989
 
897
990
  Model steps and tools run inside at-least-once A2 handlers. The returned-event
@@ -900,6 +993,12 @@ one atomic durable operation. Deterministic ids make scheduling and lifecycle
900
993
  appends effectively once, and A2 marks an incomplete model attempt
901
994
  `superseded` before starting its replacement.
902
995
 
996
+ An expired claim or superseded attempt preserves its abort reason through model,
997
+ resolver, compaction-policy, and tool calls. A2 retries that attempt without
998
+ spending its failure budget. Explicit user interruption remains terminal.
999
+ Streaming tool progress belongs to an execution attempt, so a retry can produce
1000
+ different preliminary output while preserving the tool-call id for idempotency.
1001
+
903
1002
  Provider calls and external tool side effects remain outside that transaction.
904
1003
  A process can die after an external effect succeeds and before its handler
905
1004
  completion commits, so recovery may run the call again. Give side-effecting
@@ -911,3 +1010,189 @@ continuation, but it cannot roll back that external effect.
911
1010
  Configure a [production scheduler](/guides/production) exactly as for any other A2
912
1011
  server. It recovers an incomplete generation attempt after the original
913
1012
  serverless invocation disappears.
1013
+
1014
+ ## Inbox and turn controls
1015
+
1016
+ This input protocol requires matching client and server versions. Start a new
1017
+ agent session when adopting it. Existing event logs remain readable; finish
1018
+ pending work with the runtime that created it before upgrading.
1019
+
1020
+ `inputs.message(message)` places a message in the durable inbox. The controller
1021
+ admits one input at a time, freezes its revision, and adds it to `state.messages`.
1022
+ Pending inputs appear in `state.inbox.items`; they do not enter the model prompt.
1023
+ Each item has an `id`, `revision`, `message`, and `generate` flag.
1024
+
1025
+ `state.active` identifies the logical turn, its frozen `input`, and its `phase`.
1026
+ The turn ID stays the same across model steps, tools, and a pause. Stop ends that
1027
+ turn and admits the next queued input when admission is running.
1028
+
1029
+ ```ts client/turn-controls.ts
1030
+ import { inputs, type AIState } from 'experimental-a2/ai'
1031
+ import type { UIMessage } from 'ai'
1032
+
1033
+ export function editPending(message: UIMessage, expectedRevision: number) {
1034
+ return inputs.queue.edit({ message, expectedRevision })
1035
+ }
1036
+
1037
+ export function movePending(inputId: string, beforeId: string | null) {
1038
+ return inputs.queue.move({ inputId, beforeId })
1039
+ }
1040
+
1041
+ export function removePending(inputId: string) {
1042
+ return inputs.queue.remove({ inputId })
1043
+ }
1044
+
1045
+ export function sendPendingNow(state: AIState, inputId: string) {
1046
+ return inputs.queue.sendNow({ inputId, turnId: state.active?.turnId ?? null })
1047
+ }
1048
+
1049
+ export function stopCurrent(state: AIState) {
1050
+ return state.active ? inputs.stop({ turnId: state.active.turnId }) : []
1051
+ }
1052
+
1053
+ export function steerCurrent(state: AIState, message: UIMessage) {
1054
+ return state.active
1055
+ ? inputs.steer({ turnId: state.active.turnId, message })
1056
+ : inputs.message(message)
1057
+ }
1058
+
1059
+ export const pauseAfterTurn = () => inputs.pause({ when: 'after-turn' })
1060
+ export const pauseNow = () => inputs.pause({ when: 'now' })
1061
+ export const resume = () => inputs.resume()
1062
+ ```
1063
+
1064
+ Pass these arrays to `push(...events)` from `useSession()`, or to a server
1065
+ session's `append(...events)`. `beforeId: null` moves an item to the end.
1066
+ Removal only applies to pending input. It preserves the immutable log and keeps
1067
+ the input identity reserved, so replay cannot recreate a removed prompt.
1068
+ Removing an admitted input returns `already-active`; removing an absent or
1069
+ already removed input returns `not-found`.
1070
+
1071
+ An edit that loses to admission returns `already-active`. Concurrent edits use
1072
+ `expectedRevision`; an outdated revision returns `revision-conflict`. A Stop
1073
+ aimed at a completed turn returns `stale-turn`.
1074
+
1075
+ Steer validates the new input and places it ahead of the inbox in one decision.
1076
+ If the observed turn is still active, it finishes its current model response and
1077
+ the tools produced by that response before handing off. Parallel tools all
1078
+ settle, and steering runs before the old turn starts another model step.
1079
+ Text and tool results produced after the click remain in the conversation.
1080
+ `lastSeenIndex` is not used to truncate output for steering or Send now.
1081
+
1082
+ When running work settles, unanswered approval or application-input requests
1083
+ are closed so the steering message can take over. The unapproved tool never
1084
+ executes. Provider-executed deferred results still count as running work when
1085
+ execution is authorized; a denied approval cannot block handoff.
1086
+ If the targeted turn has already finished, the input starts when idle or waits
1087
+ first behind a different active turn. Stop remains an immediate interruption.
1088
+
1089
+ Steering preserves the admission gate: steering while paused does not resume
1090
+ execution. Duplicate input identities and closed sessions still reject the command.
1091
+
1092
+ Send now uses the selected queued message's existing identity and latest
1093
+ accepted revision. It validates the selected input and the observed `turnId`,
1094
+ then moves the input first for the same handoff after the current step. Use
1095
+ `turnId: null` when no turn is active. If that expectation is stale, the command
1096
+ returns `stale-turn` without changing the queue.
1097
+
1098
+ Send now preserves pause just like steering. A paused queue keeps the selected
1099
+ input first until Resume. It accepts user-role inputs, including passive user
1100
+ input, which it marks for generation. Non-user context returns `not-user-input`;
1101
+ a conflicting response identity returns `duplicate-input`. Unsaved editor text
1102
+ is separate from the accepted queue revision; save it before sending.
1103
+
1104
+ A queued item's optional `afterStepOf` records the observed turn for steering.
1105
+ A matching active turn hands off after its current step.
1106
+ Editing preserves this intent; removing the item cancels it. Reordering the
1107
+ queue controls which input is admitted at that boundary.
1108
+
1109
+ ### Optimistic conversation display
1110
+
1111
+ With `agent().reducer` or `createReducer()`, the client state is already optimistic.
1112
+ An idle, unpaused send appears in `state.messages` immediately, with
1113
+ `state.starting: true`. Another send waits in `state.inbox.items` while that
1114
+ optimistic turn or a confirmed turn is active. Paused sessions keep new sends in
1115
+ the inbox. Render the conversation and queue from `state`; `state.active` and
1116
+ execution controls remain server-confirmed.
1117
+ Steering appears immediately in `state.messages`, after the current response,
1118
+ and stays there while that step finishes. Send now has the same display when
1119
+ an active turn is targeted. These inputs are omitted from the client inbox,
1120
+ including while paused, but remain pending on the server until admission.
1121
+ Accepted steering survives receipt and hydration in this display. Multiple
1122
+ steering inputs appear in admission order, with the most recent first.
1123
+ No additional request, subscription, or component-owned queue is needed.
1124
+
1125
+ ```tsx app/agent/[sessionId]/conversation.tsx
1126
+ 'use client'
1127
+ import { useSession } from '../session'
1128
+
1129
+ export function Conversation() {
1130
+ const { state } = useSession()
1131
+ const { messages, inbox, starting } = state
1132
+ const text = (message: (typeof messages)[number]) =>
1133
+ message.parts.filter((part) => part.type === 'text').map((part) => part.text).join(' ')
1134
+ return (
1135
+ <section>
1136
+ <div aria-label="Conversation">
1137
+ {messages.map((message) => <p key={message.id}>{text(message)}</p>)}
1138
+ {starting ? <p role="status">Starting…</p> : null}
1139
+ </div>
1140
+ <ol aria-label="Queued messages">
1141
+ {inbox.items.map((item) => <li key={item.id}>{text(item.message)}</li>)}
1142
+ </ol>
1143
+ </section>
1144
+ )
1145
+ }
1146
+ ```
1147
+
1148
+ The request echo keeps the optimistic display until `ai.control.decided` arrives. Accepted
1149
+ commits replace the confirmed base; rejected receipts remove the pending intent.
1150
+ A failed POST rolls back through the ordinary client optimistic overlay.
1151
+ Another client's accepted turn can move a speculative ordinary send back into
1152
+ the queue. Steering stays visible as a message while waiting behind that turn. `state.starting` describes an optimistic turn awaiting acceptance; it does
1153
+ not indicate a running model.
1154
+
1155
+ Server `session.state(reducer)`, reducer folds, checkpoints, and model prompts
1156
+ contain confirmed messages and inbox entries, with `starting: false`. The client
1157
+ applies pending intent and displays accepted steering only when returning its
1158
+ state, after folding events.
1159
+ A custom reducer that calls `reduceAIState()` or nests `agent.reducer.fold()`
1160
+ uses that confirmed fold; it does not inherit the built-in reducer's client display.
1161
+ `state.pendingQueueCommands` tracks unresolved queue commands incrementally, so
1162
+ the client does not scan raw history. Revisions remain server-confirmed; keep
1163
+ repeat editing and Send now disabled while an edit for that item awaits its
1164
+ receipt. Moving and removing an item do not require a revision.
1165
+
1166
+ ### Persistence and application are separate
1167
+
1168
+ `push()` confirms that the command is persisted. The `ai.control.decided` event
1169
+ confirms its application, using the submitted event's ID as `commandId`. Its
1170
+ `outcome` is `applied` or `rejected`; a rejection includes a `reason`.
1171
+ `state.receipt` exposes the most recent user command receipt. Applications that
1172
+ have several commands in flight correlate receipts through the existing event
1173
+ stream. A rejection is a normal result and does not spend a handler retry.
1174
+
1175
+ ### Pause and resume
1176
+
1177
+ Pause after the turn closes inbox admission and lets the active turn finish.
1178
+ Pause now also cancels current model and tool execution. `state.active.phase`
1179
+ becomes `pausing` while old work settles, then `paused`. Resume reopens admission
1180
+ and resumes retained work before admitting another input.
1181
+
1182
+ Resume starts a fresh model request from durable state. It does not resume the
1183
+ provider's old stream or undo completed tool effects. The controller retains
1184
+ accepted tool calls and records their outcomes while paused. A cancelled tool
1185
+ without a terminal result can execute again after resume; use the stable SDK
1186
+ `toolCallId` for external idempotency. No model or tool claim is held merely to
1187
+ keep a turn paused.
1188
+
1189
+ The controller runs on a short lane. Each decision commits accepted facts,
1190
+ compact state changes, work requests, and its receipt atomically. Workers use
1191
+ separate claims, report lifecycle batches, and wait for acceptance at model-start
1192
+ and compaction boundaries. Token progress appends directly. Explicit AI history
1193
+ reads remain bounded to a relevant generation suffix; core checkpoint rebuilding
1194
+ can replay older events when a snapshot is missing.
1195
+
1196
+ For a deferred provider-executed tool, trusted server code submits
1197
+ `inputs.toolResult(result)`. The controller checks its admitted call and records
1198
+ the result before continuing. Browser ingress rejects this server-only input.