@arnilo/prism 0.0.24 → 0.0.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -508,14 +508,144 @@ export interface SubscribeOptions {
508
508
  readonly overflow?: SubscriberOverflowPolicy;
509
509
  }
510
510
  export type AgentRunStatus = "succeeded" | "failed" | "aborted" | "suspended" | "denied";
511
- export type AgentRunInterruptionKind = "input_guardrail" | "tool_approval";
511
+ export type AgentRunInterruptionKind = "input_guardrail" | "tool_approval" | "elicitation";
512
+ export type ApprovalOutcome = "allow_once" | "allow_for_run" | "reject_once" | "reject_for_run";
513
+ export type PendingDecisionKind = "tool_approval" | "elicitation";
514
+ /** Redacted match scope for one pending or sticky decision; never contains raw tool arguments. */
515
+ export interface DecisionScope {
516
+ readonly toolName?: string;
517
+ readonly effectKind?: ToolEffectKind;
518
+ /** Redacted principal reference (tenant/kind/id); never a credential. */
519
+ readonly identity?: string;
520
+ /** Bounded argument-value constraints; deep-equal matched per key. */
521
+ readonly actionConstraints?: Readonly<Record<string, JsonValue>>;
522
+ /** SHA-256 of canonical JSON arguments; present instead of raw arguments. */
523
+ readonly argumentsHash?: string;
524
+ }
525
+ /** One redacted, unresolved approval request inside a suspended durable run. */
526
+ export interface PendingDecision {
527
+ /** Unique within the run; nested runs use supervisor-prefixed ids. */
528
+ readonly approvalId: string;
529
+ readonly kind: PendingDecisionKind;
530
+ readonly toolCallId?: string;
531
+ readonly scope: DecisionScope;
532
+ /** Bounded, redacted. */
533
+ readonly reason: string;
534
+ /** Typed payload contract for elicitation decisions. */
535
+ readonly elicitationSchema?: JsonObject;
536
+ /** Delegation chain, root-first; core-written, never client-supplied. */
537
+ readonly attribution?: {
538
+ readonly path: readonly string[];
539
+ };
540
+ }
512
541
  /** Redacted safe-boundary descriptor; never contains tool arguments. */
513
542
  export interface AgentRunInterruption {
514
543
  readonly kind: AgentRunInterruptionKind;
515
544
  readonly reason: string;
516
545
  readonly toolCallId?: string;
517
546
  readonly toolName?: string;
547
+ /** All unresolved approval requests of this suspension; absent for legacy single approvals. */
548
+ readonly pendingDecisions?: readonly PendingDecision[];
549
+ }
550
+ /** One host decision applied to one pending approval request. */
551
+ export interface RunDecision {
552
+ readonly approvalId: string;
553
+ readonly outcome: ApprovalOutcome;
554
+ /** Bounded to 2 KiB; redacted. */
555
+ readonly reason?: string;
556
+ /** Revalidated (schema, guardrails, policy) before dispatch; produces a new arguments hash. */
557
+ readonly modifiedArguments?: JsonObject;
558
+ /** Elicitation payload; validated against the pending decision's elicitationSchema. */
559
+ readonly elicitation?: JsonObject;
560
+ }
561
+ /** Run-scoped sticky decision; exact scope match, rechecked against policy, dropped at run end. */
562
+ export interface StickyDecision {
563
+ readonly scope: DecisionScope;
564
+ readonly outcome: "allow_for_run" | "reject_for_run";
565
+ readonly reason?: string;
566
+ readonly decidedAt: string;
567
+ /** Delegation path when the sticky was created for a nested-run decision. */
568
+ readonly attribution?: {
569
+ readonly path: readonly string[];
570
+ };
571
+ }
572
+ /** Root-visible link between one nested approval and the child-run approval id. */
573
+ export interface NestedRunApproval {
574
+ /** Root-visible approval id (hashed, non-enumerating across runs). */
575
+ readonly id: string;
576
+ /** Approval id as the nested run recorded it. */
577
+ readonly childApprovalId: string;
578
+ }
579
+ /** Root-visible link between a suspended nested run and the tool call that hosted it. */
580
+ export interface NestedRunRef {
581
+ readonly runId: string;
582
+ readonly sessionId?: string;
583
+ readonly toolCallId: string;
584
+ /** Redacted delegation path (child ids, root first). */
585
+ readonly path: readonly string[];
586
+ readonly approvals: readonly NestedRunApproval[];
587
+ /** Decisions persisted by a partial batch, keyed by root-visible approval id. */
588
+ readonly decisions?: Readonly<Record<string, RunDecision>>;
589
+ }
590
+ /** Outcome of resuming a nested run through the host-supplied hook. */
591
+ export type NestedRunOutcome = {
592
+ readonly status: "suspended";
593
+ readonly pendingDecisions: readonly PendingDecision[];
594
+ } | {
595
+ readonly status: "completed";
596
+ readonly value?: JsonValue;
597
+ } | {
598
+ readonly status: "failed";
599
+ readonly code: string;
600
+ readonly message: string;
601
+ };
602
+ /**
603
+ * Host hook that resumes a nested run (supervisor child) with child-visible decisions.
604
+ * Used both when a nested suspension first surfaces (sticky auto-apply) and when root
605
+ * decisions route back to the child on resume.
606
+ */
607
+ export type ResumeNestedRun = (nested: {
608
+ readonly ref: AgentRunRef;
609
+ readonly toolCallId: string;
610
+ readonly path: readonly string[];
611
+ }, decisions: readonly RunDecision[]) => Promise<NestedRunOutcome>;
612
+ /**
613
+ * Thrown by a delegated-run host (e.g. the supervisor) when a nested run suspends on
614
+ * pending decisions inside a tool execution. Core converts it into a root suspension
615
+ * with attributed, root-visible approval ids; the dispatching wrapper attaches `toolCall`.
616
+ */
617
+ export declare class AgentDelegationSuspendedError extends Error {
618
+ readonly ref: AgentRunRef;
619
+ readonly pendingDecisions: readonly PendingDecision[];
620
+ /** Redacted delegation path (child ids) used when decisions carry no attribution. */
621
+ readonly path?: readonly string[] | undefined;
622
+ readonly code = "ERR_PRISM_DELEGATION_SUSPENDED";
623
+ toolCall?: ToolCallContent;
624
+ constructor(ref: AgentRunRef, pendingDecisions: readonly PendingDecision[],
625
+ /** Redacted delegation path (child ids) used when decisions carry no attribution. */
626
+ path?: readonly string[] | undefined);
627
+ }
628
+ /** Shared decision-contract violations. Unknown and foreign approval ids share one non-enumerating error. */
629
+ export declare class AgentDecisionError extends Error {
630
+ readonly code: "ERR_PRISM_DECISION_STALE" | "ERR_PRISM_DECISION_UNKNOWN" | "ERR_PRISM_DECISION_DUPLICATE" | "ERR_PRISM_DECISION_SCOPE" | "ERR_PRISM_DECISION_INVALID" | "ERR_PRISM_DECISION_LIMIT";
631
+ constructor(code: "ERR_PRISM_DECISION_STALE" | "ERR_PRISM_DECISION_UNKNOWN" | "ERR_PRISM_DECISION_DUPLICATE" | "ERR_PRISM_DECISION_SCOPE" | "ERR_PRISM_DECISION_INVALID" | "ERR_PRISM_DECISION_LIMIT", message: string, options?: {
632
+ readonly cause?: unknown;
633
+ });
518
634
  }
635
+ export declare const DEFAULT_MAX_PENDING_DECISIONS = 32;
636
+ export declare const HARD_MAX_PENDING_DECISIONS = 128;
637
+ export declare const DEFAULT_MAX_STICKY_DECISIONS = 64;
638
+ export declare const HARD_MAX_STICKY_DECISIONS = 256;
639
+ export declare const MAX_DECISION_REASON_BYTES: number;
640
+ export declare const HARD_MAX_DECISION_REASON_BYTES: number;
641
+ export declare const MAX_ELICITATION_BYTES: number;
642
+ export declare const HARD_MAX_ELICITATION_BYTES: number;
643
+ export declare const MAX_ACTION_CONSTRAINTS = 32;
644
+ export declare const HARD_MAX_ACTION_CONSTRAINTS = 64;
645
+ /** Maximum delegation attribution depth for surfaced nested pending decisions. */
646
+ export declare const MAX_ATTRIBUTION_DEPTH = 8;
647
+ export declare const MAX_ACTION_CONSTRAINT_BYTES: number;
648
+ export declare const HARD_MAX_ACTION_CONSTRAINT_BYTES: number;
519
649
  export interface AgentRunStateOptions {
520
650
  readonly checkpoints: CheckpointStore;
521
651
  /** Host-authored immutable revision required for durable runs. */
@@ -524,6 +654,8 @@ export interface AgentRunStateOptions {
524
654
  readonly interruptBeforeTool?: boolean;
525
655
  readonly maxStateBytes?: number;
526
656
  readonly fencingToken?: number;
657
+ /** Enables sticky auto-apply when a nested suspension first surfaces during this run. */
658
+ readonly resumeNestedRun?: ResumeNestedRun;
527
659
  }
528
660
  /** Versioned, redacted checkpoint payload. Treat as opaque except status/version/interruption. */
529
661
  export interface AgentRunState {
@@ -540,8 +672,11 @@ export interface AgentRunState {
540
672
  readonly version?: number;
541
673
  }
542
674
  export interface AgentRunResume {
543
- readonly decision: "approve" | "deny";
544
675
  readonly expectedVersion: number;
676
+ /** Legacy single-approval path; `approve` allows all pending once, `deny` terminates the run denied. */
677
+ readonly decision?: "approve" | "deny";
678
+ /** Batch decision path; exactly one of decision/decisions. Applied as one atomic CAS transition. */
679
+ readonly decisions?: readonly RunDecision[];
545
680
  }
546
681
  export interface AgentRunResumeOptions {
547
682
  readonly checkpoints: CheckpointStore;
@@ -549,6 +684,8 @@ export interface AgentRunResumeOptions {
549
684
  readonly definitionRevision: string;
550
685
  readonly ownership?: OwnershipScope;
551
686
  readonly fencingToken?: number;
687
+ /** Routes root decisions for nested-run approvals back to the child (e.g. supervisor). */
688
+ readonly resumeNestedRun?: ResumeNestedRun;
552
689
  }
553
690
  /** Bounded, abortable options for `resumeAgentRunStream()`. */
554
691
  export interface AgentRunResumeStreamOptions extends AgentRunResumeOptions, SubscribeOptions {
@@ -566,6 +703,13 @@ export declare class AgentRunStateError extends Error {
566
703
  readonly code = "ERR_PRISM_AGENT_RUN_STATE";
567
704
  constructor(message: string);
568
705
  }
706
+ /** Durable-loop contract violations: hook-less custom strategy on a durable run, invalid snapshot, or revision drift. */
707
+ export declare class AgentLoopStateError extends Error {
708
+ readonly code: "ERR_PRISM_LOOP_NOT_DURABLE" | "ERR_PRISM_LOOP_SNAPSHOT" | "ERR_PRISM_LOOP_REVISION";
709
+ constructor(code: "ERR_PRISM_LOOP_NOT_DURABLE" | "ERR_PRISM_LOOP_SNAPSHOT" | "ERR_PRISM_LOOP_REVISION", message: string, options?: {
710
+ readonly cause?: unknown;
711
+ });
712
+ }
569
713
  /** Terminal result of `session.run()` / `session.prompt()`. Failed and aborted runs throw {@link AgentRunError} with this shape attached. */
570
714
  export interface AgentRunResult {
571
715
  readonly sessionId: string;
@@ -847,6 +991,19 @@ export interface ToolEffectDeclaration {
847
991
  }
848
992
  /** Runs after argument validation. It must be synchronous, deterministic, bounded, and side-effect-free. */
849
993
  export type ToolEffectClassifier = (args: JsonObject, context: ToolExecutionContext) => ToolEffectDeclaration;
994
+ /**
995
+ * Elicitation contract declared by a tool. When a durable gated run suspends on this tool,
996
+ * the pending decision has kind `elicitation` and carries this schema as its payload contract;
997
+ * the resume decision's `elicitation` payload resolves the call without executing it.
998
+ */
999
+ export interface ToolElicitationRequest {
1000
+ /** Typed payload contract; bounded to HARD_MAX_ELICITATION_BYTES when serialized. */
1001
+ readonly schema: JsonObject;
1002
+ /** Human-facing reason (e.g. the question); bounded to MAX_DECISION_REASON_BYTES. */
1003
+ readonly reason?: string;
1004
+ /** Answer-shape validation beyond structural schema checks; throw to reject the payload. */
1005
+ readonly validate?: (payload: JsonObject) => void;
1006
+ }
850
1007
  export interface ToolDefinition {
851
1008
  readonly name: string;
852
1009
  readonly description?: string;
@@ -855,6 +1012,8 @@ export interface ToolDefinition {
855
1012
  readonly exclusive?: boolean;
856
1013
  /** Optional side-effect declaration. Omitted tools retain legacy unmanaged dispatch. */
857
1014
  readonly effect?: ToolEffectDeclaration | ToolEffectClassifier;
1015
+ /** Optional elicitation contract for durable gating; return undefined to fall back to plain tool approval. */
1016
+ readonly elicitation?: (args: JsonObject, context: ToolExecutionContext) => ToolElicitationRequest | undefined;
858
1017
  execute(args: JsonObject, context: ToolExecutionContext): Promise<ToolResult> | ToolResult;
859
1018
  }
860
1019
  export interface ToolRegistry {
@@ -2022,8 +2181,13 @@ export interface LoopContext {
2022
2181
  /** Maximum independent tool calls dispatched concurrently per provider turn. Default `1`. */
2023
2182
  readonly toolConcurrency: number;
2024
2183
  assemble(nextInput: AgentInput, toolResults?: readonly ToolResult[], turn?: number): Promise<ProviderRequest>;
2025
- /** Charges a complete tool round before any call in it can start. */
2026
- chargeToolRound?(calls: readonly ToolCallContent[]): void;
2184
+ /**
2185
+ * Charges a complete tool round before any call in it can start. On durable interrupt
2186
+ * runs this is also the round-level approval gate: it collects every gated call of the
2187
+ * round into one suspension. Loops must await it; dispatch without it falls back to
2188
+ * per-call single-decision suspensions.
2189
+ */
2190
+ chargeToolRound?(calls: readonly ToolCallContent[]): void | Promise<void>;
2027
2191
  generate(request: ProviderRequest): Promise<ProviderTurnResult>;
2028
2192
  dispatchToolCall(call: ToolCallContent): Promise<ToolResult>;
2029
2193
  isToolCallExclusive?(call: ToolCallContent): boolean;
@@ -2033,10 +2197,23 @@ export interface LoopContext {
2033
2197
  hasPendingSteers?(): boolean;
2034
2198
  /** Drain pending steers into history/session. Returns true when any were applied. */
2035
2199
  applyPendingSteers?(): Promise<boolean>;
2200
+ /** Snapshot captured at the last suspension when the strategy declared snapshot/restore. Present only on resume. */
2201
+ readonly restoredLoopState?: JsonValue;
2036
2202
  }
2037
2203
  export interface AgentLoopStrategy {
2038
2204
  readonly name: string;
2205
+ /** Host-authored loop revision. Joins the durable-run fingerprint when snapshot hooks are present. */
2206
+ readonly revision?: string;
2039
2207
  run(ctx: LoopContext): Promise<Usage | undefined>;
2208
+ /**
2209
+ * Capture loop-local resumable state at suspension. Must return a JSON-compatible value;
2210
+ * core bounds and redacts it inside the durable run-state envelope. Declare together with
2211
+ * `restore`; a custom strategy without both hooks is rejected before any provider call on
2212
+ * durable runs (`AgentLoopStateError` / `ERR_PRISM_LOOP_NOT_DURABLE`).
2213
+ */
2214
+ snapshot?(): JsonValue;
2215
+ /** Rehydrate from a previously captured snapshot; must throw on drift. Called before `run` on resume. */
2216
+ restore?(snapshot: JsonValue): void;
2040
2217
  }
2041
2218
  export type AgentLoopOptions = {
2042
2219
  readonly strategy: "single-shot";
package/dist/contracts.js CHANGED
@@ -1,3 +1,47 @@
1
+ /**
2
+ * Thrown by a delegated-run host (e.g. the supervisor) when a nested run suspends on
3
+ * pending decisions inside a tool execution. Core converts it into a root suspension
4
+ * with attributed, root-visible approval ids; the dispatching wrapper attaches `toolCall`.
5
+ */
6
+ export class AgentDelegationSuspendedError extends Error {
7
+ ref;
8
+ pendingDecisions;
9
+ path;
10
+ code = "ERR_PRISM_DELEGATION_SUSPENDED";
11
+ toolCall;
12
+ constructor(ref, pendingDecisions,
13
+ /** Redacted delegation path (child ids) used when decisions carry no attribution. */
14
+ path) {
15
+ super("Delegated run suspended");
16
+ this.ref = ref;
17
+ this.pendingDecisions = pendingDecisions;
18
+ this.path = path;
19
+ this.name = "AgentDelegationSuspendedError";
20
+ }
21
+ }
22
+ /** Shared decision-contract violations. Unknown and foreign approval ids share one non-enumerating error. */
23
+ export class AgentDecisionError extends Error {
24
+ code;
25
+ constructor(code, message, options) {
26
+ super(message, options);
27
+ this.code = code;
28
+ this.name = "AgentDecisionError";
29
+ }
30
+ }
31
+ export const DEFAULT_MAX_PENDING_DECISIONS = 32;
32
+ export const HARD_MAX_PENDING_DECISIONS = 128;
33
+ export const DEFAULT_MAX_STICKY_DECISIONS = 64;
34
+ export const HARD_MAX_STICKY_DECISIONS = 256;
35
+ export const MAX_DECISION_REASON_BYTES = 2 * 1024;
36
+ export const HARD_MAX_DECISION_REASON_BYTES = 8 * 1024;
37
+ export const MAX_ELICITATION_BYTES = 16 * 1024;
38
+ export const HARD_MAX_ELICITATION_BYTES = 64 * 1024;
39
+ export const MAX_ACTION_CONSTRAINTS = 32;
40
+ export const HARD_MAX_ACTION_CONSTRAINTS = 64;
41
+ /** Maximum delegation attribution depth for surfaced nested pending decisions. */
42
+ export const MAX_ATTRIBUTION_DEPTH = 8;
43
+ export const MAX_ACTION_CONSTRAINT_BYTES = 4 * 1024;
44
+ export const HARD_MAX_ACTION_CONSTRAINT_BYTES = 16 * 1024;
1
45
  export class AgentRunStateError extends Error {
2
46
  code = "ERR_PRISM_AGENT_RUN_STATE";
3
47
  constructor(message) {
@@ -5,6 +49,15 @@ export class AgentRunStateError extends Error {
5
49
  this.name = "AgentRunStateError";
6
50
  }
7
51
  }
52
+ /** Durable-loop contract violations: hook-less custom strategy on a durable run, invalid snapshot, or revision drift. */
53
+ export class AgentLoopStateError extends Error {
54
+ code;
55
+ constructor(code, message, options) {
56
+ super(message, options);
57
+ this.code = code;
58
+ this.name = "AgentLoopStateError";
59
+ }
60
+ }
8
61
  export class AgentRunError extends Error {
9
62
  result;
10
63
  constructor(result, options) {
package/dist/index.d.ts CHANGED
@@ -2,7 +2,7 @@ export { resolveAgentDefinition } from "./agent-definitions.js";
2
2
  export { dispatchToolCallsInOrder, generateValidateReviseLoop, isAgentLoopOptions, resolveLoop, resolveToolConcurrency, singleShotLoop, } from "./agent-loops.js";
3
3
  export type { AgentRunLifecycle, AgentRunLifecycleAgent, AgentRunLifecycleOptions, AgentRunLifecycleRequest, AgentRunLifecycleStreamRequest, } from "./agent-run-lifecycle.js";
4
4
  export { createAgentRunLifecycle } from "./agent-run-lifecycle.js";
5
- export type { StoredAgentRunState } from "./agent-run-state.js";
5
+ export type { PendingToolCall, StoredAgentRunState } from "./agent-run-state.js";
6
6
  export { AGENT_RUN_STATE_NAMESPACE, AGENT_RUN_STATE_SCHEMA_VERSION, agentFingerprint, DEFAULT_MAX_AGENT_RUN_STATE_BYTES, HARD_MAX_AGENT_RUN_STATE_BYTES, loadAgentRunState, } from "./agent-run-state.js";
7
7
  export { createAgent, createAgentSession, resumeAgentRun, resumeAgentRunStream } from "./agents.js";
8
8
  export type { ArtifactApproval, ArtifactApprovalState, ArtifactCitation, ArtifactDecisionState, ArtifactDeliveryToken, ArtifactRecord, ArtifactRevision, } from "./artifacts.js";
@@ -20,8 +20,8 @@ export { assertDeclaredMediaTypeMatches, assertMediaBlocksWithinBounds, assertMe
20
20
  export type { ContextBudget, ContextBudgetMessageGroups, ContextBudgetOmission, ContextBudgetOmissionKind, ContextBudgetReport, } from "./context-budget.js";
21
21
  export { applyContextBudget, CONTEXT_BUDGET_ERROR_CODE, CONTEXT_BUDGET_REPORT_METADATA_KEY, ContextBudgetError, DEFAULT_MAX_CONTEXT_BUDGET_OMISSIONS, estimateAssemblyTokens, estimateMessageBytes, estimateMessageTokens, estimateTextBytes, estimateTextTokens, getContextBudgetReport, HARD_MAX_CONTEXT_BUDGET_BYTES, HARD_MAX_CONTEXT_BUDGET_OMISSIONS, HARD_MAX_CONTEXT_BUDGET_TOKENS, isContextBudgetError, resolveContextBudget, } from "./context-budget.js";
22
22
  export type * from "./contracts.js";
23
- export type { ProviderResolver, RealtimeCaps, RealtimeEvent, RealtimeSession, RealtimeSessionFactory, RealtimeSessionOptions, RunLimitCounters, RunLimitName, SecureAgentOptions, ToolCallAuthority, ToolEffectClassifier, ToolEffectDeclaration, ToolEffectIdempotency, ToolEffectKey, ToolEffectKind, ToolEffectRecord, ToolEffectStatus, ToolEffectStore, ToolEffectTransition, } from "./contracts.js";
24
- export { AgentRunError, AgentRunStateError, assertSessionMetadataKey, DEFAULT_MAX_PENDING_STEER_BYTES, DEFAULT_MAX_PENDING_STEERS, DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES, DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, DEFAULT_SESSION_SEARCH_LIMIT, HARD_MAX_PENDING_STEER_BYTES, HARD_MAX_PENDING_STEERS, HARD_MAX_SESSION_SEARCH_CURSOR_BYTES, HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES, HARD_MAX_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_LINEAR_BYTES, HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES, HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES, isSessionAppendConflict, isSessionEntryKind, isSessionSearchUnsupported, resolveSessionSearchQuery, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SESSION_SEARCH_UNSUPPORTED_CODE, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SessionAppendConflictError, SessionSearchUnsupportedError, } from "./contracts.js";
23
+ export type { ApprovalOutcome, DecisionScope, NestedRunApproval, NestedRunOutcome, NestedRunRef, PendingDecision, PendingDecisionKind, ProviderResolver, ResumeNestedRun, RunDecision, StickyDecision, RealtimeCaps, RealtimeEvent, RealtimeSession, RealtimeSessionFactory, RealtimeSessionOptions, RunLimitCounters, RunLimitName, SecureAgentOptions, ToolCallAuthority, ToolEffectClassifier, ToolEffectDeclaration, ToolEffectIdempotency, ToolEffectKey, ToolEffectKind, ToolEffectRecord, ToolEffectStatus, ToolEffectStore, ToolEffectTransition, ToolElicitationRequest, } from "./contracts.js";
24
+ export { AgentDecisionError, AgentDelegationSuspendedError, AgentLoopStateError, MAX_ATTRIBUTION_DEPTH, AgentRunError, AgentRunStateError, DEFAULT_MAX_PENDING_DECISIONS, DEFAULT_MAX_STICKY_DECISIONS, HARD_MAX_PENDING_DECISIONS, HARD_MAX_STICKY_DECISIONS, MAX_ACTION_CONSTRAINT_BYTES, MAX_ACTION_CONSTRAINTS, MAX_DECISION_REASON_BYTES, MAX_ELICITATION_BYTES, HARD_MAX_ACTION_CONSTRAINT_BYTES, HARD_MAX_ACTION_CONSTRAINTS, HARD_MAX_DECISION_REASON_BYTES, HARD_MAX_ELICITATION_BYTES, assertSessionMetadataKey, DEFAULT_MAX_PENDING_STEER_BYTES, DEFAULT_MAX_PENDING_STEERS, DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES, DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, DEFAULT_SESSION_SEARCH_LIMIT, HARD_MAX_PENDING_STEER_BYTES, HARD_MAX_PENDING_STEERS, HARD_MAX_SESSION_SEARCH_CURSOR_BYTES, HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES, HARD_MAX_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_LINEAR_BYTES, HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES, HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES, isSessionAppendConflict, isSessionEntryKind, isSessionSearchUnsupported, resolveSessionSearchQuery, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SESSION_SEARCH_UNSUPPORTED_CODE, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SessionAppendConflictError, SessionSearchUnsupportedError, } from "./contracts.js";
25
25
  export { parseAgentFile, parseSkillFile } from "./contribution-parsing.js";
26
26
  export type { ContributionRegistries, ContributionRegistriesOptions, ContributionRegistry, ContributionRegistryOptions, } from "./contributions.js";
27
27
  export { createContributionRegistries, createContributionRegistry, registerDiscoveredContributions } from "./contributions.js";
@@ -105,5 +105,5 @@ export type { ToolEffectErrorCode } from "./tool-effects.js";
105
105
  export type { ResolvedUseCaseModel, ResolveUseCaseModelInput, UseCaseModelBinding, } from "./use-case-model.js";
106
106
  export { resolveUseCaseModel, resolveUseCaseModelBinding, useCaseCredentialProviderId, } from "./use-case-model.js";
107
107
  export declare const name = "prism";
108
- export declare const version = "0.0.24";
108
+ export declare const version = "0.0.25";
109
109
  export declare const description = "Agent harness for AI providers, agents, sessions, and tools.";
package/dist/index.js CHANGED
@@ -10,7 +10,7 @@ export { createDefaultCompactionStrategy, isCompactionEntryData } from "./compac
10
10
  export { assertJsonObject, isJsonObject, loadConfigLayers, mergeConfigLayers } from "./config.js";
11
11
  export { assertDeclaredMediaTypeMatches, assertMediaBlocksWithinBounds, assertMessagesSupportModelCapabilities, assertModelSupportsContentBlocks, assertSsrfAllowedUrl, collectMessageContentBlocks, contentBlockInputModality, DEFAULT_MAX_AUDIO_DURATION_MS, DEFAULT_MAX_MEDIA_ITEM_BYTES, DEFAULT_MAX_MEDIA_ITEMS_PER_REQUEST, DEFAULT_MAX_MEDIA_REQUEST_BYTES, DEFAULT_MEDIA_FETCH_TIMEOUT_MS, loadBoundedBinaryResource, MediaContentError, MODEL_INPUT_CAPABILITIES, resolveMediaContentBlock, resolveMediaContentBlocks, sniffMediaMimeType, UnsupportedModalityError, } from "./content.js";
12
12
  export { applyContextBudget, CONTEXT_BUDGET_ERROR_CODE, CONTEXT_BUDGET_REPORT_METADATA_KEY, ContextBudgetError, DEFAULT_MAX_CONTEXT_BUDGET_OMISSIONS, estimateAssemblyTokens, estimateMessageBytes, estimateMessageTokens, estimateTextBytes, estimateTextTokens, getContextBudgetReport, HARD_MAX_CONTEXT_BUDGET_BYTES, HARD_MAX_CONTEXT_BUDGET_OMISSIONS, HARD_MAX_CONTEXT_BUDGET_TOKENS, isContextBudgetError, resolveContextBudget, } from "./context-budget.js";
13
- export { AgentRunError, AgentRunStateError, assertSessionMetadataKey, DEFAULT_MAX_PENDING_STEER_BYTES, DEFAULT_MAX_PENDING_STEERS, DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES, DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, DEFAULT_SESSION_SEARCH_LIMIT, HARD_MAX_PENDING_STEER_BYTES, HARD_MAX_PENDING_STEERS, HARD_MAX_SESSION_SEARCH_CURSOR_BYTES, HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES, HARD_MAX_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_LINEAR_BYTES, HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES, HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES, isSessionAppendConflict, isSessionEntryKind, isSessionSearchUnsupported, resolveSessionSearchQuery, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SESSION_SEARCH_UNSUPPORTED_CODE, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SessionAppendConflictError, SessionSearchUnsupportedError, } from "./contracts.js";
13
+ export { AgentDecisionError, AgentDelegationSuspendedError, AgentLoopStateError, MAX_ATTRIBUTION_DEPTH, AgentRunError, AgentRunStateError, DEFAULT_MAX_PENDING_DECISIONS, DEFAULT_MAX_STICKY_DECISIONS, HARD_MAX_PENDING_DECISIONS, HARD_MAX_STICKY_DECISIONS, MAX_ACTION_CONSTRAINT_BYTES, MAX_ACTION_CONSTRAINTS, MAX_DECISION_REASON_BYTES, MAX_ELICITATION_BYTES, HARD_MAX_ACTION_CONSTRAINT_BYTES, HARD_MAX_ACTION_CONSTRAINTS, HARD_MAX_DECISION_REASON_BYTES, HARD_MAX_ELICITATION_BYTES, assertSessionMetadataKey, DEFAULT_MAX_PENDING_STEER_BYTES, DEFAULT_MAX_PENDING_STEERS, DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES, DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, DEFAULT_SESSION_SEARCH_LIMIT, HARD_MAX_PENDING_STEER_BYTES, HARD_MAX_PENDING_STEERS, HARD_MAX_SESSION_SEARCH_CURSOR_BYTES, HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES, HARD_MAX_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_LINEAR_BYTES, HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES, HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES, isSessionAppendConflict, isSessionEntryKind, isSessionSearchUnsupported, resolveSessionSearchQuery, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SESSION_SEARCH_UNSUPPORTED_CODE, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SessionAppendConflictError, SessionSearchUnsupportedError, } from "./contracts.js";
14
14
  export { parseAgentFile, parseSkillFile } from "./contribution-parsing.js";
15
15
  export { createContributionRegistries, createContributionRegistry, registerDiscoveredContributions } from "./contributions.js";
16
16
  export { CONVERSATION_METADATA_KEY, ConversationError, conversationMarkerMetadata, conversationThreadFromRecord, DEFAULT_MAX_CONVERSATION_CURSOR_BYTES, decodeConversationReplayCursor, encodeConversationReplayCursor, HARD_MAX_CONVERSATION_CURSOR_BYTES, } from "./conversations.js";
@@ -57,6 +57,6 @@ export { createToolParameterValidator, createToolRegistry, dispatchToolCall, fil
57
57
  export { createMemoryToolEffectStore, ToolEffectError } from "./tool-effects.js";
58
58
  export { resolveUseCaseModel, resolveUseCaseModelBinding, useCaseCredentialProviderId, } from "./use-case-model.js";
59
59
  export const name = "prism";
60
- export const version = "0.0.24";
60
+ export const version = "0.0.25";
61
61
  export const description = "Agent harness for AI providers, agents, sessions, and tools.";
62
62
  //# sourceMappingURL=index.js.map
package/dist/tools.d.ts CHANGED
@@ -1,4 +1,4 @@
1
- import type { AgentEvent, ErrorInfo, Guardrails, JsonObject, OwnershipScope, RunLedger, ToolCallContent, ToolDefinition, ToolExecutionContext, ToolEffectStore, ToolRegistry, ToolResult } from "./contracts.js";
1
+ import type { AgentEvent, ErrorInfo, Guardrails, JsonObject, OwnershipScope, RunLedger, ToolCallContent, ToolDefinition, ToolExecutionContext, ToolEffectDeclaration, ToolEffectStore, ToolRegistry, ToolResult } from "./contracts.js";
2
2
  import type { MiddlewareRegistry } from "./middleware.js";
3
3
  import { type SecretRedactor } from "./redaction.js";
4
4
  import { type DuplicateRegistrationOptions } from "./registry-options.js";
@@ -57,3 +57,4 @@ export interface ToolRegistryOptions extends DuplicateRegistrationOptions {
57
57
  export declare function createToolRegistry(tools?: readonly ToolDefinition[], options?: ToolRegistryOptions): ToolRegistry;
58
58
  export declare function filterTools(tools: readonly ToolDefinition[], filter?: ToolFilterInput): readonly ToolDefinition[];
59
59
  export declare function dispatchToolCall(options: DispatchToolCallOptions): Promise<ToolResult>;
60
+ export declare function resolveToolEffectDeclaration(tool: ToolDefinition, args: JsonObject, context: ToolExecutionContext): ToolEffectDeclaration | undefined;
package/dist/tools.js CHANGED
@@ -153,7 +153,8 @@ export async function dispatchToolCall(options) {
153
153
  }
154
154
  catch (error) {
155
155
  await failBeforeEffect(effect, isSuspended(error) ? "failed_retryable" : "failed_terminal");
156
- if (isSuspended(error))
156
+ // Loop-state contract errors (snapshot capture) are terminal run errors, not tool errors.
157
+ if (isSuspended(error) || isLoopStateError(error) || isDelegationSuspended(error))
157
158
  throw error;
158
159
  return blocked(mediatedCall, context, "execution_denied", errorToErrorInfo(error, secrets), options, startedAt);
159
160
  }
@@ -222,6 +223,14 @@ export async function dispatchToolCall(options) {
222
223
  catch (error) {
223
224
  if (completedResult)
224
225
  return completedResult;
226
+ // Nested-run suspensions must propagate to the run loop, never become tool errors.
227
+ if (isDelegationSuspended(error)) {
228
+ if (effect && (effect.dispatched || dispatchAttempted))
229
+ await unknownEffectResult(effect, mediatedCall);
230
+ else
231
+ await failBeforeEffect(effect, "failed_retryable");
232
+ throw error;
233
+ }
225
234
  if (effect && (effect.dispatched || dispatchAttempted))
226
235
  return finishUnknownEffect(effect, mediatedCall, context, options, startedAt);
227
236
  await failBeforeEffect(effect, "failed_terminal");
@@ -298,7 +307,7 @@ async function prepareToolEffect(tool, call, context, options) {
298
307
  },
299
308
  };
300
309
  }
301
- function resolveToolEffectDeclaration(tool, args, context) {
310
+ export function resolveToolEffectDeclaration(tool, args, context) {
302
311
  const classifierContext = Object.freeze({
303
312
  sessionId: context.sessionId,
304
313
  runId: context.runId,
@@ -389,6 +398,12 @@ function effectErrorResult(call, code, message) {
389
398
  function isSuspended(error) {
390
399
  return error?.code === "ERR_PRISM_AGENT_RUN_SUSPENDED";
391
400
  }
401
+ function isLoopStateError(error) {
402
+ return typeof error?.code === "string" && error.code.startsWith("ERR_PRISM_LOOP_");
403
+ }
404
+ function isDelegationSuspended(error) {
405
+ return error?.code === "ERR_PRISM_DELEGATION_SUSPENDED";
406
+ }
392
407
  async function checkCall(call, options, startedAt) {
393
408
  const context = options.context;
394
409
  const tool = options.registry.get(call.name);
@@ -1,12 +1,12 @@
1
1
  # 0.1.0 / 1.0 Readiness Gates
2
2
 
3
- Status: **0.0.24** is the current release line (Phase 7 distributed events and recoverable tool effects); **1.0** readiness remains operator-gated, not automatic.
3
+ Status: **0.0.25** is the current release line (Phase 8 durable custom loops and human-in-the-loop); **1.0** readiness remains operator-gated, not automatic.
4
4
 
5
5
  This page distills runnable readiness gates into one command-per-gate table. The
6
6
  **Last evidence** column records the 2026-07-26 **0.0.16** baseline snapshot
7
7
  (Phase 11, Node v24.18.0, Linux x86_64). Treat it as historical floor evidence,
8
8
  not the current release tag. Re-run each gate on the target release tree before
9
- cutting 0.0.24 / 1.0. The decision to cut 1.0 stays with the operator after
9
+ cutting 0.0.25 / 1.0. The decision to cut 1.0 stays with the operator after
10
10
  operator-gated legs run in a protected environment and Phase 12 demand evidence
11
11
  exists.
12
12
 
@@ -14,15 +14,16 @@ Evidence trail: [`docs/review-coverage-2026-07-26-phase-11.md`](./review-coverag
14
14
  (addenda 0–9), [`docs/release-and-install.md`](./release-and-install.md),
15
15
  [`docs/migration.md`](./migration.md), [`docs/performance.md`](./performance.md).
16
16
 
17
- ## Current line (0.0.24)
17
+ ## Current line (0.0.25)
18
18
 
19
19
  | Item | Status |
20
20
  |---|---|
21
- | Published graph | **47** publishable manifests at **0.0.24** (`docs/release-and-install.md`) |
22
- | Phase 7 distributed events/effects | `AgentEventSource`, `ToolEffectStore`, AG-UI MCP/A2A fronting; enterprise `toolEffects`; schema v6/v7 + enterprise migration 002 |
23
- | Docs tripwires | `node --test dist/__tests__/docs.test.js` — migration section `0.0.23 → 0.0.24 distributed events and recoverable tool effects` |
24
- | Protected database evidence | `PRISM_TEST_POSTGRES_URL="$DATABASE_URL" npm run test:postgres`; `benchmark-0.0.24.json` under Task 0 ceilings |
25
- | Readiness table below | **0.0.16 measured values** remain historical network-free baseline; 0.0.24 database evidence is recorded separately |
21
+ | Published graph | **47** publishable manifests at **0.0.25** (`docs/release-and-install.md`) |
22
+ | Phase 8 durable loops / HITL | Custom-loop snapshot/restore, shared pending decisions, nested attributions, A2UI + standard AG-UI projectors |
23
+ | Docs tripwires | `node --test dist/__tests__/docs.test.js` — migration section `0.0.24 → 0.0.25 durable custom loops and human-in-the-loop` |
24
+ | Network-free Phase 8 evidence | `scripts/phase8-conformance.test.mjs`; `benchmark-0.0.25.json` under Task 0 ceilings |
25
+ | Protected database evidence (Phase 7) | `PRISM_TEST_POSTGRES_URL="$DATABASE_URL" npm run test:postgres`; `benchmark-0.0.24.json` under prior ceilings |
26
+ | Readiness table below | **0.0.16 measured values** remain historical network-free baseline; 0.0.25 loop/HITL evidence is recorded separately |
26
27
 
27
28
  ## Gate table
28
29
 
@@ -7,7 +7,7 @@ This page records Prism's compatibility review against official AG-UI `@ag-ui/co
7
7
  Official material reviewed:
8
8
 
9
9
  - [Events](https://docs.ag-ui.com/concepts/events), [messages](https://docs.ag-ui.com/concepts/messages), [tools](https://docs.ag-ui.com/concepts/tools), [state](https://docs.ag-ui.com/concepts/state), [reasoning](https://docs.ag-ui.com/concepts/reasoning), [interrupts](https://docs.ag-ui.com/concepts/interrupts), [capabilities](https://docs.ag-ui.com/concepts/capabilities), [serialization](https://docs.ag-ui.com/concepts/serialization), [server quickstart](https://docs.ag-ui.com/quickstart/server), and [protocol architecture](https://docs.ag-ui.com/concepts/architecture).
10
- - Official [MCP/A2A/AG-UI relationship](https://docs.ag-ui.com/agentic-protocols), [integrations](https://docs.ag-ui.com/integrations), [`@ag-ui/mcp-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/mcp-middleware), [`@ag-ui/mcp-apps-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/mcp-apps-middleware), [`@ag-ui/a2a`](https://github.com/ag-ui-protocol/ag-ui/tree/main/integrations/a2a/typescript), and [`@ag-ui/a2a-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/a2a-middleware).
10
+ - Official [MCP/A2A/AG-UI relationship](https://docs.ag-ui.com/agentic-protocols), [integrations](https://docs.ag-ui.com/integrations), [`@ag-ui/mcp-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/mcp-middleware), [`@ag-ui/mcp-apps-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/mcp-apps-middleware), [`@ag-ui/a2ui-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/a2ui-middleware) (Prism ships an in-package opt-in painter with frozen caps; no runtime dependency), [`@ag-ui/a2a`](https://github.com/ag-ui-protocol/ag-ui/tree/main/integrations/a2a/typescript), and [`@ag-ui/a2a-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/a2a-middleware).
11
11
  - A2A [current specification](https://a2a-protocol.org/latest/specification/) and [streaming rules](https://a2a-protocol.org/latest/topics/streaming-and-async/).
12
12
  - MCP Apps [SEP-1865](https://modelcontextprotocol.io/seps/1865-mcp-apps-interactive-user-interfaces-for-mcp) and the [`io.modelcontextprotocol/ui` draft specification](https://github.com/modelcontextprotocol/ext-apps/blob/main/specification/draft/apps.mdx).
13
13
 
package/docs/ag-ui.md CHANGED
@@ -35,7 +35,8 @@ npm install @arnilo/prism @arnilo/prism-ag-ui
35
35
  | `lifecycle` + `resolveRun` | Optional durable status/resume path. Required only for a resumed interruption. |
36
36
  | `interrupts.resume` | Optional host aggregate-policy callback for multiple AG-UI interrupts; it returns one current-version core approve/deny decision. |
37
37
  | `replay` | Optional page adapter or `createAgentEventSourceAgUiReplay(source, options)` for gap-free distributed replay/live follow. |
38
- | `projection` | Explicit safe tool/state/messages/activity/reasoning/raw/custom/interrupt projection. Omit each callback for default deny. |
38
+ | `projection` | Explicit safe tool/state/messages/activity/reasoning/raw/custom/interrupt projection. Omit each callback for default deny. Prefer `composeAgUiProjections(createMessagesFromSessionProjection(...), createStateFromStoreProjection(...), createActivityFromToolProgressProjection(), host)` for standard families. |
39
+ | `a2ui` | Opt-in A2UI painting middleware (`{ catalogId, mode, renderToolName?, allowedCatalogIds?, limits? }`). Detects `a2ui_operations` tool results and/or streams from `render_a2ui` args; paints `a2ui-surface` activity events. Absent = inert. |
39
40
  | `capabilities` | Optional host declaration narrowed to implemented SSE/projector/lifecycle features; read `handler.capabilities`. |
40
41
  | `redactor`, `limits` | Host redaction and narrowing-only finite caps. |
41
42
 
@@ -105,6 +106,43 @@ All identity, authorization, session/thread mapping, durable checkpoint lookup,
105
106
 
106
107
  Co-work projection reuses the same allow-list: `AgUiProjection.coWork(event)` may return a curated, JSON-serializable payload for a co-work event; absent it, the redacted event fields are exposed. Wire `coWorkContext` to derive thread/artifact/identity from the authorized request (never client JSON) and `coWork` to a `createCoWorkReplay()` over your durable artifact/draft/snapshot stores. The handler projects one bounded page after the run; mount a dedicated cursor-paged co-work endpoint when full pagination is needed.
107
108
 
109
+ Durable interrupts carry the shared decision batch: the fallback interrupt includes the redacted `pendingDecisions` under `metadata` and its `responseSchema` accepts either the legacy `{ decision: "approve" | "deny" }` or a `{ decisions: [{ approvalId, outcome, reason?, modifiedArguments?, elicitation? }] }` batch. All batch entries are shape- and cap-validated at the boundary (count ≤ 128, ids ≤ 128 chars, four outcomes, reason ≤ 8 KiB, payloads ≤ 64 KiB) and core re-validates each against the recorded pending set under the single CAS. `interrupts.resume` may return the batch form (`{ decisions, expectedVersion? }`); legacy `editedArgs` resume payloads still deny. ACP permission prompts offer the four outcomes (`allow_once` / `allow_always` / `reject_once` / `reject_always`) and map them onto the batch; a cancelled prompt stays deny-closed.
110
+
111
+ ### Standard projectors (opt-in)
112
+
113
+ Three batteries-included factories return `AgUiProjection` fragments. Compose with host projectors via `composeAgUiProjections(...fragments)` — **first defined callback wins** (left to right); `undefined` fragments are skipped. Absent factories keep 0.0.24 default-deny.
114
+
115
+ ```ts
116
+ createAgUiHandler({
117
+ projection: composeAgUiProjections(
118
+ createMessagesFromSessionProjection({ getMessages: () => authorizedAgUiMessages, redact }),
119
+ createStateFromStoreProjection(runStateStore),
120
+ createActivityFromToolProgressProjection(),
121
+ hostCustom,
122
+ ),
123
+ });
124
+ ```
125
+
126
+ | Factory | Emits | Notes |
127
+ | --- | --- | --- |
128
+ | `createMessagesFromSessionProjection` | `MESSAGES_SNAPSHOT` | Host `getMessages()` for authorized history, or live `message_finished` accumulation. Caps 128/1024. Redact drops closed. |
129
+ | `createStateFromStoreProjection(store)` | `STATE_SNAPSHOT` on `agent_started`; RFC 6902 `STATE_DELTA` (add/replace/remove) when `store.get()` changes | Host store; optional `subscribe` only marks dirty — no Prism watcher. Oversized/throw → drop closed. |
130
+ | `createActivityFromToolProgressProjection` | `ACTIVITY_SNAPSHOT` / `ACTIVITY_DELTA` from `tool_execution_progress` | Default `activityType: "tool-progress"`. Missing progress+metadata → drop closed. |
131
+
132
+ ### A2UI painting middleware (opt-in)
133
+
134
+ `createAgUiHandler({ a2ui: { catalogId, mode } })` paints A2UI v0.9 surfaces without a host `projection.activity` callback:
135
+
136
+ | Mode | Source | Paint |
137
+ | --- | --- | --- |
138
+ | `fixed-schema` | Tool result `{ a2ui_operations: [...] }` | First batch with `createSurface` → `ACTIVITY_SNAPSHOT` (`activityType: "a2ui-surface"`); later batches → `ACTIVITY_DELTA` |
139
+ | `streaming` | `tool_call_delta` args of `renderToolName` (default `render_a2ui`) | Progressive `ACTIVITY_SNAPSHOT` with `replace: true` only when complete ops extractable — never partial JSON |
140
+ | `both` | Both paths; streamed surfaces are not re-painted from the final envelope |
141
+
142
+ Host `catalogId` is stamped when absent; model-supplied ids outside `allowedCatalogIds` (default `[catalogId]`) are overwritten. Invalid ops emit one bounded `CUSTOM` event `prism.a2ui.error` and paint nothing. Caps: 64/512 ops per message, 64 KiB/1 MiB per op, 16/64 surfaces per run, depth 32/64.
143
+
144
+ User actions arrive as untrusted `AgUiA2UiAction` values on `input.project({ a2uiActions })` (from `forwardedProps.a2uiAction` or activity/tool-result shapes). Without `input.project` they stay default-deny — Prism never synthesizes a `log_a2ui_event` tool call (documented divergence from official `@ag-ui/a2ui-middleware`). Example: `examples/ag-ui-a2ui.ts`.
145
+
108
146
  ## Security and performance notes
109
147
 
110
148
  Authorize every start/replay/resume/proxy/follow. Treat protocol fields, MCP metadata/HTML, and A2A cards/parts as untrusted; persist run/task correlation before output and redact streams. MCP Apps requires extension acknowledgement, exact proxy origin, same-bridge visibility, approval, `ui://` HTML/MIME bounds, and sandbox CSP. It never retries UI mutations; Task 4 adds recovery.
@@ -118,7 +118,15 @@ Optional steer hooks on `LoopContext` (0.0.11): `hasPendingSteers?()` / `applyPe
118
118
 
119
119
  ## Durable runs
120
120
 
121
- `RunOptions.runState` supports only built-in loop options (`single-shot` and `generate-validate-revise`). A custom `AgentLoopStrategy` has arbitrary in-memory cursor state, so durable configuration rejects it before provider work. Built-in suspension occurs only before an input provider call or immediately before a tool side effect; completed provider turns remain in `SessionStore` history and are not repeated after `resumeAgentRun()`.
121
+ `RunOptions.runState` supports the built-in loop options (`single-shot` and `generate-validate-revise`) and custom strategies that opt into durable state. `single-shot` is durable via the runtime's pending-call mechanism and carries no loop-local state. `generate-validate-revise` snapshots `{ attempts, artifactPhase, savedSchema, pendingHistory }` at `revision: "1"`. A custom `AgentLoopStrategy` must declare both snapshot hooks or durable configuration rejects it with `AgentLoopStateError` (`ERR_PRISM_LOOP_NOT_DURABLE`) before any provider call:
122
+
123
+ | Member | Purpose |
124
+ | --- | --- |
125
+ | `revision?: string` | Host-authored loop revision. Joins the durable-run fingerprint, so a loop change without a `definitionRevision` bump fails closed on resume. |
126
+ | `snapshot?(): JsonValue` | Capture loop-local resumable state at suspension. Must be JSON-compatible; core redacts it and bounds it inside the durable run-state envelope (`maxStateBytes`, depth 32). A non-JSON value fails the run with `ERR_PRISM_LOOP_SNAPSHOT`. |
127
+ | `restore?(snapshot): void` | Rehydrate from the captured snapshot; must throw on drift. Called once before `run(ctx)` on resume. Also available as `ctx.restoredLoopState`. |
128
+
129
+ The snapshot is stored as `loopState: { name, revision, snapshot }` on the durable run state and cleared when the run reaches a terminal status. On resume, a name/revision mismatch between the stored `loopState` and the resolved strategy fails closed (`ERR_PRISM_LOOP_REVISION`), and the fingerprint check independently rejects any loop drift. Suspension occurs only before an input provider call or immediately before a tool side effect; completed provider turns remain in `SessionStore` history and are not repeated after `resumeAgentRun()`.
122
130
 
123
131
  ## Outputs / response / events
124
132
 
@@ -176,7 +176,14 @@ await agent.createSession().run("Hi", { model: overrideModel });
176
176
 
177
177
  ## Durable interruption
178
178
 
179
- Set `runState` with a host-owned `CheckpointStore`, stable `definitionRevision`, and `interruptBeforeTool: true` to suspend at a persisted pre-side-effect boundary. A suspended result has `status: "suspended"`, a redacted `interruption`, and `runState.version`; it releases session resources before returning.
179
+ Set `runState` with a host-owned `CheckpointStore`, stable `definitionRevision`, and `interruptBeforeTool: true` to suspend at a persisted pre-side-effect boundary. A suspended result has `status: "suspended"`, a redacted `interruption`, and `runState.version`; it releases session resources before returning. When a provider turn requests several tools, the round is collected into **one** suspension whose `interruption.pendingDecisions` holds one redacted `PendingDecision` per gated call (`approvalId`, kind, scope with tool name/effect kind/identity/arguments hash — never raw arguments); ungated calls still dispatch.
180
+
181
+ `resumeAgentRun` accepts exactly one of:
182
+
183
+ - `decision: "approve" | "deny"` — legacy single-approval path. `approve` allows every pending decision once; `deny` terminates the run as `denied`.
184
+ - `decisions: readonly RunDecision[]` — one atomic batch. Every entry validates against the recorded pending set (unknown/foreign `approvalId`, duplicates, stale `expectedVersion`, invalid outcomes fail the whole batch closed with `AgentDecisionError` and leave state and version untouched). Outcomes: `allow_once`, `allow_for_run`, `reject_once`, `reject_for_run`. `reject_*` continues the run with a blocked tool result carrying the bounded (2 KB) `reason`. `modifiedArguments` are revalidated (schema, then input guardrails; permission/trust re-run at dispatch) and produce a new arguments hash. `elicitation` payloads are validated against the pending decision's `elicitationSchema` (required keys plus the configured host validator) and resolve the suspended call without executing it. A batch deciding a strict subset persists the decided entries and re-suspends with the remainder pending at the bumped version.
185
+
186
+ `*_for_run` outcomes append a `StickyDecision` to the durable run state: later calls in the same run matching the scope exactly (all recorded fields) proceed or are blocked without a new suspension, policy still enforced at dispatch. Sticky decisions expire when the run reaches any terminal status. Caps: 32 pending decisions per run (hard 128), 64 sticky decisions (hard 256), 2 KB decision reasons, 16 KB elicitation payloads.
180
187
 
181
188
  ```ts
182
189
  const result = await session.run("Publish draft", {
@@ -189,7 +196,7 @@ if (result.status === "suspended") {
189
196
  }
190
197
  ```
191
198
 
192
- Resume requires exact checkpoint ownership, version, agent fingerprint, and revision. The fingerprint hashes the agent id/name, `definitionRevision`, model, instructions, system-prompt contributions, skills (name/instructions/tool names), tool definitions (name/parameters/exclusive), guardrail definitions (name/stage/revision), and loop strategy — changing any of them without bumping `definitionRevision` fails resume closed instead of silently continuing with different agent semantics. Prism CAS-claims approval before work, rechecks normal guardrail/permission/validation/limit paths, and marks a pending tool dispatched before its side effect. `createAgentRunLifecycle()` wraps the same core path for server/MCP hosts: adapters pass only authorized ownership, status returns only `{ state, version }`, and `resolveAgent()` supplies current agent/revision. `resumeStream()` uses that same claim path and bounded subscriber, so adapters do not poll or duplicate resume logic. Remote restart requires both checkpoint and session stores to be durable. A crash after that mark is ambiguous and is never replayed automatically; use host tool idempotency keyed by `runId`/`toolCallId` or resolve it manually. Checkpoints contain bounded redacted state plus session/leaf references, never provider objects, callbacks, signals, credentials, or raw secrets. State is bounded at save by `runState.maxStateBytes` (default 256 KB, at most the 1 MB hard cap); load bounds against the 1 MB hard cap only, so state saved with a raised limit stays resumable while oversized records are still rejected. Only built-in loop options are durable; custom `AgentLoopStrategy` rejects before provider work.
199
+ Resume requires exact checkpoint ownership, version, agent fingerprint, and revision. The fingerprint hashes the agent id/name, `definitionRevision`, model, instructions, system-prompt contributions, skills (name/instructions/tool names), tool definitions (name/parameters/exclusive), guardrail definitions (name/stage/revision), and loop strategy — changing any of them without bumping `definitionRevision` fails resume closed instead of silently continuing with different agent semantics. Prism CAS-claims approval before work, rechecks normal guardrail/permission/validation/limit paths, and marks a pending tool dispatched before its side effect. `createAgentRunLifecycle()` wraps the same core path for server/MCP hosts: adapters pass only authorized ownership, status returns only `{ state, version }`, and `resolveAgent()` supplies current agent/revision. `resumeStream()` uses that same claim path and bounded subscriber, so adapters do not poll or duplicate resume logic. Remote restart requires both checkpoint and session stores to be durable. A crash after that mark is ambiguous and is never replayed automatically; use host tool idempotency keyed by `runId`/`toolCallId` or resolve it manually. Checkpoints contain bounded redacted state plus session/leaf references, never provider objects, callbacks, signals, credentials, or raw secrets. State is bounded at save by `runState.maxStateBytes` (default 256 KB, at most the 1 MB hard cap); load bounds against the 1 MB hard cap only, so state saved with a raised limit stays resumable while oversized records are still rejected. Built-in loop options are durable; custom `AgentLoopStrategy` instances are durable when they declare `snapshot`/`restore` hooks (see [Agent loops § Durable runs](agent-loops.md#durable-runs)) and reject before provider work otherwise.
193
200
 
194
201
  ## Secure composition
195
202
 
@@ -143,7 +143,7 @@ Policies are ordinary host values: attach one globally through `createCodingTool
143
143
 
144
144
  The Docker reference adapter starts by recorded container ID/label, uses argument arrays only, mounts source read-only, populates a size-bounded tmpfs `/workspace`, drops all capabilities, enables `no-new-privileges`, runs with `--init`, and never exposes the Docker socket, privileged mode, or host PID/IPC namespaces. Image pull/build/update stays outside Prism. Protected real-Docker checks are opt-in via `PRISM_TEST_DOCKER_SANDBOX=1` with host-supplied `PRISM_TEST_DOCKER_BIN` and digest-pinned `PRISM_TEST_DOCKER_IMAGE`.
145
145
 
146
- Callback approval remains process-local. For approval that must survive restart, wrap the action in an opted-in workflow `toolNode({ approval: { reason, data?, resumeSchema? } })`. The workflow persists `suspended` state before any tool side effect. After explicit approve, it recomputes the action and invokes this package's current `ExecutionPolicy`; durable approval never populates or bypasses the process-local approval cache. Adapters should emit chunks through `request.onData` as they arrive and honor `request.signal`/`request.timeout`; buffering is unnecessary. Coding-agent composes caller abort with its total-output controller, so ignoring the supplied signal defeats process termination even though Prism stops retaining output at the cap. Default caching is `none`; use run-scoped caching only when repeated approval within one run is desired, and session scope only when that wider lifecycle is intentional.
146
+ Callback approval remains process-local. For approval that must survive restart, wrap the action in an opted-in workflow `toolNode({ approval: { reason, data?, resumeSchema? } })`. `ask_user_decision` also maps onto the shared decision model: inside a durable gated agent run its call suspends as a kind-`elicitation` pending decision whose schema carries the choice contract (option-id enums) plus the full question/options UX payload on a Prism-owned schema extension property; the resume decision's `elicitation` payload (a `selectedId`/`selectedIds`/`customText` answer) is validated against the schema and the tool-level answer-shape rules, then resolves the call without invoking the blocking `ask()` callback. The process-local `ask()` path and the workflow suspend/resume path are unchanged. The workflow persists `suspended` state before any tool side effect. After explicit approve, it recomputes the action and invokes this package's current `ExecutionPolicy`; durable approval never populates or bypasses the process-local approval cache. Adapters should emit chunks through `request.onData` as they arrive and honor `request.signal`/`request.timeout`; buffering is unnecessary. Coding-agent composes caller abort with its total-output controller, so ignoring the supplied signal defeats process termination even though Prism stops retaining output at the cap. Default caching is `none`; use run-scoped caching only when repeated approval within one run is desired, and session scope only when that wider lifecycle is intentional.
147
147
 
148
148
  ## Security and performance notes
149
149