@arnilo/prism 0.0.24 → 0.0.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/dist/agent-loops.js +37 -5
- package/dist/agent-run-state.d.ts +27 -1
- package/dist/agent-run-state.js +80 -4
- package/dist/agents.js +884 -77
- package/dist/contracts.d.ts +181 -4
- package/dist/contracts.js +53 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +2 -2
- package/dist/tools.d.ts +2 -1
- package/dist/tools.js +17 -2
- package/docs/0.1.0-readiness.md +9 -8
- package/docs/ag-ui-adoption.md +1 -1
- package/docs/ag-ui.md +39 -1
- package/docs/agent-loops.md +9 -1
- package/docs/agent-session-runtime.md +9 -2
- package/docs/coding-security.md +1 -1
- package/docs/index.md +6 -6
- package/docs/mcp-tools.md +2 -0
- package/docs/migration.md +24 -0
- package/docs/performance.md +7 -6
- package/docs/release-and-install.md +33 -12
- package/docs/server.md +1 -0
- package/docs/supervisors.md +4 -0
- package/docs/workflows.md +1 -1
- package/package.json +2 -2
package/dist/contracts.d.ts
CHANGED
|
@@ -508,14 +508,144 @@ export interface SubscribeOptions {
|
|
|
508
508
|
readonly overflow?: SubscriberOverflowPolicy;
|
|
509
509
|
}
|
|
510
510
|
export type AgentRunStatus = "succeeded" | "failed" | "aborted" | "suspended" | "denied";
|
|
511
|
-
export type AgentRunInterruptionKind = "input_guardrail" | "tool_approval";
|
|
511
|
+
export type AgentRunInterruptionKind = "input_guardrail" | "tool_approval" | "elicitation";
|
|
512
|
+
export type ApprovalOutcome = "allow_once" | "allow_for_run" | "reject_once" | "reject_for_run";
|
|
513
|
+
export type PendingDecisionKind = "tool_approval" | "elicitation";
|
|
514
|
+
/** Redacted match scope for one pending or sticky decision; never contains raw tool arguments. */
|
|
515
|
+
export interface DecisionScope {
|
|
516
|
+
readonly toolName?: string;
|
|
517
|
+
readonly effectKind?: ToolEffectKind;
|
|
518
|
+
/** Redacted principal reference (tenant/kind/id); never a credential. */
|
|
519
|
+
readonly identity?: string;
|
|
520
|
+
/** Bounded argument-value constraints; deep-equal matched per key. */
|
|
521
|
+
readonly actionConstraints?: Readonly<Record<string, JsonValue>>;
|
|
522
|
+
/** SHA-256 of canonical JSON arguments; present instead of raw arguments. */
|
|
523
|
+
readonly argumentsHash?: string;
|
|
524
|
+
}
|
|
525
|
+
/** One redacted, unresolved approval request inside a suspended durable run. */
|
|
526
|
+
export interface PendingDecision {
|
|
527
|
+
/** Unique within the run; nested runs use supervisor-prefixed ids. */
|
|
528
|
+
readonly approvalId: string;
|
|
529
|
+
readonly kind: PendingDecisionKind;
|
|
530
|
+
readonly toolCallId?: string;
|
|
531
|
+
readonly scope: DecisionScope;
|
|
532
|
+
/** Bounded, redacted. */
|
|
533
|
+
readonly reason: string;
|
|
534
|
+
/** Typed payload contract for elicitation decisions. */
|
|
535
|
+
readonly elicitationSchema?: JsonObject;
|
|
536
|
+
/** Delegation chain, root-first; core-written, never client-supplied. */
|
|
537
|
+
readonly attribution?: {
|
|
538
|
+
readonly path: readonly string[];
|
|
539
|
+
};
|
|
540
|
+
}
|
|
512
541
|
/** Redacted safe-boundary descriptor; never contains tool arguments. */
|
|
513
542
|
export interface AgentRunInterruption {
|
|
514
543
|
readonly kind: AgentRunInterruptionKind;
|
|
515
544
|
readonly reason: string;
|
|
516
545
|
readonly toolCallId?: string;
|
|
517
546
|
readonly toolName?: string;
|
|
547
|
+
/** All unresolved approval requests of this suspension; absent for legacy single approvals. */
|
|
548
|
+
readonly pendingDecisions?: readonly PendingDecision[];
|
|
549
|
+
}
|
|
550
|
+
/** One host decision applied to one pending approval request. */
|
|
551
|
+
export interface RunDecision {
|
|
552
|
+
readonly approvalId: string;
|
|
553
|
+
readonly outcome: ApprovalOutcome;
|
|
554
|
+
/** Bounded to 2 KiB; redacted. */
|
|
555
|
+
readonly reason?: string;
|
|
556
|
+
/** Revalidated (schema, guardrails, policy) before dispatch; produces a new arguments hash. */
|
|
557
|
+
readonly modifiedArguments?: JsonObject;
|
|
558
|
+
/** Elicitation payload; validated against the pending decision's elicitationSchema. */
|
|
559
|
+
readonly elicitation?: JsonObject;
|
|
560
|
+
}
|
|
561
|
+
/** Run-scoped sticky decision; exact scope match, rechecked against policy, dropped at run end. */
|
|
562
|
+
export interface StickyDecision {
|
|
563
|
+
readonly scope: DecisionScope;
|
|
564
|
+
readonly outcome: "allow_for_run" | "reject_for_run";
|
|
565
|
+
readonly reason?: string;
|
|
566
|
+
readonly decidedAt: string;
|
|
567
|
+
/** Delegation path when the sticky was created for a nested-run decision. */
|
|
568
|
+
readonly attribution?: {
|
|
569
|
+
readonly path: readonly string[];
|
|
570
|
+
};
|
|
571
|
+
}
|
|
572
|
+
/** Root-visible link between one nested approval and the child-run approval id. */
|
|
573
|
+
export interface NestedRunApproval {
|
|
574
|
+
/** Root-visible approval id (hashed, non-enumerating across runs). */
|
|
575
|
+
readonly id: string;
|
|
576
|
+
/** Approval id as the nested run recorded it. */
|
|
577
|
+
readonly childApprovalId: string;
|
|
578
|
+
}
|
|
579
|
+
/** Root-visible link between a suspended nested run and the tool call that hosted it. */
|
|
580
|
+
export interface NestedRunRef {
|
|
581
|
+
readonly runId: string;
|
|
582
|
+
readonly sessionId?: string;
|
|
583
|
+
readonly toolCallId: string;
|
|
584
|
+
/** Redacted delegation path (child ids, root first). */
|
|
585
|
+
readonly path: readonly string[];
|
|
586
|
+
readonly approvals: readonly NestedRunApproval[];
|
|
587
|
+
/** Decisions persisted by a partial batch, keyed by root-visible approval id. */
|
|
588
|
+
readonly decisions?: Readonly<Record<string, RunDecision>>;
|
|
589
|
+
}
|
|
590
|
+
/** Outcome of resuming a nested run through the host-supplied hook. */
|
|
591
|
+
export type NestedRunOutcome = {
|
|
592
|
+
readonly status: "suspended";
|
|
593
|
+
readonly pendingDecisions: readonly PendingDecision[];
|
|
594
|
+
} | {
|
|
595
|
+
readonly status: "completed";
|
|
596
|
+
readonly value?: JsonValue;
|
|
597
|
+
} | {
|
|
598
|
+
readonly status: "failed";
|
|
599
|
+
readonly code: string;
|
|
600
|
+
readonly message: string;
|
|
601
|
+
};
|
|
602
|
+
/**
|
|
603
|
+
* Host hook that resumes a nested run (supervisor child) with child-visible decisions.
|
|
604
|
+
* Used both when a nested suspension first surfaces (sticky auto-apply) and when root
|
|
605
|
+
* decisions route back to the child on resume.
|
|
606
|
+
*/
|
|
607
|
+
export type ResumeNestedRun = (nested: {
|
|
608
|
+
readonly ref: AgentRunRef;
|
|
609
|
+
readonly toolCallId: string;
|
|
610
|
+
readonly path: readonly string[];
|
|
611
|
+
}, decisions: readonly RunDecision[]) => Promise<NestedRunOutcome>;
|
|
612
|
+
/**
|
|
613
|
+
* Thrown by a delegated-run host (e.g. the supervisor) when a nested run suspends on
|
|
614
|
+
* pending decisions inside a tool execution. Core converts it into a root suspension
|
|
615
|
+
* with attributed, root-visible approval ids; the dispatching wrapper attaches `toolCall`.
|
|
616
|
+
*/
|
|
617
|
+
export declare class AgentDelegationSuspendedError extends Error {
|
|
618
|
+
readonly ref: AgentRunRef;
|
|
619
|
+
readonly pendingDecisions: readonly PendingDecision[];
|
|
620
|
+
/** Redacted delegation path (child ids) used when decisions carry no attribution. */
|
|
621
|
+
readonly path?: readonly string[] | undefined;
|
|
622
|
+
readonly code = "ERR_PRISM_DELEGATION_SUSPENDED";
|
|
623
|
+
toolCall?: ToolCallContent;
|
|
624
|
+
constructor(ref: AgentRunRef, pendingDecisions: readonly PendingDecision[],
|
|
625
|
+
/** Redacted delegation path (child ids) used when decisions carry no attribution. */
|
|
626
|
+
path?: readonly string[] | undefined);
|
|
627
|
+
}
|
|
628
|
+
/** Shared decision-contract violations. Unknown and foreign approval ids share one non-enumerating error. */
|
|
629
|
+
export declare class AgentDecisionError extends Error {
|
|
630
|
+
readonly code: "ERR_PRISM_DECISION_STALE" | "ERR_PRISM_DECISION_UNKNOWN" | "ERR_PRISM_DECISION_DUPLICATE" | "ERR_PRISM_DECISION_SCOPE" | "ERR_PRISM_DECISION_INVALID" | "ERR_PRISM_DECISION_LIMIT";
|
|
631
|
+
constructor(code: "ERR_PRISM_DECISION_STALE" | "ERR_PRISM_DECISION_UNKNOWN" | "ERR_PRISM_DECISION_DUPLICATE" | "ERR_PRISM_DECISION_SCOPE" | "ERR_PRISM_DECISION_INVALID" | "ERR_PRISM_DECISION_LIMIT", message: string, options?: {
|
|
632
|
+
readonly cause?: unknown;
|
|
633
|
+
});
|
|
518
634
|
}
|
|
635
|
+
export declare const DEFAULT_MAX_PENDING_DECISIONS = 32;
|
|
636
|
+
export declare const HARD_MAX_PENDING_DECISIONS = 128;
|
|
637
|
+
export declare const DEFAULT_MAX_STICKY_DECISIONS = 64;
|
|
638
|
+
export declare const HARD_MAX_STICKY_DECISIONS = 256;
|
|
639
|
+
export declare const MAX_DECISION_REASON_BYTES: number;
|
|
640
|
+
export declare const HARD_MAX_DECISION_REASON_BYTES: number;
|
|
641
|
+
export declare const MAX_ELICITATION_BYTES: number;
|
|
642
|
+
export declare const HARD_MAX_ELICITATION_BYTES: number;
|
|
643
|
+
export declare const MAX_ACTION_CONSTRAINTS = 32;
|
|
644
|
+
export declare const HARD_MAX_ACTION_CONSTRAINTS = 64;
|
|
645
|
+
/** Maximum delegation attribution depth for surfaced nested pending decisions. */
|
|
646
|
+
export declare const MAX_ATTRIBUTION_DEPTH = 8;
|
|
647
|
+
export declare const MAX_ACTION_CONSTRAINT_BYTES: number;
|
|
648
|
+
export declare const HARD_MAX_ACTION_CONSTRAINT_BYTES: number;
|
|
519
649
|
export interface AgentRunStateOptions {
|
|
520
650
|
readonly checkpoints: CheckpointStore;
|
|
521
651
|
/** Host-authored immutable revision required for durable runs. */
|
|
@@ -524,6 +654,8 @@ export interface AgentRunStateOptions {
|
|
|
524
654
|
readonly interruptBeforeTool?: boolean;
|
|
525
655
|
readonly maxStateBytes?: number;
|
|
526
656
|
readonly fencingToken?: number;
|
|
657
|
+
/** Enables sticky auto-apply when a nested suspension first surfaces during this run. */
|
|
658
|
+
readonly resumeNestedRun?: ResumeNestedRun;
|
|
527
659
|
}
|
|
528
660
|
/** Versioned, redacted checkpoint payload. Treat as opaque except status/version/interruption. */
|
|
529
661
|
export interface AgentRunState {
|
|
@@ -540,8 +672,11 @@ export interface AgentRunState {
|
|
|
540
672
|
readonly version?: number;
|
|
541
673
|
}
|
|
542
674
|
export interface AgentRunResume {
|
|
543
|
-
readonly decision: "approve" | "deny";
|
|
544
675
|
readonly expectedVersion: number;
|
|
676
|
+
/** Legacy single-approval path; `approve` allows all pending once, `deny` terminates the run denied. */
|
|
677
|
+
readonly decision?: "approve" | "deny";
|
|
678
|
+
/** Batch decision path; exactly one of decision/decisions. Applied as one atomic CAS transition. */
|
|
679
|
+
readonly decisions?: readonly RunDecision[];
|
|
545
680
|
}
|
|
546
681
|
export interface AgentRunResumeOptions {
|
|
547
682
|
readonly checkpoints: CheckpointStore;
|
|
@@ -549,6 +684,8 @@ export interface AgentRunResumeOptions {
|
|
|
549
684
|
readonly definitionRevision: string;
|
|
550
685
|
readonly ownership?: OwnershipScope;
|
|
551
686
|
readonly fencingToken?: number;
|
|
687
|
+
/** Routes root decisions for nested-run approvals back to the child (e.g. supervisor). */
|
|
688
|
+
readonly resumeNestedRun?: ResumeNestedRun;
|
|
552
689
|
}
|
|
553
690
|
/** Bounded, abortable options for `resumeAgentRunStream()`. */
|
|
554
691
|
export interface AgentRunResumeStreamOptions extends AgentRunResumeOptions, SubscribeOptions {
|
|
@@ -566,6 +703,13 @@ export declare class AgentRunStateError extends Error {
|
|
|
566
703
|
readonly code = "ERR_PRISM_AGENT_RUN_STATE";
|
|
567
704
|
constructor(message: string);
|
|
568
705
|
}
|
|
706
|
+
/** Durable-loop contract violations: hook-less custom strategy on a durable run, invalid snapshot, or revision drift. */
|
|
707
|
+
export declare class AgentLoopStateError extends Error {
|
|
708
|
+
readonly code: "ERR_PRISM_LOOP_NOT_DURABLE" | "ERR_PRISM_LOOP_SNAPSHOT" | "ERR_PRISM_LOOP_REVISION";
|
|
709
|
+
constructor(code: "ERR_PRISM_LOOP_NOT_DURABLE" | "ERR_PRISM_LOOP_SNAPSHOT" | "ERR_PRISM_LOOP_REVISION", message: string, options?: {
|
|
710
|
+
readonly cause?: unknown;
|
|
711
|
+
});
|
|
712
|
+
}
|
|
569
713
|
/** Terminal result of `session.run()` / `session.prompt()`. Failed and aborted runs throw {@link AgentRunError} with this shape attached. */
|
|
570
714
|
export interface AgentRunResult {
|
|
571
715
|
readonly sessionId: string;
|
|
@@ -847,6 +991,19 @@ export interface ToolEffectDeclaration {
|
|
|
847
991
|
}
|
|
848
992
|
/** Runs after argument validation. It must be synchronous, deterministic, bounded, and side-effect-free. */
|
|
849
993
|
export type ToolEffectClassifier = (args: JsonObject, context: ToolExecutionContext) => ToolEffectDeclaration;
|
|
994
|
+
/**
|
|
995
|
+
* Elicitation contract declared by a tool. When a durable gated run suspends on this tool,
|
|
996
|
+
* the pending decision has kind `elicitation` and carries this schema as its payload contract;
|
|
997
|
+
* the resume decision's `elicitation` payload resolves the call without executing it.
|
|
998
|
+
*/
|
|
999
|
+
export interface ToolElicitationRequest {
|
|
1000
|
+
/** Typed payload contract; bounded to HARD_MAX_ELICITATION_BYTES when serialized. */
|
|
1001
|
+
readonly schema: JsonObject;
|
|
1002
|
+
/** Human-facing reason (e.g. the question); bounded to MAX_DECISION_REASON_BYTES. */
|
|
1003
|
+
readonly reason?: string;
|
|
1004
|
+
/** Answer-shape validation beyond structural schema checks; throw to reject the payload. */
|
|
1005
|
+
readonly validate?: (payload: JsonObject) => void;
|
|
1006
|
+
}
|
|
850
1007
|
export interface ToolDefinition {
|
|
851
1008
|
readonly name: string;
|
|
852
1009
|
readonly description?: string;
|
|
@@ -855,6 +1012,8 @@ export interface ToolDefinition {
|
|
|
855
1012
|
readonly exclusive?: boolean;
|
|
856
1013
|
/** Optional side-effect declaration. Omitted tools retain legacy unmanaged dispatch. */
|
|
857
1014
|
readonly effect?: ToolEffectDeclaration | ToolEffectClassifier;
|
|
1015
|
+
/** Optional elicitation contract for durable gating; return undefined to fall back to plain tool approval. */
|
|
1016
|
+
readonly elicitation?: (args: JsonObject, context: ToolExecutionContext) => ToolElicitationRequest | undefined;
|
|
858
1017
|
execute(args: JsonObject, context: ToolExecutionContext): Promise<ToolResult> | ToolResult;
|
|
859
1018
|
}
|
|
860
1019
|
export interface ToolRegistry {
|
|
@@ -2022,8 +2181,13 @@ export interface LoopContext {
|
|
|
2022
2181
|
/** Maximum independent tool calls dispatched concurrently per provider turn. Default `1`. */
|
|
2023
2182
|
readonly toolConcurrency: number;
|
|
2024
2183
|
assemble(nextInput: AgentInput, toolResults?: readonly ToolResult[], turn?: number): Promise<ProviderRequest>;
|
|
2025
|
-
/**
|
|
2026
|
-
|
|
2184
|
+
/**
|
|
2185
|
+
* Charges a complete tool round before any call in it can start. On durable interrupt
|
|
2186
|
+
* runs this is also the round-level approval gate: it collects every gated call of the
|
|
2187
|
+
* round into one suspension. Loops must await it; dispatch without it falls back to
|
|
2188
|
+
* per-call single-decision suspensions.
|
|
2189
|
+
*/
|
|
2190
|
+
chargeToolRound?(calls: readonly ToolCallContent[]): void | Promise<void>;
|
|
2027
2191
|
generate(request: ProviderRequest): Promise<ProviderTurnResult>;
|
|
2028
2192
|
dispatchToolCall(call: ToolCallContent): Promise<ToolResult>;
|
|
2029
2193
|
isToolCallExclusive?(call: ToolCallContent): boolean;
|
|
@@ -2033,10 +2197,23 @@ export interface LoopContext {
|
|
|
2033
2197
|
hasPendingSteers?(): boolean;
|
|
2034
2198
|
/** Drain pending steers into history/session. Returns true when any were applied. */
|
|
2035
2199
|
applyPendingSteers?(): Promise<boolean>;
|
|
2200
|
+
/** Snapshot captured at the last suspension when the strategy declared snapshot/restore. Present only on resume. */
|
|
2201
|
+
readonly restoredLoopState?: JsonValue;
|
|
2036
2202
|
}
|
|
2037
2203
|
export interface AgentLoopStrategy {
|
|
2038
2204
|
readonly name: string;
|
|
2205
|
+
/** Host-authored loop revision. Joins the durable-run fingerprint when snapshot hooks are present. */
|
|
2206
|
+
readonly revision?: string;
|
|
2039
2207
|
run(ctx: LoopContext): Promise<Usage | undefined>;
|
|
2208
|
+
/**
|
|
2209
|
+
* Capture loop-local resumable state at suspension. Must return a JSON-compatible value;
|
|
2210
|
+
* core bounds and redacts it inside the durable run-state envelope. Declare together with
|
|
2211
|
+
* `restore`; a custom strategy without both hooks is rejected before any provider call on
|
|
2212
|
+
* durable runs (`AgentLoopStateError` / `ERR_PRISM_LOOP_NOT_DURABLE`).
|
|
2213
|
+
*/
|
|
2214
|
+
snapshot?(): JsonValue;
|
|
2215
|
+
/** Rehydrate from a previously captured snapshot; must throw on drift. Called before `run` on resume. */
|
|
2216
|
+
restore?(snapshot: JsonValue): void;
|
|
2040
2217
|
}
|
|
2041
2218
|
export type AgentLoopOptions = {
|
|
2042
2219
|
readonly strategy: "single-shot";
|
package/dist/contracts.js
CHANGED
|
@@ -1,3 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Thrown by a delegated-run host (e.g. the supervisor) when a nested run suspends on
|
|
3
|
+
* pending decisions inside a tool execution. Core converts it into a root suspension
|
|
4
|
+
* with attributed, root-visible approval ids; the dispatching wrapper attaches `toolCall`.
|
|
5
|
+
*/
|
|
6
|
+
export class AgentDelegationSuspendedError extends Error {
|
|
7
|
+
ref;
|
|
8
|
+
pendingDecisions;
|
|
9
|
+
path;
|
|
10
|
+
code = "ERR_PRISM_DELEGATION_SUSPENDED";
|
|
11
|
+
toolCall;
|
|
12
|
+
constructor(ref, pendingDecisions,
|
|
13
|
+
/** Redacted delegation path (child ids) used when decisions carry no attribution. */
|
|
14
|
+
path) {
|
|
15
|
+
super("Delegated run suspended");
|
|
16
|
+
this.ref = ref;
|
|
17
|
+
this.pendingDecisions = pendingDecisions;
|
|
18
|
+
this.path = path;
|
|
19
|
+
this.name = "AgentDelegationSuspendedError";
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
/** Shared decision-contract violations. Unknown and foreign approval ids share one non-enumerating error. */
|
|
23
|
+
export class AgentDecisionError extends Error {
|
|
24
|
+
code;
|
|
25
|
+
constructor(code, message, options) {
|
|
26
|
+
super(message, options);
|
|
27
|
+
this.code = code;
|
|
28
|
+
this.name = "AgentDecisionError";
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
export const DEFAULT_MAX_PENDING_DECISIONS = 32;
|
|
32
|
+
export const HARD_MAX_PENDING_DECISIONS = 128;
|
|
33
|
+
export const DEFAULT_MAX_STICKY_DECISIONS = 64;
|
|
34
|
+
export const HARD_MAX_STICKY_DECISIONS = 256;
|
|
35
|
+
export const MAX_DECISION_REASON_BYTES = 2 * 1024;
|
|
36
|
+
export const HARD_MAX_DECISION_REASON_BYTES = 8 * 1024;
|
|
37
|
+
export const MAX_ELICITATION_BYTES = 16 * 1024;
|
|
38
|
+
export const HARD_MAX_ELICITATION_BYTES = 64 * 1024;
|
|
39
|
+
export const MAX_ACTION_CONSTRAINTS = 32;
|
|
40
|
+
export const HARD_MAX_ACTION_CONSTRAINTS = 64;
|
|
41
|
+
/** Maximum delegation attribution depth for surfaced nested pending decisions. */
|
|
42
|
+
export const MAX_ATTRIBUTION_DEPTH = 8;
|
|
43
|
+
export const MAX_ACTION_CONSTRAINT_BYTES = 4 * 1024;
|
|
44
|
+
export const HARD_MAX_ACTION_CONSTRAINT_BYTES = 16 * 1024;
|
|
1
45
|
export class AgentRunStateError extends Error {
|
|
2
46
|
code = "ERR_PRISM_AGENT_RUN_STATE";
|
|
3
47
|
constructor(message) {
|
|
@@ -5,6 +49,15 @@ export class AgentRunStateError extends Error {
|
|
|
5
49
|
this.name = "AgentRunStateError";
|
|
6
50
|
}
|
|
7
51
|
}
|
|
52
|
+
/** Durable-loop contract violations: hook-less custom strategy on a durable run, invalid snapshot, or revision drift. */
|
|
53
|
+
export class AgentLoopStateError extends Error {
|
|
54
|
+
code;
|
|
55
|
+
constructor(code, message, options) {
|
|
56
|
+
super(message, options);
|
|
57
|
+
this.code = code;
|
|
58
|
+
this.name = "AgentLoopStateError";
|
|
59
|
+
}
|
|
60
|
+
}
|
|
8
61
|
export class AgentRunError extends Error {
|
|
9
62
|
result;
|
|
10
63
|
constructor(result, options) {
|
package/dist/index.d.ts
CHANGED
|
@@ -2,7 +2,7 @@ export { resolveAgentDefinition } from "./agent-definitions.js";
|
|
|
2
2
|
export { dispatchToolCallsInOrder, generateValidateReviseLoop, isAgentLoopOptions, resolveLoop, resolveToolConcurrency, singleShotLoop, } from "./agent-loops.js";
|
|
3
3
|
export type { AgentRunLifecycle, AgentRunLifecycleAgent, AgentRunLifecycleOptions, AgentRunLifecycleRequest, AgentRunLifecycleStreamRequest, } from "./agent-run-lifecycle.js";
|
|
4
4
|
export { createAgentRunLifecycle } from "./agent-run-lifecycle.js";
|
|
5
|
-
export type { StoredAgentRunState } from "./agent-run-state.js";
|
|
5
|
+
export type { PendingToolCall, StoredAgentRunState } from "./agent-run-state.js";
|
|
6
6
|
export { AGENT_RUN_STATE_NAMESPACE, AGENT_RUN_STATE_SCHEMA_VERSION, agentFingerprint, DEFAULT_MAX_AGENT_RUN_STATE_BYTES, HARD_MAX_AGENT_RUN_STATE_BYTES, loadAgentRunState, } from "./agent-run-state.js";
|
|
7
7
|
export { createAgent, createAgentSession, resumeAgentRun, resumeAgentRunStream } from "./agents.js";
|
|
8
8
|
export type { ArtifactApproval, ArtifactApprovalState, ArtifactCitation, ArtifactDecisionState, ArtifactDeliveryToken, ArtifactRecord, ArtifactRevision, } from "./artifacts.js";
|
|
@@ -20,8 +20,8 @@ export { assertDeclaredMediaTypeMatches, assertMediaBlocksWithinBounds, assertMe
|
|
|
20
20
|
export type { ContextBudget, ContextBudgetMessageGroups, ContextBudgetOmission, ContextBudgetOmissionKind, ContextBudgetReport, } from "./context-budget.js";
|
|
21
21
|
export { applyContextBudget, CONTEXT_BUDGET_ERROR_CODE, CONTEXT_BUDGET_REPORT_METADATA_KEY, ContextBudgetError, DEFAULT_MAX_CONTEXT_BUDGET_OMISSIONS, estimateAssemblyTokens, estimateMessageBytes, estimateMessageTokens, estimateTextBytes, estimateTextTokens, getContextBudgetReport, HARD_MAX_CONTEXT_BUDGET_BYTES, HARD_MAX_CONTEXT_BUDGET_OMISSIONS, HARD_MAX_CONTEXT_BUDGET_TOKENS, isContextBudgetError, resolveContextBudget, } from "./context-budget.js";
|
|
22
22
|
export type * from "./contracts.js";
|
|
23
|
-
export type { ProviderResolver, RealtimeCaps, RealtimeEvent, RealtimeSession, RealtimeSessionFactory, RealtimeSessionOptions, RunLimitCounters, RunLimitName, SecureAgentOptions, ToolCallAuthority, ToolEffectClassifier, ToolEffectDeclaration, ToolEffectIdempotency, ToolEffectKey, ToolEffectKind, ToolEffectRecord, ToolEffectStatus, ToolEffectStore, ToolEffectTransition, } from "./contracts.js";
|
|
24
|
-
export { AgentRunError, AgentRunStateError, assertSessionMetadataKey, DEFAULT_MAX_PENDING_STEER_BYTES, DEFAULT_MAX_PENDING_STEERS, DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES, DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, DEFAULT_SESSION_SEARCH_LIMIT, HARD_MAX_PENDING_STEER_BYTES, HARD_MAX_PENDING_STEERS, HARD_MAX_SESSION_SEARCH_CURSOR_BYTES, HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES, HARD_MAX_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_LINEAR_BYTES, HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES, HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES, isSessionAppendConflict, isSessionEntryKind, isSessionSearchUnsupported, resolveSessionSearchQuery, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SESSION_SEARCH_UNSUPPORTED_CODE, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SessionAppendConflictError, SessionSearchUnsupportedError, } from "./contracts.js";
|
|
23
|
+
export type { ApprovalOutcome, DecisionScope, NestedRunApproval, NestedRunOutcome, NestedRunRef, PendingDecision, PendingDecisionKind, ProviderResolver, ResumeNestedRun, RunDecision, StickyDecision, RealtimeCaps, RealtimeEvent, RealtimeSession, RealtimeSessionFactory, RealtimeSessionOptions, RunLimitCounters, RunLimitName, SecureAgentOptions, ToolCallAuthority, ToolEffectClassifier, ToolEffectDeclaration, ToolEffectIdempotency, ToolEffectKey, ToolEffectKind, ToolEffectRecord, ToolEffectStatus, ToolEffectStore, ToolEffectTransition, ToolElicitationRequest, } from "./contracts.js";
|
|
24
|
+
export { AgentDecisionError, AgentDelegationSuspendedError, AgentLoopStateError, MAX_ATTRIBUTION_DEPTH, AgentRunError, AgentRunStateError, DEFAULT_MAX_PENDING_DECISIONS, DEFAULT_MAX_STICKY_DECISIONS, HARD_MAX_PENDING_DECISIONS, HARD_MAX_STICKY_DECISIONS, MAX_ACTION_CONSTRAINT_BYTES, MAX_ACTION_CONSTRAINTS, MAX_DECISION_REASON_BYTES, MAX_ELICITATION_BYTES, HARD_MAX_ACTION_CONSTRAINT_BYTES, HARD_MAX_ACTION_CONSTRAINTS, HARD_MAX_DECISION_REASON_BYTES, HARD_MAX_ELICITATION_BYTES, assertSessionMetadataKey, DEFAULT_MAX_PENDING_STEER_BYTES, DEFAULT_MAX_PENDING_STEERS, DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES, DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, DEFAULT_SESSION_SEARCH_LIMIT, HARD_MAX_PENDING_STEER_BYTES, HARD_MAX_PENDING_STEERS, HARD_MAX_SESSION_SEARCH_CURSOR_BYTES, HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES, HARD_MAX_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_LINEAR_BYTES, HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES, HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES, isSessionAppendConflict, isSessionEntryKind, isSessionSearchUnsupported, resolveSessionSearchQuery, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SESSION_SEARCH_UNSUPPORTED_CODE, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SessionAppendConflictError, SessionSearchUnsupportedError, } from "./contracts.js";
|
|
25
25
|
export { parseAgentFile, parseSkillFile } from "./contribution-parsing.js";
|
|
26
26
|
export type { ContributionRegistries, ContributionRegistriesOptions, ContributionRegistry, ContributionRegistryOptions, } from "./contributions.js";
|
|
27
27
|
export { createContributionRegistries, createContributionRegistry, registerDiscoveredContributions } from "./contributions.js";
|
|
@@ -105,5 +105,5 @@ export type { ToolEffectErrorCode } from "./tool-effects.js";
|
|
|
105
105
|
export type { ResolvedUseCaseModel, ResolveUseCaseModelInput, UseCaseModelBinding, } from "./use-case-model.js";
|
|
106
106
|
export { resolveUseCaseModel, resolveUseCaseModelBinding, useCaseCredentialProviderId, } from "./use-case-model.js";
|
|
107
107
|
export declare const name = "prism";
|
|
108
|
-
export declare const version = "0.0.
|
|
108
|
+
export declare const version = "0.0.25";
|
|
109
109
|
export declare const description = "Agent harness for AI providers, agents, sessions, and tools.";
|
package/dist/index.js
CHANGED
|
@@ -10,7 +10,7 @@ export { createDefaultCompactionStrategy, isCompactionEntryData } from "./compac
|
|
|
10
10
|
export { assertJsonObject, isJsonObject, loadConfigLayers, mergeConfigLayers } from "./config.js";
|
|
11
11
|
export { assertDeclaredMediaTypeMatches, assertMediaBlocksWithinBounds, assertMessagesSupportModelCapabilities, assertModelSupportsContentBlocks, assertSsrfAllowedUrl, collectMessageContentBlocks, contentBlockInputModality, DEFAULT_MAX_AUDIO_DURATION_MS, DEFAULT_MAX_MEDIA_ITEM_BYTES, DEFAULT_MAX_MEDIA_ITEMS_PER_REQUEST, DEFAULT_MAX_MEDIA_REQUEST_BYTES, DEFAULT_MEDIA_FETCH_TIMEOUT_MS, loadBoundedBinaryResource, MediaContentError, MODEL_INPUT_CAPABILITIES, resolveMediaContentBlock, resolveMediaContentBlocks, sniffMediaMimeType, UnsupportedModalityError, } from "./content.js";
|
|
12
12
|
export { applyContextBudget, CONTEXT_BUDGET_ERROR_CODE, CONTEXT_BUDGET_REPORT_METADATA_KEY, ContextBudgetError, DEFAULT_MAX_CONTEXT_BUDGET_OMISSIONS, estimateAssemblyTokens, estimateMessageBytes, estimateMessageTokens, estimateTextBytes, estimateTextTokens, getContextBudgetReport, HARD_MAX_CONTEXT_BUDGET_BYTES, HARD_MAX_CONTEXT_BUDGET_OMISSIONS, HARD_MAX_CONTEXT_BUDGET_TOKENS, isContextBudgetError, resolveContextBudget, } from "./context-budget.js";
|
|
13
|
-
export { AgentRunError, AgentRunStateError, assertSessionMetadataKey, DEFAULT_MAX_PENDING_STEER_BYTES, DEFAULT_MAX_PENDING_STEERS, DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES, DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, DEFAULT_SESSION_SEARCH_LIMIT, HARD_MAX_PENDING_STEER_BYTES, HARD_MAX_PENDING_STEERS, HARD_MAX_SESSION_SEARCH_CURSOR_BYTES, HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES, HARD_MAX_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_LINEAR_BYTES, HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES, HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES, isSessionAppendConflict, isSessionEntryKind, isSessionSearchUnsupported, resolveSessionSearchQuery, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SESSION_SEARCH_UNSUPPORTED_CODE, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SessionAppendConflictError, SessionSearchUnsupportedError, } from "./contracts.js";
|
|
13
|
+
export { AgentDecisionError, AgentDelegationSuspendedError, AgentLoopStateError, MAX_ATTRIBUTION_DEPTH, AgentRunError, AgentRunStateError, DEFAULT_MAX_PENDING_DECISIONS, DEFAULT_MAX_STICKY_DECISIONS, HARD_MAX_PENDING_DECISIONS, HARD_MAX_STICKY_DECISIONS, MAX_ACTION_CONSTRAINT_BYTES, MAX_ACTION_CONSTRAINTS, MAX_DECISION_REASON_BYTES, MAX_ELICITATION_BYTES, HARD_MAX_ACTION_CONSTRAINT_BYTES, HARD_MAX_ACTION_CONSTRAINTS, HARD_MAX_DECISION_REASON_BYTES, HARD_MAX_ELICITATION_BYTES, assertSessionMetadataKey, DEFAULT_MAX_PENDING_STEER_BYTES, DEFAULT_MAX_PENDING_STEERS, DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES, DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, DEFAULT_SESSION_SEARCH_LIMIT, HARD_MAX_PENDING_STEER_BYTES, HARD_MAX_PENDING_STEERS, HARD_MAX_SESSION_SEARCH_CURSOR_BYTES, HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES, HARD_MAX_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_LINEAR_BYTES, HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES, HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES, isSessionAppendConflict, isSessionEntryKind, isSessionSearchUnsupported, resolveSessionSearchQuery, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SESSION_SEARCH_UNSUPPORTED_CODE, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SessionAppendConflictError, SessionSearchUnsupportedError, } from "./contracts.js";
|
|
14
14
|
export { parseAgentFile, parseSkillFile } from "./contribution-parsing.js";
|
|
15
15
|
export { createContributionRegistries, createContributionRegistry, registerDiscoveredContributions } from "./contributions.js";
|
|
16
16
|
export { CONVERSATION_METADATA_KEY, ConversationError, conversationMarkerMetadata, conversationThreadFromRecord, DEFAULT_MAX_CONVERSATION_CURSOR_BYTES, decodeConversationReplayCursor, encodeConversationReplayCursor, HARD_MAX_CONVERSATION_CURSOR_BYTES, } from "./conversations.js";
|
|
@@ -57,6 +57,6 @@ export { createToolParameterValidator, createToolRegistry, dispatchToolCall, fil
|
|
|
57
57
|
export { createMemoryToolEffectStore, ToolEffectError } from "./tool-effects.js";
|
|
58
58
|
export { resolveUseCaseModel, resolveUseCaseModelBinding, useCaseCredentialProviderId, } from "./use-case-model.js";
|
|
59
59
|
export const name = "prism";
|
|
60
|
-
export const version = "0.0.
|
|
60
|
+
export const version = "0.0.25";
|
|
61
61
|
export const description = "Agent harness for AI providers, agents, sessions, and tools.";
|
|
62
62
|
//# sourceMappingURL=index.js.map
|
package/dist/tools.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { AgentEvent, ErrorInfo, Guardrails, JsonObject, OwnershipScope, RunLedger, ToolCallContent, ToolDefinition, ToolExecutionContext, ToolEffectStore, ToolRegistry, ToolResult } from "./contracts.js";
|
|
1
|
+
import type { AgentEvent, ErrorInfo, Guardrails, JsonObject, OwnershipScope, RunLedger, ToolCallContent, ToolDefinition, ToolExecutionContext, ToolEffectDeclaration, ToolEffectStore, ToolRegistry, ToolResult } from "./contracts.js";
|
|
2
2
|
import type { MiddlewareRegistry } from "./middleware.js";
|
|
3
3
|
import { type SecretRedactor } from "./redaction.js";
|
|
4
4
|
import { type DuplicateRegistrationOptions } from "./registry-options.js";
|
|
@@ -57,3 +57,4 @@ export interface ToolRegistryOptions extends DuplicateRegistrationOptions {
|
|
|
57
57
|
export declare function createToolRegistry(tools?: readonly ToolDefinition[], options?: ToolRegistryOptions): ToolRegistry;
|
|
58
58
|
export declare function filterTools(tools: readonly ToolDefinition[], filter?: ToolFilterInput): readonly ToolDefinition[];
|
|
59
59
|
export declare function dispatchToolCall(options: DispatchToolCallOptions): Promise<ToolResult>;
|
|
60
|
+
export declare function resolveToolEffectDeclaration(tool: ToolDefinition, args: JsonObject, context: ToolExecutionContext): ToolEffectDeclaration | undefined;
|
package/dist/tools.js
CHANGED
|
@@ -153,7 +153,8 @@ export async function dispatchToolCall(options) {
|
|
|
153
153
|
}
|
|
154
154
|
catch (error) {
|
|
155
155
|
await failBeforeEffect(effect, isSuspended(error) ? "failed_retryable" : "failed_terminal");
|
|
156
|
-
|
|
156
|
+
// Loop-state contract errors (snapshot capture) are terminal run errors, not tool errors.
|
|
157
|
+
if (isSuspended(error) || isLoopStateError(error) || isDelegationSuspended(error))
|
|
157
158
|
throw error;
|
|
158
159
|
return blocked(mediatedCall, context, "execution_denied", errorToErrorInfo(error, secrets), options, startedAt);
|
|
159
160
|
}
|
|
@@ -222,6 +223,14 @@ export async function dispatchToolCall(options) {
|
|
|
222
223
|
catch (error) {
|
|
223
224
|
if (completedResult)
|
|
224
225
|
return completedResult;
|
|
226
|
+
// Nested-run suspensions must propagate to the run loop, never become tool errors.
|
|
227
|
+
if (isDelegationSuspended(error)) {
|
|
228
|
+
if (effect && (effect.dispatched || dispatchAttempted))
|
|
229
|
+
await unknownEffectResult(effect, mediatedCall);
|
|
230
|
+
else
|
|
231
|
+
await failBeforeEffect(effect, "failed_retryable");
|
|
232
|
+
throw error;
|
|
233
|
+
}
|
|
225
234
|
if (effect && (effect.dispatched || dispatchAttempted))
|
|
226
235
|
return finishUnknownEffect(effect, mediatedCall, context, options, startedAt);
|
|
227
236
|
await failBeforeEffect(effect, "failed_terminal");
|
|
@@ -298,7 +307,7 @@ async function prepareToolEffect(tool, call, context, options) {
|
|
|
298
307
|
},
|
|
299
308
|
};
|
|
300
309
|
}
|
|
301
|
-
function resolveToolEffectDeclaration(tool, args, context) {
|
|
310
|
+
export function resolveToolEffectDeclaration(tool, args, context) {
|
|
302
311
|
const classifierContext = Object.freeze({
|
|
303
312
|
sessionId: context.sessionId,
|
|
304
313
|
runId: context.runId,
|
|
@@ -389,6 +398,12 @@ function effectErrorResult(call, code, message) {
|
|
|
389
398
|
function isSuspended(error) {
|
|
390
399
|
return error?.code === "ERR_PRISM_AGENT_RUN_SUSPENDED";
|
|
391
400
|
}
|
|
401
|
+
function isLoopStateError(error) {
|
|
402
|
+
return typeof error?.code === "string" && error.code.startsWith("ERR_PRISM_LOOP_");
|
|
403
|
+
}
|
|
404
|
+
function isDelegationSuspended(error) {
|
|
405
|
+
return error?.code === "ERR_PRISM_DELEGATION_SUSPENDED";
|
|
406
|
+
}
|
|
392
407
|
async function checkCall(call, options, startedAt) {
|
|
393
408
|
const context = options.context;
|
|
394
409
|
const tool = options.registry.get(call.name);
|
package/docs/0.1.0-readiness.md
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
# 0.1.0 / 1.0 Readiness Gates
|
|
2
2
|
|
|
3
|
-
Status: **0.0.
|
|
3
|
+
Status: **0.0.25** is the current release line (Phase 8 durable custom loops and human-in-the-loop); **1.0** readiness remains operator-gated, not automatic.
|
|
4
4
|
|
|
5
5
|
This page distills runnable readiness gates into one command-per-gate table. The
|
|
6
6
|
**Last evidence** column records the 2026-07-26 **0.0.16** baseline snapshot
|
|
7
7
|
(Phase 11, Node v24.18.0, Linux x86_64). Treat it as historical floor evidence,
|
|
8
8
|
not the current release tag. Re-run each gate on the target release tree before
|
|
9
|
-
cutting 0.0.
|
|
9
|
+
cutting 0.0.25 / 1.0. The decision to cut 1.0 stays with the operator after
|
|
10
10
|
operator-gated legs run in a protected environment and Phase 12 demand evidence
|
|
11
11
|
exists.
|
|
12
12
|
|
|
@@ -14,15 +14,16 @@ Evidence trail: [`docs/review-coverage-2026-07-26-phase-11.md`](./review-coverag
|
|
|
14
14
|
(addenda 0–9), [`docs/release-and-install.md`](./release-and-install.md),
|
|
15
15
|
[`docs/migration.md`](./migration.md), [`docs/performance.md`](./performance.md).
|
|
16
16
|
|
|
17
|
-
## Current line (0.0.
|
|
17
|
+
## Current line (0.0.25)
|
|
18
18
|
|
|
19
19
|
| Item | Status |
|
|
20
20
|
|---|---|
|
|
21
|
-
| Published graph | **47** publishable manifests at **0.0.
|
|
22
|
-
| Phase
|
|
23
|
-
| Docs tripwires | `node --test dist/__tests__/docs.test.js` — migration section `0.0.
|
|
24
|
-
|
|
|
25
|
-
|
|
|
21
|
+
| Published graph | **47** publishable manifests at **0.0.25** (`docs/release-and-install.md`) |
|
|
22
|
+
| Phase 8 durable loops / HITL | Custom-loop snapshot/restore, shared pending decisions, nested attributions, A2UI + standard AG-UI projectors |
|
|
23
|
+
| Docs tripwires | `node --test dist/__tests__/docs.test.js` — migration section `0.0.24 → 0.0.25 durable custom loops and human-in-the-loop` |
|
|
24
|
+
| Network-free Phase 8 evidence | `scripts/phase8-conformance.test.mjs`; `benchmark-0.0.25.json` under Task 0 ceilings |
|
|
25
|
+
| Protected database evidence (Phase 7) | `PRISM_TEST_POSTGRES_URL="$DATABASE_URL" npm run test:postgres`; `benchmark-0.0.24.json` under prior ceilings |
|
|
26
|
+
| Readiness table below | **0.0.16 measured values** remain historical network-free baseline; 0.0.25 loop/HITL evidence is recorded separately |
|
|
26
27
|
|
|
27
28
|
## Gate table
|
|
28
29
|
|
package/docs/ag-ui-adoption.md
CHANGED
|
@@ -7,7 +7,7 @@ This page records Prism's compatibility review against official AG-UI `@ag-ui/co
|
|
|
7
7
|
Official material reviewed:
|
|
8
8
|
|
|
9
9
|
- [Events](https://docs.ag-ui.com/concepts/events), [messages](https://docs.ag-ui.com/concepts/messages), [tools](https://docs.ag-ui.com/concepts/tools), [state](https://docs.ag-ui.com/concepts/state), [reasoning](https://docs.ag-ui.com/concepts/reasoning), [interrupts](https://docs.ag-ui.com/concepts/interrupts), [capabilities](https://docs.ag-ui.com/concepts/capabilities), [serialization](https://docs.ag-ui.com/concepts/serialization), [server quickstart](https://docs.ag-ui.com/quickstart/server), and [protocol architecture](https://docs.ag-ui.com/concepts/architecture).
|
|
10
|
-
- Official [MCP/A2A/AG-UI relationship](https://docs.ag-ui.com/agentic-protocols), [integrations](https://docs.ag-ui.com/integrations), [`@ag-ui/mcp-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/mcp-middleware), [`@ag-ui/mcp-apps-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/mcp-apps-middleware), [`@ag-ui/a2a`](https://github.com/ag-ui-protocol/ag-ui/tree/main/integrations/a2a/typescript), and [`@ag-ui/a2a-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/a2a-middleware).
|
|
10
|
+
- Official [MCP/A2A/AG-UI relationship](https://docs.ag-ui.com/agentic-protocols), [integrations](https://docs.ag-ui.com/integrations), [`@ag-ui/mcp-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/mcp-middleware), [`@ag-ui/mcp-apps-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/mcp-apps-middleware), [`@ag-ui/a2ui-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/a2ui-middleware) (Prism ships an in-package opt-in painter with frozen caps; no runtime dependency), [`@ag-ui/a2a`](https://github.com/ag-ui-protocol/ag-ui/tree/main/integrations/a2a/typescript), and [`@ag-ui/a2a-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/a2a-middleware).
|
|
11
11
|
- A2A [current specification](https://a2a-protocol.org/latest/specification/) and [streaming rules](https://a2a-protocol.org/latest/topics/streaming-and-async/).
|
|
12
12
|
- MCP Apps [SEP-1865](https://modelcontextprotocol.io/seps/1865-mcp-apps-interactive-user-interfaces-for-mcp) and the [`io.modelcontextprotocol/ui` draft specification](https://github.com/modelcontextprotocol/ext-apps/blob/main/specification/draft/apps.mdx).
|
|
13
13
|
|
package/docs/ag-ui.md
CHANGED
|
@@ -35,7 +35,8 @@ npm install @arnilo/prism @arnilo/prism-ag-ui
|
|
|
35
35
|
| `lifecycle` + `resolveRun` | Optional durable status/resume path. Required only for a resumed interruption. |
|
|
36
36
|
| `interrupts.resume` | Optional host aggregate-policy callback for multiple AG-UI interrupts; it returns one current-version core approve/deny decision. |
|
|
37
37
|
| `replay` | Optional page adapter or `createAgentEventSourceAgUiReplay(source, options)` for gap-free distributed replay/live follow. |
|
|
38
|
-
| `projection` | Explicit safe tool/state/messages/activity/reasoning/raw/custom/interrupt projection. Omit each callback for default deny. |
|
|
38
|
+
| `projection` | Explicit safe tool/state/messages/activity/reasoning/raw/custom/interrupt projection. Omit each callback for default deny. Prefer `composeAgUiProjections(createMessagesFromSessionProjection(...), createStateFromStoreProjection(...), createActivityFromToolProgressProjection(), host)` for standard families. |
|
|
39
|
+
| `a2ui` | Opt-in A2UI painting middleware (`{ catalogId, mode, renderToolName?, allowedCatalogIds?, limits? }`). Detects `a2ui_operations` tool results and/or streams from `render_a2ui` args; paints `a2ui-surface` activity events. Absent = inert. |
|
|
39
40
|
| `capabilities` | Optional host declaration narrowed to implemented SSE/projector/lifecycle features; read `handler.capabilities`. |
|
|
40
41
|
| `redactor`, `limits` | Host redaction and narrowing-only finite caps. |
|
|
41
42
|
|
|
@@ -105,6 +106,43 @@ All identity, authorization, session/thread mapping, durable checkpoint lookup,
|
|
|
105
106
|
|
|
106
107
|
Co-work projection reuses the same allow-list: `AgUiProjection.coWork(event)` may return a curated, JSON-serializable payload for a co-work event; absent it, the redacted event fields are exposed. Wire `coWorkContext` to derive thread/artifact/identity from the authorized request (never client JSON) and `coWork` to a `createCoWorkReplay()` over your durable artifact/draft/snapshot stores. The handler projects one bounded page after the run; mount a dedicated cursor-paged co-work endpoint when full pagination is needed.
|
|
107
108
|
|
|
109
|
+
Durable interrupts carry the shared decision batch: the fallback interrupt includes the redacted `pendingDecisions` under `metadata` and its `responseSchema` accepts either the legacy `{ decision: "approve" | "deny" }` or a `{ decisions: [{ approvalId, outcome, reason?, modifiedArguments?, elicitation? }] }` batch. All batch entries are shape- and cap-validated at the boundary (count ≤ 128, ids ≤ 128 chars, four outcomes, reason ≤ 8 KiB, payloads ≤ 64 KiB) and core re-validates each against the recorded pending set under the single CAS. `interrupts.resume` may return the batch form (`{ decisions, expectedVersion? }`); legacy `editedArgs` resume payloads still deny. ACP permission prompts offer the four outcomes (`allow_once` / `allow_always` / `reject_once` / `reject_always`) and map them onto the batch; a cancelled prompt stays deny-closed.
|
|
110
|
+
|
|
111
|
+
### Standard projectors (opt-in)
|
|
112
|
+
|
|
113
|
+
Three batteries-included factories return `AgUiProjection` fragments. Compose with host projectors via `composeAgUiProjections(...fragments)` — **first defined callback wins** (left to right); `undefined` fragments are skipped. Absent factories keep 0.0.24 default-deny.
|
|
114
|
+
|
|
115
|
+
```ts
|
|
116
|
+
createAgUiHandler({
|
|
117
|
+
projection: composeAgUiProjections(
|
|
118
|
+
createMessagesFromSessionProjection({ getMessages: () => authorizedAgUiMessages, redact }),
|
|
119
|
+
createStateFromStoreProjection(runStateStore),
|
|
120
|
+
createActivityFromToolProgressProjection(),
|
|
121
|
+
hostCustom,
|
|
122
|
+
),
|
|
123
|
+
});
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
| Factory | Emits | Notes |
|
|
127
|
+
| --- | --- | --- |
|
|
128
|
+
| `createMessagesFromSessionProjection` | `MESSAGES_SNAPSHOT` | Host `getMessages()` for authorized history, or live `message_finished` accumulation. Caps 128/1024. Redact drops closed. |
|
|
129
|
+
| `createStateFromStoreProjection(store)` | `STATE_SNAPSHOT` on `agent_started`; RFC 6902 `STATE_DELTA` (add/replace/remove) when `store.get()` changes | Host store; optional `subscribe` only marks dirty — no Prism watcher. Oversized/throw → drop closed. |
|
|
130
|
+
| `createActivityFromToolProgressProjection` | `ACTIVITY_SNAPSHOT` / `ACTIVITY_DELTA` from `tool_execution_progress` | Default `activityType: "tool-progress"`. Missing progress+metadata → drop closed. |
|
|
131
|
+
|
|
132
|
+
### A2UI painting middleware (opt-in)
|
|
133
|
+
|
|
134
|
+
`createAgUiHandler({ a2ui: { catalogId, mode } })` paints A2UI v0.9 surfaces without a host `projection.activity` callback:
|
|
135
|
+
|
|
136
|
+
| Mode | Source | Paint |
|
|
137
|
+
| --- | --- | --- |
|
|
138
|
+
| `fixed-schema` | Tool result `{ a2ui_operations: [...] }` | First batch with `createSurface` → `ACTIVITY_SNAPSHOT` (`activityType: "a2ui-surface"`); later batches → `ACTIVITY_DELTA` |
|
|
139
|
+
| `streaming` | `tool_call_delta` args of `renderToolName` (default `render_a2ui`) | Progressive `ACTIVITY_SNAPSHOT` with `replace: true` only when complete ops extractable — never partial JSON |
|
|
140
|
+
| `both` | Both paths; streamed surfaces are not re-painted from the final envelope |
|
|
141
|
+
|
|
142
|
+
Host `catalogId` is stamped when absent; model-supplied ids outside `allowedCatalogIds` (default `[catalogId]`) are overwritten. Invalid ops emit one bounded `CUSTOM` event `prism.a2ui.error` and paint nothing. Caps: 64/512 ops per message, 64 KiB/1 MiB per op, 16/64 surfaces per run, depth 32/64.
|
|
143
|
+
|
|
144
|
+
User actions arrive as untrusted `AgUiA2UiAction` values on `input.project({ a2uiActions })` (from `forwardedProps.a2uiAction` or activity/tool-result shapes). Without `input.project` they stay default-deny — Prism never synthesizes a `log_a2ui_event` tool call (documented divergence from official `@ag-ui/a2ui-middleware`). Example: `examples/ag-ui-a2ui.ts`.
|
|
145
|
+
|
|
108
146
|
## Security and performance notes
|
|
109
147
|
|
|
110
148
|
Authorize every start/replay/resume/proxy/follow. Treat protocol fields, MCP metadata/HTML, and A2A cards/parts as untrusted; persist run/task correlation before output and redact streams. MCP Apps requires extension acknowledgement, exact proxy origin, same-bridge visibility, approval, `ui://` HTML/MIME bounds, and sandbox CSP. It never retries UI mutations; Task 4 adds recovery.
|
package/docs/agent-loops.md
CHANGED
|
@@ -118,7 +118,15 @@ Optional steer hooks on `LoopContext` (0.0.11): `hasPendingSteers?()` / `applyPe
|
|
|
118
118
|
|
|
119
119
|
## Durable runs
|
|
120
120
|
|
|
121
|
-
`RunOptions.runState` supports
|
|
121
|
+
`RunOptions.runState` supports the built-in loop options (`single-shot` and `generate-validate-revise`) and custom strategies that opt into durable state. `single-shot` is durable via the runtime's pending-call mechanism and carries no loop-local state. `generate-validate-revise` snapshots `{ attempts, artifactPhase, savedSchema, pendingHistory }` at `revision: "1"`. A custom `AgentLoopStrategy` must declare both snapshot hooks or durable configuration rejects it with `AgentLoopStateError` (`ERR_PRISM_LOOP_NOT_DURABLE`) before any provider call:
|
|
122
|
+
|
|
123
|
+
| Member | Purpose |
|
|
124
|
+
| --- | --- |
|
|
125
|
+
| `revision?: string` | Host-authored loop revision. Joins the durable-run fingerprint, so a loop change without a `definitionRevision` bump fails closed on resume. |
|
|
126
|
+
| `snapshot?(): JsonValue` | Capture loop-local resumable state at suspension. Must be JSON-compatible; core redacts it and bounds it inside the durable run-state envelope (`maxStateBytes`, depth 32). A non-JSON value fails the run with `ERR_PRISM_LOOP_SNAPSHOT`. |
|
|
127
|
+
| `restore?(snapshot): void` | Rehydrate from the captured snapshot; must throw on drift. Called once before `run(ctx)` on resume. Also available as `ctx.restoredLoopState`. |
|
|
128
|
+
|
|
129
|
+
The snapshot is stored as `loopState: { name, revision, snapshot }` on the durable run state and cleared when the run reaches a terminal status. On resume, a name/revision mismatch between the stored `loopState` and the resolved strategy fails closed (`ERR_PRISM_LOOP_REVISION`), and the fingerprint check independently rejects any loop drift. Suspension occurs only before an input provider call or immediately before a tool side effect; completed provider turns remain in `SessionStore` history and are not repeated after `resumeAgentRun()`.
|
|
122
130
|
|
|
123
131
|
## Outputs / response / events
|
|
124
132
|
|
|
@@ -176,7 +176,14 @@ await agent.createSession().run("Hi", { model: overrideModel });
|
|
|
176
176
|
|
|
177
177
|
## Durable interruption
|
|
178
178
|
|
|
179
|
-
Set `runState` with a host-owned `CheckpointStore`, stable `definitionRevision`, and `interruptBeforeTool: true` to suspend at a persisted pre-side-effect boundary. A suspended result has `status: "suspended"`, a redacted `interruption`, and `runState.version`; it releases session resources before returning.
|
|
179
|
+
Set `runState` with a host-owned `CheckpointStore`, stable `definitionRevision`, and `interruptBeforeTool: true` to suspend at a persisted pre-side-effect boundary. A suspended result has `status: "suspended"`, a redacted `interruption`, and `runState.version`; it releases session resources before returning. When a provider turn requests several tools, the round is collected into **one** suspension whose `interruption.pendingDecisions` holds one redacted `PendingDecision` per gated call (`approvalId`, kind, scope with tool name/effect kind/identity/arguments hash — never raw arguments); ungated calls still dispatch.
|
|
180
|
+
|
|
181
|
+
`resumeAgentRun` accepts exactly one of:
|
|
182
|
+
|
|
183
|
+
- `decision: "approve" | "deny"` — legacy single-approval path. `approve` allows every pending decision once; `deny` terminates the run as `denied`.
|
|
184
|
+
- `decisions: readonly RunDecision[]` — one atomic batch. Every entry validates against the recorded pending set (unknown/foreign `approvalId`, duplicates, stale `expectedVersion`, invalid outcomes fail the whole batch closed with `AgentDecisionError` and leave state and version untouched). Outcomes: `allow_once`, `allow_for_run`, `reject_once`, `reject_for_run`. `reject_*` continues the run with a blocked tool result carrying the bounded (2 KB) `reason`. `modifiedArguments` are revalidated (schema, then input guardrails; permission/trust re-run at dispatch) and produce a new arguments hash. `elicitation` payloads are validated against the pending decision's `elicitationSchema` (required keys plus the configured host validator) and resolve the suspended call without executing it. A batch deciding a strict subset persists the decided entries and re-suspends with the remainder pending at the bumped version.
|
|
185
|
+
|
|
186
|
+
`*_for_run` outcomes append a `StickyDecision` to the durable run state: later calls in the same run matching the scope exactly (all recorded fields) proceed or are blocked without a new suspension, policy still enforced at dispatch. Sticky decisions expire when the run reaches any terminal status. Caps: 32 pending decisions per run (hard 128), 64 sticky decisions (hard 256), 2 KB decision reasons, 16 KB elicitation payloads.
|
|
180
187
|
|
|
181
188
|
```ts
|
|
182
189
|
const result = await session.run("Publish draft", {
|
|
@@ -189,7 +196,7 @@ if (result.status === "suspended") {
|
|
|
189
196
|
}
|
|
190
197
|
```
|
|
191
198
|
|
|
192
|
-
Resume requires exact checkpoint ownership, version, agent fingerprint, and revision. The fingerprint hashes the agent id/name, `definitionRevision`, model, instructions, system-prompt contributions, skills (name/instructions/tool names), tool definitions (name/parameters/exclusive), guardrail definitions (name/stage/revision), and loop strategy — changing any of them without bumping `definitionRevision` fails resume closed instead of silently continuing with different agent semantics. Prism CAS-claims approval before work, rechecks normal guardrail/permission/validation/limit paths, and marks a pending tool dispatched before its side effect. `createAgentRunLifecycle()` wraps the same core path for server/MCP hosts: adapters pass only authorized ownership, status returns only `{ state, version }`, and `resolveAgent()` supplies current agent/revision. `resumeStream()` uses that same claim path and bounded subscriber, so adapters do not poll or duplicate resume logic. Remote restart requires both checkpoint and session stores to be durable. A crash after that mark is ambiguous and is never replayed automatically; use host tool idempotency keyed by `runId`/`toolCallId` or resolve it manually. Checkpoints contain bounded redacted state plus session/leaf references, never provider objects, callbacks, signals, credentials, or raw secrets. State is bounded at save by `runState.maxStateBytes` (default 256 KB, at most the 1 MB hard cap); load bounds against the 1 MB hard cap only, so state saved with a raised limit stays resumable while oversized records are still rejected.
|
|
199
|
+
Resume requires exact checkpoint ownership, version, agent fingerprint, and revision. The fingerprint hashes the agent id/name, `definitionRevision`, model, instructions, system-prompt contributions, skills (name/instructions/tool names), tool definitions (name/parameters/exclusive), guardrail definitions (name/stage/revision), and loop strategy — changing any of them without bumping `definitionRevision` fails resume closed instead of silently continuing with different agent semantics. Prism CAS-claims approval before work, rechecks normal guardrail/permission/validation/limit paths, and marks a pending tool dispatched before its side effect. `createAgentRunLifecycle()` wraps the same core path for server/MCP hosts: adapters pass only authorized ownership, status returns only `{ state, version }`, and `resolveAgent()` supplies current agent/revision. `resumeStream()` uses that same claim path and bounded subscriber, so adapters do not poll or duplicate resume logic. Remote restart requires both checkpoint and session stores to be durable. A crash after that mark is ambiguous and is never replayed automatically; use host tool idempotency keyed by `runId`/`toolCallId` or resolve it manually. Checkpoints contain bounded redacted state plus session/leaf references, never provider objects, callbacks, signals, credentials, or raw secrets. State is bounded at save by `runState.maxStateBytes` (default 256 KB, at most the 1 MB hard cap); load bounds against the 1 MB hard cap only, so state saved with a raised limit stays resumable while oversized records are still rejected. Built-in loop options are durable; custom `AgentLoopStrategy` instances are durable when they declare `snapshot`/`restore` hooks (see [Agent loops § Durable runs](agent-loops.md#durable-runs)) and reject before provider work otherwise.
|
|
193
200
|
|
|
194
201
|
## Secure composition
|
|
195
202
|
|
package/docs/coding-security.md
CHANGED
|
@@ -143,7 +143,7 @@ Policies are ordinary host values: attach one globally through `createCodingTool
|
|
|
143
143
|
|
|
144
144
|
The Docker reference adapter starts by recorded container ID/label, uses argument arrays only, mounts source read-only, populates a size-bounded tmpfs `/workspace`, drops all capabilities, enables `no-new-privileges`, runs with `--init`, and never exposes the Docker socket, privileged mode, or host PID/IPC namespaces. Image pull/build/update stays outside Prism. Protected real-Docker checks are opt-in via `PRISM_TEST_DOCKER_SANDBOX=1` with host-supplied `PRISM_TEST_DOCKER_BIN` and digest-pinned `PRISM_TEST_DOCKER_IMAGE`.
|
|
145
145
|
|
|
146
|
-
Callback approval remains process-local. For approval that must survive restart, wrap the action in an opted-in workflow `toolNode({ approval: { reason, data?, resumeSchema? } })`. The workflow persists `suspended` state before any tool side effect. After explicit approve, it recomputes the action and invokes this package's current `ExecutionPolicy`; durable approval never populates or bypasses the process-local approval cache. Adapters should emit chunks through `request.onData` as they arrive and honor `request.signal`/`request.timeout`; buffering is unnecessary. Coding-agent composes caller abort with its total-output controller, so ignoring the supplied signal defeats process termination even though Prism stops retaining output at the cap. Default caching is `none`; use run-scoped caching only when repeated approval within one run is desired, and session scope only when that wider lifecycle is intentional.
|
|
146
|
+
Callback approval remains process-local. For approval that must survive restart, wrap the action in an opted-in workflow `toolNode({ approval: { reason, data?, resumeSchema? } })`. `ask_user_decision` also maps onto the shared decision model: inside a durable gated agent run its call suspends as a kind-`elicitation` pending decision whose schema carries the choice contract (option-id enums) plus the full question/options UX payload on a Prism-owned schema extension property; the resume decision's `elicitation` payload (a `selectedId`/`selectedIds`/`customText` answer) is validated against the schema and the tool-level answer-shape rules, then resolves the call without invoking the blocking `ask()` callback. The process-local `ask()` path and the workflow suspend/resume path are unchanged. The workflow persists `suspended` state before any tool side effect. After explicit approve, it recomputes the action and invokes this package's current `ExecutionPolicy`; durable approval never populates or bypasses the process-local approval cache. Adapters should emit chunks through `request.onData` as they arrive and honor `request.signal`/`request.timeout`; buffering is unnecessary. Coding-agent composes caller abort with its total-output controller, so ignoring the supplied signal defeats process termination even though Prism stops retaining output at the cap. Default caching is `none`; use run-scoped caching only when repeated approval within one run is desired, and session scope only when that wider lifecycle is intentional.
|
|
147
147
|
|
|
148
148
|
## Security and performance notes
|
|
149
149
|
|