@arnilo/prism 0.0.24 → 0.0.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,25 @@
1
1
  # Changelog
2
2
 
3
+ ## [0.0.25] - 2026-08-06
4
+
5
+ ### Added
6
+ - Durable custom `AgentLoopStrategy` hooks: optional `revision` / `snapshot` / `restore`; `AgentLoopStateError` fail-closed codes; fingerprint includes loop `{name,revision}`.
7
+ - Shared pending-decision model: parallel approvals, batch CAS `decisions`, sticky allow/reject for run, modified arguments, elicitation; nested supervisor attribution.
8
+ - Protocol mappings: AG-UI/ACP/server batch resume, MCP elicitation helpers, coding `ask_user_decision` elicitation hook.
9
+ - Opt-in A2UI painting middleware + standard AG-UI projectors (`messages`/`state`/`activity`).
10
+ - Network-free Phase 8 conformance + `benchmark-0.0.25.json` evidence; examples `durable-loops-and-approvals.ts`, `ag-ui-a2ui.ts`.
11
+
12
+ ### Changed
13
+ - Publishable graph remains **47** manifests at **0.0.25**.
14
+ - Fingerprint loop entry shape `string` → `{name,revision}` (0.0.24 persisted durable runs fail closed on resume).
15
+
16
+ ### Breaking (minor, pre-1.0)
17
+ - Custom loops on durable runs need snapshot/restore hooks or `ERR_PRISM_LOOP_NOT_DURABLE`.
18
+ - Resume prefers `decisions: RunDecision[]`; legacy binary `decision` remains but is exclusive with the batch path.
19
+ - ACP permission offers four outcomes; `reject_once` is blocked-continue (cancelled stays terminal deny).
20
+
21
+ See [docs/migration.md](docs/migration.md) for the 0.0.24 → 0.0.25 guide.
22
+
3
23
  ## [0.0.24] - 2026-08-04
4
24
 
5
25
  ### Added
@@ -1,3 +1,4 @@
1
+ import { AgentLoopStateError } from "./contracts.js";
1
2
  import { createId } from "./ids.js";
2
3
  import { inputMessages } from "./input.js";
3
4
  import { artifactStructuredOutputRequest, withoutStructuredOutput } from "./structured-output.js";
@@ -21,6 +22,8 @@ function toolResultMessage(result) {
21
22
  // fires artifact_* events here as a noop seam — single-shot emits zero.
22
23
  export const singleShotLoop = {
23
24
  name: "single-shot",
25
+ // Durable via the runtime's pending-call mechanism; no loop-local state to snapshot.
26
+ revision: "1",
24
27
  async run(ctx) {
25
28
  let usage;
26
29
  let toolRounds = 0;
@@ -74,16 +77,39 @@ export function generateValidateReviseLoop(opts) {
74
77
  const max = opts.maxRevisions ?? 3;
75
78
  const repairer = opts.repairer ?? defaultRepairer();
76
79
  const finalOnly = opts.structuredOutputTiming === "final-turn-only" && opts.toolCalls === "bounded";
80
+ // ponytail: per-run state hoisted to factory scope so snapshot/restore can capture it;
81
+ // resolveLoop invokes this factory once per run, so there is no cross-run leakage.
82
+ let attempts = 0;
83
+ let artifactPhase = !finalOnly;
84
+ let savedSchema;
85
+ let pendingHistory = [];
77
86
  return {
78
87
  name: "generate-validate-revise",
88
+ revision: "1",
89
+ snapshot() {
90
+ return {
91
+ attempts,
92
+ artifactPhase,
93
+ savedSchema: savedSchema ?? null,
94
+ pendingHistory,
95
+ };
96
+ },
97
+ restore(snapshot) {
98
+ const state = snapshot;
99
+ if (typeof state.attempts !== "number" || !Number.isInteger(state.attempts) || typeof state.artifactPhase !== "boolean") {
100
+ throw new AgentLoopStateError("ERR_PRISM_LOOP_SNAPSHOT", "generate-validate-revise snapshot drift");
101
+ }
102
+ attempts = state.attempts;
103
+ artifactPhase = state.artifactPhase;
104
+ savedSchema = (state.savedSchema ?? undefined);
105
+ // Repair messages were appended to the session before suspension, so the rebuilt
106
+ // history already carries them; re-applying pendingHistory would duplicate them.
107
+ pendingHistory = [];
108
+ },
79
109
  async run(ctx) {
80
110
  let usage;
81
111
  let nextInput = ctx.input;
82
- let pendingHistory = [];
83
112
  let toolRounds = 0;
84
- let attempts = 0;
85
- let artifactPhase = !finalOnly;
86
- let savedSchema;
87
113
  for (let turn = 1; attempts <= max; turn += 1) {
88
114
  throwIfAborted(ctx.signal);
89
115
  await ctx.applyPendingSteers?.();
@@ -219,7 +245,7 @@ export function dispatchableToolCalls(calls) {
219
245
  export async function dispatchToolCallsInOrder(calls, ctx) {
220
246
  if (calls.length === 0)
221
247
  return;
222
- ctx.chargeToolRound?.(calls);
248
+ await ctx.chargeToolRound?.(calls);
223
249
  const concurrency = calls.some((call) => ctx.isToolCallExclusive?.(call)) ? 1 : Math.max(1, Math.min(ctx.toolConcurrency, calls.length));
224
250
  if (concurrency === 1) {
225
251
  for (const call of calls) {
@@ -243,10 +269,16 @@ export async function dispatchToolCallsInOrder(calls, ctx) {
243
269
  await Promise.all(workers);
244
270
  for (const result of results) {
245
271
  throwIfAborted(ctx.signal);
272
+ if (!result)
273
+ continue;
246
274
  await appendToolResultMessage(result, ctx);
247
275
  }
248
276
  }
249
277
  async function appendToolResultMessage(result, ctx) {
278
+ // Approval-gated calls return a marker instead of a real result; the transcript must not
279
+ // record a phantom tool_result for a call that never dispatched.
280
+ if (result.metadata?.approvalPending === true)
281
+ return;
250
282
  const message = toolResultMessage(result);
251
283
  ctx.history.push(message);
252
284
  await ctx.appendMessage(message);
@@ -1,19 +1,44 @@
1
- import type { Agent, AgentRunInterruption, AgentRunRef, AgentRunState, AgentRunStateOptions, AgentRunStatusResult, CheckpointRecord, CheckpointStore, Message, ModelConfig, OwnershipScope, RunLimitCounters, ToolCallContent } from "./contracts.js";
1
+ import type { Agent, AgentRunInterruption, AgentRunRef, AgentRunState, AgentRunStateOptions, AgentRunStatusResult, CheckpointRecord, CheckpointStore, JsonValue, Message, ModelConfig, NestedRunRef, OwnershipScope, RunDecision, RunLimitCounters, StickyDecision, ToolCallContent } from "./contracts.js";
2
2
  import type { SecretRedactor } from "./redaction.js";
3
3
  export declare const AGENT_RUN_STATE_NAMESPACE = "prism.agent-run";
4
4
  export declare const AGENT_RUN_STATE_SCHEMA_VERSION: 1;
5
5
  export declare const DEFAULT_MAX_AGENT_RUN_STATE_BYTES: number;
6
6
  export declare const HARD_MAX_AGENT_RUN_STATE_BYTES: number;
7
+ /** One gated tool call awaiting or holding a decision inside a suspended durable run. */
8
+ export interface PendingToolCall {
9
+ readonly call: ToolCallContent;
10
+ readonly status: "ready" | "dispatched";
11
+ readonly approvalId: string;
12
+ /** Decision persisted by a partial batch; applied when the run finally resumes. */
13
+ readonly decision?: RunDecision;
14
+ }
7
15
  export interface StoredAgentRunState extends AgentRunState {
8
16
  readonly input?: readonly Message[];
17
+ /** Legacy single gated call (pre-0.0.25 checkpoints). New states write `pendingCalls`. */
9
18
  readonly pending?: {
10
19
  readonly call: ToolCallContent;
11
20
  readonly status: "ready" | "dispatched";
12
21
  };
22
+ /** Gated calls of the current suspension, in provider-turn order. */
23
+ readonly pendingCalls?: readonly PendingToolCall[];
24
+ /** Suspended nested runs (supervisor children) whose pending decisions surface at this root. */
25
+ readonly nestedRuns?: readonly NestedRunRef[];
26
+ /** Run-scoped sticky decisions; exact scope match, dropped at any terminal status. */
27
+ readonly stickyDecisions?: readonly StickyDecision[];
13
28
  readonly interruptBeforeTool?: boolean;
14
29
  readonly counters: RunLimitCounters;
15
30
  readonly deadlineAt: string;
31
+ /** Loop-local durable state captured by the strategy's snapshot hook at suspension. */
32
+ readonly loopState?: {
33
+ readonly name: string;
34
+ readonly revision: string;
35
+ readonly snapshot: JsonValue;
36
+ };
16
37
  }
38
+ /** Revision stamps of the built-in loops; custom strategies declare their own `revision`. */
39
+ export declare const BUILT_IN_LOOP_REVISIONS: Readonly<Record<string, string>>;
40
+ /** Validate a strategy snapshot as JSON-compatible and package it for the durable envelope. */
41
+ export declare function boundedLoopSnapshot(name: string, revision: string, snapshot: JsonValue): StoredAgentRunState["loopState"];
17
42
  export declare function agentFingerprint(agent: Agent, revision: string): string;
18
43
  export declare function agentId(agent: Agent): string;
19
44
  export declare function validateRunStateOptions(options: AgentRunStateOptions): void;
@@ -48,6 +73,7 @@ export declare function initialAgentRunState(input: {
48
73
  readonly interruption?: AgentRunInterruption;
49
74
  readonly messages?: readonly Message[];
50
75
  readonly pending?: StoredAgentRunState["pending"];
76
+ readonly pendingCalls?: StoredAgentRunState["pendingCalls"];
51
77
  readonly interruptBeforeTool?: boolean;
52
78
  }): StoredAgentRunState;
53
79
  export declare function parseAgentRunState(value: unknown, version?: number): StoredAgentRunState;
@@ -1,11 +1,50 @@
1
1
  import { createHash } from "node:crypto";
2
- import { AgentRunStateError } from "./contracts.js";
2
+ import { AgentLoopStateError, AgentRunStateError } from "./contracts.js";
3
3
  export const AGENT_RUN_STATE_NAMESPACE = "prism.agent-run";
4
4
  export const AGENT_RUN_STATE_SCHEMA_VERSION = 1;
5
5
  export const DEFAULT_MAX_AGENT_RUN_STATE_BYTES = 256 * 1024;
6
6
  export const HARD_MAX_AGENT_RUN_STATE_BYTES = 1024 * 1024;
7
7
  const MAX_DEPTH = 32;
8
8
  const MAX_PROPERTIES = 256;
9
+ /** Revision stamps of the built-in loops; custom strategies declare their own `revision`. */
10
+ export const BUILT_IN_LOOP_REVISIONS = {
11
+ "single-shot": "1",
12
+ "generate-validate-revise": "1",
13
+ };
14
+ /** Validate a strategy snapshot as JSON-compatible and package it for the durable envelope. */
15
+ export function boundedLoopSnapshot(name, revision, snapshot) {
16
+ try {
17
+ assertJsonValue(snapshot, 0);
18
+ }
19
+ catch (error) {
20
+ if (error instanceof AgentLoopStateError)
21
+ throw error;
22
+ throw new AgentLoopStateError("ERR_PRISM_LOOP_SNAPSHOT", "Loop snapshot must be JSON-compatible", { cause: error });
23
+ }
24
+ return { name, revision, snapshot };
25
+ }
26
+ function assertJsonValue(value, depth) {
27
+ if (depth > MAX_DEPTH)
28
+ throw new AgentLoopStateError("ERR_PRISM_LOOP_SNAPSHOT", `Loop snapshot exceeds depth ${MAX_DEPTH}`);
29
+ switch (typeof value) {
30
+ case "string":
31
+ case "boolean":
32
+ return;
33
+ case "number":
34
+ if (!Number.isFinite(value))
35
+ throw new AgentLoopStateError("ERR_PRISM_LOOP_SNAPSHOT", "Loop snapshot numbers must be finite");
36
+ return;
37
+ case "object": {
38
+ if (value === null)
39
+ return;
40
+ for (const item of Object.values(value))
41
+ assertJsonValue(item, depth + 1);
42
+ return;
43
+ }
44
+ default:
45
+ throw new AgentLoopStateError("ERR_PRISM_LOOP_SNAPSHOT", "Loop snapshot must be JSON-compatible");
46
+ }
47
+ }
9
48
  export function agentFingerprint(agent, revision) {
10
49
  const config = agent.config;
11
50
  const tools = !config.tools ? [] : "list" in config.tools ? config.tools.list() : config.tools;
@@ -35,9 +74,10 @@ export function agentFingerprint(agent, revision) {
35
74
  effect: typeof tool.effect === "function" ? "classifier" : tool.effect,
36
75
  })),
37
76
  guardrails: guardrails.map((guardrail) => ({ name: guardrail.name, stage: guardrail.stage, revision: guardrail.revision })),
77
+ // Loop revision participates so a loop change without a definitionRevision bump fails closed.
38
78
  loop: typeof config.loop === "object" && config.loop && "strategy" in config.loop
39
- ? config.loop.strategy
40
- : (config.loop?.name ?? "single-shot"),
79
+ ? { name: config.loop.strategy, revision: BUILT_IN_LOOP_REVISIONS[config.loop.strategy] ?? null }
80
+ : { name: config.loop?.name ?? "single-shot", revision: config.loop?.revision ?? BUILT_IN_LOOP_REVISIONS["single-shot"] },
41
81
  });
42
82
  return createHash("sha256").update(value).digest("hex");
43
83
  }
@@ -85,7 +125,7 @@ export function statusFromState(state, version) {
85
125
  return { state: publicState({ ...state, version }), version };
86
126
  }
87
127
  export function publicState(state) {
88
- const { input: _input, pending: _pending, interruptBeforeTool: _interruptBeforeTool, counters: _counters, deadlineAt: _deadlineAt, ...publicValue } = state;
128
+ const { input: _input, pending: _pending, pendingCalls: _pendingCalls, nestedRuns: _nestedRuns, interruptBeforeTool: _interruptBeforeTool, counters: _counters, deadlineAt: _deadlineAt, ...publicValue } = state;
89
129
  return publicValue;
90
130
  }
91
131
  export function initialAgentRunState(input) {
@@ -103,6 +143,7 @@ export function initialAgentRunState(input) {
103
143
  interruption: input.interruption,
104
144
  input: input.messages,
105
145
  pending: input.pending,
146
+ pendingCalls: input.pendingCalls,
106
147
  interruptBeforeTool: input.interruptBeforeTool,
107
148
  counters: input.counters,
108
149
  deadlineAt: input.deadlineAt,
@@ -125,6 +166,41 @@ export function parseAgentRunState(value, version) {
125
166
  !state.deadlineAt) {
126
167
  throw new AgentRunStateError("Malformed agent run state");
127
168
  }
169
+ if (state.pendingCalls !== undefined &&
170
+ (!Array.isArray(state.pendingCalls) ||
171
+ state.pendingCalls.some((entry) => !entry ||
172
+ typeof entry !== "object" ||
173
+ !entry.call ||
174
+ typeof entry.approvalId !== "string" ||
175
+ (entry.status !== "ready" && entry.status !== "dispatched")))) {
176
+ throw new AgentRunStateError("Malformed agent run pending calls");
177
+ }
178
+ if (state.stickyDecisions !== undefined &&
179
+ (!Array.isArray(state.stickyDecisions) ||
180
+ state.stickyDecisions.some((entry) => !entry ||
181
+ typeof entry !== "object" ||
182
+ !entry.scope ||
183
+ (entry.outcome !== "allow_for_run" && entry.outcome !== "reject_for_run")))) {
184
+ throw new AgentRunStateError("Malformed agent run sticky decisions");
185
+ }
186
+ if (state.nestedRuns !== undefined &&
187
+ (!Array.isArray(state.nestedRuns) ||
188
+ state.nestedRuns.some((entry) => !entry ||
189
+ typeof entry !== "object" ||
190
+ typeof entry.runId !== "string" ||
191
+ typeof entry.toolCallId !== "string" ||
192
+ !Array.isArray(entry.path) ||
193
+ !Array.isArray(entry.approvals) ||
194
+ entry.approvals.some((approval) => typeof approval?.id !== "string" || typeof approval?.childApprovalId !== "string")))) {
195
+ throw new AgentRunStateError("Malformed agent run nested runs");
196
+ }
197
+ if (state.loopState !== undefined &&
198
+ (typeof state.loopState !== "object" ||
199
+ typeof state.loopState.name !== "string" ||
200
+ typeof state.loopState.revision !== "string" ||
201
+ !("snapshot" in state.loopState))) {
202
+ throw new AgentRunStateError("Malformed agent run loop state");
203
+ }
128
204
  // Load bounds against the hard cap, not the default: the configured maxStateBytes is a
129
205
  // save-side policy knob, while the load-side bound is only a DoS ceiling. States saved
130
206
  // with a raised maxStateBytes must remain resumable.