@arnilo/prism 0.0.24 → 0.0.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/dist/agent-loops.js +37 -5
- package/dist/agent-run-state.d.ts +27 -1
- package/dist/agent-run-state.js +80 -4
- package/dist/agents.js +884 -77
- package/dist/contracts.d.ts +181 -4
- package/dist/contracts.js +53 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +2 -2
- package/dist/tools.d.ts +2 -1
- package/dist/tools.js +17 -2
- package/docs/0.1.0-readiness.md +9 -8
- package/docs/ag-ui-adoption.md +1 -1
- package/docs/ag-ui.md +39 -1
- package/docs/agent-loops.md +9 -1
- package/docs/agent-session-runtime.md +9 -2
- package/docs/coding-security.md +1 -1
- package/docs/index.md +6 -6
- package/docs/mcp-tools.md +2 -0
- package/docs/migration.md +24 -0
- package/docs/performance.md +7 -6
- package/docs/release-and-install.md +33 -12
- package/docs/server.md +1 -0
- package/docs/supervisors.md +4 -0
- package/docs/workflows.md +1 -1
- package/package.json +2 -2
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,25 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [0.0.25] - 2026-08-06
|
|
4
|
+
|
|
5
|
+
### Added
|
|
6
|
+
- Durable custom `AgentLoopStrategy` hooks: optional `revision` / `snapshot` / `restore`; `AgentLoopStateError` fail-closed codes; fingerprint includes loop `{name,revision}`.
|
|
7
|
+
- Shared pending-decision model: parallel approvals, batch CAS `decisions`, sticky allow/reject for run, modified arguments, elicitation; nested supervisor attribution.
|
|
8
|
+
- Protocol mappings: AG-UI/ACP/server batch resume, MCP elicitation helpers, coding `ask_user_decision` elicitation hook.
|
|
9
|
+
- Opt-in A2UI painting middleware + standard AG-UI projectors (`messages`/`state`/`activity`).
|
|
10
|
+
- Network-free Phase 8 conformance + `benchmark-0.0.25.json` evidence; examples `durable-loops-and-approvals.ts`, `ag-ui-a2ui.ts`.
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
- Publishable graph remains **47** manifests at **0.0.25**.
|
|
14
|
+
- Fingerprint loop entry shape `string` → `{name,revision}` (0.0.24 persisted durable runs fail closed on resume).
|
|
15
|
+
|
|
16
|
+
### Breaking (minor, pre-1.0)
|
|
17
|
+
- Custom loops on durable runs need snapshot/restore hooks or `ERR_PRISM_LOOP_NOT_DURABLE`.
|
|
18
|
+
- Resume prefers `decisions: RunDecision[]`; legacy binary `decision` remains but is exclusive with the batch path.
|
|
19
|
+
- ACP permission offers four outcomes; `reject_once` is blocked-continue (cancelled stays terminal deny).
|
|
20
|
+
|
|
21
|
+
See [docs/migration.md](docs/migration.md) for the 0.0.24 → 0.0.25 guide.
|
|
22
|
+
|
|
3
23
|
## [0.0.24] - 2026-08-04
|
|
4
24
|
|
|
5
25
|
### Added
|
package/dist/agent-loops.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { AgentLoopStateError } from "./contracts.js";
|
|
1
2
|
import { createId } from "./ids.js";
|
|
2
3
|
import { inputMessages } from "./input.js";
|
|
3
4
|
import { artifactStructuredOutputRequest, withoutStructuredOutput } from "./structured-output.js";
|
|
@@ -21,6 +22,8 @@ function toolResultMessage(result) {
|
|
|
21
22
|
// fires artifact_* events here as a noop seam — single-shot emits zero.
|
|
22
23
|
export const singleShotLoop = {
|
|
23
24
|
name: "single-shot",
|
|
25
|
+
// Durable via the runtime's pending-call mechanism; no loop-local state to snapshot.
|
|
26
|
+
revision: "1",
|
|
24
27
|
async run(ctx) {
|
|
25
28
|
let usage;
|
|
26
29
|
let toolRounds = 0;
|
|
@@ -74,16 +77,39 @@ export function generateValidateReviseLoop(opts) {
|
|
|
74
77
|
const max = opts.maxRevisions ?? 3;
|
|
75
78
|
const repairer = opts.repairer ?? defaultRepairer();
|
|
76
79
|
const finalOnly = opts.structuredOutputTiming === "final-turn-only" && opts.toolCalls === "bounded";
|
|
80
|
+
// ponytail: per-run state hoisted to factory scope so snapshot/restore can capture it;
|
|
81
|
+
// resolveLoop invokes this factory once per run, so there is no cross-run leakage.
|
|
82
|
+
let attempts = 0;
|
|
83
|
+
let artifactPhase = !finalOnly;
|
|
84
|
+
let savedSchema;
|
|
85
|
+
let pendingHistory = [];
|
|
77
86
|
return {
|
|
78
87
|
name: "generate-validate-revise",
|
|
88
|
+
revision: "1",
|
|
89
|
+
snapshot() {
|
|
90
|
+
return {
|
|
91
|
+
attempts,
|
|
92
|
+
artifactPhase,
|
|
93
|
+
savedSchema: savedSchema ?? null,
|
|
94
|
+
pendingHistory,
|
|
95
|
+
};
|
|
96
|
+
},
|
|
97
|
+
restore(snapshot) {
|
|
98
|
+
const state = snapshot;
|
|
99
|
+
if (typeof state.attempts !== "number" || !Number.isInteger(state.attempts) || typeof state.artifactPhase !== "boolean") {
|
|
100
|
+
throw new AgentLoopStateError("ERR_PRISM_LOOP_SNAPSHOT", "generate-validate-revise snapshot drift");
|
|
101
|
+
}
|
|
102
|
+
attempts = state.attempts;
|
|
103
|
+
artifactPhase = state.artifactPhase;
|
|
104
|
+
savedSchema = (state.savedSchema ?? undefined);
|
|
105
|
+
// Repair messages were appended to the session before suspension, so the rebuilt
|
|
106
|
+
// history already carries them; re-applying pendingHistory would duplicate them.
|
|
107
|
+
pendingHistory = [];
|
|
108
|
+
},
|
|
79
109
|
async run(ctx) {
|
|
80
110
|
let usage;
|
|
81
111
|
let nextInput = ctx.input;
|
|
82
|
-
let pendingHistory = [];
|
|
83
112
|
let toolRounds = 0;
|
|
84
|
-
let attempts = 0;
|
|
85
|
-
let artifactPhase = !finalOnly;
|
|
86
|
-
let savedSchema;
|
|
87
113
|
for (let turn = 1; attempts <= max; turn += 1) {
|
|
88
114
|
throwIfAborted(ctx.signal);
|
|
89
115
|
await ctx.applyPendingSteers?.();
|
|
@@ -219,7 +245,7 @@ export function dispatchableToolCalls(calls) {
|
|
|
219
245
|
export async function dispatchToolCallsInOrder(calls, ctx) {
|
|
220
246
|
if (calls.length === 0)
|
|
221
247
|
return;
|
|
222
|
-
ctx.chargeToolRound?.(calls);
|
|
248
|
+
await ctx.chargeToolRound?.(calls);
|
|
223
249
|
const concurrency = calls.some((call) => ctx.isToolCallExclusive?.(call)) ? 1 : Math.max(1, Math.min(ctx.toolConcurrency, calls.length));
|
|
224
250
|
if (concurrency === 1) {
|
|
225
251
|
for (const call of calls) {
|
|
@@ -243,10 +269,16 @@ export async function dispatchToolCallsInOrder(calls, ctx) {
|
|
|
243
269
|
await Promise.all(workers);
|
|
244
270
|
for (const result of results) {
|
|
245
271
|
throwIfAborted(ctx.signal);
|
|
272
|
+
if (!result)
|
|
273
|
+
continue;
|
|
246
274
|
await appendToolResultMessage(result, ctx);
|
|
247
275
|
}
|
|
248
276
|
}
|
|
249
277
|
async function appendToolResultMessage(result, ctx) {
|
|
278
|
+
// Approval-gated calls return a marker instead of a real result; the transcript must not
|
|
279
|
+
// record a phantom tool_result for a call that never dispatched.
|
|
280
|
+
if (result.metadata?.approvalPending === true)
|
|
281
|
+
return;
|
|
250
282
|
const message = toolResultMessage(result);
|
|
251
283
|
ctx.history.push(message);
|
|
252
284
|
await ctx.appendMessage(message);
|
|
@@ -1,19 +1,44 @@
|
|
|
1
|
-
import type { Agent, AgentRunInterruption, AgentRunRef, AgentRunState, AgentRunStateOptions, AgentRunStatusResult, CheckpointRecord, CheckpointStore, Message, ModelConfig, OwnershipScope, RunLimitCounters, ToolCallContent } from "./contracts.js";
|
|
1
|
+
import type { Agent, AgentRunInterruption, AgentRunRef, AgentRunState, AgentRunStateOptions, AgentRunStatusResult, CheckpointRecord, CheckpointStore, JsonValue, Message, ModelConfig, NestedRunRef, OwnershipScope, RunDecision, RunLimitCounters, StickyDecision, ToolCallContent } from "./contracts.js";
|
|
2
2
|
import type { SecretRedactor } from "./redaction.js";
|
|
3
3
|
export declare const AGENT_RUN_STATE_NAMESPACE = "prism.agent-run";
|
|
4
4
|
export declare const AGENT_RUN_STATE_SCHEMA_VERSION: 1;
|
|
5
5
|
export declare const DEFAULT_MAX_AGENT_RUN_STATE_BYTES: number;
|
|
6
6
|
export declare const HARD_MAX_AGENT_RUN_STATE_BYTES: number;
|
|
7
|
+
/** One gated tool call awaiting or holding a decision inside a suspended durable run. */
|
|
8
|
+
export interface PendingToolCall {
|
|
9
|
+
readonly call: ToolCallContent;
|
|
10
|
+
readonly status: "ready" | "dispatched";
|
|
11
|
+
readonly approvalId: string;
|
|
12
|
+
/** Decision persisted by a partial batch; applied when the run finally resumes. */
|
|
13
|
+
readonly decision?: RunDecision;
|
|
14
|
+
}
|
|
7
15
|
export interface StoredAgentRunState extends AgentRunState {
|
|
8
16
|
readonly input?: readonly Message[];
|
|
17
|
+
/** Legacy single gated call (pre-0.0.25 checkpoints). New states write `pendingCalls`. */
|
|
9
18
|
readonly pending?: {
|
|
10
19
|
readonly call: ToolCallContent;
|
|
11
20
|
readonly status: "ready" | "dispatched";
|
|
12
21
|
};
|
|
22
|
+
/** Gated calls of the current suspension, in provider-turn order. */
|
|
23
|
+
readonly pendingCalls?: readonly PendingToolCall[];
|
|
24
|
+
/** Suspended nested runs (supervisor children) whose pending decisions surface at this root. */
|
|
25
|
+
readonly nestedRuns?: readonly NestedRunRef[];
|
|
26
|
+
/** Run-scoped sticky decisions; exact scope match, dropped at any terminal status. */
|
|
27
|
+
readonly stickyDecisions?: readonly StickyDecision[];
|
|
13
28
|
readonly interruptBeforeTool?: boolean;
|
|
14
29
|
readonly counters: RunLimitCounters;
|
|
15
30
|
readonly deadlineAt: string;
|
|
31
|
+
/** Loop-local durable state captured by the strategy's snapshot hook at suspension. */
|
|
32
|
+
readonly loopState?: {
|
|
33
|
+
readonly name: string;
|
|
34
|
+
readonly revision: string;
|
|
35
|
+
readonly snapshot: JsonValue;
|
|
36
|
+
};
|
|
16
37
|
}
|
|
38
|
+
/** Revision stamps of the built-in loops; custom strategies declare their own `revision`. */
|
|
39
|
+
export declare const BUILT_IN_LOOP_REVISIONS: Readonly<Record<string, string>>;
|
|
40
|
+
/** Validate a strategy snapshot as JSON-compatible and package it for the durable envelope. */
|
|
41
|
+
export declare function boundedLoopSnapshot(name: string, revision: string, snapshot: JsonValue): StoredAgentRunState["loopState"];
|
|
17
42
|
export declare function agentFingerprint(agent: Agent, revision: string): string;
|
|
18
43
|
export declare function agentId(agent: Agent): string;
|
|
19
44
|
export declare function validateRunStateOptions(options: AgentRunStateOptions): void;
|
|
@@ -48,6 +73,7 @@ export declare function initialAgentRunState(input: {
|
|
|
48
73
|
readonly interruption?: AgentRunInterruption;
|
|
49
74
|
readonly messages?: readonly Message[];
|
|
50
75
|
readonly pending?: StoredAgentRunState["pending"];
|
|
76
|
+
readonly pendingCalls?: StoredAgentRunState["pendingCalls"];
|
|
51
77
|
readonly interruptBeforeTool?: boolean;
|
|
52
78
|
}): StoredAgentRunState;
|
|
53
79
|
export declare function parseAgentRunState(value: unknown, version?: number): StoredAgentRunState;
|
package/dist/agent-run-state.js
CHANGED
|
@@ -1,11 +1,50 @@
|
|
|
1
1
|
import { createHash } from "node:crypto";
|
|
2
|
-
import { AgentRunStateError } from "./contracts.js";
|
|
2
|
+
import { AgentLoopStateError, AgentRunStateError } from "./contracts.js";
|
|
3
3
|
export const AGENT_RUN_STATE_NAMESPACE = "prism.agent-run";
|
|
4
4
|
export const AGENT_RUN_STATE_SCHEMA_VERSION = 1;
|
|
5
5
|
export const DEFAULT_MAX_AGENT_RUN_STATE_BYTES = 256 * 1024;
|
|
6
6
|
export const HARD_MAX_AGENT_RUN_STATE_BYTES = 1024 * 1024;
|
|
7
7
|
const MAX_DEPTH = 32;
|
|
8
8
|
const MAX_PROPERTIES = 256;
|
|
9
|
+
/** Revision stamps of the built-in loops; custom strategies declare their own `revision`. */
|
|
10
|
+
export const BUILT_IN_LOOP_REVISIONS = {
|
|
11
|
+
"single-shot": "1",
|
|
12
|
+
"generate-validate-revise": "1",
|
|
13
|
+
};
|
|
14
|
+
/** Validate a strategy snapshot as JSON-compatible and package it for the durable envelope. */
|
|
15
|
+
export function boundedLoopSnapshot(name, revision, snapshot) {
|
|
16
|
+
try {
|
|
17
|
+
assertJsonValue(snapshot, 0);
|
|
18
|
+
}
|
|
19
|
+
catch (error) {
|
|
20
|
+
if (error instanceof AgentLoopStateError)
|
|
21
|
+
throw error;
|
|
22
|
+
throw new AgentLoopStateError("ERR_PRISM_LOOP_SNAPSHOT", "Loop snapshot must be JSON-compatible", { cause: error });
|
|
23
|
+
}
|
|
24
|
+
return { name, revision, snapshot };
|
|
25
|
+
}
|
|
26
|
+
function assertJsonValue(value, depth) {
|
|
27
|
+
if (depth > MAX_DEPTH)
|
|
28
|
+
throw new AgentLoopStateError("ERR_PRISM_LOOP_SNAPSHOT", `Loop snapshot exceeds depth ${MAX_DEPTH}`);
|
|
29
|
+
switch (typeof value) {
|
|
30
|
+
case "string":
|
|
31
|
+
case "boolean":
|
|
32
|
+
return;
|
|
33
|
+
case "number":
|
|
34
|
+
if (!Number.isFinite(value))
|
|
35
|
+
throw new AgentLoopStateError("ERR_PRISM_LOOP_SNAPSHOT", "Loop snapshot numbers must be finite");
|
|
36
|
+
return;
|
|
37
|
+
case "object": {
|
|
38
|
+
if (value === null)
|
|
39
|
+
return;
|
|
40
|
+
for (const item of Object.values(value))
|
|
41
|
+
assertJsonValue(item, depth + 1);
|
|
42
|
+
return;
|
|
43
|
+
}
|
|
44
|
+
default:
|
|
45
|
+
throw new AgentLoopStateError("ERR_PRISM_LOOP_SNAPSHOT", "Loop snapshot must be JSON-compatible");
|
|
46
|
+
}
|
|
47
|
+
}
|
|
9
48
|
export function agentFingerprint(agent, revision) {
|
|
10
49
|
const config = agent.config;
|
|
11
50
|
const tools = !config.tools ? [] : "list" in config.tools ? config.tools.list() : config.tools;
|
|
@@ -35,9 +74,10 @@ export function agentFingerprint(agent, revision) {
|
|
|
35
74
|
effect: typeof tool.effect === "function" ? "classifier" : tool.effect,
|
|
36
75
|
})),
|
|
37
76
|
guardrails: guardrails.map((guardrail) => ({ name: guardrail.name, stage: guardrail.stage, revision: guardrail.revision })),
|
|
77
|
+
// Loop revision participates so a loop change without a definitionRevision bump fails closed.
|
|
38
78
|
loop: typeof config.loop === "object" && config.loop && "strategy" in config.loop
|
|
39
|
-
? config.loop.strategy
|
|
40
|
-
:
|
|
79
|
+
? { name: config.loop.strategy, revision: BUILT_IN_LOOP_REVISIONS[config.loop.strategy] ?? null }
|
|
80
|
+
: { name: config.loop?.name ?? "single-shot", revision: config.loop?.revision ?? BUILT_IN_LOOP_REVISIONS["single-shot"] },
|
|
41
81
|
});
|
|
42
82
|
return createHash("sha256").update(value).digest("hex");
|
|
43
83
|
}
|
|
@@ -85,7 +125,7 @@ export function statusFromState(state, version) {
|
|
|
85
125
|
return { state: publicState({ ...state, version }), version };
|
|
86
126
|
}
|
|
87
127
|
export function publicState(state) {
|
|
88
|
-
const { input: _input, pending: _pending, interruptBeforeTool: _interruptBeforeTool, counters: _counters, deadlineAt: _deadlineAt, ...publicValue } = state;
|
|
128
|
+
const { input: _input, pending: _pending, pendingCalls: _pendingCalls, nestedRuns: _nestedRuns, interruptBeforeTool: _interruptBeforeTool, counters: _counters, deadlineAt: _deadlineAt, ...publicValue } = state;
|
|
89
129
|
return publicValue;
|
|
90
130
|
}
|
|
91
131
|
export function initialAgentRunState(input) {
|
|
@@ -103,6 +143,7 @@ export function initialAgentRunState(input) {
|
|
|
103
143
|
interruption: input.interruption,
|
|
104
144
|
input: input.messages,
|
|
105
145
|
pending: input.pending,
|
|
146
|
+
pendingCalls: input.pendingCalls,
|
|
106
147
|
interruptBeforeTool: input.interruptBeforeTool,
|
|
107
148
|
counters: input.counters,
|
|
108
149
|
deadlineAt: input.deadlineAt,
|
|
@@ -125,6 +166,41 @@ export function parseAgentRunState(value, version) {
|
|
|
125
166
|
!state.deadlineAt) {
|
|
126
167
|
throw new AgentRunStateError("Malformed agent run state");
|
|
127
168
|
}
|
|
169
|
+
if (state.pendingCalls !== undefined &&
|
|
170
|
+
(!Array.isArray(state.pendingCalls) ||
|
|
171
|
+
state.pendingCalls.some((entry) => !entry ||
|
|
172
|
+
typeof entry !== "object" ||
|
|
173
|
+
!entry.call ||
|
|
174
|
+
typeof entry.approvalId !== "string" ||
|
|
175
|
+
(entry.status !== "ready" && entry.status !== "dispatched")))) {
|
|
176
|
+
throw new AgentRunStateError("Malformed agent run pending calls");
|
|
177
|
+
}
|
|
178
|
+
if (state.stickyDecisions !== undefined &&
|
|
179
|
+
(!Array.isArray(state.stickyDecisions) ||
|
|
180
|
+
state.stickyDecisions.some((entry) => !entry ||
|
|
181
|
+
typeof entry !== "object" ||
|
|
182
|
+
!entry.scope ||
|
|
183
|
+
(entry.outcome !== "allow_for_run" && entry.outcome !== "reject_for_run")))) {
|
|
184
|
+
throw new AgentRunStateError("Malformed agent run sticky decisions");
|
|
185
|
+
}
|
|
186
|
+
if (state.nestedRuns !== undefined &&
|
|
187
|
+
(!Array.isArray(state.nestedRuns) ||
|
|
188
|
+
state.nestedRuns.some((entry) => !entry ||
|
|
189
|
+
typeof entry !== "object" ||
|
|
190
|
+
typeof entry.runId !== "string" ||
|
|
191
|
+
typeof entry.toolCallId !== "string" ||
|
|
192
|
+
!Array.isArray(entry.path) ||
|
|
193
|
+
!Array.isArray(entry.approvals) ||
|
|
194
|
+
entry.approvals.some((approval) => typeof approval?.id !== "string" || typeof approval?.childApprovalId !== "string")))) {
|
|
195
|
+
throw new AgentRunStateError("Malformed agent run nested runs");
|
|
196
|
+
}
|
|
197
|
+
if (state.loopState !== undefined &&
|
|
198
|
+
(typeof state.loopState !== "object" ||
|
|
199
|
+
typeof state.loopState.name !== "string" ||
|
|
200
|
+
typeof state.loopState.revision !== "string" ||
|
|
201
|
+
!("snapshot" in state.loopState))) {
|
|
202
|
+
throw new AgentRunStateError("Malformed agent run loop state");
|
|
203
|
+
}
|
|
128
204
|
// Load bounds against the hard cap, not the default: the configured maxStateBytes is a
|
|
129
205
|
// save-side policy knob, while the load-side bound is only a DoS ceiling. States saved
|
|
130
206
|
// with a raised maxStateBytes must remain resumable.
|