@arnilo/prism 0.0.23 → 0.0.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +39 -3
- package/dist/agent-event-source.d.ts +11 -0
- package/dist/agent-event-source.js +512 -0
- package/dist/agent-loops.js +37 -5
- package/dist/agent-run-state.d.ts +27 -1
- package/dist/agent-run-state.js +86 -5
- package/dist/agents.js +890 -78
- package/dist/contracts.d.ts +338 -4
- package/dist/contracts.js +53 -0
- package/dist/index.d.ts +9 -4
- package/dist/index.js +5 -2
- package/dist/testing/agent-event-source-conformance.d.ts +4 -0
- package/dist/testing/agent-event-source-conformance.js +54 -0
- package/dist/testing/persistence-schema.d.ts +2 -2
- package/dist/testing/persistence-schema.js +58 -21
- package/dist/testing/tool-effect-store-conformance.d.ts +9 -0
- package/dist/testing/tool-effect-store-conformance.js +85 -0
- package/dist/tool-effects.d.ts +15 -0
- package/dist/tool-effects.js +338 -0
- package/dist/tools.d.ts +4 -1
- package/dist/tools.js +219 -9
- package/docs/0.1.0-readiness.md +10 -9
- package/docs/a2a.md +6 -2
- package/docs/ag-ui-adoption.md +77 -0
- package/docs/ag-ui.md +77 -42
- package/docs/agent-events.md +5 -1
- package/docs/agent-loops.md +9 -1
- package/docs/agent-session-runtime.md +9 -2
- package/docs/browser-automation.md +2 -0
- package/docs/coding-agent-tools.md +2 -0
- package/docs/coding-security.md +1 -1
- package/docs/database-persistence.md +2 -0
- package/docs/enterprise-postgres-state.md +5 -1
- package/docs/host-security.md +8 -1
- package/docs/index.md +13 -11
- package/docs/mcp-tools.md +19 -2
- package/docs/migration.md +46 -0
- package/docs/performance.md +25 -0
- package/docs/postgres-persistence.md +5 -2
- package/docs/public-contracts.md +2 -0
- package/docs/release-and-install.md +70 -690
- package/docs/server.md +10 -6
- package/docs/sqlite-persistence.md +10 -2
- package/docs/supervisors.md +6 -0
- package/docs/tool-effects.md +95 -0
- package/docs/tools.md +4 -0
- package/docs/work-tools.md +4 -0
- package/docs/workflows.md +1 -1
- package/package.json +11 -3
package/dist/tools.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { AgentEvent, ErrorInfo, Guardrails, JsonObject, OwnershipScope, RunLedger, ToolCallContent, ToolDefinition, ToolExecutionContext, ToolRegistry, ToolResult } from "./contracts.js";
|
|
1
|
+
import type { AgentEvent, ErrorInfo, Guardrails, JsonObject, OwnershipScope, RunLedger, ToolCallContent, ToolDefinition, ToolExecutionContext, ToolEffectDeclaration, ToolEffectStore, ToolRegistry, ToolResult } from "./contracts.js";
|
|
2
2
|
import type { MiddlewareRegistry } from "./middleware.js";
|
|
3
3
|
import { type SecretRedactor } from "./redaction.js";
|
|
4
4
|
import { type DuplicateRegistrationOptions } from "./registry-options.js";
|
|
@@ -42,6 +42,8 @@ export interface DispatchToolCallOptions {
|
|
|
42
42
|
readonly trust?: TrustPolicy;
|
|
43
43
|
readonly redactor?: SecretRedactor;
|
|
44
44
|
readonly ledger?: RunLedger;
|
|
45
|
+
/** Optional shared recovery store. Only declared optional/required effects use it. */
|
|
46
|
+
readonly effectStore?: ToolEffectStore;
|
|
45
47
|
readonly ownership?: OwnershipScope;
|
|
46
48
|
/** Host-verified identity; asserted active before tool side effects when present. */
|
|
47
49
|
readonly identity?: import("./identity.js").AgentIdentity;
|
|
@@ -55,3 +57,4 @@ export interface ToolRegistryOptions extends DuplicateRegistrationOptions {
|
|
|
55
57
|
export declare function createToolRegistry(tools?: readonly ToolDefinition[], options?: ToolRegistryOptions): ToolRegistry;
|
|
56
58
|
export declare function filterTools(tools: readonly ToolDefinition[], filter?: ToolFilterInput): readonly ToolDefinition[];
|
|
57
59
|
export declare function dispatchToolCall(options: DispatchToolCallOptions): Promise<ToolResult>;
|
|
60
|
+
export declare function resolveToolEffectDeclaration(tool: ToolDefinition, args: JsonObject, context: ToolExecutionContext): ToolEffectDeclaration | undefined;
|
package/dist/tools.js
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import { isJsonObject } from "./config.js";
|
|
2
2
|
import { GuardrailError, runGuardrails } from "./guardrails.js";
|
|
3
|
-
import { assertIdentityActive, assertIdentityMatchesOwnership } from "./identity.js";
|
|
3
|
+
import { assertIdentityActive, assertIdentityMatchesOwnership, ownershipFromIdentity } from "./identity.js";
|
|
4
4
|
import { createId } from "./ids.js";
|
|
5
5
|
import { errorToErrorInfo, redactRunLedgerRecord, redactSecrets } from "./redaction.js";
|
|
6
6
|
import { assertCanRegister } from "./registry-options.js";
|
|
7
7
|
import { assertPermission, assertTrusted } from "./security.js";
|
|
8
|
+
import { deriveToolEffectKey, toolEffectArgumentsHash, ToolEffectError } from "./tool-effects.js";
|
|
8
9
|
/** Wrap a schema adapter as the existing `ToolValidator` seam used by dispatch and the agent runtime. */
|
|
9
10
|
export function createToolParameterValidator(validator, options = {}) {
|
|
10
11
|
const missingSchema = options.missingSchema ?? "allow";
|
|
@@ -89,8 +90,9 @@ export async function dispatchToolCall(options) {
|
|
|
89
90
|
const postcheck = await checkCall(mediatedCall, options, startedAt);
|
|
90
91
|
if (postcheck)
|
|
91
92
|
return postcheck;
|
|
92
|
-
const
|
|
93
|
-
|
|
93
|
+
const { idempotencyKey: _untrustedKey, ...baseContext } = options.context;
|
|
94
|
+
let context = {
|
|
95
|
+
...baseContext,
|
|
94
96
|
toolCallId: mediatedCall.id,
|
|
95
97
|
identity: options.identity ?? options.context.identity,
|
|
96
98
|
progress: async (progress, metadata) => {
|
|
@@ -135,17 +137,38 @@ export async function dispatchToolCall(options) {
|
|
|
135
137
|
const validation = await options.validate?.(tool, mediatedCall.arguments, context);
|
|
136
138
|
if (validation)
|
|
137
139
|
return blocked(mediatedCall, context, "validation_failed", toErrorInfo(validation, secrets), options, startedAt);
|
|
140
|
+
let effect;
|
|
141
|
+
try {
|
|
142
|
+
const prepared = await prepareToolEffect(tool, mediatedCall, context, options);
|
|
143
|
+
if (prepared.result)
|
|
144
|
+
return prepared.result;
|
|
145
|
+
context = prepared.context;
|
|
146
|
+
effect = prepared.effect;
|
|
147
|
+
}
|
|
148
|
+
catch (error) {
|
|
149
|
+
return blocked(mediatedCall, context, "execution_denied", errorToErrorInfo(error, secrets), options, startedAt);
|
|
150
|
+
}
|
|
138
151
|
try {
|
|
139
152
|
await options.beforeExecute?.(mediatedCall, tool, context);
|
|
140
153
|
}
|
|
141
154
|
catch (error) {
|
|
142
|
-
|
|
155
|
+
await failBeforeEffect(effect, isSuspended(error) ? "failed_retryable" : "failed_terminal");
|
|
156
|
+
// Loop-state contract errors (snapshot capture) are terminal run errors, not tool errors.
|
|
157
|
+
if (isSuspended(error) || isLoopStateError(error) || isDelegationSuspended(error))
|
|
143
158
|
throw error;
|
|
144
159
|
return blocked(mediatedCall, context, "execution_denied", errorToErrorInfo(error, secrets), options, startedAt);
|
|
145
160
|
}
|
|
146
|
-
|
|
147
|
-
|
|
161
|
+
let completedResult;
|
|
162
|
+
let dispatchAttempted = false;
|
|
148
163
|
try {
|
|
164
|
+
await options.emit?.({ type: "tool_execution_started", sessionId: context.sessionId, runId: context.runId, call: mediatedCall });
|
|
165
|
+
await appendToolCallRecord(options, "started", mediatedCall, startedAt, {});
|
|
166
|
+
if (effect) {
|
|
167
|
+
dispatchAttempted = true;
|
|
168
|
+
const record = await effect.store.markDispatched(transition(effect));
|
|
169
|
+
effect.expectedVersion = record.version;
|
|
170
|
+
effect.dispatched = true;
|
|
171
|
+
}
|
|
149
172
|
const raw = await tool.execute(mediatedCall.arguments, context);
|
|
150
173
|
const mediatedResult = await (options.middleware?.run("tool_result", raw) ?? raw);
|
|
151
174
|
const outputGuards = await runGuardrails({
|
|
@@ -166,16 +189,51 @@ export async function dispatchToolCall(options) {
|
|
|
166
189
|
if (outputGuards.terminal) {
|
|
167
190
|
if (outputGuards.terminal.action !== "block")
|
|
168
191
|
throw new GuardrailError(outputGuards.terminal);
|
|
192
|
+
if (effect)
|
|
193
|
+
return finishUnknownEffect(effect, mediatedCall, context, options, startedAt);
|
|
169
194
|
return blocked(mediatedCall, context, "guardrail_blocked", { message: "Tool result blocked by guardrail" }, options, startedAt);
|
|
170
195
|
}
|
|
196
|
+
if (effect && mediatedResult.error)
|
|
197
|
+
return finishUnknownEffect(effect, mediatedCall, context, options, startedAt);
|
|
171
198
|
const result = options.redactor?.redact(mediatedResult) ?? mediatedResult;
|
|
199
|
+
if (effect) {
|
|
200
|
+
try {
|
|
201
|
+
const record = await effect.store.complete({ ...transition(effect), result });
|
|
202
|
+
effect.expectedVersion = record.version;
|
|
203
|
+
effect.completed = true;
|
|
204
|
+
completedResult = record.result ?? result;
|
|
205
|
+
}
|
|
206
|
+
catch {
|
|
207
|
+
return finishUnknownEffect(effect, mediatedCall, context, options, startedAt);
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
completedResult ??= result;
|
|
172
211
|
const finishedAt = new Date().toISOString();
|
|
173
212
|
const metadata = toolExecutionMetadata(startedAt, "finished");
|
|
174
|
-
await options.emit?.({
|
|
175
|
-
|
|
176
|
-
|
|
213
|
+
await options.emit?.({
|
|
214
|
+
type: "tool_execution_finished",
|
|
215
|
+
sessionId: context.sessionId,
|
|
216
|
+
runId: context.runId,
|
|
217
|
+
result: completedResult,
|
|
218
|
+
metadata,
|
|
219
|
+
});
|
|
220
|
+
await appendToolCallRecord(options, "finished", mediatedCall, startedAt, { finishedAt, result: completedResult });
|
|
221
|
+
return completedResult;
|
|
177
222
|
}
|
|
178
223
|
catch (error) {
|
|
224
|
+
if (completedResult)
|
|
225
|
+
return completedResult;
|
|
226
|
+
// Nested-run suspensions must propagate to the run loop, never become tool errors.
|
|
227
|
+
if (isDelegationSuspended(error)) {
|
|
228
|
+
if (effect && (effect.dispatched || dispatchAttempted))
|
|
229
|
+
await unknownEffectResult(effect, mediatedCall);
|
|
230
|
+
else
|
|
231
|
+
await failBeforeEffect(effect, "failed_retryable");
|
|
232
|
+
throw error;
|
|
233
|
+
}
|
|
234
|
+
if (effect && (effect.dispatched || dispatchAttempted))
|
|
235
|
+
return finishUnknownEffect(effect, mediatedCall, context, options, startedAt);
|
|
236
|
+
await failBeforeEffect(effect, "failed_terminal");
|
|
179
237
|
if (error instanceof GuardrailError)
|
|
180
238
|
throw error;
|
|
181
239
|
const info = errorToErrorInfo(error, secrets);
|
|
@@ -194,6 +252,158 @@ export async function dispatchToolCall(options) {
|
|
|
194
252
|
return result;
|
|
195
253
|
}
|
|
196
254
|
}
|
|
255
|
+
async function prepareToolEffect(tool, call, context, options) {
|
|
256
|
+
const identity = context.identity;
|
|
257
|
+
const declaration = resolveToolEffectDeclaration(tool, call.arguments, context);
|
|
258
|
+
if (!declaration || declaration.kind === "none" || declaration.idempotency === "none")
|
|
259
|
+
return { context };
|
|
260
|
+
if (!identity && declaration.idempotency === "unsupported")
|
|
261
|
+
return { context };
|
|
262
|
+
if (!identity)
|
|
263
|
+
throw new ToolEffectError("ERR_PRISM_TOOL_EFFECT_CONFLICT", "verified identity is required for a durable tool effect");
|
|
264
|
+
const ownership = ownershipFromIdentity(identity);
|
|
265
|
+
const argumentsHash = toolEffectArgumentsHash(call.arguments);
|
|
266
|
+
const base = {
|
|
267
|
+
identity,
|
|
268
|
+
ownership,
|
|
269
|
+
sessionId: context.sessionId,
|
|
270
|
+
runId: context.runId,
|
|
271
|
+
toolCallId: call.id,
|
|
272
|
+
toolName: call.name,
|
|
273
|
+
argumentsHash,
|
|
274
|
+
};
|
|
275
|
+
const key = { ...base, key: deriveToolEffectKey(base), signal: context.signal };
|
|
276
|
+
const keyedContext = { ...context, idempotencyKey: key.key };
|
|
277
|
+
if (declaration.idempotency === "tool_managed" || declaration.idempotency === "unsupported")
|
|
278
|
+
return { context: keyedContext };
|
|
279
|
+
const store = options.effectStore;
|
|
280
|
+
if (!store) {
|
|
281
|
+
if (declaration.idempotency === "required")
|
|
282
|
+
throw new ToolEffectError("ERR_PRISM_TOOL_EFFECT_REQUIRED", "durable tool effect store is required");
|
|
283
|
+
return { context: keyedContext };
|
|
284
|
+
}
|
|
285
|
+
let begun;
|
|
286
|
+
try {
|
|
287
|
+
begun = await store.begin(key);
|
|
288
|
+
}
|
|
289
|
+
catch (error) {
|
|
290
|
+
if (error instanceof ToolEffectError)
|
|
291
|
+
throw error;
|
|
292
|
+
throw new ToolEffectError("ERR_PRISM_TOOL_EFFECT_UNKNOWN", "tool effect claim outcome is unknown");
|
|
293
|
+
}
|
|
294
|
+
if (begun.outcome === "existing")
|
|
295
|
+
return { context: keyedContext, result: replayEffectResult(begun.record) };
|
|
296
|
+
if (!begun.record.claimToken)
|
|
297
|
+
throw new ToolEffectError("ERR_PRISM_TOOL_EFFECT_UNKNOWN", "tool effect claim outcome is unknown");
|
|
298
|
+
return {
|
|
299
|
+
context: keyedContext,
|
|
300
|
+
effect: {
|
|
301
|
+
store,
|
|
302
|
+
key,
|
|
303
|
+
claimToken: begun.record.claimToken,
|
|
304
|
+
expectedVersion: begun.record.version,
|
|
305
|
+
dispatched: false,
|
|
306
|
+
completed: false,
|
|
307
|
+
},
|
|
308
|
+
};
|
|
309
|
+
}
|
|
310
|
+
export function resolveToolEffectDeclaration(tool, args, context) {
|
|
311
|
+
const classifierContext = Object.freeze({
|
|
312
|
+
sessionId: context.sessionId,
|
|
313
|
+
runId: context.runId,
|
|
314
|
+
toolCallId: context.toolCallId,
|
|
315
|
+
signal: context.signal,
|
|
316
|
+
metadata: context.metadata,
|
|
317
|
+
});
|
|
318
|
+
const declaration = typeof tool.effect === "function" ? tool.effect(args, classifierContext) : tool.effect;
|
|
319
|
+
if (!declaration)
|
|
320
|
+
return undefined;
|
|
321
|
+
if (!["none", "local_mutation", "external_mutation"].includes(declaration.kind) ||
|
|
322
|
+
!["none", "optional", "required", "tool_managed", "unsupported"].includes(declaration.idempotency) ||
|
|
323
|
+
(declaration.kind === "none" && declaration.idempotency !== "none")) {
|
|
324
|
+
throw new ToolEffectError("ERR_PRISM_TOOL_EFFECT_LIMIT", "tool effect declaration is invalid");
|
|
325
|
+
}
|
|
326
|
+
return declaration;
|
|
327
|
+
}
|
|
328
|
+
function replayEffectResult(record) {
|
|
329
|
+
if (record.status === "completed") {
|
|
330
|
+
if (record.result)
|
|
331
|
+
return record.result;
|
|
332
|
+
throw new ToolEffectError("ERR_PRISM_TOOL_EFFECT_COMPLETED", "tool effect already completed without replayable result");
|
|
333
|
+
}
|
|
334
|
+
if (record.status === "dispatched" || record.status === "unknown") {
|
|
335
|
+
throw new ToolEffectError("ERR_PRISM_TOOL_EFFECT_UNKNOWN", "tool effect outcome requires reconciliation");
|
|
336
|
+
}
|
|
337
|
+
throw new ToolEffectError("ERR_PRISM_TOOL_EFFECT_CONFLICT", "tool effect is not dispatchable");
|
|
338
|
+
}
|
|
339
|
+
function transition(effect) {
|
|
340
|
+
return { ...effect.key, claimToken: effect.claimToken, expectedVersion: effect.expectedVersion };
|
|
341
|
+
}
|
|
342
|
+
async function failBeforeEffect(effect, status) {
|
|
343
|
+
if (!effect || effect.dispatched || effect.completed)
|
|
344
|
+
return;
|
|
345
|
+
try {
|
|
346
|
+
await effect.store.fail({
|
|
347
|
+
...transition(effect),
|
|
348
|
+
status,
|
|
349
|
+
failure: { code: "ERR_PRISM_TOOL_EFFECT_PRE_DISPATCH" },
|
|
350
|
+
});
|
|
351
|
+
}
|
|
352
|
+
catch {
|
|
353
|
+
// No effect was invoked. A stale/failed pre-dispatch transition only delays a later safe retry.
|
|
354
|
+
}
|
|
355
|
+
}
|
|
356
|
+
async function unknownEffectResult(effect, call) {
|
|
357
|
+
try {
|
|
358
|
+
let claim = effect.dispatched
|
|
359
|
+
? { claimToken: effect.claimToken, version: effect.expectedVersion }
|
|
360
|
+
: undefined;
|
|
361
|
+
if (!claim) {
|
|
362
|
+
const current = await effect.store.get(effect.key);
|
|
363
|
+
if (current?.status === "dispatched" && current.claimToken)
|
|
364
|
+
claim = { claimToken: current.claimToken, version: current.version };
|
|
365
|
+
}
|
|
366
|
+
if (claim) {
|
|
367
|
+
await effect.store.markUnknown({
|
|
368
|
+
...effect.key,
|
|
369
|
+
claimToken: claim.claimToken,
|
|
370
|
+
expectedVersion: claim.version,
|
|
371
|
+
failure: { code: "ERR_PRISM_TOOL_EFFECT_UNKNOWN" },
|
|
372
|
+
});
|
|
373
|
+
}
|
|
374
|
+
}
|
|
375
|
+
catch {
|
|
376
|
+
// A post-dispatch persistence error is itself ambiguous; never expose or retry it.
|
|
377
|
+
}
|
|
378
|
+
return effectErrorResult(call, "ERR_PRISM_TOOL_EFFECT_UNKNOWN", "tool effect outcome requires reconciliation");
|
|
379
|
+
}
|
|
380
|
+
async function finishUnknownEffect(effect, call, context, options, startedAt) {
|
|
381
|
+
const result = await unknownEffectResult(effect, call);
|
|
382
|
+
const error = result.error;
|
|
383
|
+
const finishedAt = new Date().toISOString();
|
|
384
|
+
const metadata = toolExecutionMetadata(startedAt, "error");
|
|
385
|
+
try {
|
|
386
|
+
await options.emit?.({ type: "tool_execution_error", sessionId: context.sessionId, runId: context.runId, call, error, metadata });
|
|
387
|
+
await appendToolCallRecord(options, "error", call, startedAt, { finishedAt, result });
|
|
388
|
+
}
|
|
389
|
+
catch {
|
|
390
|
+
// The effect is already ambiguous; exposure/ledger failures cannot make it safe to retry.
|
|
391
|
+
}
|
|
392
|
+
return result;
|
|
393
|
+
}
|
|
394
|
+
function effectErrorResult(call, code, message) {
|
|
395
|
+
const error = new ToolEffectError(code, message);
|
|
396
|
+
return { toolCallId: call.id, name: call.name, error: errorToErrorInfo(error) };
|
|
397
|
+
}
|
|
398
|
+
function isSuspended(error) {
|
|
399
|
+
return error?.code === "ERR_PRISM_AGENT_RUN_SUSPENDED";
|
|
400
|
+
}
|
|
401
|
+
function isLoopStateError(error) {
|
|
402
|
+
return typeof error?.code === "string" && error.code.startsWith("ERR_PRISM_LOOP_");
|
|
403
|
+
}
|
|
404
|
+
function isDelegationSuspended(error) {
|
|
405
|
+
return error?.code === "ERR_PRISM_DELEGATION_SUSPENDED";
|
|
406
|
+
}
|
|
197
407
|
async function checkCall(call, options, startedAt) {
|
|
198
408
|
const context = options.context;
|
|
199
409
|
const tool = options.registry.get(call.name);
|
package/docs/0.1.0-readiness.md
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
# 0.1.0 / 1.0 Readiness Gates
|
|
2
2
|
|
|
3
|
-
Status: **0.0.
|
|
3
|
+
Status: **0.0.25** is the current release line (Phase 8 durable custom loops and human-in-the-loop); **1.0** readiness remains operator-gated, not automatic.
|
|
4
4
|
|
|
5
5
|
This page distills runnable readiness gates into one command-per-gate table. The
|
|
6
6
|
**Last evidence** column records the 2026-07-26 **0.0.16** baseline snapshot
|
|
7
7
|
(Phase 11, Node v24.18.0, Linux x86_64). Treat it as historical floor evidence,
|
|
8
8
|
not the current release tag. Re-run each gate on the target release tree before
|
|
9
|
-
cutting 0.0.
|
|
9
|
+
cutting 0.0.25 / 1.0. The decision to cut 1.0 stays with the operator after
|
|
10
10
|
operator-gated legs run in a protected environment and Phase 12 demand evidence
|
|
11
11
|
exists.
|
|
12
12
|
|
|
@@ -14,15 +14,16 @@ Evidence trail: [`docs/review-coverage-2026-07-26-phase-11.md`](./review-coverag
|
|
|
14
14
|
(addenda 0–9), [`docs/release-and-install.md`](./release-and-install.md),
|
|
15
15
|
[`docs/migration.md`](./migration.md), [`docs/performance.md`](./performance.md).
|
|
16
16
|
|
|
17
|
-
## Current line (0.0.
|
|
17
|
+
## Current line (0.0.25)
|
|
18
18
|
|
|
19
19
|
| Item | Status |
|
|
20
20
|
|---|---|
|
|
21
|
-
| Published graph | **47** publishable manifests at **0.0.
|
|
22
|
-
| Phase
|
|
23
|
-
| Docs tripwires | `node --test dist/__tests__/docs.test.js` — migration section `0.0.
|
|
24
|
-
|
|
|
25
|
-
|
|
|
21
|
+
| Published graph | **47** publishable manifests at **0.0.25** (`docs/release-and-install.md`) |
|
|
22
|
+
| Phase 8 durable loops / HITL | Custom-loop snapshot/restore, shared pending decisions, nested attributions, A2UI + standard AG-UI projectors |
|
|
23
|
+
| Docs tripwires | `node --test dist/__tests__/docs.test.js` — migration section `0.0.24 → 0.0.25 durable custom loops and human-in-the-loop` |
|
|
24
|
+
| Network-free Phase 8 evidence | `scripts/phase8-conformance.test.mjs`; `benchmark-0.0.25.json` under Task 0 ceilings |
|
|
25
|
+
| Protected database evidence (Phase 7) | `PRISM_TEST_POSTGRES_URL="$DATABASE_URL" npm run test:postgres`; `benchmark-0.0.24.json` under prior ceilings |
|
|
26
|
+
| Readiness table below | **0.0.16 measured values** remain historical network-free baseline; 0.0.25 loop/HITL evidence is recorded separately |
|
|
26
27
|
|
|
27
28
|
## Gate table
|
|
28
29
|
|
|
@@ -80,7 +81,7 @@ Deterministic budgets (CI gate, `scripts/budget-gate.test.mjs`):
|
|
|
80
81
|
| Root unpacked bytes | 2,043,402 | +5% | 2.1 MB (within) |
|
|
81
82
|
| Root file count | 270 | +5% | 270 |
|
|
82
83
|
| Cold-startup import | 38 ms | ceiling 250 ms | ~38 ms |
|
|
83
|
-
| Aggregate packed (47 manifests, reference only) | 1,217,694 | +10% | remeasure for the 0.0.
|
|
84
|
+
| Aggregate packed (47 manifests, reference only) | 1,217,694 | +10% | remeasure for the 0.0.24 graph before release |
|
|
84
85
|
|
|
85
86
|
Benchmark medians (on-demand evidence, `scripts/benchmark-0.0.16.mjs`, ±25%):
|
|
86
87
|
|
package/docs/a2a.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
`@arnilo/prism-supervisor` implements bounded A2A 1.0 over the JSON-RPC/HTTPS binding. Supported operations: `SendMessage`, `SendStreamingMessage`, `GetTask`, `ListTasks`, `CancelTask`, `SubscribeToTask`, push-notification-config create/get/list/delete, and `GetExtendedAgentCard`. Agent Cards retain explicit ES256 verification. gRPC, HTTP+JSON, discovery registries, automatic JWK/OAuth fetching, and an internal task worker/store are absent.
|
|
5
|
+
`@arnilo/prism-supervisor` implements bounded A2A 1.0 over the JSON-RPC/HTTPS binding. Supported operations: `SendMessage`, `SendStreamingMessage`, `GetTask`, `ListTasks`, `CancelTask`, `SubscribeToTask`, push-notification-config create/get/list/delete, and `GetExtendedAgentCard`. `client.streamMessage()` additionally exposes verified rich task/message events for frontend adapters while legacy `stream()` remains text-compatible. Agent Cards retain explicit ES256 verification. gRPC, HTTP+JSON, discovery registries, automatic JWK/OAuth fetching, and an internal task worker/store are absent.
|
|
6
6
|
|
|
7
7
|
## When to use it
|
|
8
8
|
|
|
@@ -59,12 +59,15 @@ Streams use ordered SSE frames with `id:` and JSON-RPC `result` containing one `
|
|
|
59
59
|
Client APIs:
|
|
60
60
|
|
|
61
61
|
- `send()` / `stream()` preserve text-to-`AgentRunResult` compatibility.
|
|
62
|
+
- `streamMessage(message)` exposes bounded verified `A2AStreamEvent` task/message records without discarding artifact/data parts.
|
|
62
63
|
- `sendMessage()` returns rich/durable `A2ATask`.
|
|
63
64
|
- `getTask()`, `listTasks()`, `cancelTask()`, `subscribeToTask()` operate on durable tasks.
|
|
64
65
|
- `createPushConfig()`, `getPushConfig()`, `listPushConfigs()`, `deletePushConfig()` expose declared push config operations.
|
|
65
66
|
|
|
66
67
|
Every protocol request sends/negotiates `A2A-Version: 1.0`. Client endpoint/card URLs require exact allow-listed HTTPS and `redirect: "error"`. Cards are parsed then optionally verified against host-pinned keys; no key URL is fetched.
|
|
67
68
|
|
|
69
|
+
`createA2AAgentEventSource({ source, resolveTask, map })` supplies only the durable `subscribe` seam for a host-owned `A2ATaskLifecycle`. It resolves task→exact Prism run under authorization, consumes `AgentEventSource.subscribe()`, and uses each opaque source cursor as stable A2A `eventId`. With no cursor, the first mapped update must be a full Task, matching A2A streaming rules. It creates no task store or worker. Standard `SubscribeToTask` has no `afterEventId`; Prism retains that bounded field as an explicitly documented reconnect extension.
|
|
70
|
+
|
|
68
71
|
## Request/response example
|
|
69
72
|
|
|
70
73
|
```json
|
|
@@ -73,7 +76,7 @@ Every protocol request sends/negotiates `A2A-Version: 1.0`. Client endpoint/card
|
|
|
73
76
|
|
|
74
77
|
## Extension and configuration notes
|
|
75
78
|
|
|
76
|
-
Handler requires `card.capabilities.pushNotifications` to exactly match supplied `push`; mismatch fails construction, preserving signed-card integrity and preventing false capability claims. Streaming remains available for direct text invocation. Push adapter owns exact-owner persistence, signing/auth credentials, and network transport. Host explicitly calls `deliverA2APushEvent()` from its durable update path; helper bounds event, timeout (10s default/60s hard), attempts (1 default/3 hard), and passes stable event ID as idempotency key to host `A2APushDelivery`. It starts no hidden sender and performs no network itself. Config handling validates IDs/count/bytes and requires same explicit URL policy used for URL parts. Returned push configs omit token and authentication credentials.
|
|
79
|
+
Handler requires `card.capabilities.pushNotifications` to exactly match supplied `push`; mismatch fails construction, preserving signed-card integrity and preventing false capability claims. `createAgUiA2AAdapter({ client, select, correlate, projectPart })` in `@arnilo/prism-ag-ui` fronts one host-selected verified client: host selects new/follow task mode, persists exact run/thread/task correlation before output, and may project non-text/tool/A2UI parts. It never discovers agents, opens a local session, or replaces this direct A2A API. Streaming remains available for direct text invocation. Push adapter owns exact-owner persistence, signing/auth credentials, and network transport. Host explicitly calls `deliverA2APushEvent()` from its durable update path; helper bounds event, timeout (10s default/60s hard), attempts (1 default/3 hard), and passes stable event ID as idempotency key to host `A2APushDelivery`. It starts no hidden sender and performs no network itself. Config handling validates IDs/count/bytes and requires same explicit URL policy used for URL parts. Returned push configs omit token and authentication credentials.
|
|
77
80
|
|
|
78
81
|
Defaults/hard caps include: request 64 KiB/1 MiB; response 1/8 MiB; event 64 KiB/1 MiB; stream 10/64 MiB and 10k/100k events; replay 1k/10k events; concurrency 16/256; timeout 120s/30m; IDs 256/4096 B; parts 32/256; part/raw 1/8 MiB; data 256 KiB/4 MiB; artifacts 32/256; history/page 100/1000; cursor 4/16 KiB; push configs 10/100. Hosts may narrow limits.
|
|
79
82
|
|
|
@@ -95,3 +98,4 @@ Defaults/hard caps include: request 64 KiB/1 MiB; response 1/8 MiB; event 64 KiB
|
|
|
95
98
|
- [Workflows](workflows.md)
|
|
96
99
|
- [Host security](host-security.md)
|
|
97
100
|
- [Frontend interoperability (AG-UI and ACP)](ag-ui.md): browser/editor protocol adapters over a Prism session; not an A2A card, task lifecycle, or remote-agent transport.
|
|
101
|
+
- [AG-UI adoption evaluation](ag-ui-adoption.md): official AG-UI A2A fronting assessment and shipped explicit adapter.
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
# AG-UI adoption evaluation
|
|
2
|
+
|
|
3
|
+
## What it does
|
|
4
|
+
|
|
5
|
+
This page records Prism's compatibility review against official AG-UI `@ag-ui/core` **0.0.57** and the official repository at commit [`a40b5c0`](https://github.com/ag-ui-protocol/ag-ui/commit/a40b5c0824564eb2f9ab9edf2be43f355f42a3b8). It separates shipped transport/replay support from remaining work needed to claim full AG-UI support, including AG-UI fronting MCP and A2A agents.
|
|
6
|
+
|
|
7
|
+
Official material reviewed:
|
|
8
|
+
|
|
9
|
+
- [Events](https://docs.ag-ui.com/concepts/events), [messages](https://docs.ag-ui.com/concepts/messages), [tools](https://docs.ag-ui.com/concepts/tools), [state](https://docs.ag-ui.com/concepts/state), [reasoning](https://docs.ag-ui.com/concepts/reasoning), [interrupts](https://docs.ag-ui.com/concepts/interrupts), [capabilities](https://docs.ag-ui.com/concepts/capabilities), [serialization](https://docs.ag-ui.com/concepts/serialization), [server quickstart](https://docs.ag-ui.com/quickstart/server), and [protocol architecture](https://docs.ag-ui.com/concepts/architecture).
|
|
10
|
+
- Official [MCP/A2A/AG-UI relationship](https://docs.ag-ui.com/agentic-protocols), [integrations](https://docs.ag-ui.com/integrations), [`@ag-ui/mcp-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/mcp-middleware), [`@ag-ui/mcp-apps-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/mcp-apps-middleware), [`@ag-ui/a2ui-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/a2ui-middleware) (Prism ships an in-package opt-in painter with frozen caps; no runtime dependency), [`@ag-ui/a2a`](https://github.com/ag-ui-protocol/ag-ui/tree/main/integrations/a2a/typescript), and [`@ag-ui/a2a-middleware`](https://github.com/ag-ui-protocol/ag-ui/tree/main/middlewares/a2a-middleware).
|
|
11
|
+
- A2A [current specification](https://a2a-protocol.org/latest/specification/) and [streaming rules](https://a2a-protocol.org/latest/topics/streaming-and-async/).
|
|
12
|
+
- MCP Apps [SEP-1865](https://modelcontextprotocol.io/seps/1865-mcp-apps-interactive-user-interfaces-for-mcp) and the [`io.modelcontextprotocol/ui` draft specification](https://github.com/modelcontextprotocol/ext-apps/blob/main/specification/draft/apps.mdx).
|
|
13
|
+
|
|
14
|
+
## When to use it
|
|
15
|
+
|
|
16
|
+
Use this matrix when selecting Prism for an AG-UI client or planning protocol work. Tasks 3A and 3B complete full AG-UI 0.0.57 request/event compatibility plus explicit hardened MCP, MCP Apps, and remote A2A fronting. These are opt-in adapters over existing Prism clients, not an alternate runtime or discovery path.
|
|
17
|
+
|
|
18
|
+
## Inputs / request
|
|
19
|
+
|
|
20
|
+
Official `RunAgentInput` fields—lineage, all message roles/history, state, tools, context, props, media, and resume—are schema/bound checked then have no authority until `input.project` returns host-selected messages. Client tools remain client handoffs; state/props/media never grant identity, ownership, or server tools. Resume is exact `${runId}:${version}` CAS; edited arguments deny.
|
|
21
|
+
|
|
22
|
+
## Outputs / response / events
|
|
23
|
+
|
|
24
|
+
Prism emits current standard lifecycle, step, text, tool, state, messages, activity, reasoning, raw, and custom families when its mapper or an explicit projector can prove them. Deprecated `THINKING_*` and convenience chunk output are intentionally absent. SSE is baseline; capabilities truthfully narrow to configured replay/projectors/lifecycle, and source cursors remain bounded `prismCursor` metadata for reconnect.
|
|
25
|
+
|
|
26
|
+
## Request/response example
|
|
27
|
+
|
|
28
|
+
```json
|
|
29
|
+
{
|
|
30
|
+
"type": "TEXT_MESSAGE_CONTENT",
|
|
31
|
+
"messageId": "message-1",
|
|
32
|
+
"delta": "hello",
|
|
33
|
+
"prismEventId": "event-42",
|
|
34
|
+
"prismCursor": "opaque-owner-run-bound-cursor"
|
|
35
|
+
}
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
## Implementation example
|
|
39
|
+
|
|
40
|
+
```ts
|
|
41
|
+
import { createAgentEventSourceAgUiReplay, createAgUiHandler } from "@arnilo/prism-ag-ui";
|
|
42
|
+
|
|
43
|
+
const replay = createAgentEventSourceAgUiReplay(persistence.events, {
|
|
44
|
+
resolveRun: hostResolveProtocolRun,
|
|
45
|
+
ownership: (authorization) => authorization.ownership,
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
const handle = createAgUiHandler({ authorize, sessionFactory, lifecycle, resolveRun, replay });
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## Extension and configuration notes
|
|
52
|
+
|
|
53
|
+
### A2A adoption
|
|
54
|
+
|
|
55
|
+
A2A stays separately mounted. `createA2AAgentEventSource()` maps durable runs to task events; `afterEventId` remains Prism-only.
|
|
56
|
+
|
|
57
|
+
`createAgUiA2AAdapter({ client, select, correlate, projectPart })` fronts one verified host-selected client: task text/status becomes AG-UI text/activity, correlation persists before output, and non-text/tool/A2UI needs a schema-validated projector. It uses streaming when declared; fallback accepts only a terminal task. Client origin/card/auth/bounds/abort checks remain active.
|
|
58
|
+
|
|
59
|
+
### MCP adoption through AG-UI
|
|
60
|
+
|
|
61
|
+
Prism adapts its hardened MCP bridge; no official middleware or second runtime. `connectMcpTools({ mcpApps: true })` requires `io.modelcontextprotocol/ui` acknowledgement, retains nested UI metadata over deprecated flat metadata, hides app-only tools, and bounds linked `ui://` HTML through `bridge.apps`. `createAgUiMcpAdapter()` selects model-visible tools for normal core dispatch; `createAgUiMcpAppHandler()` reauthorizes one-bridge initialize/ping/logging/tool/resource calls with approval and visibility; sandbox helper returns fixed iframe/CSP constraints. No generic proxy, cross-server call, raw HTML rendering, or automatic mutation retry.
|
|
62
|
+
|
|
63
|
+
## Security and performance notes
|
|
64
|
+
|
|
65
|
+
- Every replay/reconnect reauthorizes, then source access uses exact host ownership and host-resolved internal run IDs. Cursor content never selects ownership.
|
|
66
|
+
- Durable streams are at-least-once. Clients deduplicate `prismEventId`; source cursors resume strictly after a durable record.
|
|
67
|
+
- Client tools, state, context, forwarded properties, media URLs/data, remote A2A parts, MCP metadata, HTML, iframe messages, and reasoning blobs are untrusted. Projectors must use existing Prism media URL/SSRF/MIME policy before any resolution.
|
|
68
|
+
- `input.project` and all output projectors are allow-lists; all generic JSON has byte/depth/property/array caps and prototype-pollution keys fail before host callbacks. Tool handoffs are client-only and cannot widen identity, ownership, permissions, or active server tools.
|
|
69
|
+
- MCP Apps requires sandbox-origin separation, restrictive CSP, declared-domain ceilings, audited JSON-RPC, app/tool visibility checks, and user approval for UI-initiated mutations. The shipped proxy does not retry those mutations; Task 4 adds generic durable effect recovery.
|
|
70
|
+
|
|
71
|
+
## Related APIs
|
|
72
|
+
|
|
73
|
+
- [Frontend interoperability](ag-ui.md): shipped Prism AG-UI/ACP API.
|
|
74
|
+
- [Agent events](agent-events.md): source events and durable delivery.
|
|
75
|
+
- [Web-standard server](server.md): SSE `Last-Event-ID` reconnect route.
|
|
76
|
+
- [A2A interoperability](a2a.md): separately mounted A2A lifecycle and durable source adapter.
|
|
77
|
+
- [MCP bridge/server](mcp-tools.md): hardened MCP transport and capability boundary reused by future AG-UI MCP support.
|