@arnilo/prism 0.2.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/dist/cache-telemetry.js +4 -1
- package/dist/content.d.ts +2 -2
- package/dist/content.js +30 -50
- package/dist/contracts-core.d.ts +32 -2
- package/dist/contracts-core.js +20 -0
- package/dist/conversations.d.ts +2 -0
- package/dist/conversations.js +1 -0
- package/dist/event-multiplexer.d.ts +13 -0
- package/dist/event-multiplexer.js +39 -16
- package/dist/index.d.ts +5 -3
- package/dist/index.js +5 -3
- package/dist/oauth-device-code.d.ts +58 -0
- package/dist/oauth-device-code.js +119 -0
- package/dist/pinned-fetch.d.ts +25 -0
- package/dist/pinned-fetch.js +246 -0
- package/dist/providers/openai-compatible.d.ts +2 -2
- package/dist/providers/openai-compatible.js +2 -2
- package/dist/providers/transport.d.ts +14 -1
- package/dist/providers/transport.js +43 -0
- package/dist/testing/persistence-schema.d.ts +1 -1
- package/dist/testing/persistence-schema.js +32 -28
- package/dist/testing/state-concurrency-conformance.d.ts +145 -0
- package/dist/testing/state-concurrency-conformance.js +348 -0
- package/docs/agent-events.md +1 -1
- package/docs/conversations.md +5 -2
- package/docs/credential-storage.md +1 -0
- package/docs/credentials-and-redaction.md +1 -0
- package/docs/database-persistence.md +4 -0
- package/docs/enterprise-postgres-state.md +4 -3
- package/docs/index.md +17 -17
- package/docs/mcp-tools.md +1 -1
- package/docs/migration.md +83 -0
- package/docs/model-routing.md +13 -10
- package/docs/multimodal-content.md +1 -0
- package/docs/policy-and-audit.md +1 -0
- package/docs/provider-caching.md +4 -1
- package/docs/provider-primitives.md +17 -0
- package/docs/providers/azure.md +1 -1
- package/docs/providers/bedrock.md +1 -0
- package/docs/providers/openai-compatible.md +4 -3
- package/docs/providers/vertex.md +1 -1
- package/docs/public-contracts.md +3 -3
- package/docs/release-and-install.md +39 -0
- package/docs/workflows.md +2 -1
- package/package.json +8 -4
|
@@ -0,0 +1,348 @@
|
|
|
1
|
+
// ponytail: runner-free multi-process state-concurrency probe for memory and
|
|
2
|
+
// durable stores (plan 022 Task 4). Deterministic barriers only: every probe
|
|
3
|
+
// awaits the conflicting op and then asserts state — never setTimeout/sleeps.
|
|
4
|
+
import assert from "node:assert/strict";
|
|
5
|
+
import { isSessionMetadataConflict } from "../contracts-core.js";
|
|
6
|
+
/**
|
|
7
|
+
* Run the state-concurrency probes for every provided store family. Returns
|
|
8
|
+
* the executed probe names so gates can assert coverage. Deterministic:
|
|
9
|
+
* concurrent ops are awaited through `Promise.allSettled` and state is
|
|
10
|
+
* asserted afterward; no timing-only sleeps anywhere in this file.
|
|
11
|
+
*/
|
|
12
|
+
export async function assertStateConcurrencyConforms(factories) {
|
|
13
|
+
const executed = [];
|
|
14
|
+
if (factories.checkpoints) {
|
|
15
|
+
const checkpoints = await factories.checkpoints();
|
|
16
|
+
await checkpointCasProbe(checkpoints);
|
|
17
|
+
executed.push("checkpoint-cas");
|
|
18
|
+
await approvalDeterminismProbe(checkpoints);
|
|
19
|
+
executed.push("approval-determinism");
|
|
20
|
+
}
|
|
21
|
+
if (factories.events) {
|
|
22
|
+
await cursorResumeProbe(factories.events);
|
|
23
|
+
executed.push("cursor-resume");
|
|
24
|
+
}
|
|
25
|
+
if (factories.sessions) {
|
|
26
|
+
await conversationMetadataCasProbe(await factories.sessions());
|
|
27
|
+
executed.push("conversation-metadata-cas");
|
|
28
|
+
}
|
|
29
|
+
if (factories.idempotency) {
|
|
30
|
+
await idempotencyRetryProbe(await factories.idempotency());
|
|
31
|
+
executed.push("idempotency-retry");
|
|
32
|
+
}
|
|
33
|
+
if (factories.routerState) {
|
|
34
|
+
await reservationOversubscriptionProbe(await factories.routerState.create());
|
|
35
|
+
executed.push("router-reservation");
|
|
36
|
+
if (factories.routerState.nowInjected) {
|
|
37
|
+
await unknownOutcomeProbe(await factories.routerState.create());
|
|
38
|
+
executed.push("unknown-outcome");
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
return executed;
|
|
42
|
+
}
|
|
43
|
+
const ownership = { tenantId: "tenant-a", accountId: "account-a", userId: "user-a" };
|
|
44
|
+
function identity(tenantId) {
|
|
45
|
+
return {
|
|
46
|
+
tenantId,
|
|
47
|
+
userId: "user-a",
|
|
48
|
+
principal: { kind: "agent", id: "agent-a" },
|
|
49
|
+
scopes: ["work:mutate"],
|
|
50
|
+
issuedAt: "2026-08-03T00:00:00.000Z",
|
|
51
|
+
verified: true,
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* Checkpoint CAS: exact-version and fencing-token guards. A stale
|
|
56
|
+
* `expectedVersion` is rejected; a lower `fencingToken` cannot replace a
|
|
57
|
+
* fenced record; a higher fence wins.
|
|
58
|
+
*/
|
|
59
|
+
async function checkpointCasProbe(checkpoints) {
|
|
60
|
+
const key = { namespace: "state-concurrency", key: "checkpoint-cas", ...ownership };
|
|
61
|
+
await checkpoints.saveCheckpoint({ ...key, version: 1, value: { step: 1 } });
|
|
62
|
+
await checkpoints.saveCheckpoint({ ...key, version: 2, expectedVersion: 1, value: { step: 2 } });
|
|
63
|
+
await assertRejectsCode(() => checkpoints.saveCheckpoint({ ...key, version: 3, expectedVersion: 1, value: { step: 3 } }), "ERR_PRISM_CHECKPOINT_CONFLICT", "stale expectedVersion must be rejected");
|
|
64
|
+
await checkpoints.saveCheckpoint({ ...key, version: 3, expectedVersion: 2, fencingToken: 5, value: { step: 3 } });
|
|
65
|
+
await assertRejectsCode(() => checkpoints.saveCheckpoint({ ...key, version: 4, expectedVersion: 3, fencingToken: 4, value: { step: 4 } }), "ERR_PRISM_CHECKPOINT_CONFLICT", "lower fence must be rejected");
|
|
66
|
+
const fenced = await checkpoints.saveCheckpoint({ ...key, version: 4, expectedVersion: 3, fencingToken: 6, value: { step: 4 } });
|
|
67
|
+
assert.equal(fenced.fencingToken, 6, "higher fence must win");
|
|
68
|
+
const loaded = await checkpoints.loadCheckpoint(key);
|
|
69
|
+
assert.equal(loaded?.version, 4, "winning version must persist");
|
|
70
|
+
assert.equal(loaded?.fencingToken, 6, "winning fence must persist");
|
|
71
|
+
await assertRejects(() => checkpoints.loadCheckpoint({ ...key, tenantId: "tenant-b" }), /ownership|tenant/i, "foreign checkpoint access must fail closed");
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Approval determinism: concurrent approve/deny of the same pending decision
|
|
75
|
+
* resolves to exactly one terminal state; a stale decision discriminant
|
|
76
|
+
* (the `expectedVersion` a resumer read) is rejected — the 0.2.0 resume fix
|
|
77
|
+
* contract: resume requires `record.version === expectedVersion`.
|
|
78
|
+
*/
|
|
79
|
+
async function approvalDeterminismProbe(checkpoints) {
|
|
80
|
+
const key = { namespace: "state-concurrency", key: "approval", ...ownership };
|
|
81
|
+
const suspended = {
|
|
82
|
+
status: "suspended",
|
|
83
|
+
interruption: {
|
|
84
|
+
kind: "tool_approval",
|
|
85
|
+
reason: "review",
|
|
86
|
+
pendingDecisions: [{ approvalId: "a1", kind: "tool_approval", scope: { toolName: "fs.write", argumentsHash: "h1" } }],
|
|
87
|
+
},
|
|
88
|
+
};
|
|
89
|
+
await checkpoints.saveCheckpoint({ ...key, version: 1, value: suspended });
|
|
90
|
+
const approve = { status: "running", decision: { approvalId: "a1", outcome: "allow_once" } };
|
|
91
|
+
const deny = { status: "denied", decision: { approvalId: "a1", outcome: "reject_once" } };
|
|
92
|
+
const results = await Promise.allSettled([
|
|
93
|
+
checkpoints.saveCheckpoint({ ...key, version: 2, expectedVersion: 1, fencingToken: 1, value: approve }),
|
|
94
|
+
checkpoints.saveCheckpoint({ ...key, version: 2, expectedVersion: 1, fencingToken: 1, value: deny }),
|
|
95
|
+
]);
|
|
96
|
+
const fulfilled = results.filter((result) => result.status === "fulfilled");
|
|
97
|
+
const rejected = results.filter((result) => result.status === "rejected");
|
|
98
|
+
assert.equal(fulfilled.length, 1, "concurrent approve/deny must admit exactly one terminal write");
|
|
99
|
+
assert.equal(rejected.length, 1, "concurrent approve/deny must reject exactly one writer");
|
|
100
|
+
assert.equal(rejected[0].reason?.code, "ERR_PRISM_CHECKPOINT_CONFLICT", "loser must fail with checkpoint conflict");
|
|
101
|
+
const final = await checkpoints.loadCheckpoint(key);
|
|
102
|
+
assert.equal(final?.version, 2, "terminal state must be written exactly once");
|
|
103
|
+
assert.ok(deepEqual(final?.value, approve) || deepEqual(final?.value, deny), "final state must be exactly one winner's payload, never a mix");
|
|
104
|
+
await assertRejectsCode(() => checkpoints.saveCheckpoint({ ...key, version: 3, expectedVersion: 1, value: { status: "running" } }), "ERR_PRISM_CHECKPOINT_CONFLICT", "a stale decision discriminant (version read before the race) must be rejected");
|
|
105
|
+
}
|
|
106
|
+
/**
|
|
107
|
+
* Replay-cursor resume: a cursor taken from the last consumed position resumes
|
|
108
|
+
* at the next event, never the stream head. Durable factories additionally
|
|
109
|
+
* close and re-open the store (restart) and resume from the same cursor.
|
|
110
|
+
*/
|
|
111
|
+
async function cursorResumeProbe(factory) {
|
|
112
|
+
const input = { ownership, sessionId: "concurrency-session", runId: "concurrency-run" };
|
|
113
|
+
const source = await factory.create();
|
|
114
|
+
await source.append(event("event-1", "agent_started", input));
|
|
115
|
+
await source.append(event("event-2", "turn_started", input, "2026-01-01T00:00:01.000Z"));
|
|
116
|
+
await source.append(event("event-3", "agent_finished", input, "2026-01-01T00:00:02.000Z"));
|
|
117
|
+
const first = await source.page({ ...input, limit: 1 });
|
|
118
|
+
assert.equal(first.items[0]?.record.id, "event-1", "first page must start at the head");
|
|
119
|
+
const second = await source.page({ ...input, after: first.items[0].cursor, limit: 1 });
|
|
120
|
+
assert.equal(second.items[0]?.record.id, "event-2", "second page must continue after the first cursor");
|
|
121
|
+
const resumed = await source.page({ ...input, after: second.items[0].cursor, limit: 10 });
|
|
122
|
+
assert.deepEqual(resumed.items.map((envelope) => envelope.record.id), ["event-3"], "resume from the last-acked cursor must not replay the head");
|
|
123
|
+
assert.equal(resumed.terminal, true, "resumed page must reach the terminal event");
|
|
124
|
+
const foreign = await source.page({ ...input, ownership: { ...ownership, tenantId: "tenant-b" }, limit: 10 });
|
|
125
|
+
assert.equal(foreign.items.length, 0, "ownership must never cross tenants");
|
|
126
|
+
if (factory.reopenable) {
|
|
127
|
+
await source.close?.();
|
|
128
|
+
const restarted = await factory.create();
|
|
129
|
+
const afterRestart = await restarted.page({ ...input, after: second.items[0].cursor, limit: 10 });
|
|
130
|
+
assert.deepEqual(afterRestart.items.map((envelope) => envelope.record.id), ["event-3"], "reopened store must resume from the last-acked cursor, not the head");
|
|
131
|
+
assert.equal(afterRestart.terminal, true, "reopened page must reach the terminal event");
|
|
132
|
+
await restarted.close?.();
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
/** Idempotency: a retry with the same key returns the recorded outcome, never a duplicate effect. */
|
|
136
|
+
async function idempotencyRetryProbe(idempotency) {
|
|
137
|
+
const mutation = { identity: identity("tenant-a"), key: "idem-1", op: "example.mutate" };
|
|
138
|
+
const parallel = await Promise.all(Array.from({ length: 8 }, () => idempotency.begin(mutation)));
|
|
139
|
+
const acquired = parallel.filter((result) => result.outcome === "acquired");
|
|
140
|
+
const existing = parallel.filter((result) => result.outcome === "existing");
|
|
141
|
+
assert.equal(acquired.length, 1, "parallel begins must admit exactly one claim");
|
|
142
|
+
assert.equal(existing.length, 7, "parallel begins must report the rest as existing");
|
|
143
|
+
const claim = acquired[0].record;
|
|
144
|
+
assert.ok(claim.claimToken, "acquired claim must carry a token");
|
|
145
|
+
const completed = await idempotency.complete({
|
|
146
|
+
...mutation,
|
|
147
|
+
claimToken: claim.claimToken,
|
|
148
|
+
expectedVersion: claim.version,
|
|
149
|
+
result: { draftId: "draft-1" },
|
|
150
|
+
});
|
|
151
|
+
assert.equal(completed.status, "completed", "completed mutation must be terminal");
|
|
152
|
+
const retry = await idempotency.begin(mutation);
|
|
153
|
+
assert.equal(retry.outcome, "existing", "retry must never re-acquire");
|
|
154
|
+
assert.equal(retry.record.result?.draftId, "draft-1", "retry must return the recorded outcome, not a duplicate effect");
|
|
155
|
+
assert.equal(await idempotency.get({ ...mutation, identity: identity("tenant-b") }), undefined, "ownership must never cross tenants");
|
|
156
|
+
await assertRejectsCode(() => idempotency.complete({
|
|
157
|
+
...mutation,
|
|
158
|
+
claimToken: claim.claimToken,
|
|
159
|
+
expectedVersion: claim.version,
|
|
160
|
+
result: { draftId: "draft-2" },
|
|
161
|
+
}), "ERR_PRISM_WORK_IDEMPOTENCY_CONFLICT", "a stale-version second complete must be rejected");
|
|
162
|
+
}
|
|
163
|
+
/** Router reservation: parallel admissions cannot oversubscribe; commit/release reconcile actuals. */
|
|
164
|
+
async function reservationOversubscriptionProbe(router) {
|
|
165
|
+
const key = {
|
|
166
|
+
tenantId: "tenant-a",
|
|
167
|
+
accountId: "account-a",
|
|
168
|
+
userId: "user-a",
|
|
169
|
+
principalId: "principal-a",
|
|
170
|
+
provider: "benchmark",
|
|
171
|
+
model: "reserved",
|
|
172
|
+
};
|
|
173
|
+
const parallel = await Promise.all(Array.from({ length: 4 }, () => router.reserveBudget({ key, tokens: 26, maxTokens: 100, windowMs: 60_000, reservationTtlMs: 60_000, now: 0 })));
|
|
174
|
+
const admitted = parallel.filter((result) => result.admitted);
|
|
175
|
+
const denied = parallel.filter((result) => !result.admitted);
|
|
176
|
+
assert.equal(admitted.length, 3, "parallel reservations must admit exactly N-1");
|
|
177
|
+
assert.equal(denied.length, 1, "parallel reservations must deny exactly one");
|
|
178
|
+
assert.ok(denied[0].retryAfterMs !== undefined, "denial must carry a retry-after hint");
|
|
179
|
+
await assertRejectsCode(() => router.commitBudget({
|
|
180
|
+
key,
|
|
181
|
+
reservationId: admitted[2].reservationId,
|
|
182
|
+
fencingToken: "stale",
|
|
183
|
+
tokens: 1,
|
|
184
|
+
windowMs: 60_000,
|
|
185
|
+
now: 1_000,
|
|
186
|
+
}), "ERR_PRISM_MODEL_ROUTER_STATE", "a stale/foreign fencing token must be rejected");
|
|
187
|
+
const committed = await router.commitBudget({
|
|
188
|
+
key,
|
|
189
|
+
reservationId: admitted[0].reservationId,
|
|
190
|
+
fencingToken: admitted[0].fencingToken,
|
|
191
|
+
tokens: 10,
|
|
192
|
+
windowMs: 60_000,
|
|
193
|
+
now: 1_000,
|
|
194
|
+
});
|
|
195
|
+
assert.equal(committed.unknownUsage, false, "live commit must not report unknown usage");
|
|
196
|
+
await router.releaseBudget({
|
|
197
|
+
key,
|
|
198
|
+
reservationId: admitted[1].reservationId,
|
|
199
|
+
fencingToken: admitted[1].fencingToken,
|
|
200
|
+
windowMs: 60_000,
|
|
201
|
+
now: 1_000,
|
|
202
|
+
});
|
|
203
|
+
assert.deepEqual(await router.readBudget({ key, windowMs: 60_000, now: 1_000 }), {
|
|
204
|
+
tokens: 10,
|
|
205
|
+
costUsd: 0,
|
|
206
|
+
});
|
|
207
|
+
}
|
|
208
|
+
/**
|
|
209
|
+
* Unknown-outcome recovery: an abandoned reservation (crash before commit)
|
|
210
|
+
* reconciles to the reserved amount with `unknownUsage: true` — never a
|
|
211
|
+
* silent drop. Deterministic: expiry is driven through the injected `now`
|
|
212
|
+
* (memory stores); the durable leg runs in the Task 1 enterprise-conformance
|
|
213
|
+
* integration probe because durable expiry uses the database clock. The
|
|
214
|
+
* redacted `unknown_usage` diagnostic emission is asserted at router level in
|
|
215
|
+
* the model-router suite (Task 1) — the store seam reports the flag only.
|
|
216
|
+
*/
|
|
217
|
+
async function unknownOutcomeProbe(router) {
|
|
218
|
+
const key = {
|
|
219
|
+
tenantId: "tenant-a",
|
|
220
|
+
accountId: "account-a",
|
|
221
|
+
userId: "user-a",
|
|
222
|
+
principalId: "principal-a",
|
|
223
|
+
provider: "benchmark",
|
|
224
|
+
model: "ttl",
|
|
225
|
+
};
|
|
226
|
+
const reserved = await router.reserveBudget({ key, tokens: 7, maxTokens: 100, windowMs: 60_000, reservationTtlMs: 60_000, now: 0 });
|
|
227
|
+
assert.ok(reserved.admitted, "reservation must admit within capacity");
|
|
228
|
+
const late = await router.commitBudget({
|
|
229
|
+
key,
|
|
230
|
+
reservationId: reserved.reservationId,
|
|
231
|
+
fencingToken: reserved.fencingToken,
|
|
232
|
+
tokens: 1,
|
|
233
|
+
windowMs: 60_000,
|
|
234
|
+
now: 61_001,
|
|
235
|
+
});
|
|
236
|
+
assert.equal(late.unknownUsage, true, "a late commit after expiry must reconcile as unknown usage");
|
|
237
|
+
assert.deepEqual(await router.readBudget({ key, windowMs: 60_000, now: 61_001 }), {
|
|
238
|
+
tokens: 7,
|
|
239
|
+
costUsd: 0,
|
|
240
|
+
});
|
|
241
|
+
}
|
|
242
|
+
/**
|
|
243
|
+
* Conversation metadata CAS: concurrent create/branch/archive-style writes
|
|
244
|
+
* admit exactly one winner per version and reject the rest with
|
|
245
|
+
* `metadata_conflict`; a deleted row is never resurrected; ownership guards
|
|
246
|
+
* hold inside the same guarded write.
|
|
247
|
+
*/
|
|
248
|
+
async function conversationMetadataCasProbe(sessions) {
|
|
249
|
+
if (!sessions.appendSession)
|
|
250
|
+
throw new Error("state-concurrency sessions probe requires appendSession");
|
|
251
|
+
const base = {
|
|
252
|
+
id: "conversation-1",
|
|
253
|
+
tenantId: "tenant-a",
|
|
254
|
+
accountId: "account-a",
|
|
255
|
+
userId: "user-a",
|
|
256
|
+
createdAt: "2026-08-03T00:00:00.000Z",
|
|
257
|
+
updatedAt: "2026-08-03T00:00:00.000Z",
|
|
258
|
+
metadata: { prismConversation: { state: "active", title: "root" } },
|
|
259
|
+
};
|
|
260
|
+
const created = await sessions.appendSession({ ...base, expectedVersion: 0 });
|
|
261
|
+
assert.equal(created?.version, 1, "create-only write must land at version 1");
|
|
262
|
+
const creates = await Promise.allSettled(Array.from({ length: 8 }, (_, index) => sessions.appendSession({
|
|
263
|
+
...base,
|
|
264
|
+
expectedVersion: 0,
|
|
265
|
+
metadata: { prismConversation: { state: "active", title: `racer-${index}` } },
|
|
266
|
+
})));
|
|
267
|
+
const createWinners = creates.filter((result) => result.status === "fulfilled");
|
|
268
|
+
const createLosers = creates.filter((result) => result.status === "rejected");
|
|
269
|
+
assert.equal(createWinners.length, 0, "duplicate create-only writes must never overwrite the winner");
|
|
270
|
+
assert.equal(createLosers.length, 8, "every duplicate create must conflict");
|
|
271
|
+
for (const result of createLosers) {
|
|
272
|
+
assert.ok(isSessionMetadataConflict(result.reason), "duplicate create must be metadata_conflict");
|
|
273
|
+
}
|
|
274
|
+
const racerMetadata = (index) => ({
|
|
275
|
+
prismConversation: {
|
|
276
|
+
state: "active",
|
|
277
|
+
title: `branch-${index}`,
|
|
278
|
+
refs: [{ leafId: `leaf-${index}`, createdAt: "2026-08-03T00:00:01.000Z" }],
|
|
279
|
+
},
|
|
280
|
+
});
|
|
281
|
+
const updates = await Promise.allSettled(Array.from({ length: 8 }, (_, index) => sessions.appendSession({ ...base, expectedVersion: 1, metadata: racerMetadata(index) })));
|
|
282
|
+
const updateWinners = updates.filter((result) => result.status === "fulfilled");
|
|
283
|
+
const updateLosers = updates.filter((result) => result.status === "rejected");
|
|
284
|
+
assert.equal(updateWinners.length, 1, "concurrent CAS updates must admit exactly one winner");
|
|
285
|
+
assert.equal(updateLosers.length, 7, "concurrent CAS updates must reject the rest");
|
|
286
|
+
const winnerVersion = updateWinners[0].value?.version;
|
|
287
|
+
assert.equal(winnerVersion, 2, "winner must land at version 2");
|
|
288
|
+
for (const result of updateLosers) {
|
|
289
|
+
const reason = result.reason;
|
|
290
|
+
assert.ok(isSessionMetadataConflict(reason), "CAS loser must be metadata_conflict");
|
|
291
|
+
assert.equal(reason.conflict?.currentVersion, 2, "conflict must report the current version");
|
|
292
|
+
}
|
|
293
|
+
const winnerIndex = updates.findIndex((result) => result.status === "fulfilled");
|
|
294
|
+
const page = await sessions.querySessions({ id: base.id, tenantId: "tenant-a", accountId: "account-a", userId: "user-a" });
|
|
295
|
+
assert.equal(page.items.length, 1, "session must resolve to a single record");
|
|
296
|
+
assert.equal(page.items[0]?.version, 2, "stored version must reflect the winner's write");
|
|
297
|
+
assert.deepEqual(page.items[0]?.metadata, racerMetadata(winnerIndex), "stored metadata must be exactly the winner's marker");
|
|
298
|
+
await assertRejectsCode(() => sessions.appendSession({
|
|
299
|
+
...base,
|
|
300
|
+
tenantId: "tenant-b",
|
|
301
|
+
expectedVersion: 1,
|
|
302
|
+
metadata: { prismConversation: { state: "archived" } },
|
|
303
|
+
}), "metadata_conflict", "cross-ownership CAS write must be rejected in the same guarded statement");
|
|
304
|
+
}
|
|
305
|
+
function event(id, type, input, timestamp = "2026-01-01T00:00:00.000Z") {
|
|
306
|
+
const payload = type === "turn_started"
|
|
307
|
+
? { type, sessionId: input.sessionId, runId: input.runId, turn: 1 }
|
|
308
|
+
: { type, sessionId: input.sessionId, runId: input.runId };
|
|
309
|
+
return { id, ...input.ownership, sessionId: input.sessionId, runId: input.runId, type, timestamp, event: payload, redacted: true };
|
|
310
|
+
}
|
|
311
|
+
function deepEqual(actual, expected) {
|
|
312
|
+
if (Object.is(actual, expected))
|
|
313
|
+
return true;
|
|
314
|
+
if (typeof actual !== "object" || typeof expected !== "object" || actual === null || expected === null)
|
|
315
|
+
return false;
|
|
316
|
+
const actualEntries = Object.entries(actual);
|
|
317
|
+
const expectedEntries = Object.entries(expected);
|
|
318
|
+
if (actualEntries.length !== expectedEntries.length)
|
|
319
|
+
return false;
|
|
320
|
+
for (const [key, value] of expectedEntries) {
|
|
321
|
+
if (!deepEqual(actual[key], value))
|
|
322
|
+
return false;
|
|
323
|
+
}
|
|
324
|
+
return true;
|
|
325
|
+
}
|
|
326
|
+
async function assertRejectsCode(action, code, message) {
|
|
327
|
+
try {
|
|
328
|
+
await action();
|
|
329
|
+
}
|
|
330
|
+
catch (error) {
|
|
331
|
+
if (error?.code === code)
|
|
332
|
+
return;
|
|
333
|
+
throw new Error(`${message}; expected code ${code}, received ${String(error?.code)}: ${String(error)}`);
|
|
334
|
+
}
|
|
335
|
+
throw new Error(`${message}; expected a rejection with code ${code}`);
|
|
336
|
+
}
|
|
337
|
+
async function assertRejects(action, pattern, message) {
|
|
338
|
+
try {
|
|
339
|
+
await action();
|
|
340
|
+
}
|
|
341
|
+
catch (error) {
|
|
342
|
+
if (pattern.test(String(error)))
|
|
343
|
+
return;
|
|
344
|
+
throw new Error(`${message}; rejection did not match ${pattern}: ${String(error)}`);
|
|
345
|
+
}
|
|
346
|
+
throw new Error(`${message}; expected a rejection matching ${pattern}`);
|
|
347
|
+
}
|
|
348
|
+
//# sourceMappingURL=state-concurrency-conformance.js.map
|
package/docs/agent-events.md
CHANGED
|
@@ -45,7 +45,7 @@ const nc = await connect({ servers: process.env.NATS_URL });
|
|
|
45
45
|
const source = createNatsAgentEventSource({ connection: await createNatsJetStream(nc), stream: "prism_agent_events" });
|
|
46
46
|
```
|
|
47
47
|
|
|
48
|
-
One subject per run (`prism.agent-events.<tenant>.<session>.<run>`); the JetStream per-subject sequence is the per-run event sequence. `append` is idempotent by `record.id` within the stream's dedupe window; `page`/`subscribe` replay per subject from HMAC-signed cursors; `subscribe` uses a durable pull consumer with explicit acks (at-least-once, 30s redelivery, dedupe by `record.id`); `cleanup` deletes ownership-scoped messages older than `before`. The host provisions the stream (subjects `prism.agent-events.>`, retention limits, dedupe window). Inert on import; network-free tests use an in-memory fake of the narrow `NatsJetStream` seam.
|
|
48
|
+
One subject per run (`prism.agent-events.<tenant>.<session>.<run>`); the JetStream per-subject sequence is the per-run event sequence. `append` is idempotent by `record.id` within the stream's dedupe window; `page`/`subscribe` replay per subject from HMAC-signed cursors; `subscribe` uses a durable pull consumer with explicit acks (at-least-once, 30s redelivery, dedupe by `record.id`) and a **restart-stable durable identity** — the consumer name is `prism_<hmac16>` of `tenantId|sessionId|runId` (no random suffix), so a crashed subscribe leaves a consumer that a restarting subscribe reuses at its last-acked position instead of replaying from the stream head. Clean stops still delete the consumer; pre-0.2.2 orphaned random-suffixed consumers are reclaimed by hosts via `deleteConsumer`/consumer enumeration. `cleanup` deletes ownership-scoped messages older than `before`. The host provisions the stream (subjects `prism.agent-events.>`, retention limits, dedupe window). Inert on import; network-free tests use an in-memory fake of the narrow `NatsJetStream` seam.
|
|
49
49
|
|
|
50
50
|
## Inputs / request
|
|
51
51
|
|
package/docs/conversations.md
CHANGED
|
@@ -113,13 +113,16 @@ Behavior notes:
|
|
|
113
113
|
- Replay cursors are thread-bound: a cursor minted for one thread is rejected on another (`cursor_thread_mismatch`).
|
|
114
114
|
- Ledger rows from runs that had no redactor are never served by replay/export (fail-closed skip).
|
|
115
115
|
- Export truncates at page granularity when the next page would exceed `exportBytes`; a single page larger than `exportBytes` cannot be exported (raise the cap or page via `replay`).
|
|
116
|
-
- Branch refs live in thread metadata
|
|
116
|
+
- Branch refs live in thread metadata; every `branch()`/`archive()` write is an optimistic compare-and-swap against the stored session `version`, so concurrent callers cannot lose a ref or resurrect stale state: the loser of a `branch`+`branch`, `branch`+`archive`, or `create`+`create` race gets `metadata_conflict` (HTTP 409) and re-reads/retries if it chooses.
|
|
117
|
+
- `create` with an explicit `id` writes with `expectedVersion: 0` (create-only): a duplicate create never overwrites the winner's metadata and returns the existing thread.
|
|
118
|
+
- `branch()` enforces `maxActiveBranches` on its read snapshot; the version guard inside the same write makes the cap exact even under concurrency (a concurrent branch cannot slip past the cap), and the marker keeps branch refs append-only.
|
|
119
|
+
- `archive()` on an already-archived thread is a no-op; a stale `branch`/`archive` racing a delete fails `not_found` (the row is gone) and the write path never re-creates a deleted thread — delete wins.
|
|
117
120
|
- Deletion purges the whole session ledger (entries, runs, events, tool calls, usage, branches, search rows) through `lifecycle.applyRetention`; legal holds block deletion and report `held: true`.
|
|
118
121
|
|
|
119
122
|
## Security and performance notes
|
|
120
123
|
|
|
121
124
|
- Every operation starts from host-verified ownership (and optional `AgentIdentity`, which must project onto ownership without widening); wrong-user access returns not-found, never leaked existence.
|
|
122
|
-
- `appendSession` upserts set ownership columns only on create; metadata/`updatedAt` on update — ownership is immutable after create.
|
|
125
|
+
- `appendSession` upserts set ownership columns only on create; metadata/`updatedAt` on update — ownership is immutable after create, and the CAS write additionally rejects a mismatched ownership in the same guarded statement.
|
|
123
126
|
- Replay/export serve `redacted: true` ledger rows only and pass through the service redactor; no local paths, raw tool payloads, or secrets are emitted.
|
|
124
127
|
- All loops are bounded by the frozen caps above; review/agent turns consume shared `RunLimits` via the host's `runOptions`.
|
|
125
128
|
- No new permission surface: conversations reuse session/event/identity/redaction/lifecycle seams (roadmap gate 8).
|
|
@@ -248,6 +248,7 @@ const providers = createOpenAIProviderPackage({ apiKey });
|
|
|
248
248
|
- Never log passphrases, derived keys, or decrypted credential payloads.
|
|
249
249
|
- Optional host KMS: `encryptWithHostKms` / `decryptWithHostKms` wrap a random AES-256-GCM DEK via host `HostKms.wrapKey`/`unwrapKey` (timeout ≤ 60 s). Envelope sizes reuse vault/file caps. Keys are never logged. `createMemoryHostKms` is for tests only.
|
|
250
250
|
- Storage is not OAuth eligibility. A durable store may persist credentials for a provider only after the host selects a provider-authorized flow; it must not be used to piggyback on a vendor CLI or consumer subscription.
|
|
251
|
+
- OIDC JWKS fetches (0.2.1) are DNS-pinned through the core `pinnedFetch` primitive: one resolve per request, every resolved address SSRF-checked before the connect (rebinding defense), redirects rejected outright, and the JWKS body read under a hard byte ceiling; oversized documents fail closed as a parse error, never as a rotatable transport failure.
|
|
251
252
|
- Live keychain tests are opt-in (`PRISM_TEST_KEYCHAIN=1`); default `npm test` stays offline.
|
|
252
253
|
|
|
253
254
|
## MCP authentication boundary
|
|
@@ -124,6 +124,7 @@ A future provider-local OAuth adapter needs published permission for third-party
|
|
|
124
124
|
- `resolveCredentialValue()` and `createExplicitCredentialResolver()` do not cache values. Add host-side caching only if a real credential source needs it.
|
|
125
125
|
- `refreshOAuthCredential()` only calls the supplied OAuth provider and optional store; it has no built-in persistence or retry loop.
|
|
126
126
|
- OpenAI Codex device-code OAuth polls inside `createOpenAICodexOAuthProvider().login()` with bounded delays and abort support via `OAuthLoginCallbacks.signal`. Token-endpoint failures redact authorization codes, PKCE verifiers, device/user codes, and access/refresh tokens when those values are known.
|
|
127
|
+
- The shared bounded device/token flow lives in core `pollDeviceCodeToken` (0.2.1) and is used by the OpenAI Codex provider and the credentials-node OAuth 2.0 provider (Microsoft 365 / Google Workspace). It owns the RFC 8628 device-code request and poll loop (`authorization_pending` continue, `slow_down` +5s backoff, expiry deadline, abort), reads every response body under the shared byte ceiling, parses success bodies with a fail-closed shape gate (an `access_token` string is required), and redacts device/user codes, authorization codes, PKCE verifiers, and tokens from every thrown error. Adapter-specific fields (message prefix, extra token params, account binding) are plain options, never subclasses.
|
|
127
128
|
|
|
128
129
|
## Related APIs
|
|
129
130
|
|
|
@@ -76,6 +76,8 @@ Important shapes:
|
|
|
76
76
|
|
|
77
77
|
Optional session-record write seam (0.0.14): `appendSession?(record: SessionRecord)` upserts a session row — ownership columns are set on create only, `metadata`/`updatedAt` on update — so hosts (e.g. the [conversation service](conversations.md)) can durably mark and title sessions without entry writes. `SessionQuery` gained two bounded filters for the same seam: `id` (exact session lookup) and `metadataKey` (sessions whose `metadata` object contains a top-level key, validated by `assertSessionMetadataKey`). SQLite implements it with `json_extract`, PostgreSQL with a `jsonb` existence check; both keep ownership filtering intact.
|
|
78
78
|
|
|
79
|
+
Metadata CAS (0.2.2): the write seam accepts an additive `expectedVersion` guard and returns the new `{ version }`. `expectedVersion: 0` means create-only (a duplicate insert is rejected with `SessionMetadataConflictError`, code `metadata_conflict`), a positive number means the stored `version` must match exactly (update-only — a deleted row is never re-created), and omitting it keeps legacy last-write-wins. Both adapters implement the guard inside the single upsert statement (PostgreSQL `INSERT … SELECT … WHERE … ON CONFLICT … DO UPDATE … WHERE version = $n`; SQLite the same shape with null-safe ownership `IS` comparisons); conflicts carry the id and versions only, never metadata content. The `version` column ships as migration `008_session_version` (`INTEGER NOT NULL DEFAULT 0` plus a one-time backfill to 1) — additive and forward-only, so pre-0.2.2 databases upgrade in place and non-CAS callers behave byte-identically.
|
|
80
|
+
|
|
79
81
|
Artifact co-work review (0.0.14) reuses the generic `CheckpointStore` rather than adding a dedicated table: the [artifact service](work-artifacts-and-review.md) stores each artifact as a versioned checkpoint value (namespace `prism.artifact`, key `threadId:artifactId`, category `artifact`). The checkpoint `version` is the compare-and-swap counter that resolves concurrent reviewers; revision numbers, approvals, and `lastValidatedVersion` live inside the JSON value. SQLite/Postgres already persist checkpoints durably, so there is no separate artifact schema or migration, and records carry metadata/hashes/refs only — never file bodies.
|
|
80
82
|
|
|
81
83
|
## Outputs / response / events
|
|
@@ -372,6 +374,7 @@ Recommended migration practices:
|
|
|
372
374
|
- Use a sequential or timestamped migration naming convention.
|
|
373
375
|
- Store applied migrations in `prism_migrations` with `name`, `version`, `applied_at`, `applied_by`, and `checksum`.
|
|
374
376
|
- Make entry-kind and schema-version changes additive when possible; new kinds and versions fail closed in the JSONL parser, so DB schemas should accept the same additive expansion.
|
|
377
|
+
- Additive columns ship as forward-only migrations (`ALTER TABLE … ADD COLUMN IF NOT EXISTS` + a one-time backfill, e.g. `008_session_version` for the `prism_sessions.version` CAS column).
|
|
375
378
|
- Index new query columns before deploying code that uses them.
|
|
376
379
|
- Back-fill redacted flags and ownership columns before enforcing tenant isolation.
|
|
377
380
|
|
|
@@ -448,6 +451,7 @@ const dbStore: ProductionPersistenceStore = {
|
|
|
448
451
|
- `ProductionPersistenceStore.feedback?: RunFeedbackStore` exposes immutable append, bounded owned query, and owned deletion. First-party adapters store migration-003 rows in `prism_run_feedback`, FK-link `run_id`, and index owner/run/trace creation cursors.
|
|
449
452
|
- Hosts choose the database, schema, transaction, and indexing strategy. The contracts specify query and checkpoint capability shapes.
|
|
450
453
|
- `SessionStore` (`append`/`list`/`get`/optional `readBranchPath`) can be implemented on top of `ProductionPersistenceStore` or kept separate.
|
|
454
|
+
- **State-concurrency conformance (0.2.2):** durable adapters must pass `assertStateConcurrencyConforms` from `@arnilo/prism/testing/state-concurrency-conformance` against both the memory stores and their own implementation (approval determinism, checkpoint CAS, replay-cursor resume, idempotency retry, router reservation, conversation metadata CAS, unknown-outcome recovery). The harness uses deterministic barriers only — no timing-only sleeps — and runs the memory leg in the default `npm test` and the durable legs in `test:postgres`/`test:nats`; `scripts/phase22-conformance.test.mjs` asserts every store leg executed (missing protected environment records a named BLOCKED GATE, never a green skip).
|
|
451
455
|
- Cursor values and idempotency keys are host-defined and opaque to Prism.
|
|
452
456
|
- First-party SQLite/PostgreSQL adapters expose `persistence.checkpoints` and `persistence.leases`, backed by package-owned `prism_checkpoints` / `prism_leases` tables. `@arnilo/prism-workflows` consumes them for durable resume, human suspension, multi-process coordination, Phase 11 schedule records/fire leases, shared state, and replay lineage; workflow code owns no SQL table. `suspended`/`denied`, schedules, state history, and replay lineage remain namespaces/categories plus bounded checkpoint JSON values, so Phases 8 and 11 need no database migration.
|
|
453
457
|
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
| Tool effects | `toolEffects` | Durable `ToolEffectStore` claim/CAS for recoverable tool side effects (migration 002). |
|
|
14
14
|
| Tool effects | `toolEffects` | Durable `ToolEffectStore` claim/CAS for recoverable tool side effects (migration 002). |
|
|
15
15
|
|
|
16
|
-
`createPostgresEnterpriseState()` opens a host-supplied or adapter-owned `pg` pool, verifies/applies checksum-protected enterprise migrations (`001_enterprise_state`, `002_tool_effects`), and returns those stores plus explicit cleanup and close operations. Importing it performs no I/O. It is separate from session/run persistence in [`@arnilo/prism-session-store-postgres`](postgres-persistence.md).
|
|
16
|
+
`createPostgresEnterpriseState()` opens a host-supplied or adapter-owned `pg` pool, verifies/applies checksum-protected enterprise migrations (`001_enterprise_state`, `002_tool_effects`, `003_router_reservations`), and returns those stores plus explicit cleanup and close operations. Importing it performs no I/O. It is separate from session/run persistence in [`@arnilo/prism-session-store-postgres`](postgres-persistence.md).
|
|
17
17
|
|
|
18
18
|
## When to use it
|
|
19
19
|
|
|
@@ -85,7 +85,7 @@ Model-router state is asynchronous and owner/principal/provider/model scoped. Su
|
|
|
85
85
|
}
|
|
86
86
|
```
|
|
87
87
|
|
|
88
|
-
A migration creates `prism_policy_decisions`, `prism_evaluations`, `prism_work_idempotency`, three `prism_model_router_*` tables, and its separate `prism_enterprise_migrations` history. Startup serializes per-schema setup with an advisory transaction lock and rejects checksum or catalog drift rather than silently repairing it.
|
|
88
|
+
A migration creates `prism_policy_decisions`, `prism_evaluations`, `prism_work_idempotency`, three `prism_model_router_*` tables, and its separate `prism_enterprise_migrations` history. Migration `003_router_reservations` adds the nullable-by-default `reservations` JSONB column to `prism_model_router_budgets` (atomic reservation slots for router admission; 0.2.1 readers ignore it). Startup serializes per-schema setup with an advisory transaction lock and rejects checksum or catalog drift rather than silently repairing it.
|
|
89
89
|
|
|
90
90
|
## Implementation example
|
|
91
91
|
|
|
@@ -152,7 +152,8 @@ export async function recordEnterpriseState(state: PostgresEnterpriseState) {
|
|
|
152
152
|
|
|
153
153
|
## Extension and configuration notes
|
|
154
154
|
|
|
155
|
-
- `createModelRouter({ resolver, stateStore: state.modelRouter })` keeps allow-list, residency, fallback, and diagnostics behavior in `@arnilo/prism-model-router`; this package only supplies durable state.
|
|
155
|
+
- `createModelRouter({ resolver, stateStore: state.modelRouter })` keeps allow-list, residency, fallback, and diagnostics behavior in `@arnilo/prism-model-router`; this package only supplies durable state. Router admission reservations (`reserveBudget`/`commitBudget`/`releaseBudget` on `state.modelRouter`) live in the `reservations` JSONB column of `prism_model_router_budgets`: one atomic UPSERT per admission, fencing-token-guarded commit/release in a SERIALIZABLE transaction, and TTL reconciliation as unknown usage; see [Model routing](model-routing.md).
|
|
156
|
+
- Rate/budget/circuit tables are capped like the memory store: `consumeRate`/`readBudget`/`addUsage`/`reserveBudget` accept `maxRateKeys`/`maxBudgetKeys` (the router passes its resolved limits) and evict the least-recently-used row on new-key insert — never the row just inserted, never a budget row holding an active reservation — else fail closed with `ERR_PRISM_MODEL_ROUTER_STATE`. Cleanup prunes expired reservations within its bounded batch.
|
|
156
157
|
- Policy/evaluation/query public contracts stay in their owning packages. This package exports only `createPostgresEnterpriseState`, its options/result types, and `EnterprisePostgresError`; it has no SQL, DDL, codec, queryable, or migration subpath.
|
|
157
158
|
- The fixed schema has no generic key/value table and no background cleanup scheduler. Schedule `state.cleanup()` from an authorized host job, size its bounded batch for the deployment, and monitor unknown work rows for reconciliation. Run protected integration checks with `PRISM_TEST_POSTGRES_URL="$DATABASE_URL" npm run test:postgres`; the command rejects an absent URL instead of silently skipping database coverage.
|
|
158
159
|
- The OPA adapter (`@arnilo/prism-policy/opa`, 0.0.28) records decisions into the same `state.policy` store unchanged via `evaluateAndAppend` — see [Policy and audit](policy-and-audit.md#opa-external-policy-adapter-arniloprism-policyopa-008).
|