@arnilo/prism 0.0.10 → 0.0.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/agent-loops.d.ts +1 -0
- package/dist/agent-loops.js +37 -2
- package/dist/agents.js +125 -6
- package/dist/context-budget.d.ts +63 -0
- package/dist/context-budget.js +235 -0
- package/dist/contracts.d.ts +105 -0
- package/dist/contracts.js +77 -0
- package/dist/index.d.ts +6 -4
- package/dist/index.js +4 -3
- package/dist/input.d.ts +3 -0
- package/dist/input.js +71 -28
- package/dist/node/session-store-jsonl.js +4 -1
- package/dist/rpc.js +13 -2
- package/dist/session-stores.d.ts +7 -2
- package/dist/session-stores.js +174 -4
- package/dist/structured-output.d.ts +5 -1
- package/dist/structured-output.js +18 -0
- package/dist/testing/persistence-schema.d.ts +1 -1
- package/dist/testing/persistence-schema.js +8 -2
- package/dist/testing/session-store-conformance.d.ts +6 -0
- package/dist/testing/session-store-conformance.js +36 -1
- package/docs/agent-loops.md +8 -1
- package/docs/agent-session-runtime.md +5 -1
- package/docs/cli-rpc.md +2 -1
- package/docs/coding-agent-tools.md +68 -1
- package/docs/evaluations.md +1 -1
- package/docs/index.md +12 -11
- package/docs/input-and-prompt-assembly.md +4 -1
- package/docs/migration.md +16 -0
- package/docs/node-jsonl-session-store.md +1 -1
- package/docs/performance.md +17 -0
- package/docs/postgres-persistence.md +3 -3
- package/docs/provider-packages.md +1 -1
- package/docs/providers/anthropic.md +92 -0
- package/docs/providers/google.md +87 -0
- package/docs/public-contracts.md +4 -0
- package/docs/release-and-install.md +159 -70
- package/docs/review-coverage-2026-07-22-phase-6.md +209 -0
- package/docs/session-store-conformance.md +2 -0
- package/docs/session-stores.md +40 -1
- package/docs/sqlite-persistence.md +3 -3
- package/docs/structured-output.md +7 -1
- package/docs/workflows.md +2 -0
- package/package.json +2 -2
package/dist/session-stores.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { SESSION_APPEND_CONFLICT_CODE, SessionAppendConflictError } from "./contracts.js";
|
|
1
|
+
import { DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, SESSION_APPEND_CONFLICT_CODE, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SessionAppendConflictError, SessionSearchUnsupportedError, resolveSessionSearchQuery, } from "./contracts.js";
|
|
2
2
|
import { createId } from "./ids.js";
|
|
3
3
|
export function createSessionEntry(options) {
|
|
4
4
|
const { createId, now, ...entry } = options;
|
|
@@ -93,16 +93,17 @@ function rebuildSessionContextCore(entries, options = {}) {
|
|
|
93
93
|
}
|
|
94
94
|
return { leafId: branch.at(-1)?.id, entries: branch, messages, summaries };
|
|
95
95
|
}
|
|
96
|
-
export function createMemorySessionStore(initialEntries = []) {
|
|
96
|
+
export function createMemorySessionStore(initialEntries = [], options = {}) {
|
|
97
97
|
const byId = new Map();
|
|
98
98
|
const bySession = new Map();
|
|
99
99
|
const leafBySession = new Map();
|
|
100
100
|
const idempotencySeen = new Set();
|
|
101
|
+
const mode = options.sessionSearchMode ?? "linear";
|
|
101
102
|
for (const entry of initialEntries)
|
|
102
103
|
add(entry);
|
|
103
104
|
return {
|
|
104
|
-
async append(entry,
|
|
105
|
-
add(entry,
|
|
105
|
+
async append(entry, appendOptions) {
|
|
106
|
+
add(entry, appendOptions);
|
|
106
107
|
},
|
|
107
108
|
async list(sessionId) {
|
|
108
109
|
return (bySession.get(sessionId) ?? []).map(cloneEntry);
|
|
@@ -111,6 +112,11 @@ export function createMemorySessionStore(initialEntries = []) {
|
|
|
111
112
|
const entry = byId.get(id);
|
|
112
113
|
return entry ? cloneEntry(entry) : undefined;
|
|
113
114
|
},
|
|
115
|
+
async searchSessions(query) {
|
|
116
|
+
if (mode === "unsupported")
|
|
117
|
+
throw new SessionSearchUnsupportedError();
|
|
118
|
+
return searchMemorySessionsLinear(bySession, leafBySession, query);
|
|
119
|
+
},
|
|
114
120
|
};
|
|
115
121
|
function add(entry, options) {
|
|
116
122
|
// ponytail: idempotency dedup keyed on (session, key, expectedParentId) so a
|
|
@@ -146,6 +152,170 @@ export function createMemorySessionStore(initialEntries = []) {
|
|
|
146
152
|
leafBySession.set(entry.sessionId, entry.id);
|
|
147
153
|
}
|
|
148
154
|
}
|
|
155
|
+
function searchMemorySessionsLinear(bySession, leafBySession, query) {
|
|
156
|
+
const q = resolveSessionSearchQuery(query);
|
|
157
|
+
q.signal?.throwIfAborted();
|
|
158
|
+
let sessionsScanned = 0;
|
|
159
|
+
let entriesScanned = 0;
|
|
160
|
+
let bytesScanned = 0;
|
|
161
|
+
const matches = [];
|
|
162
|
+
for (const [sessionId, entries] of bySession) {
|
|
163
|
+
if (sessionsScanned >= DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS)
|
|
164
|
+
break;
|
|
165
|
+
if (entriesScanned >= DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES)
|
|
166
|
+
break;
|
|
167
|
+
if (bytesScanned >= DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES)
|
|
168
|
+
break;
|
|
169
|
+
q.signal?.throwIfAborted();
|
|
170
|
+
sessionsScanned += 1;
|
|
171
|
+
let updatedAt = "";
|
|
172
|
+
let label;
|
|
173
|
+
let summary;
|
|
174
|
+
let workspaceRoot;
|
|
175
|
+
let tenantId;
|
|
176
|
+
let accountId;
|
|
177
|
+
let userId;
|
|
178
|
+
let matchedLabel = false;
|
|
179
|
+
let matchedSummary = false;
|
|
180
|
+
let matchedQuery = false;
|
|
181
|
+
let matchedProvider = false;
|
|
182
|
+
let matchedModel = false;
|
|
183
|
+
let snippetSource;
|
|
184
|
+
for (const entry of entries) {
|
|
185
|
+
if (entriesScanned >= DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES)
|
|
186
|
+
break;
|
|
187
|
+
if (bytesScanned >= DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES)
|
|
188
|
+
break;
|
|
189
|
+
entriesScanned += 1;
|
|
190
|
+
const text = entrySearchText(entry);
|
|
191
|
+
bytesScanned += utf8Bytes(text) + utf8Bytes(entry.label) + utf8Bytes(entry.summary);
|
|
192
|
+
if (entry.timestamp > updatedAt)
|
|
193
|
+
updatedAt = entry.timestamp;
|
|
194
|
+
if (entry.label)
|
|
195
|
+
label = entry.label;
|
|
196
|
+
if (entry.summary)
|
|
197
|
+
summary = entry.summary;
|
|
198
|
+
const meta = entry.metadata;
|
|
199
|
+
if (meta) {
|
|
200
|
+
if (typeof meta[SESSION_SEARCH_WORKSPACE_METADATA_KEY] === "string") {
|
|
201
|
+
workspaceRoot = meta[SESSION_SEARCH_WORKSPACE_METADATA_KEY];
|
|
202
|
+
}
|
|
203
|
+
if (typeof meta.tenantId === "string")
|
|
204
|
+
tenantId = meta.tenantId;
|
|
205
|
+
if (typeof meta.accountId === "string")
|
|
206
|
+
accountId = meta.accountId;
|
|
207
|
+
if (typeof meta.userId === "string")
|
|
208
|
+
userId = meta.userId;
|
|
209
|
+
}
|
|
210
|
+
if (q.label && entry.label?.includes(q.label))
|
|
211
|
+
matchedLabel = true;
|
|
212
|
+
if (q.summary && entry.summary?.includes(q.summary))
|
|
213
|
+
matchedSummary = true;
|
|
214
|
+
if (q.query) {
|
|
215
|
+
const hay = `${entry.label ?? ""}\n${entry.summary ?? ""}\n${text}`;
|
|
216
|
+
if (hay.includes(q.query)) {
|
|
217
|
+
matchedQuery = true;
|
|
218
|
+
snippetSource ??= entry.label ?? entry.summary ?? text;
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
if (q.provider && (entry.model?.provider === q.provider || metaProvider(entry) === q.provider))
|
|
222
|
+
matchedProvider = true;
|
|
223
|
+
if (q.model && entry.model?.model === q.model)
|
|
224
|
+
matchedModel = true;
|
|
225
|
+
}
|
|
226
|
+
if (q.workspaceRoot && workspaceRoot !== q.workspaceRoot)
|
|
227
|
+
continue;
|
|
228
|
+
if (q.tenantId && tenantId !== q.tenantId)
|
|
229
|
+
continue;
|
|
230
|
+
if (q.accountId && accountId !== q.accountId)
|
|
231
|
+
continue;
|
|
232
|
+
if (q.userId && userId !== q.userId)
|
|
233
|
+
continue;
|
|
234
|
+
if (q.label && !matchedLabel)
|
|
235
|
+
continue;
|
|
236
|
+
if (q.summary && !matchedSummary)
|
|
237
|
+
continue;
|
|
238
|
+
if (q.query && !matchedQuery)
|
|
239
|
+
continue;
|
|
240
|
+
if (q.provider && !matchedProvider)
|
|
241
|
+
continue;
|
|
242
|
+
if (q.model && !matchedModel)
|
|
243
|
+
continue;
|
|
244
|
+
if (q.fromUpdatedAt && updatedAt < q.fromUpdatedAt)
|
|
245
|
+
continue;
|
|
246
|
+
if (q.toUpdatedAt && updatedAt > q.toUpdatedAt)
|
|
247
|
+
continue;
|
|
248
|
+
matches.push({
|
|
249
|
+
sessionId,
|
|
250
|
+
leafId: leafBySession.get(sessionId),
|
|
251
|
+
updatedAt: updatedAt || undefined,
|
|
252
|
+
label,
|
|
253
|
+
summary,
|
|
254
|
+
snippet: clipSnippet(snippetSource ?? label ?? summary),
|
|
255
|
+
metadata: workspaceRoot !== undefined ? { [SESSION_SEARCH_WORKSPACE_METADATA_KEY]: workspaceRoot } : undefined,
|
|
256
|
+
});
|
|
257
|
+
}
|
|
258
|
+
matches.sort((a, b) => compareSearchHit(a, b, q.order));
|
|
259
|
+
const afterCursor = q.cursor ? decodeSearchCursor(q.cursor) : undefined;
|
|
260
|
+
const filtered = afterCursor
|
|
261
|
+
? matches.filter((hit) => isAfterSearchCursor(hit, afterCursor, q.order))
|
|
262
|
+
: matches;
|
|
263
|
+
const page = filtered.slice(0, q.limit);
|
|
264
|
+
const last = page.at(-1);
|
|
265
|
+
return {
|
|
266
|
+
items: page,
|
|
267
|
+
nextCursor: filtered.length > q.limit && last?.updatedAt
|
|
268
|
+
? encodeSearchCursor(last.updatedAt, last.sessionId)
|
|
269
|
+
: undefined,
|
|
270
|
+
};
|
|
271
|
+
}
|
|
272
|
+
function entrySearchText(entry) {
|
|
273
|
+
if (!entry.message?.content)
|
|
274
|
+
return "";
|
|
275
|
+
const parts = [];
|
|
276
|
+
for (const block of entry.message.content) {
|
|
277
|
+
if (block.type === "text" && typeof block.text === "string")
|
|
278
|
+
parts.push(block.text);
|
|
279
|
+
}
|
|
280
|
+
return parts.join("\n");
|
|
281
|
+
}
|
|
282
|
+
function metaProvider(entry) {
|
|
283
|
+
const value = entry.metadata?.provider;
|
|
284
|
+
return typeof value === "string" ? value : undefined;
|
|
285
|
+
}
|
|
286
|
+
function utf8Bytes(value) {
|
|
287
|
+
return value ? new TextEncoder().encode(value).byteLength : 0;
|
|
288
|
+
}
|
|
289
|
+
function clipSnippet(value) {
|
|
290
|
+
if (!value)
|
|
291
|
+
return undefined;
|
|
292
|
+
const encoded = new TextEncoder().encode(value);
|
|
293
|
+
if (encoded.byteLength <= DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES)
|
|
294
|
+
return value;
|
|
295
|
+
return new TextDecoder().decode(encoded.slice(0, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES));
|
|
296
|
+
}
|
|
297
|
+
function encodeSearchCursor(updatedAt, sessionId) {
|
|
298
|
+
return `${updatedAt}\t${sessionId}`;
|
|
299
|
+
}
|
|
300
|
+
function decodeSearchCursor(cursor) {
|
|
301
|
+
const tab = cursor.indexOf("\t");
|
|
302
|
+
if (tab <= 0 || tab === cursor.length - 1)
|
|
303
|
+
throw new TypeError("SessionSearchQuery.cursor is invalid");
|
|
304
|
+
return { updatedAt: cursor.slice(0, tab), sessionId: cursor.slice(tab + 1) };
|
|
305
|
+
}
|
|
306
|
+
function compareSearchHit(a, b, order) {
|
|
307
|
+
const aAt = a.updatedAt ?? "";
|
|
308
|
+
const bAt = b.updatedAt ?? "";
|
|
309
|
+
const cmp = aAt < bAt ? -1 : aAt > bAt ? 1 : a.sessionId < b.sessionId ? -1 : a.sessionId > b.sessionId ? 1 : 0;
|
|
310
|
+
return order === "asc" ? cmp : -cmp;
|
|
311
|
+
}
|
|
312
|
+
function isAfterSearchCursor(hit, cursor, order) {
|
|
313
|
+
const at = hit.updatedAt ?? "";
|
|
314
|
+
if (order === "asc") {
|
|
315
|
+
return at > cursor.updatedAt || (at === cursor.updatedAt && hit.sessionId > cursor.sessionId);
|
|
316
|
+
}
|
|
317
|
+
return at < cursor.updatedAt || (at === cursor.updatedAt && hit.sessionId < cursor.sessionId);
|
|
318
|
+
}
|
|
149
319
|
function cloneEntry(entry) {
|
|
150
320
|
return structuredClone(entry);
|
|
151
321
|
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { AgentConfig, ModelCapabilities, ModelConfig, ProviderRequestOptions, RunOptions, StructuredOutputOptions } from "./contracts.js";
|
|
1
|
+
import type { AgentConfig, ModelCapabilities, ModelConfig, ProviderRequest, ProviderRequestOptions, RunOptions, StructuredOutputOptions } from "./contracts.js";
|
|
2
2
|
export declare const DEFAULT_MAX_STRUCTURED_OUTPUT_SCHEMA_BYTES = 65536;
|
|
3
3
|
export declare const DEFAULT_MAX_STRUCTURED_OUTPUT_NAME_LENGTH = 128;
|
|
4
4
|
export declare class StructuredOutputError extends Error {
|
|
@@ -9,3 +9,7 @@ export declare function modelSupportsStructuredOutput(capabilities?: ModelCapabi
|
|
|
9
9
|
export declare function validateStructuredOutputOptions(options: StructuredOutputOptions): StructuredOutputOptions;
|
|
10
10
|
export declare function assertStructuredOutputRequestSupported(model: ModelConfig, options?: ProviderRequestOptions): void;
|
|
11
11
|
export declare function resolveRunProviderOptions(runOptions: Pick<RunOptions, "providerOptions" | "loop">, config: Pick<AgentConfig, "providerOptions" | "loop">): ProviderRequestOptions | undefined;
|
|
12
|
+
/** Strip native schema so a tool-eligible turn can call tools freely. */
|
|
13
|
+
export declare function withoutStructuredOutput(request: ProviderRequest): ProviderRequest;
|
|
14
|
+
/** Artifact/revision turn: keep/restore schema and withdraw tools. */
|
|
15
|
+
export declare function artifactStructuredOutputRequest(request: ProviderRequest, schema?: StructuredOutputOptions): ProviderRequest;
|
|
@@ -56,4 +56,22 @@ export function resolveRunProviderOptions(runOptions, config) {
|
|
|
56
56
|
structuredOutput: validateStructuredOutputOptions(loop.structuredOutput),
|
|
57
57
|
});
|
|
58
58
|
}
|
|
59
|
+
/** Strip native schema so a tool-eligible turn can call tools freely. */
|
|
60
|
+
export function withoutStructuredOutput(request) {
|
|
61
|
+
if (!request.options?.structuredOutput)
|
|
62
|
+
return request;
|
|
63
|
+
const { structuredOutput: _drop, ...options } = request.options;
|
|
64
|
+
return { ...request, options: Object.keys(options).length > 0 ? options : undefined };
|
|
65
|
+
}
|
|
66
|
+
/** Artifact/revision turn: keep/restore schema and withdraw tools. */
|
|
67
|
+
export function artifactStructuredOutputRequest(request, schema) {
|
|
68
|
+
const structuredOutput = schema ?? request.options?.structuredOutput;
|
|
69
|
+
if (!structuredOutput)
|
|
70
|
+
return { ...request, tools: undefined };
|
|
71
|
+
return {
|
|
72
|
+
...request,
|
|
73
|
+
tools: undefined,
|
|
74
|
+
options: { ...request.options, structuredOutput },
|
|
75
|
+
};
|
|
76
|
+
}
|
|
59
77
|
//# sourceMappingURL=structured-output.js.map
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { PersistencePage, SessionEntry, SessionEntryQuery } from "../contracts.js";
|
|
2
2
|
/** Current shared persistence schema version for production database adapters. */
|
|
3
|
-
export declare const PERSISTENCE_SCHEMA_VERSION =
|
|
3
|
+
export declare const PERSISTENCE_SCHEMA_VERSION = 4;
|
|
4
4
|
export type PersistenceTableName = "prism_tenants" | "prism_accounts" | "prism_users" | "prism_agent_definitions" | "prism_sessions" | "prism_branches" | "prism_session_entries" | "prism_session_append_idempotency" | "prism_runs" | "prism_agent_events" | "prism_tool_calls" | "prism_usage" | "prism_run_feedback" | "prism_retention_policies" | "prism_migrations";
|
|
5
5
|
export type PersistenceColumnType = "text" | "integer" | "number" | "boolean" | "json" | "timestamp";
|
|
6
6
|
export interface PersistenceColumnDefinition {
|
|
@@ -4,7 +4,7 @@ import { createHash } from "node:crypto";
|
|
|
4
4
|
// this module defines the shared table/index/pagination/migration expectations
|
|
5
5
|
// adapter authors implement and test against before shipping dialect-specific DDL.
|
|
6
6
|
/** Current shared persistence schema version for production database adapters. */
|
|
7
|
-
export const PERSISTENCE_SCHEMA_VERSION =
|
|
7
|
+
export const PERSISTENCE_SCHEMA_VERSION = 4;
|
|
8
8
|
/** Guidance adapters must follow: values are bound parameters, never interpolated. */
|
|
9
9
|
export const PARAMETERIZED_QUERY_GUIDANCE = "Bind every user-supplied value (session ids, idempotency keys, tenant ids, timestamps, JSON payloads) as a query parameter. Quote/validate schema and table identifiers only; never interpolate untrusted strings into SQL text.";
|
|
10
10
|
const TENANT_COLUMNS = [
|
|
@@ -315,7 +315,12 @@ function migrationStep(version, name, description) {
|
|
|
315
315
|
}
|
|
316
316
|
: version === 2
|
|
317
317
|
? { table: "prism_usage", columns: ["scope", "turn", "attempt"], indexes: ["prism_usage_session_scope_recorded_idx"] }
|
|
318
|
-
:
|
|
318
|
+
: version === 3
|
|
319
|
+
? { tables: ["prism_run_feedback"], indexes: model.indexes.filter((index) => index.name.startsWith("prism_run_feedback_")).map((index) => index.name) }
|
|
320
|
+
: version === 4
|
|
321
|
+
// Adapter-local FTS objects (SQLite FTS5 / Postgres tsvector) map to this canonical name.
|
|
322
|
+
? { search: ["prism_session_search"], indexes: ["prism_sessions_updated_id_idx"] }
|
|
323
|
+
: (() => { throw new Error(`Unknown migration version ${version}`); })();
|
|
319
324
|
return {
|
|
320
325
|
version,
|
|
321
326
|
name,
|
|
@@ -332,6 +337,7 @@ export function createPersistenceMigrationContract() {
|
|
|
332
337
|
migrationStep(1, "001_init", "Create core session, branch, entry, idempotency, run, ledger, and migration tables."),
|
|
333
338
|
migrationStep(2, "002_usage_scope", "Distinguish provider-turn usage from aggregate run totals."),
|
|
334
339
|
migrationStep(3, "003_run_feedback", "Add immutable ownership-scoped run/trace feedback and evaluation links."),
|
|
340
|
+
migrationStep(4, "004_session_search", "Add bounded session search indexes and adapter-local FTS objects."),
|
|
335
341
|
],
|
|
336
342
|
lockGuidance: "Acquire a dialect-specific migration lock before applying steps (PostgreSQL advisory lock; SQLite exclusive transaction). Only one process should migrate at a time.",
|
|
337
343
|
leastPrivilegeGuidance: "Run migrations with a DDL-capable role; use a separate least-privilege runtime role limited to INSERT/SELECT/UPDATE on adapter tables. Never grant migration credentials to the agent runtime.",
|
|
@@ -10,6 +10,12 @@ export interface SessionStoreConformanceOptions {
|
|
|
10
10
|
* when the store does not implement `readBranchPath`.
|
|
11
11
|
*/
|
|
12
12
|
readonly exerciseReadBranchPath?: boolean;
|
|
13
|
+
/**
|
|
14
|
+
* When true, exercises optional `searchSessions` (empty page, limit cap,
|
|
15
|
+
* invalid limit/query rejection via `resolveSessionSearchQuery` semantics).
|
|
16
|
+
* Skipped when the store does not implement `searchSessions`.
|
|
17
|
+
*/
|
|
18
|
+
readonly exerciseSearchSessions?: boolean;
|
|
13
19
|
/** When true, appends concurrent children of the same parent (fork allowed). */
|
|
14
20
|
readonly exerciseConcurrentParentAppend?: boolean;
|
|
15
21
|
/**
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
// stores already satisfy. Mirrors the assertion shape already repeated across
|
|
6
6
|
// src/__tests__/session-stores.test.ts and node-session-store-jsonl.test.ts so
|
|
7
7
|
// adapter authors do not re-derive them. Throws plain Error; no test runner.
|
|
8
|
-
import { isSessionAppendConflict } from "../contracts.js";
|
|
8
|
+
import { HARD_MAX_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, isSessionAppendConflict, resolveSessionSearchQuery, } from "../contracts.js";
|
|
9
9
|
/**
|
|
10
10
|
* Assert that a `SessionStore` implementation satisfies the core adapter
|
|
11
11
|
* contract: round-trip append/list, duplicate-entry-id rejection,
|
|
@@ -63,6 +63,9 @@ export async function assertSessionStoreConforms(store, options = {}) {
|
|
|
63
63
|
throw new Error(`readBranchPath must return the ancestor chain root→leaf in order; got ${JSON.stringify(ids)}`);
|
|
64
64
|
}
|
|
65
65
|
}
|
|
66
|
+
if (options.exerciseSearchSessions && typeof store.searchSessions === "function") {
|
|
67
|
+
await assertSessionStoreSearchSessions(store);
|
|
68
|
+
}
|
|
66
69
|
await assertSessionStoreBranchIsolation(store, options);
|
|
67
70
|
if (options.exerciseConcurrentParentAppend) {
|
|
68
71
|
await assertConcurrentParentAppendAllowed(store, sessionId, now, make);
|
|
@@ -104,6 +107,38 @@ export async function runSessionStoreConformance(factory, options = {}) {
|
|
|
104
107
|
label: "dup",
|
|
105
108
|
}, { idempotencyKey: "reopen-idem", expectedParentId: parent?.id }), (error) => isSessionAppendConflict(error) && error.conflict.idempotencyDuplicate === true, "Restarted store must still deduplicate an exact idempotency retry");
|
|
106
109
|
}
|
|
110
|
+
async function assertSessionStoreSearchSessions(store) {
|
|
111
|
+
const search = store.searchSessions;
|
|
112
|
+
await reject(() => search({ limit: 0 }), (error) => error instanceof TypeError, "searchSessions must reject non-positive limit");
|
|
113
|
+
await reject(() => search({ limit: Number.NaN }), (error) => error instanceof TypeError, "searchSessions must reject NaN limit");
|
|
114
|
+
await reject(() => search({ limit: HARD_MAX_SESSION_SEARCH_LIMIT + 1 }), (error) => error instanceof TypeError, "searchSessions must reject oversize limit");
|
|
115
|
+
await reject(() => search({ query: "x".repeat(HARD_MAX_SESSION_SEARCH_QUERY_BYTES + 1) }), (error) => error instanceof TypeError, "searchSessions must reject oversize query string");
|
|
116
|
+
const empty = await search(resolveSessionSearchQuery({
|
|
117
|
+
workspaceRoot: "__prism_conformance_empty__",
|
|
118
|
+
limit: 5,
|
|
119
|
+
}));
|
|
120
|
+
if (!Array.isArray(empty.items)) {
|
|
121
|
+
throw new Error("searchSessions must return a PersistencePage with an items array");
|
|
122
|
+
}
|
|
123
|
+
const resolved = resolveSessionSearchQuery({ limit: 1 });
|
|
124
|
+
const page = await search(resolved);
|
|
125
|
+
if (page.items.length > resolved.limit) {
|
|
126
|
+
throw new Error(`searchSessions must honor limit; got ${page.items.length} items for limit ${resolved.limit}`);
|
|
127
|
+
}
|
|
128
|
+
for (const hit of page.items) {
|
|
129
|
+
if (typeof hit.sessionId !== "string" || !hit.sessionId) {
|
|
130
|
+
throw new Error("searchSessions hits must include a non-empty sessionId");
|
|
131
|
+
}
|
|
132
|
+
if ("credential" in hit || "apiKey" in hit || "password" in hit) {
|
|
133
|
+
throw new Error("searchSessions hits must not include credential fields");
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
const unscoped = await search({ limit: 100 });
|
|
137
|
+
const owned = await search({ tenantId: "__missing_tenant__", limit: 100 });
|
|
138
|
+
if (owned.items.length > unscoped.items.length) {
|
|
139
|
+
throw new Error("searchSessions ownership filter must not return more hits than an unscoped search");
|
|
140
|
+
}
|
|
141
|
+
}
|
|
107
142
|
async function assertSessionStoreBranchIsolation(store, options) {
|
|
108
143
|
const sessionId = options.sessionId ?? "conformance";
|
|
109
144
|
const otherSessionId = options.otherSessionId ?? `${sessionId}-other`;
|
package/docs/agent-loops.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
Agent loops make the agent's per-run turn-control flow a replaceable strategy without forking the runtime. The runtime owns provider calls, retry, abort, store appends, redaction, and event emission; a loop only orchestrates those shared primitives through a `LoopContext`. The default `singleShotLoop` is the former inline turn loop extracted verbatim — assemble → generate → append assistant message → optional tool dispatch → next turn. `generateValidateReviseLoop` is the first alternative loop: generate → parse → validate → revise up to a budget.
|
|
5
|
+
Agent loops make the agent's per-run turn-control flow a replaceable strategy without forking the runtime. The runtime owns provider calls, retry, abort, store appends, redaction, and event emission; a loop only orchestrates those shared primitives through a `LoopContext`. The default `singleShotLoop` is the former inline turn loop extracted verbatim — optional drain pending steers → assemble → generate → append assistant message → optional tool dispatch → next turn (continue while steers remain even if the provider returned no tool calls). `generateValidateReviseLoop` is the first alternative loop: generate → parse → validate → revise up to a budget.
|
|
6
6
|
|
|
7
7
|
Loops are opt-in. When no `loop` is configured, the runtime runs `singleShotLoop` and behavior is bit-for-bit with the pre-loop runtime.
|
|
8
8
|
|
|
@@ -58,6 +58,7 @@ await session.run(input, {
|
|
|
58
58
|
repairer: hostRepairer, // optional; default stringifies validation.errors[].message
|
|
59
59
|
maxRevisions: 3, // optional; default 3
|
|
60
60
|
toolCalls: "bounded", // optional; default "disabled"; uses limits.maxToolRounds
|
|
61
|
+
structuredOutputTiming: "final-turn-only", // optional; default "every-turn"
|
|
61
62
|
},
|
|
62
63
|
});
|
|
63
64
|
|
|
@@ -82,6 +83,10 @@ type AgentLoopOptions =
|
|
|
82
83
|
readonly maxRevisions?: number;
|
|
83
84
|
/** Default "disabled". "bounded" dispatches sequentially up to limits.maxToolRounds. */
|
|
84
85
|
readonly toolCalls?: "disabled" | "bounded";
|
|
86
|
+
readonly structuredOutput?: StructuredOutputOptions;
|
|
87
|
+
readonly structuredOutputMode?: "native" | "artifact-loop";
|
|
88
|
+
/** Default "every-turn". "final-turn-only" omits schema while tools may run. */
|
|
89
|
+
readonly structuredOutputTiming?: "every-turn" | "final-turn-only";
|
|
85
90
|
};
|
|
86
91
|
```
|
|
87
92
|
|
|
@@ -96,6 +101,8 @@ Host callback contracts (all generic over host `T`):
|
|
|
96
101
|
| `ArtifactContext` | `{ sessionId, runId, turn, signal, metadata }` — passed to every callback. |
|
|
97
102
|
| `ArtifactParseResult<T>` | `{ ok: boolean; value?: T; error?: string }`. |
|
|
98
103
|
|
|
104
|
+
Optional steer hooks on `LoopContext` (0.0.11): `hasPendingSteers?()` / `applyPendingSteers?()`. Hosts/custom loops that omit them keep pre-steer behavior; built-in loops drain at turn start.
|
|
105
|
+
|
|
99
106
|
`LoopContext` (what the runtime builds for the loop each run):
|
|
100
107
|
|
|
101
108
|
| Field | Purpose |
|
|
@@ -78,7 +78,11 @@ Provider `thinking`/`reasoning` content emitted during a turn is preserved as `t
|
|
|
78
78
|
|
|
79
79
|
Missing providers fail closed: `run()` emits `error` and rejects before calling any provider. Provider `error` events emit session `error` and reject unless configured retry handles a transient provider-turn failure before output. Unknown tools fail closed through the tool harness and do not execute. Tool exceptions emit `tool_execution_error`, return an error `ToolResult`, and may still continue to the next provider turn.
|
|
80
80
|
|
|
81
|
-
Only one `run()` may be active per session. Concurrent `run()` calls emit `error` and reject immediately; Prism does not queue
|
|
81
|
+
Only one `run()` may be active per session. Concurrent `run()` / `prompt` / `followUp` calls emit `error` and reject immediately; Prism does not queue second prompts. Manual `compact()` also rejects while a run is active.
|
|
82
|
+
|
|
83
|
+
### Mid-run steer (0.0.11)
|
|
84
|
+
|
|
85
|
+
`session.steer(input, options?)` enqueues user text into the **same** active run (fail closed when no run). Default: inject at the next turn boundary (after tool rounds / before next provider assemble). `options.softInterrupt: true` aborts only the current provider stream, then continues the same `runId` with steered text. Pending queue caps: **8** messages / **64 KiB** UTF-8 total (`DEFAULT_MAX_PENDING_STEERS` / `DEFAULT_MAX_PENDING_STEER_BYTES`); overflow throws. Steered messages pass input guardrails + normal session append/redaction. Loops drain via optional `LoopContext.hasPendingSteers` / `applyPendingSteers`.
|
|
82
86
|
|
|
83
87
|
`session.abort(reason)` aborts the active run. The abort signal is passed to input assembly, provider requests, and tool execution; if a tool/provider path aborts after a tool call, Prism does not start another provider turn.
|
|
84
88
|
|
package/docs/cli-rpc.md
CHANGED
|
@@ -106,7 +106,7 @@ Branch-aware session commands return live handle details:
|
|
|
106
106
|
|
|
107
107
|
`sessionId` identifies the durable session. `leafId` is the selected branch tip. `handleId` is the RPC map key used by `switchSession`; forks that share the same `sessionId` get stable ids like `session-1#2` so the parent handle is not overwritten.
|
|
108
108
|
|
|
109
|
-
Invalid CLI flags return exit code `2`. Invalid JSON, missing ids, unknown RPC commands,
|
|
109
|
+
Invalid CLI flags return exit code `2`. Invalid JSON, missing ids, unknown RPC commands, unknown command contributions, and runtime failures return `ok: false` response envelopes without executing unknown tools or commands. `steer` with no active run (or overflow) returns `ok: false`.
|
|
110
110
|
|
|
111
111
|
## Request/response example
|
|
112
112
|
|
|
@@ -128,6 +128,7 @@ Invalid CLI flags return exit code `2`. Invalid JSON, missing ids, unknown RPC c
|
|
|
128
128
|
- `state`, `messages`, `setModel`, `switchSession`, `forkSession`, `cloneSession`, `checkout`, and registered `command` requests are processed immediately.
|
|
129
129
|
- `compact` is fail-closed: if the current session has an active run, it returns `ok: false` because the session rejects compaction during a run.
|
|
130
130
|
- A second `prompt` or `followUp` for the same session while it already has an active run returns `ok: false` immediately instead of blocking the input loop.
|
|
131
|
+
- `steer` enqueues mid-run user text for the active session (`params.input`, optional `params.softInterrupt`). Fails closed when no active run or when the pending steer queue overflows (8 messages / 64 KiB). Soft interrupt aborts the current provider stream only; the run continues.
|
|
131
132
|
|
|
132
133
|
Events streamed during a run keep the original prompt request id, even when an `abort` with a different request id cancels the run. The completion or error response for the prompt also uses the original prompt request id.
|
|
133
134
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
`@arnilo/prism-coding-agent` is an optional first-party package that provides host shell/filesystem/repository tools as Prism `ToolDefinition` objects. It ships six default coding tools — `shell`, `read`, `write`, `edit`, `repo_list`, `repo_search` — plus
|
|
5
|
+
`@arnilo/prism-coding-agent` is an optional first-party package that provides host shell/filesystem/repository tools as Prism `ToolDefinition` objects. It ships six default coding tools — `shell`, `read`, `write`, `edit`, `repo_list`, `repo_search` — plus opt-in structured Git/check set (`createGitTools`), opt-in `createAskUserDecisionTool({ ask })`, and bounded coding-plan/checkpoint helpers. The tools are **inert** until a host imports them and registers them into a `ToolRegistry`. Hosts may register any subset, omit aggregators entirely, or mix first-party tools with host-owned `ToolDefinition`s. Behavior for shell/read/write/edit is a behavioral port of the pi coding agent's tools, adapted to Prism's `ToolDefinition` / `ToolResult` contracts (no `@earendil-works/*` or `typebox` dependencies; only `diff` plus the Node standard library). List/search/Git are native Prism tools with no glob/ripgrep/Git-library dependency.
|
|
6
6
|
|
|
7
7
|
| Export | Purpose |
|
|
8
8
|
| --- | --- |
|
|
@@ -17,6 +17,7 @@
|
|
|
17
17
|
| `createAllTools(cwd, options?)` | Identical to `createCodingTools` (Git tools remain opt-in via `createGitTools`). |
|
|
18
18
|
| `createGitTools(cwd, options?)` | Opt-in Git tools (`git_status`/`git_diff`/`git_branch`/`git_worktree`/`git_apply`/`git_commit`/`git_pr_handoff`) plus optional `coding_check`. |
|
|
19
19
|
| `createCodingCheckTool(cwd, options)` | Named host-declared checks; model selects only a name. |
|
|
20
|
+
| `createAskUserDecisionTool(options)` | Opt-in user decision tool (`ask_user_decision`); host supplies `ask` callback. Not in default aggregators. |
|
|
20
21
|
| `createLocalRepositoryOperations(limits?)` | Default streaming Node filesystem backend for list/search. |
|
|
21
22
|
| `createGitOperations(options)` | Typed Git operations backend (argument arrays, safe config, finite output). |
|
|
22
23
|
| `buildCodingCheckpointMetadata` / `validateCodingCheckpointMetadata` / `assertCodingResumeAllowed` | Bounded durable coding-task metadata for workflow `state.coding` (no second runtime). |
|
|
@@ -231,6 +232,72 @@ const gitTools = createGitTools(workspaceRoot, {
|
|
|
231
232
|
});
|
|
232
233
|
```
|
|
233
234
|
|
|
235
|
+
### Ask-user decision (`createAskUserDecisionTool`)
|
|
236
|
+
|
|
237
|
+
Opt-in `ask_user_decision` for ambiguous, high-impact direction choices. Model must pass a question plus 2+ options, each with **exactly 3 pros and 3 cons**. Host supplies `ask` (blocks until the user picks). Not in `createCodingTools` / `createAllTools` / `createReadOnlyTools`.
|
|
238
|
+
|
|
239
|
+
| Mode | How |
|
|
240
|
+
| --- | --- |
|
|
241
|
+
| Single (default) | `selectionMode: "single"` → host returns `{ selectedId }` (or length-1 `selectedIds`) |
|
|
242
|
+
| Multi | `selectionMode: "multiple"` → `{ selectedIds: [...] }` (non-empty, known ids) |
|
|
243
|
+
| Free-text | `allowCustom: true` → host may return `{ customText }` **XOR** selection (never both) |
|
|
244
|
+
| Blocking tool | `createAskUserDecisionTool({ ask })` — in-process UI callback |
|
|
245
|
+
| Durable workflow | `suspendAskUserDecision(request)` + `createAskUserDecisionResumeValidator()` / `validateAskUserDecisionResume` on `resumeWorkflow` |
|
|
246
|
+
| Agent durable adapter | `validateAskUserDecisionAgentResume({ request, answer })` — same validation; **no** new `AgentRunInterruption` kinds in 0.0.11 |
|
|
247
|
+
|
|
248
|
+
Custom-text caps match question defaults (2 KiB / hard 8 KiB). Options default max 6 (hard 16).
|
|
249
|
+
|
|
250
|
+
```ts
|
|
251
|
+
import { createToolRegistry } from "@arnilo/prism";
|
|
252
|
+
import {
|
|
253
|
+
createAskUserDecisionTool,
|
|
254
|
+
createCodingTools,
|
|
255
|
+
suspendAskUserDecision,
|
|
256
|
+
createAskUserDecisionResumeValidator,
|
|
257
|
+
} from "@arnilo/prism-coding-agent";
|
|
258
|
+
|
|
259
|
+
const tools = createToolRegistry([
|
|
260
|
+
...createCodingTools(workspaceRoot),
|
|
261
|
+
createAskUserDecisionTool({
|
|
262
|
+
ask: async ({ question, options, selectionMode, allowCustom }) =>
|
|
263
|
+
ui.ask({ question, options, selectionMode, allowCustom }),
|
|
264
|
+
}),
|
|
265
|
+
]);
|
|
266
|
+
|
|
267
|
+
// Workflow node:
|
|
268
|
+
return suspendAskUserDecision({
|
|
269
|
+
question: "Ship sqlite or postgres?",
|
|
270
|
+
options: [/* ≥2 with 3 pros + 3 cons each */],
|
|
271
|
+
selectionMode: "single",
|
|
272
|
+
allowCustom: false,
|
|
273
|
+
});
|
|
274
|
+
// resumeWorkflow(..., { validateResume: createAskUserDecisionResumeValidator() })
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
### Goal → verify helper (`runCodingGoalVerify`)
|
|
278
|
+
|
|
279
|
+
Thin composition over existing plan Markdown, named checks, workflow `suspend`/`resumeWorkflow`, and bounded PR handoff. **No Goal table / second runtime.** Peer `@arnilo/prism-workflows`. Example: `examples/coding-goal-verify.ts`.
|
|
280
|
+
|
|
281
|
+
```ts
|
|
282
|
+
import { runCodingGoalVerify } from "@arnilo/prism-coding-agent";
|
|
283
|
+
|
|
284
|
+
const result = await runCodingGoalVerify({
|
|
285
|
+
goal: "Fix the flake",
|
|
286
|
+
cwd: process.cwd(),
|
|
287
|
+
taskId: "flake-1",
|
|
288
|
+
baseBranch: "main",
|
|
289
|
+
branch: "fix/flake",
|
|
290
|
+
checkNames: ["test"],
|
|
291
|
+
checkDefinitions: { test: { file: "/usr/bin/npm", args: ["test"] } },
|
|
292
|
+
runCheck: hostRunCheck,
|
|
293
|
+
buildHandoff: hostBuildHandoff,
|
|
294
|
+
approval: { validateResume: hostValidate },
|
|
295
|
+
checkpoints,
|
|
296
|
+
ownership,
|
|
297
|
+
redactor,
|
|
298
|
+
});
|
|
299
|
+
```
|
|
300
|
+
|
|
234
301
|
### Durable coding plans and checkpoints
|
|
235
302
|
|
|
236
303
|
There is no `CodingRun`, todo database, or second approval engine. Persist executable plan/todos as ordinary workspace Markdown (for example `plans/<task>.md`) and store only bounded metadata under workflow `state.coding`:
|
package/docs/evaluations.md
CHANGED
|
@@ -152,5 +152,5 @@ Fixtures reuse `@arnilo/prism-evals` (`defineDataset` / `defineScorer` / `scoreR
|
|
|
152
152
|
- [Runs and usage ledger](runs-and-usage.md): run/session identity for score linkage
|
|
153
153
|
- [Observability](observability.md): use `onTraceReference` or bounded `traceId(runId)` to supply `ScoreRunOptions.traceId`; evaluation telemetry emits no reason/explanation content
|
|
154
154
|
- [Coding agent tools](coding-agent-tools.md) / [Browser automation](browser-automation.md) / [Workflows](workflows.md): network-free coding-task composition at `examples/durable-coding-workflow.ts`; adversarial coding/browser eval example at `examples/coding-browser-evaluation.ts`
|
|
155
|
-
- [Performance limits](performance.md): `scripts/benchmark-0.0.10.mjs` workspace-mode evidence and `scripts/benchmark-0.0.9.mjs` coding/browser evidence fields
|
|
155
|
+
- [Performance limits](performance.md): `scripts/benchmark-0.0.11.mjs` search/budget evidence, `scripts/benchmark-0.0.10.mjs` workspace-mode evidence, and `scripts/benchmark-0.0.9.mjs` coding/browser evidence fields
|
|
156
156
|
- [Release and install](release-and-install.md): optional package install and protected sandbox-browser workflow
|