@arnilo/prism 0.1.2 → 0.1.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +10 -0
  2. package/dist/agent-approval.d.ts +49 -0
  3. package/dist/agent-approval.js +178 -0
  4. package/dist/agent-run-lifecycle.d.ts +7 -1
  5. package/dist/agent-run-lifecycle.js +213 -3
  6. package/dist/agent-run-state.d.ts +11 -0
  7. package/dist/agent-run-state.js +23 -0
  8. package/dist/agent-session.d.ts +98 -0
  9. package/dist/agent-session.js +1806 -0
  10. package/dist/agent-tool-dispatch.d.ts +13 -0
  11. package/dist/agent-tool-dispatch.js +92 -0
  12. package/dist/agents.d.ts +16 -9
  13. package/dist/agents.js +14 -2254
  14. package/dist/contracts-core.d.ts +1402 -0
  15. package/dist/contracts-core.js +119 -0
  16. package/dist/contracts-protocol.d.ts +596 -0
  17. package/dist/contracts-protocol.js +2 -0
  18. package/dist/contracts-run-state.d.ts +285 -0
  19. package/dist/contracts-run-state.js +77 -0
  20. package/dist/contracts.d.ts +7 -2261
  21. package/dist/contracts.js +7 -192
  22. package/dist/index.d.ts +1 -1
  23. package/dist/index.js +1 -1
  24. package/docs/0.1.0-readiness.md +3 -1
  25. package/docs/agent-identity.md +1 -1
  26. package/docs/agent-session-runtime.md +3 -1
  27. package/docs/browser-automation.md +13 -9
  28. package/docs/coding-agent-tools.md +13 -0
  29. package/docs/context-and-skills.md +2 -2
  30. package/docs/index.md +3 -14
  31. package/docs/migration.md +11 -5
  32. package/docs/performance.md +14 -4
  33. package/docs/persistence-credentials-multimodality-primitives.md +1 -1
  34. package/docs/provider-primitives.md +1 -1
  35. package/docs/public-contracts.md +2 -0
  36. package/docs/release-and-install.md +65 -0
  37. package/docs/session-stores.md +2 -2
  38. package/docs/thinking-and-reasoning.md +1 -1
  39. package/docs/tool-execution-primitives.md +2 -2
  40. package/docs/use-case-model-selection.md +1 -1
  41. package/docs/workflow-orchestration-primitives.md +1 -1
  42. package/package.json +4 -3
package/dist/contracts.js CHANGED
@@ -1,195 +1,10 @@
1
1
  /**
2
- * Thrown by a delegated-run host (e.g. the supervisor) when a nested run suspends on
3
- * pending decisions inside a tool execution. Core converts it into a root suspension
4
- * with attributed, root-visible approval ids; the dispatching wrapper attaches `toolCall`.
2
+ * Contracts barrel (0.1.4 god-module split): re-exports the full public
3
+ * contracts surface from the three concern-split modules
4
+ * (contracts-core / contracts-run-state / contracts-protocol) so the
5
+ * import surface of `./contracts.js` is unchanged.
5
6
  */
6
- export class AgentDelegationSuspendedError extends Error {
7
- ref;
8
- pendingDecisions;
9
- path;
10
- code = "ERR_PRISM_DELEGATION_SUSPENDED";
11
- toolCall;
12
- constructor(ref, pendingDecisions,
13
- /** Redacted delegation path (child ids) used when decisions carry no attribution. */
14
- path) {
15
- super("Delegated run suspended");
16
- this.ref = ref;
17
- this.pendingDecisions = pendingDecisions;
18
- this.path = path;
19
- this.name = "AgentDelegationSuspendedError";
20
- }
21
- }
22
- /** Shared decision-contract violations. Unknown and foreign approval ids share one non-enumerating error. */
23
- export class AgentDecisionError extends Error {
24
- code;
25
- constructor(code, message, options) {
26
- super(message, options);
27
- this.code = code;
28
- this.name = "AgentDecisionError";
29
- }
30
- }
31
- export const DEFAULT_MAX_PENDING_DECISIONS = 32;
32
- export const HARD_MAX_PENDING_DECISIONS = 128;
33
- export const DEFAULT_MAX_STICKY_DECISIONS = 64;
34
- export const HARD_MAX_STICKY_DECISIONS = 256;
35
- export const MAX_DECISION_REASON_BYTES = 2 * 1024;
36
- export const HARD_MAX_DECISION_REASON_BYTES = 8 * 1024;
37
- export const MAX_ELICITATION_BYTES = 16 * 1024;
38
- export const HARD_MAX_ELICITATION_BYTES = 64 * 1024;
39
- export const MAX_ACTION_CONSTRAINTS = 32;
40
- export const HARD_MAX_ACTION_CONSTRAINTS = 64;
41
- /** Maximum delegation attribution depth for surfaced nested pending decisions. */
42
- export const MAX_ATTRIBUTION_DEPTH = 8;
43
- export const MAX_ACTION_CONSTRAINT_BYTES = 4 * 1024;
44
- export const HARD_MAX_ACTION_CONSTRAINT_BYTES = 16 * 1024;
45
- export class AgentRunStateError extends Error {
46
- code = "ERR_PRISM_AGENT_RUN_STATE";
47
- constructor(message) {
48
- super(message);
49
- this.name = "AgentRunStateError";
50
- }
51
- }
52
- /** Durable-loop contract violations: hook-less custom strategy on a durable run, invalid snapshot, or revision drift. */
53
- export class AgentLoopStateError extends Error {
54
- code;
55
- constructor(code, message, options) {
56
- super(message, options);
57
- this.code = code;
58
- this.name = "AgentLoopStateError";
59
- }
60
- }
61
- export class AgentRunError extends Error {
62
- result;
63
- constructor(result, options) {
64
- super(result.error?.message ?? (result.status === "aborted" ? "Agent run aborted" : "Agent run failed"), options);
65
- this.name = "AgentRunError";
66
- this.result = result;
67
- }
68
- }
69
- /** Mid-run steer queue: default pending message count (fail closed at this cap). */
70
- export const DEFAULT_MAX_PENDING_STEERS = 8;
71
- /** Absolute pending steer count ceiling if hosts later expose overrides. */
72
- export const HARD_MAX_PENDING_STEERS = 32;
73
- /** Mid-run steer queue: default total UTF-8 byte budget across pending messages. */
74
- export const DEFAULT_MAX_PENDING_STEER_BYTES = 64 * 1024;
75
- /** Absolute pending steer byte ceiling if hosts later expose overrides. */
76
- export const HARD_MAX_PENDING_STEER_BYTES = 256 * 1024;
77
- export const SESSION_ENTRY_KINDS = [
78
- "message",
79
- "event",
80
- "summary",
81
- "metadata",
82
- "model_change",
83
- "label",
84
- "custom",
85
- "compaction",
86
- ];
87
- const SESSION_ENTRY_KIND_SET = new Set(SESSION_ENTRY_KINDS);
88
- export const SESSION_ENTRY_SCHEMA_VERSION = 1;
89
- export function isSessionEntryKind(value) {
90
- return typeof value === "string" && SESSION_ENTRY_KIND_SET.has(value);
91
- }
92
- /** Host-written `SessionRecord.metadata` / session metadata key for workspace filtering. */
93
- export const SESSION_SEARCH_WORKSPACE_METADATA_KEY = "workspaceRoot";
94
- export const DEFAULT_SESSION_SEARCH_LIMIT = 20;
95
- export const HARD_MAX_SESSION_SEARCH_LIMIT = 100;
96
- export const DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES = 4 * 1024;
97
- export const HARD_MAX_SESSION_SEARCH_QUERY_BYTES = 16 * 1024;
98
- export const DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES = 512;
99
- export const HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES = 4 * 1024;
100
- export const DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES = 1 * 1024;
101
- export const HARD_MAX_SESSION_SEARCH_CURSOR_BYTES = 4 * 1024;
102
- export const DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS = 1_000;
103
- export const HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS = 5_000;
104
- export const DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES = 10_000;
105
- export const HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES = 50_000;
106
- export const DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES = 8 * 1024 * 1024;
107
- export const HARD_MAX_SESSION_SEARCH_LINEAR_BYTES = 64 * 1024 * 1024;
108
- export const DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES = 1_000;
109
- export const HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES = 5_000;
110
- /**
111
- * O(1) validation before any scan/query. Applies default page limit; rejects NaN,
112
- * non-positive limits, oversize query/cursor/filter strings, and invalid order.
113
- */
114
- export function resolveSessionSearchQuery(query) {
115
- const limit = query.limit === undefined ? DEFAULT_SESSION_SEARCH_LIMIT : query.limit;
116
- if (!Number.isSafeInteger(limit) || limit < 1 || limit > HARD_MAX_SESSION_SEARCH_LIMIT) {
117
- throw new TypeError(`SessionSearchQuery.limit must be a safe integer from 1 to ${HARD_MAX_SESSION_SEARCH_LIMIT}`);
118
- }
119
- const order = query.order ?? "desc";
120
- if (order !== "asc" && order !== "desc") {
121
- throw new TypeError('SessionSearchQuery.order must be "asc" or "desc"');
122
- }
123
- assertSearchStringBytes(query.query, "query", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
124
- assertSearchStringBytes(query.cursor, "cursor", HARD_MAX_SESSION_SEARCH_CURSOR_BYTES);
125
- assertSearchStringBytes(query.workspaceRoot, "workspaceRoot", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
126
- assertSearchStringBytes(query.provider, "provider", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
127
- assertSearchStringBytes(query.model, "model", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
128
- assertSearchStringBytes(query.label, "label", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
129
- assertSearchStringBytes(query.summary, "summary", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
130
- assertSearchStringBytes(query.tenantId, "tenantId", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
131
- assertSearchStringBytes(query.accountId, "accountId", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
132
- assertSearchStringBytes(query.userId, "userId", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
133
- assertSearchStringBytes(query.fromUpdatedAt, "fromUpdatedAt", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
134
- assertSearchStringBytes(query.toUpdatedAt, "toUpdatedAt", HARD_MAX_SESSION_SEARCH_QUERY_BYTES);
135
- return { ...query, limit, order };
136
- }
137
- function assertSearchStringBytes(value, name, hardMax) {
138
- if (value === undefined)
139
- return;
140
- if (typeof value !== "string") {
141
- throw new TypeError(`SessionSearchQuery.${name} must be a string`);
142
- }
143
- // ponytail: UTF-8 byte length via TextEncoder; upgrade only if a non-Unicode host appears.
144
- const bytes = new TextEncoder().encode(value).byteLength;
145
- if (bytes > hardMax) {
146
- throw new TypeError(`SessionSearchQuery.${name} exceeds ${hardMax} bytes`);
147
- }
148
- }
149
- export const SESSION_SEARCH_UNSUPPORTED_CODE = "session_search_unsupported";
150
- /** Thrown when a store opts out of `searchSessions` (memory `unsupported`, JSONL). */
151
- export class SessionSearchUnsupportedError extends Error {
152
- code = SESSION_SEARCH_UNSUPPORTED_CODE;
153
- constructor(message = "session search is unsupported by this store") {
154
- super(message);
155
- this.name = "SessionSearchUnsupportedError";
156
- }
157
- }
158
- export function isSessionSearchUnsupported(error) {
159
- return error instanceof Error && error.code === SESSION_SEARCH_UNSUPPORTED_CODE;
160
- }
161
- /** Stable error code carried by `SessionAppendConflictError`. */
162
- export const SESSION_APPEND_CONFLICT_CODE = "session_append_conflict";
163
- /**
164
- * Thrown when `SessionStore.append` rejects an entry under `SessionAppendOptions`
165
- * (dangling/stale `expectedParentId`, stricter adapter CAS failure, or duplicate
166
- * idempotency key for the same parent). Recognize via the stable `code` and
167
- * `isSessionAppendConflict`, not message text.
168
- */
169
- export class SessionAppendConflictError extends Error {
170
- conflict;
171
- code = SESSION_APPEND_CONFLICT_CODE;
172
- constructor(conflict) {
173
- const detail = conflict.idempotencyDuplicate
174
- ? `idempotency key already used`
175
- : conflict.currentLeafId !== undefined
176
- ? `expected parent ${conflict.expectedParentId ?? "<none>"} does not match current leaf ${conflict.currentLeafId}`
177
- : `expected parent ${conflict.expectedParentId ?? "<none>"} is unavailable`;
178
- super(`session append conflict: ${detail}`);
179
- this.conflict = conflict;
180
- this.name = "SessionAppendConflictError";
181
- }
182
- }
183
- /** Type guard keyed off the stable `code` (works across bundles; not message text). */
184
- export function isSessionAppendConflict(error) {
185
- return error instanceof Error && error.code === SESSION_APPEND_CONFLICT_CODE;
186
- }
187
- const SESSION_METADATA_KEY_PATTERN = /^[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}$/;
188
- /** Validate a top-level `SessionRecord.metadata` key used by `SessionQuery.metadataKey` filters. */
189
- export function assertSessionMetadataKey(key) {
190
- if (typeof key !== "string" || !SESSION_METADATA_KEY_PATTERN.test(key)) {
191
- throw new RangeError("metadataKey must match /^[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}$/");
192
- }
193
- return key;
194
- }
7
+ export * from "./contracts-core.js";
8
+ export * from "./contracts-run-state.js";
9
+ export * from "./contracts-protocol.js";
195
10
  //# sourceMappingURL=contracts.js.map
package/dist/index.d.ts CHANGED
@@ -105,5 +105,5 @@ export { createToolParameterValidator, createToolRegistry, dispatchToolCall, fil
105
105
  export type { ResolvedUseCaseModel, ResolveUseCaseModelInput, UseCaseModelBinding, } from "./use-case-model.js";
106
106
  export { resolveUseCaseModel, resolveUseCaseModelBinding, useCaseCredentialProviderId, } from "./use-case-model.js";
107
107
  export declare const name = "prism";
108
- export declare const version = "0.1.2";
108
+ export declare const version = "0.1.4";
109
109
  export declare const description = "Agent harness for AI providers, agents, sessions, and tools.";
package/dist/index.js CHANGED
@@ -57,6 +57,6 @@ export { DEFAULT_TOOL_RESULT_FOLD_MAX_SUMMARY_BYTES, DEFAULT_TOOL_RESULT_FOLD_MI
57
57
  export { createToolParameterValidator, createToolRegistry, dispatchToolCall, filterTools } from "./tools.js";
58
58
  export { resolveUseCaseModel, resolveUseCaseModelBinding, useCaseCredentialProviderId, } from "./use-case-model.js";
59
59
  export const name = "prism";
60
- export const version = "0.1.2";
60
+ export const version = "0.1.4";
61
61
  export const description = "Agent harness for AI providers, agents, sessions, and tools.";
62
62
  //# sourceMappingURL=index.js.map
@@ -10,10 +10,12 @@ the release tree before cutting 1.0. The decision to cut 1.0 stays with the
10
10
  operator after operator-gated legs run in a protected environment and Phase
11
11
  12 demand evidence exists.
12
12
 
13
- Evidence trail: [`docs/review-coverage-2026-07-26-phase-11.md`](./review-coverage-2026-07-26-phase-11.md)
13
+ Evidence trail: [`docs/_evidence/review-coverage-2026-07-26-phase-11.md`](./_evidence/review-coverage-2026-07-26-phase-11.md)
14
14
  (addenda 0–9), [`docs/release-and-install.md`](./release-and-install.md),
15
15
  [`docs/migration.md`](./migration.md), [`docs/performance.md`](./performance.md),
16
16
  [`docs/public-contracts.md`](./public-contracts.md) (frozen 0.1.x contract).
17
+ The per-phase review-coverage evidence archive lives in [`docs/_evidence/`](./_evidence/)
18
+ (plans 067–079, releases 0.0.4–0.0.16; tarball-excluded, kept in-repo for audit).
17
19
  Historical release lines (0.0.16 floor → 0.0.27 Phase 10 ACP interop → 0.1.0)
18
20
  keep their per-phase evidence in the pages above; this page records the 0.1.1
19
21
  snapshot (plan 013) with the 0.1.0 table below as the previous line.
@@ -130,7 +130,7 @@ Identity is optional. Hosts that only set `ownership` keep prior behavior. When
130
130
  - Delegation only narrows scopes; tenant/account/user cannot widen on propagation.
131
131
  - Credential refs never expand to secrets in events, ledgers, or telemetry attributes.
132
132
  - Checks are O(fields) and network-free in core; remote auth stays in the host verifier.
133
- - Raising hard caps requires updating `docs/review-coverage-2026-07-23-phase-8.md`, tests, and docs.
133
+ - Raising hard caps requires updating `docs/_evidence/review-coverage-2026-07-23-phase-8.md`, tests, and docs.
134
134
 
135
135
  ## Related APIs
136
136
 
@@ -133,6 +133,8 @@ const clone = await session.clone({ id: "s2" });
133
133
 
134
134
  ## Extension and configuration notes
135
135
 
136
+ **Internal file structure (0.1.4).** Since 0.1.4 the runtime is spread across sibling modules behind the `src/agents.ts` barrel: `src/agent-session.ts` (the `RuntimeAgentSession` class, session factories, and shared session helpers), `src/agent-run-lifecycle.ts` (resume lifecycle: `resumeAgentRun`/`resumeAgentRunStream`), `src/agent-approval.ts` (pending-decision and approval helpers), `src/agent-tool-dispatch.ts` (elicitation and tool-policy helpers), `src/agent-run-state.ts` (run-state persistence + agent fingerprint), and `src/agent-loops.ts`/`src/compaction.ts`. The public import surface is unchanged — `createAgent`/`createAgentSession`/`resumeAgentRun`/`resumeAgentRunStream` still resolve from the package entry.
137
+
136
138
  The runtime calls `assembleProviderInput()` on every turn and uses only runtime-consumed values supplied on `AgentConfig`: `instructions`, `systemPrompt`, `inputBuilder`, `promptBuilder`, `inputLayout`, `context`, selected `skills`, active `tools`, `middleware`, `resourceLoader`, metadata, `compaction`, `retry`, and `RunOptions.model`/`systemPrompt`/`inputLayout`/`compaction`/`retry`. Contributions remain inert until a host passes selected values into the agent config.
137
139
 
138
140
  `AgentConfig` no longer accepts inert `extensions`, `settings`, or `credentials` fields. Load extensions with `createExtensionKernel()` before building config; read settings in the host before passing concrete runtime options; resolve credentials at the provider edge and pass exact secret values to redaction when needed.
@@ -196,7 +198,7 @@ if (result.status === "suspended") {
196
198
  }
197
199
  ```
198
200
 
199
- Resume requires exact checkpoint ownership, version, agent fingerprint, and revision. The fingerprint hashes the agent id/name, `definitionRevision`, model, instructions, system-prompt contributions, skills (name/instructions/tool names), tool definitions (name/parameters/exclusive), guardrail definitions (name/stage/revision), and loop strategy — changing any of them without bumping `definitionRevision` fails resume closed instead of silently continuing with different agent semantics. Prism CAS-claims approval before work, rechecks normal guardrail/permission/validation/limit paths, and marks a pending tool dispatched before its side effect. `createAgentRunLifecycle()` wraps the same core path for server/MCP hosts: adapters pass only authorized ownership, status returns only `{ state, version }`, and `resolveAgent()` supplies current agent/revision. `resumeStream()` uses that same claim path and bounded subscriber, so adapters do not poll or duplicate resume logic. Remote restart requires both checkpoint and session stores to be durable. A crash after that mark is ambiguous and is never replayed automatically; use host tool idempotency keyed by `runId`/`toolCallId` or resolve it manually. Checkpoints contain bounded redacted state plus session/leaf references, never provider objects, callbacks, signals, credentials, or raw secrets. State is bounded at save by `runState.maxStateBytes` (default 256 KB, at most the 1 MB hard cap); load bounds against the 1 MB hard cap only, so state saved with a raised limit stays resumable while oversized records are still rejected. Built-in loop options are durable; custom `AgentLoopStrategy` instances are durable when they declare `snapshot`/`restore` hooks (see [Agent loops § Durable runs](agent-loops.md#durable-runs)) and reject before provider work otherwise.
201
+ Resume requires exact checkpoint ownership, version, agent fingerprint, and revision. The fingerprint hashes the agent id/name, `definitionRevision`, model, instructions, system-prompt contributions, skills (name/instructions/tool names), tool definitions (name/parameters/exclusive), guardrail definitions (name/stage/revision), and loop strategy — changing any of them without bumping `definitionRevision` fails resume closed instead of silently continuing with different agent semantics. Prism CAS-claims approval before work, rechecks normal guardrail/permission/validation/limit paths, and marks a pending tool dispatched before its side effect. `createAgentRunLifecycle()` wraps the same core path for server/MCP hosts: adapters pass only authorized ownership, status returns only `{ state, version }`, and `resolveAgent()` supplies current agent/revision. `resumeStream()` uses that same claim path and bounded subscriber, so adapters do not poll or duplicate resume logic. Remote restart requires both checkpoint and session stores to be durable. A crash after that mark is ambiguous and is never replayed automatically; use host tool idempotency keyed by `runId`/`toolCallId` or resolve it manually. Checkpoints contain bounded redacted state plus session/leaf references, never provider objects, callbacks, signals, credentials, or raw secrets. State is bounded at save by `runState.maxStateBytes` (default 256 KB, at most the 1 MB hard cap); load bounds against the 1 MB hard cap only, so state saved with a raised limit stays resumable while oversized records are still rejected. Since 0.1.3 (plan 015 Task 4), durable runs may opt in to session-state persistence with `persistSessionState: true` on both the run and resume options: the loaded-skill **name catalog** (≤64 names, ≤256 chars each) rides the checkpoint and is restored into the resumed session's `LoadedSkillSet`; skill **bodies are never persisted** and re-resolve from the live registry via `load_skill`. Default off keeps the checkpoint shape byte-identical to 0.1.2. Built-in loop options are durable; custom `AgentLoopStrategy` instances are durable when they declare `snapshot`/`restore` hooks (see [Agent loops § Durable runs](agent-loops.md#durable-runs)) and reject before provider work otherwise.
200
202
 
201
203
  ## Secure composition
202
204
 
@@ -2,11 +2,11 @@
2
2
 
3
3
  ## What it does
4
4
 
5
- `@arnilo/prism-browser` exposes exactly four exclusive model-facing tools—`browser_open`, `browser_snapshot`, `browser_act`, and `browser_close`—over a host-supplied Playwright `Browser`. Prism creates one non-persistent `BrowserContext` per run, serializes actions, returns bounded AI-mode accessibility snapshots with snapshot-scoped refs, enforces egress/side-effect/upload/download/screenshot policy, and closes context/pages/listeners/quarantined downloads on close, abort, or manager disposal.
5
+ `@arnilo/prism-browser` exposes six exclusive model-facing tools—`browser_open`, `browser_snapshot`, `browser_act`, `browser_close`, `browser_evaluate`, and `browser_observe`—over a host-supplied Playwright `Browser`. Prism creates one non-persistent `BrowserContext` per run, serializes actions, returns bounded AI-mode accessibility snapshots with snapshot-scoped refs, enforces egress/side-effect/upload/download/screenshot policy, and closes context/pages/listeners/quarantined downloads on close, abort, or manager disposal. Since 0.1.4 the package also rides playwright-core's existing CDP transport for bounded page evaluation, console/network observation, and network/emulation control on Chromium hosts — zero new dependencies, Prism still never launches or downloads browsers.
6
6
 
7
7
  ## When to use it
8
8
 
9
- Use when an agent must interact with JavaScript-heavy or authenticated pages that search/fetch cannot cover. Prefer `@arnilo/prism-web-tools` for ordinary public retrieval. Do not use this package as a browser launcher, MCP proxy, visual planner, CDP console, or persistent profile manager.
9
+ Use when an agent must interact with JavaScript-heavy or authenticated pages that search/fetch cannot cover, or needs bounded page-context evaluation and console/network observation on Chromium. Prefer `@arnilo/prism-web-tools` for ordinary public retrieval. Do not use this package as a browser launcher, MCP proxy, visual planner, general-purpose CDP console (domains are allowlisted), or persistent profile manager.
10
10
 
11
11
  ## Inputs / request
12
12
 
@@ -14,14 +14,16 @@ Use when an agent must interact with JavaScript-heavy or authenticated pages tha
14
14
  | --- | --- | --- |
15
15
  | `browser_open` | optional absolute `http(s)` `url` | host Playwright `Browser` or `BrowserManager`, limits, `ExecutionPolicy`, `networkPolicy`, uploads/downloads |
16
16
  | `browser_snapshot` | optional `pageId` | same manager/context |
17
- | `browser_act` | `action` plus action-specific fields (`target`, `snapshotId`, `url`, `text`, `values`, `paths`, `downloadId`, `dialogResponse`, `pageId`, `clip`, …) | policy checked before side effects |
17
+ | `browser_act` | `action` plus action-specific fields (`target`, `snapshotId`, `url`, `text`, `values`, `paths`, `downloadId`, `dialogResponse`, `pageId`, `clip`, `patterns`, `offline`, `latencyMs`, `downloadKbps`, `uploadKbps`, `reset`, `width`, `height`, `mobile`, `deviceScaleFactor`, `userAgent`, …) | policy checked before side effects |
18
+ | `browser_evaluate` | optional `pageId`, required `expression`, optional `awaitPromise`/`timeoutMs` | CDP `Runtime.evaluate` on a per-page session (Chromium hosts) |
19
+ | `browser_observe` | optional `pageId` | CDP Runtime/Network ring, drain-on-read |
18
20
  | `browser_close` | none | closes only the run-owned context, never the host Browser process |
19
21
 
20
- `createBrowserTools({ browser, executionPolicy?, limits?, networkPolicy?, uploads?, downloads?, beforeSideEffect? })` builds the four tools. `createBrowserManager(...)` exposes host lifecycle helpers `closeRun(runId)` / `close()` and `listDownloads(runId)`.
22
+ `createBrowserTools({ browser, executionPolicy?, limits?, networkPolicy?, uploads?, downloads?, beforeSideEffect?, cdp? })` builds the six tools. `createBrowserManager(...)` exposes host lifecycle helpers `closeRun(runId)` / `close()` and `listDownloads(runId)`.
21
23
 
22
- Targets accepted by `browser_act`: snapshot `ref`, `role`(+`name`), `label`, `testId`, or `text`. CSS, XPath, selector strings, `page.evaluate`, CDP/devtools, extensions, and persistent/local profiles are unsupported.
24
+ Targets accepted by `browser_act`: snapshot `ref`, `role`(+`name`), `label`, `testId`, `text`, and (0.1.4) raw `{ css }` / `{ xpath }`. `selector` strings and `evaluate` target keys stay denied; page-context JS goes through the policy-gated `browser_evaluate`. Extensions and persistent/local profiles are unsupported.
23
25
 
24
- `browser_act` actions: `navigate`, `click`, `type`, `fill`, `select`, `check`, `uncheck`, `scroll`, `wait`, `dialog`, `select_page`, `upload`, `screenshot`, `download_release`.
26
+ `browser_act` actions: `navigate`, `click`, `type`, `fill`, `select`, `check`, `uncheck`, `scroll`, `wait`, `dialog`, `select_page`, `upload`, `screenshot`, `download_release`, and (0.1.4, CDP) `block_urls`, `unblock_urls`, `throttle`, `emulate`.
25
27
 
26
28
  ## Outputs / response / events
27
29
 
@@ -105,16 +107,18 @@ await browser.close();
105
107
  - Uploads require absolute paths under `uploads.roots` (realpath-contained; symlink escapes rejected). Downloads stream into `downloads.quarantine` with SHA-256/MIME/name metadata; `download_release` requires host `approveRelease`. Screenshots return bounded `ImageContent`.
106
108
  - Observation (`snapshot`, `wait`, open-without-url, `close`) vs mutation/high-impact (`navigate`, click/form, dialog accept, upload, download release, popup select) is classified for `ExecutionPolicy` / `beforeSideEffect`.
107
109
  - `createSharedSandboxBrowserOptions()` aligns browser uploads/downloads with Task 1 sandbox `/workspace` and `/downloads`. `assertBrowserSandboxNetwork()` in `@arnilo/prism-coding-security` fails closed for custom Docker networks without browser egress attestation.
108
- - Raw CSS is absent from production defaults. Ref resolution uses Playwright’s built-in `aria-ref=` selector with a package-owned snapshot ref table for staleness checks.
110
+ - Raw CSS/XPath: since 0.1.4 `{ css }` / `{ xpath }` targets resolve via Playwright's selector engine (`locator(css)` / `locator("xpath=…")`); ref resolution keeps the built-in `aria-ref=` selector with a package-owned snapshot ref table for staleness checks.
111
+ - CDP capabilities (0.1.4): `browser_evaluate` (bounded `Runtime.evaluate`), `browser_observe` (Runtime console/exception + Network request/response/failed events in a bounded ring with drain-on-read), and `block_urls`/`unblock_urls` (`Network.setBlockedURLs`), `throttle` (`Network.emulateNetworkConditions`), `emulate` (`Emulation.setDeviceMetricsOverride` + optional `setUserAgentOverride`). All CDP sessions are per-page via `context.newCDPSession(page)` and are detached on run close — network/emulation changes are run-scoped and reset with `browser_close`. `BrowserCdpOptions.mode` (`auto` | `on` | `off`, default `auto`) gates the surface: non-Chromium hosts or mode `off` return `ERR_PRISM_BROWSER_CDP_UNAVAILABLE` without affecting Playwright-only tools. Domains are limited to the Runtime/Network/Emulation allowlist — cookies, tracing, performance profiles, IndexedDB, and worker debugging are not exposed. CDP is not an egress bypass: page network still routes through the run's routing/blocking and `networkPolicy`.
112
+ - CDP bounds (0.1.4): evaluate expression ≤ `maxActionInputBytes` (64 KiB default / 256 KiB hard) and result capped at `maxEvaluateResultBytes` (64 KiB / 256 KiB) with truncation marking; `browser_observe` rings capped at `maxConsoleEntries` (200/500) and `maxNetworkRequests`; `block_urls` patterns ≤ `maxBlockedUrlPatterns` (32/128); throttle latency ≤ 120 s and throughput ≤ 1 Gbps; emulate dimensions ≤ 16 384 and scale ≤ 10, user agent ≤ 2 KiB. Evaluate is classified high-impact (arbitrary page-context code execution): `ExecutionPolicy` approval and the `beforeSideEffect` hook are mandatory, it charges the action budget, and results are marked `untrusted_external`. `browser_observe` is observation-only (no side-effect hook, no action charge) and **never captures request/response bodies, cookies, or auth headers** — only bounded URL/method/status/error-text/arg previews.
109
113
  - Verified-state checkpoints (0.0.14): `createBrowserCheckpointLedger()` records navigation state — URL, a domain-state hash, and host-owned data refs — never serialized browser internals (cookies/storage/contexts), which are fragile and secret-bearing. Frozen caps: URL 8 KiB/16 KiB, domain-state hash 256 B/1 KiB, host-data ref 2 KiB/8 KiB (refs only, never bodies), 16/64 checkpoints per run (oldest evicted). After any resume/interruption `markResumed(runId)` marks state stale; `assertVerifiedBeforeSideEffect(runId)` fails closed until the host reloads + `verify()`s, so side effects never replay on stale state. Checkpoints are run-scoped: a conversation thread composes through the run it owns, reusing the manager's sandbox/egress/approval/limit policy above.
110
114
 
111
115
  Observation tools declare `kind: none`; mutations are `external_mutation`/`unsupported` and fail closed on stale checkpoint state. See [tool effects](tool-effects.md).
112
116
 
113
117
  ## Security and performance notes
114
118
 
115
- Import is inert. Construction fails clearly when neither `browser` nor `manager` is supplied. Browser installation, launch, version, and control endpoint are host-owned. Prism never exposes `page.evaluate`, init scripts, CDP, extensions, persistent profiles, or model-supplied Playwright launch options. Secrets and storage state must not appear in snapshots, tool results, logs, or checkpoints. Finite caps charge before context/page/action/queue/snapshot/network/artifact retention; snapshots retain no unbounded DOM, console, request, response, or trace history. Unreleased downloads are deleted on context close.
119
+ Import is inert. Construction fails clearly when neither `browser` nor `manager` is supplied. Browser installation, launch, version, and control endpoint are host-owned. Prism never exposes init scripts, extensions, persistent profiles, or model-supplied Playwright launch options; CDP exposure is limited to the allowlisted Runtime/Network/Emulation surface above (evaluate is policy-gated arbitrary code execution — treat results as untrusted). Secrets and storage state must not appear in snapshots, tool results, logs, or checkpoints. Finite caps charge before context/page/action/queue/snapshot/network/artifact retention; snapshots retain no unbounded DOM, console, request, response, or trace history. Unreleased downloads are deleted on context close.
116
120
 
117
- Default tests use fake Playwright APIs only. Protected live gate: `PRISM_LIVE_PLAYWRIGHT=1` (or `PRISM_TEST_PLAYWRIGHT=1`) `npm run test:live -w @arnilo/prism-browser` exercises a local loopback hostile HTML fixture for snapshot refs, stale-ref rejection, CSS denial, private/file deny, upload containment, screenshot bounds, and download quarantine/release. Missing browser binaries fail closed when the gate is enabled. Adversarial network-free fixtures live in `eval-fixtures.test.ts`; see [Evaluations](evaluations.md) and `examples/coding-browser-evaluation.ts`.
121
+ Default tests use fake Playwright APIs only. Protected live gate: `PRISM_LIVE_PLAYWRIGHT=1` (or `PRISM_TEST_PLAYWRIGHT=1`) `npm run test:live -w @arnilo/prism-browser` exercises a local loopback hostile HTML fixture for snapshot refs, stale-ref rejection, css/xpath targets, private/file deny, upload containment, screenshot bounds, download quarantine/release, and the CDP leg (real evaluate, observe, and emulate). Missing browser binaries fail closed when the gate is enabled. Adversarial network-free fixtures live in `eval-fixtures.test.ts`; see [Evaluations](evaluations.md) and `examples/coding-browser-evaluation.ts`.
118
122
 
119
123
  ## Related APIs
120
124
 
@@ -206,6 +206,19 @@ const write = createWriteTool(cwd, { requireReadBeforeWrite: true, readPathSet:
206
206
  const edit = createEditTool(cwd, { requireReadBeforeWrite: true, readPathSet: readPaths });
207
207
  ```
208
208
 
209
+ Since 0.1.3 (plan 015 Task 4) hosts may opt in to persisting the set across restarts via the host-owned `CheckpointStore`:
210
+
211
+ ```ts
212
+ import { createReadPathSet, createReadPathSetPersistence } from "@arnilo/prism-coding-agent";
213
+
214
+ const readPaths = createReadPathSet();
215
+ const persistence = createReadPathSetPersistence({ checkpoints, key: sessionId, ownership });
216
+ await persistence.restore(readPaths); // on session attach (returns restored count)
217
+ await persistence.save(readPaths); // after reads, before session close
218
+ ```
219
+
220
+ Names only (paths are bounded at 1024 entries / 1024 chars each; larger sets fail closed with no partial write). Records live under the `prism.coding-agent.read-path-set` namespace keyed by session id, and `ownership` is part of the trust boundary: restoring under a different tenant/user throws instead of leaking paths. Default is **off** — the set stays in-memory unless the host wires the helper explicitly.
221
+
209
222
  ### `edit`
210
223
 
211
224
  Precise text replacement in an existing file via exact-then-fuzzy matching.
@@ -170,7 +170,7 @@ Catalog caps: **64** entries default / **256** hard; descriptions **512 B** defa
170
170
  ```ts
171
171
  import { assembleProviderInput, createLoadedSkillSet } from "@arnilo/prism";
172
172
 
173
- const loaded = createLoadedSkillSet(); // session-owned; not checkpoint-persisted in 0.0.20
173
+ const loaded = createLoadedSkillSet(); // session-owned; opt-in checkpoint-persisted via runState.persistSessionState (names only) since 0.1.3
174
174
  const request = await assembleProviderInput({
175
175
  model,
176
176
  input: "Hi",
@@ -256,7 +256,7 @@ Use `activateAllCapabilities: true` only as a temporary all-skills/all-tools com
256
256
  - Skill registry lookup is `Map`-backed, and selection is linear in requested skills plus active tools. Strict duplicate mode adds one O(1) `Map.has()` check during registration only.
257
257
  - Progressive catalog render is O(active skills) with byte/count caps; `load_skill` lookup is O(1). Budget eviction over context/skills is O(n log n) worst case.
258
258
  - `load_skill` cannot grant tools; loaded instructions are untrusted text bounded by hard caps. `toolResultFold` summarizer output is untrusted and capped; failures keep raw tool results.
259
- - Loaded-skill names are session-scoped in memory only in 0.0.20 — not checkpoint-persisted; new sessions start catalog-only until reload.
259
+ - Loaded-skill names are session-scoped. Since 0.1.3 (plan 015 Task 4) a durable run may opt in to persistence with `runState.persistSessionState: true` (and the same flag on resume options): the name catalog rides the run-state checkpoint (≤64 names, ≤256 chars each, charged against `maxStateBytes`) and is restored into the session `LoadedSkillSet` on resume, so progressive disclosure survives restart. **Bodies are never persisted** — they re-resolve from the live skill registry the next time the model loads the skill. Default off: checkpoint shape is identical to 0.1.2.
260
260
  - These helpers perform no provider calls, tool execution, resource loading, package discovery, filesystem/network access, retries, timers, or watchers by themselves.
261
261
  - Context and skill output is host/extension data. Do not include secrets unless the host explicitly accepts that prompt exposure.
262
262
  - Active tools remain host-supplied; skills and middleware do not activate tools or grant permissions. Use `duplicate: "error"` when loading third-party skills to prevent silent name shadowing.
package/docs/index.md CHANGED
@@ -72,7 +72,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
72
72
  - [Web search, fetch, and extraction](web-tools.md): optional host-selected Brave/Exa discovery and Firecrawl Markdown/schema tools with native fetch, stable citations, late credentials, finite limits, and explicit untrusted-content boundaries.
73
73
  - [Work tools](work-tools.md): optional `@arnilo/prism-work-tools` identity-scoped M365 + GWS connectors (hard-coded CLI argv, draft-then-approve, state-machine idempotency, shared result shapes); 0.0.14 adds a late-bound per-identity `tokenProvider` (env-only, fail-closed).
74
74
  - [Work connectors](work-connectors.md): connector principles, capability gates, scoped OAuth establishment (0.0.14), and out-of-scope boundaries (Slack/Teams channels not shipped) for Microsoft 365 / Google Workspace.
75
- - [Browser automation](browser-automation.md): optional `@arnilo/prism-browser` with host-supplied Playwright contexts, AI-mode snapshots/refs, ordered `browser_open`/`browser_snapshot`/`browser_act`/`browser_close`, egress/side-effect/upload/download/screenshot policy, finite page/action/snapshot/network/artifact caps, and 0.0.14 verified-state checkpoints with reload/verify-before-side-effect.
75
+ - [Browser automation](browser-automation.md): optional `@arnilo/prism-browser` with host-supplied Playwright contexts, AI-mode snapshots/refs, ordered `browser_open`/`browser_snapshot`/`browser_act`/`browser_close` plus (0.1.4) `browser_evaluate`/`browser_observe` and CDP `block_urls`/`unblock_urls`/`throttle`/`emulate` act actions on Chromium hosts, egress/side-effect/upload/download/screenshot policy, finite page/action/snapshot/network/artifact caps, and 0.0.14 verified-state checkpoints with reload/verify-before-side-effect.
76
76
  - [Device adapters](device-adapters.md): deny-by-default realtime voice / desktop-control contract + conformance (0.0.14); no vendor package — admission fails closed without explicit consent+sandbox+approval, stream bounds, shared `RunLimits`, redacted telemetry.
77
77
  - [Coding agent tools](coding-agent-tools.md): optional `shell`, `read`, `write`, `edit`, `repo_list`, `repo_search`, `glob`, `delete`, and `move` definitions plus opt-in `createGitTools()` / `coding_check`, opt-in `createAskUserDecisionTool` (single/multi/free-text + durable suspend glue), and `runCodingGoalVerify`; durable plan/todo Markdown helpers with workflow `state.coding` checkpoint metadata; streamed text pages, `repo_search` `outputMode`, bounded glob, optional read-before-write, optional Git-aware (`createGitAwareRepositoryOperations`) ignore-aware enumeration with native fallback, finite Git/check/plan/ask caps, bounded image/edit reads and write/edit payloads, finite shell wall/total-output limits, secure host-owned spill cleanup, pluggable bounded operation contracts, per-path mutation serialization, and optional `ExecutionPolicy`. No PDF/trash/PTY in the 0.0.21 baseline; Phase 9 adds optional language intelligence (separate page). Limits do not sandbox host access—gate with permission/trust policy and `@arnilo/prism-coding-security`.
78
78
  - [Language intelligence](language-intelligence.md): optional host-activated `createLanguageIntelligence` — bounded in-package LSP 3.17 JSON-RPC client (Content-Length framing), host-selected server command/args per language, workspace symbols/definitions/references/diagnostics/hover/rename; lazy spawn; URI root confinement; rename gated by `ExecutionPolicy` + atomic write/mutation queue; frozen message/diagnostic/pending/result/timeout/server caps. No `vscode-languageserver-protocol` dependency.
@@ -129,18 +129,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
129
129
  - [Ponytail behavior integration](ponytail.md): optional `@arnilo/prism-ponytail` — upstream Ponytail skills/commands, `ponytail-mode` injector, session `ponytail-mode` persistence; resolves peer `@dietrichgebert/ponytail` or `upstreamPath`; opt-in (not in code/sdk profiles).
130
130
 
131
131
  ## Release and install
132
- - [Release and install](release-and-install.md): current **0.1.2** 49-package graph (root + 48 workspace packages; plan 014 Alibaba provider enrichment on the frozen 0.1.x line — embeddings, video input, verified compatible-mode surface decision table; plan 013 post-release hardening — build single-flight, MCP SSE relay test, combined coverage summary, canonical manifest-count narrative, ACP modes/config persistence guidance; Phase 12 release-candidate hardening; plan 012 — freeze manifest, compatibility matrix, upgrade matrix, packed-install e2e journeys, restart-recovery evidence, capacity envelopes, security policy), exact-peer/install/tarball rules, deterministic resumable publication and publish dry-run, frozen 0.1.x compatibility and support matrix (Node/PostgreSQL/platform/provider/protocol pins and unsupported combinations, machine-checked against `scripts/phase12-freeze-manifest.json`), protected PostgreSQL gate, pinned supply-chain gates, offline tests, the 0.0.15 provider/AI-SDK/RAG/memory protected live-canary matrix, and sandbox-browser Docker/Playwright gates.
132
+ - [Release and install](release-and-install.md): current **0.1.4** 49-package graph (root + 48 workspace packages; plan 016 internal god-module split — `agents.ts`/`contracts.ts` reorganized behind barrel re-exports with a byte-identical public entry surface, measured tree-shaking improvement in `scripts/phase16-baseline.json`, and additive `@arnilo/prism-browser` Chrome DevTools Protocol capabilities — `browser_evaluate`/`browser_observe` and `block_urls`/`unblock_urls`/`throttle`/`emulate` act actions; plan 015 dead-code and deprecation hygiene on the frozen 0.1.x line — parameterized benchmark runner `scripts/benchmark.mjs` absorbing the per-version runners, archived review-coverage evidence in `docs/_evidence/`, non-blocking unused-code sweep `npm run sweep:unused`, opt-in checkpoint persistence for loaded-skill names and read-path sets; plan 014 Alibaba provider enrichment — embeddings, video input, verified compatible-mode surface decision table; plan 013 post-release hardening — build single-flight, MCP SSE relay test, combined coverage summary, canonical manifest-count narrative, ACP modes/config persistence guidance; Phase 12 release-candidate hardening; plan 012 — freeze manifest, compatibility matrix, upgrade matrix, packed-install e2e journeys, restart-recovery evidence, capacity envelopes, security policy), exact-peer/install/tarball rules, deterministic resumable publication and publish dry-run, frozen 0.1.x compatibility and support matrix (Node/PostgreSQL/platform/provider/protocol pins and unsupported combinations, machine-checked against `scripts/phase12-freeze-manifest.json`), protected PostgreSQL gate, pinned supply-chain gates, offline tests, the 0.0.15 provider/AI-SDK/RAG/memory protected live-canary matrix, and sandbox-browser Docker/Playwright gates.
133
133
  - [0.1.0 / 1.0 readiness gates](0.1.0-readiness.md): command-per-gate 1.0 readiness table — frozen API surface + compat gate, migration/docs tripwires, budget table, live-suite matrix, security matrix, current-line status (**0.0.23** published target), signed-publication/live-canary prerequisites for 1.0, and Phase 12 demand-evidence entry criteria.
134
- - [Review coverage (2026-07-26 Phase 11)](review-coverage-2026-07-26-phase-11.md): Plan 079 evidence freeze — baseline size/startup/benchmark budgets, hotspot domain extraction table, confirmed duplication survivors (redactor/cleanJson/row-codecs/checkpoints/exec-runner/approval/ownership), profile adoption recommendations, and tarball artifact-diet findings for 0.0.16.
135
- - [Review coverage (2026-07-26 Phase 10)](review-coverage-2026-07-26-phase-10.md): Plan 078 evidence freeze — OpenAI hosted tools/continuation/realtime, AI SDK version matrix, remaining provider metadata parity, RAG replaceSource/loaders/parsers/reranker/provenance/ingestion-status, memory export/rebuild/conformance, and 0.0.15 (43 → 43 manifests; no new package) release gates.
136
- - [Review coverage (2026-07-25 Phase 9)](review-coverage-2026-07-25-phase-9.md): Plan 077 evidence freeze — conversation service, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth, browser checkpoint composition, and deny-by-default device contracts for 0.0.14 (41 → 43 manifests; only the two provider packages are new).
137
- - [Review coverage (2026-07-23 Phase 8)](review-coverage-2026-07-23-phase-8.md): Plan 076 evidence freeze — enterprise identity/policy/router packages, Azure/Bedrock/Vertex adapters, server deployment seams, persistence lifecycle hooks, and M365/GWS work-connector bounds for 0.0.13.
138
- - [Review coverage (2026-07-22 Phase 7)](review-coverage-2026-07-22-phase-7.md): Plan 075 evidence freeze — AG-UI/ACP package boundary, streamed durable resume, bounded replay/projection, coding compaction preset, and provider-authorized OAuth policy for 0.0.12.
139
- - [Review coverage (2026-07-22 Phase 6)](review-coverage-2026-07-22-phase-6.md): Plan 074 evidence freeze — SessionIndex/search, contextBudget, native Anthropic/Google packages, goal→verify, steer, ask_user_decision (multi/free-text/suspend), finite limits, threats, and 0.0.11 release gates.
140
- - [Review coverage (2026-07-21 Phase 5)](review-coverage-2026-07-21-phase-5.md): Plan 073 evidence freeze — unified workspace modes, primitive ownership, reused finite limits, threats, and 0.0.10 release gates.
141
- - [Review coverage (2026-07-20 Phase 4)](review-coverage-2026-07-20-phase-4.md): Plan 072 evidence freeze — revised coding/browser-only scope, external revisions, primitive ownership, finite limits, threats, and 0.0.9 release gates.
142
- - [Review coverage (2026-07-19 Phase 3)](review-coverage-2026-07-19-phase-3.md): Plan 070 evidence freeze — exact protocol/vendor references, capability/primitive/limit matrices, supported boundaries, and 0.0.8 release evidence.
143
- - [Review coverage (2026-07-17 provider validation)](review-coverage-2026-07-17-provider-validation.md): Plan 067 evidence freeze — P0–P2 re-verification owners, seven first-party provider packages mapped to official-doc URLs, Pi secondary refs, cache/thinking/discovery surfaces, credential canaries, and use-case model-binding inventory.
144
- - [Review coverage (2026-07-15)](review-coverage-2026-07-15.md): frozen 0.0.5 finding/feature ownership, existing-primitive inventory, package decisions, threat boundaries, exclusions, and measured Phase 0 baseline.
145
- - [Review coverage (2026-07-14)](review-coverage-2026-07-14.md): traceability matrix linking review findings and bug-report fixes to plan tasks, tests, and documentation for release 0.0.4.
134
+ - [Review coverage archive](_evidence/): per-phase evidence freezes (plans 067–079, releases 0.0.4–0.0.16) — traceability matrices, provider validation, capability/primitive/limit matrices, benchmark budgets, and artifact-diet findings; tarball-excluded, kept in-repo for audit.
146
135
 
package/docs/migration.md CHANGED
@@ -1,5 +1,11 @@
1
1
  # Migration guide
2
2
 
3
+ ## 0.1.3 → 0.1.4 internal reorganization behind barrel re-exports (no migration)
4
+
5
+ Release **0.1.4** (plan 016) is an **internal file reorganization behind barrel re-exports**: the root `src/agents.ts` and `src/contracts.ts` god-modules were split by concern into sibling modules (`contracts-core` / `contracts-run-state` / `contracts-protocol` behind the `contracts.ts` barrel; `agent-session` / `agent-run-lifecycle` / `agent-approval` / `agent-tool-dispatch` / `agent-run-state` / `agent-loops` / `compaction` behind the `agents.ts` barrel). **Public declaration surface unchanged** — the root entry surface is byte-identical to 0.1.3 (zero added/removed/changed on the public entry; the only union-surface additions are 14 internal cross-module helper exports that are not consumer-importable, see `scripts/compat-baseline/arnilo__prism.txt`). The optional `@arnilo/prism-browser` package extends additively with Chrome DevTools Protocol capabilities (0.1.4): `browser_evaluate`, `browser_observe`, and the `block_urls`/`unblock_urls`/`throttle`/`emulate` act actions on Chromium hosts, plus raw `{ css }`/`{ xpath }` targets — new exports and two optional structural interface members only, zero removals. **Store compatibility: compatible** — no persisted shape, event schema, or default behavior changed (no runtime path changed; the split is declaration-level). no migration step; rollback = restore the 0.1.3 manifests/tag (stores never change). The next line, **0.1.5**, is the documented **breaking cut** (deprecated-option removal); its migration section will list the removed symbols (the public-but-unused export candidates from `scripts/dead-exports.mjs`).
6
+
7
+ Release **0.1.3** (plan 015) is the dead-code and deprecation hygiene patch on the frozen 0.1.x line: benchmark-runner consolidation (one parameterized `scripts/benchmark.mjs --scenario <name>` replaces the per-version runners; 16 orphaned `benchmark-0.0.{8..16}` runner/test files removed, all `benchmark-*.json` evidence kept), the 12 `docs/review-coverage-2026-07-*.md` evidence files archived to the tarball-excluded `docs/_evidence/`, a non-blocking unused-code sweep (`npm run sweep:unused`, always exits 0, report to `scripts/unused-sweep-report.txt`), and opt-in checkpoint persistence (`persistSessionState: true` on durable run/resume options persists the loaded-skill name catalog ≤64 names in the run-state checkpoint and restores it on resume — bodies re-resolve from the live registry; `createReadPathSetPersistence` in `@arnilo/prism-coding-agent` persists the read-before-write path set through the host `CheckpointStore`, ≤1024 paths, ownership-scoped). **Store compatibility: compatible** — the persisted run-state schema stays at version 1 (the optional `sessionState` field is absent by default, so 0.1.2 checkpoints parse unchanged and opt-out checkpoints are byte-identical); no upgrade or rollback step exists (rollback = restore the 0.1.2 manifests/tag; stores never change). Declaration surface is additive-only vs the frozen 0.1.x contract (`scripts/compat-baseline` regenerated at 0.1.3 with zero breaking deltas, enforced by `node scripts/release.mjs gate`). No breaking defaults.
8
+
3
9
  ## 0.1.0 → 0.1.1 post-release hardening (additive, no migration)
4
10
 
5
11
  Release **0.1.1** (plan 013) is a hardening patch on the frozen 0.1.x line: five scoped fixes — build single-flight (`npm run clean` removed from `npm run build`, standalone), deterministic MCP SSE relay test (`relayStatelessBody` internal export in `@arnilo/prism-mcp`, not in the package entry surface), combined core + workspace coverage summary (`scripts/coverage-summary.mjs`), canonical manifest-count narrative (49 publishable manifests = root + 48 workspace packages), and ACP modes/config ownership-scoped persistence guidance (the agent never persists `modeId`/`configValues`; host stores MUST key by `sessions.ownership`). **Store compatibility: compatible** — no persisted shape, event schema, or default behavior changed; the 0.0.28 → 0.1.0 → 0.1.1 lines all stay on the same checksum-protected contract, so no upgrade or rollback step exists (rollback = restore the 0.1.0 manifests/tag; stores never change). Declaration surface is additive-only vs the frozen 0.1.x contract (`scripts/compat-baseline` regenerated at 0.1.1 with zero breaking deltas, enforced by `node scripts/release.mjs gate`). No breaking defaults.
@@ -261,7 +267,7 @@ Prism 0.0.6 preserves documented 0.0.3 agent construction except for two intenti
261
267
 
262
268
  ## 0.0.15 → 0.0.16 simplification, shared survivors, and release gates (additive, pre-release)
263
269
 
264
- Release **0.0.16** is a simplification/readiness release: no runtime behavior changes, no package retired, and the only public-surface change is one additive export plus one internal package. The published root tarball is smaller and the release now runs offline pre-publish gates. See [Phase 11 evidence](review-coverage-2026-07-26-phase-11.md).
270
+ Release **0.0.16** is a simplification/readiness release: no runtime behavior changes, no package retired, and the only public-surface change is one additive export plus one internal package. The published root tarball is smaller and the release now runs offline pre-publish gates. See [Phase 11 evidence](_evidence/review-coverage-2026-07-26-phase-11.md).
265
271
 
266
272
  ### New shared export: `resolveRedactor` (additive)
267
273
 
@@ -318,7 +324,7 @@ RAG retrieval now optionally accepts host-owned `Reranker`; it receives redacted
318
324
 
319
325
  ## 0.0.13 → 0.0.14 personal/work-agent conversations, co-work review, and channel/device gates (additive, pre-release)
320
326
 
321
- Release **0.0.14** is strictly additive: every surface extends a shipped package and reuses the AG-UI adapter shipped in 0.0.12. The only new packages are two optional provider adapters (41 → 43 manifests): `@arnilo/prism-provider-alibaba` and `@arnilo/prism-provider-ollama`, both enrolled via the `@arnilo/prism-providers` family. No permission broadening — channel/device/co-work features cannot widen consent, memory, network, file, browser, connector, or tool permissions (roadmap gate 8). See [Phase 9 evidence](review-coverage-2026-07-25-phase-9.md).
327
+ Release **0.0.14** is strictly additive: every surface extends a shipped package and reuses the AG-UI adapter shipped in 0.0.12. The only new packages are two optional provider adapters (41 → 43 manifests): `@arnilo/prism-provider-alibaba` and `@arnilo/prism-provider-ollama`, both enrolled via the `@arnilo/prism-providers` family. No permission broadening — channel/device/co-work features cannot widen consent, memory, network, file, browser, connector, or tool permissions (roadmap gate 8). See [Phase 9 evidence](_evidence/review-coverage-2026-07-25-phase-9.md).
322
328
 
323
329
  | Surface | Before (0.0.13) | After (0.0.14) |
324
330
  | --- | --- | --- |
@@ -350,7 +356,7 @@ Optional `@arnilo/prism-policy` records allow/deny/modify/approval decisions wit
350
356
  | Model governance | Host wraps resolver ad hoc | Optional `@arnilo/prism-model-router` before provider I/O |
351
357
  | Work connectors | n/a | Optional `@arnilo/prism-work-tools` M365 + GWS; draft-then-approve; hard-coded CLI argv |
352
358
 
353
- **Deferred to 0.0.14+:** conversation storage/service, Studio/control plane, internal auth DB, Redis/SQS queue adapters, local Office binaries. See [Phase 8 evidence](review-coverage-2026-07-23-phase-8.md).
359
+ **Deferred to 0.0.14+:** conversation storage/service, Studio/control plane, internal auth DB, Redis/SQS queue adapters, local Office binaries. See [Phase 8 evidence](_evidence/review-coverage-2026-07-23-phase-8.md).
354
360
 
355
361
  Benchmark placeholder: `node scripts/benchmark-0.0.13.mjs` (release Task 10). Caps documented in [Performance limits](performance.md).
356
362
 
@@ -378,7 +384,7 @@ Release **0.0.12** adds optional `@arnilo/prism-ag-ui` (root AG-UI and stable `.
378
384
 
379
385
  **Host actions:** install the optional package only when a frontend protocol is needed; keep authorization, session/thread/run mapping, durable correlation, storage, redaction, and projection in the host. Reject frontend tools and state unless an explicit host policy accepts them. For a durable approval, persist protocol-run correlation before exposing the exact `${runId}:${version}` interrupt, then resume through the lifecycle with current ownership/version. Configure a redacted `ProductionPersistenceStore` before enabling replay. Use `createCodingCompactionStrategy()` only when the host already supplies a summary provider/model.
380
386
 
381
- AG-UI defaults/hard caps: request 64 KiB/1 MiB; projected event 64 KiB/1 MiB; replay page 100/500; subscriber queue 128/4096; stream 10k/100k events and 10/64 MiB; wall time 120 seconds/30 minutes. Benchmark results remain a release-gate placeholder: `node scripts/benchmark-0.0.12.mjs` lands in Task 8. See [Frontend interoperability](ag-ui.md), [LLM compaction package](compaction-llm.md), and [Phase 7 evidence](review-coverage-2026-07-22-phase-7.md).
387
+ AG-UI defaults/hard caps: request 64 KiB/1 MiB; projected event 64 KiB/1 MiB; replay page 100/500; subscriber queue 128/4096; stream 10k/100k events and 10/64 MiB; wall time 120 seconds/30 minutes. Benchmark results remain a release-gate placeholder: `node scripts/benchmark-0.0.12.mjs` lands in Task 8. See [Frontend interoperability](ag-ui.md), [LLM compaction package](compaction-llm.md), and [Phase 7 evidence](_evidence/review-coverage-2026-07-22-phase-7.md).
382
388
 
383
389
  ## 0.0.10 → 0.0.11 coding harness fundamentals (additive)
384
390
 
@@ -394,7 +400,7 @@ Release **0.0.11** adds SessionIndex/search, assembler `contextBudget`, native A
394
400
  | Ask user | n/a | Opt-in `createAskUserDecisionTool`; durable `suspendAskUserDecision` (no new agent interruption kinds) |
395
401
  | Structured output + tools | Native schema attached every GVR provider turn | Opt-in `structuredOutputTiming: "final-turn-only"` (default `"every-turn"`): tool-eligible turns omit schema; artifact/revision turns schema-on / tools-off |
396
402
 
397
- **Host actions:** reopen SQLite/Postgres stores so migration 004 applies; set `metadata.workspaceRoot` when filtering by workspace; wire Anthropic/Google packages explicitly; do not expect JSONL search. Benchmarks: `scripts/benchmark-0.0.11.mjs` (lands with release Task 13). See [Phase 6 evidence](review-coverage-2026-07-22-phase-6.md).
403
+ **Host actions:** reopen SQLite/Postgres stores so migration 004 applies; set `metadata.workspaceRoot` when filtering by workspace; wire Anthropic/Google packages explicitly; do not expect JSONL search. Benchmarks: `scripts/benchmark-0.0.11.mjs` (lands with release Task 13). See [Phase 6 evidence](_evidence/review-coverage-2026-07-22-phase-6.md).
398
404
 
399
405
  ## 0.0.9 / 0.0.96 → 0.0.10 coding workspace modes (breaking composition)
400
406
 
@@ -28,6 +28,16 @@ node scripts/benchmark-0.1.0.mjs --out scripts/benchmark-0.1.0.json
28
28
  PRISM_TEST_POSTGRES_URL="postgresql://…" node scripts/benchmark-0.1.0.mjs --out scripts/benchmark-0.1.0.json # adds protected legs
29
29
  ```
30
30
 
31
+ ## 0.1.4 tree-shake measurement (static-reachability proxy)
32
+
33
+ The 0.1.4 god-module split (agents/contracts → per-concern modules behind barrels) is
34
+ measured by `scripts/phase16-tree-shake.mjs`: `dist/agents.js`/`dist/contracts.js` byte
35
+ sizes, `dist/*.js` module count, and a static-import reachability count from the minimal
36
+ entry proxy, recorded in `scripts/phase16-baseline.json` (task 3 of plan 016). Static
37
+ graph reachability is an upper-bound proxy, not a real bundle — a byte-accurate
38
+ tree-shake budget needs a bundler and actual consumer code, deferred behind a 0.1.7 DX
39
+ demand gate.
40
+
31
41
  **Pass/fail thresholds.** Network-free rows fail above the frozen ceiling in
32
42
  the table below; protected PostgreSQL rows fail above their per-phase
33
43
  budgets.json ceilings (50/100 ms per the approved budget contract); startup
@@ -165,7 +175,7 @@ The same run accepted 1,000 rate claims, accumulated 16,000 budget tokens, grant
165
175
  Release 0.0.16 is a simplification/readiness release: it added no performance-affecting code, so the six network-free scenario medians are held at the 0.0.15 baseline and the win is a smaller published artifact. Budgets live in `scripts/budgets.json` (measured baselines + tolerance) and are enforced two ways:
166
176
 
167
177
  - **Fast gate (every `npm test`)** — `scripts/budget-gate.test.mjs` re-packs the root tarball (`npm pack --dry-run --json`) and fails if packed bytes, unpacked bytes, or file count exceed baseline + 5%, and fails if cold-process `import('./dist/index.js')` exceeds the 250 ms sanity ceiling. Negative fixtures prove an inflated/regressed value fails.
168
- - **Release evidence runner** — `node scripts/benchmark-0.0.16.mjs` re-measures root pack + startup, spawns `benchmark-0.0.15.mjs` for the six scenario medians (reused unchanged), compares every value to `budgets.json` (throughput floor / latency ceiling at ±25%), prints the evidence report below, and exits non-zero on any regression.
178
+ - **Release evidence runner** — `node scripts/benchmark-0.0.16.mjs` re-measures root pack + startup, spawns `benchmark-0.0.15.mjs` for the six scenario medians (reused unchanged), compares every value to `budgets.json` (throughput floor / latency ceiling at ±25%), prints the evidence report below, and exits non-zero on any regression. *(0.1.3, plan 015 Task 1: the per-version runners were consolidated into the parameterized runner `scripts/benchmark.mjs --scenario <name>`; the 0.0.16 evidence below is the historical record, budgets.json medians unchanged.)*
169
179
 
170
180
  **Artifact diet (the 0.0.16 finding).** The Task 1 tarball deny list dropped the historical `docs/review-coverage-*.md` (11 files, 283,022 bytes) from the root package: the root tarball went from **659,478 packed / 2,310,686 unpacked / 281 files** (0.0.15) to a budgeted **≈575,680 packed / 2,043,402 unpacked / 270 files**. The per-release `scripts/benchmark-0.0.*.mjs` history never shipped in artifacts (root `files` is `dist`/`docs`/`templates`/`CHANGELOG.md` only — zero `scripts/` entries packed), so no archive move was needed; `benchmark-0.0.16.mjs` consolidates the current evidence behind one budget-gating runner.
171
181
 
@@ -248,7 +258,7 @@ No network, credentials, provider summary call, durable database, or live subscr
248
258
 
249
259
  ## Release 0.0.11 session search / context budget / steer caps
250
260
 
251
- Finite caps (defaults / hard) — full matrix in [Phase 6 evidence](review-coverage-2026-07-22-phase-6.md):
261
+ Finite caps (defaults / hard) — full matrix in [Phase 6 evidence](_evidence/review-coverage-2026-07-22-phase-6.md):
252
262
 
253
263
  | Resource | Default / hard |
254
264
  | --- | --- |
@@ -315,7 +325,7 @@ Durable coding plan/checkpoint defaults/hard caps: plan Markdown 256 KiB/1 MiB;
315
325
 
316
326
  Browser automation defaults/hard caps from `@arnilo/prism-browser`: pages 4/16; actions 100/256; queued actions 16/64; snapshot refs 2,000/10,000; depth 30/100; snapshot bytes 256 KiB/2 MiB; navigation 30 s/120 s; action 10 s/60 s; wait 30 s/120 s; run wall 20 min/30 min; popups 4/16; dialogs 16/64; listeners 64/256; action input 64 KiB/256 KiB; close grace 5 s/30 s; network requests 1,000/10,000 with 10/32 redirects per request and 8/32 WebSockets; screenshots 16/64 with 16/64 megapixels and 10 MiB/32 MiB encoded; uploads 8/32 files, 16 MiB/64 MiB each, 64 MiB/256 MiB aggregate; downloads 8/32 files, 32 MiB/256 MiB each, 64 MiB/512 MiB aggregate. Caps charge before context/page/action/queue/snapshot/network/artifact retention. Host supplies Playwright and egress proxy attestation; package import launches nothing.
317
327
 
318
- 0.0.14 co-work defaults/hard caps (frozen in [Phase 9 evidence](review-coverage-2026-07-25-phase-9.md)): conversation thread list pages 50/200, active branches per thread 16/64, replay/export page 100/500 events; artifact revisions per artifact 32/128, artifacts per thread 64/256, metadata record 8/64 KiB, preview 16/64 KiB, citations 32/128 (2/8 KiB each), delivery-link TTL 5 min/24 h, delivery token 4/16 KiB, compare exactly 2 revisions; memory retention batch 500/5000; proactive capability TTL 24 h/31 d, capability token record 16 KiB; browser checkpoint URL 8 KiB/16 KiB, domain-state hash 256 B/1 KiB, host-data ref 2 KiB/8 KiB, 16/64 checkpoints per run; device stream chunk 1 MiB/8 MiB, concurrent device sessions per identity 1/4 (device wall/turns/tool calls consume shared `RunLimits`). All caps charge before persist/emit and fail closed on overflow. Benchmark placeholder: `node scripts/benchmark-0.0.14.mjs` (release Task 12) reports conversation replay, memory injection/consent, artifact revision/delivery, AG-UI co-work mapping, and connector refresh overhead against these budgets.
328
+ 0.0.14 co-work defaults/hard caps (frozen in [Phase 9 evidence](_evidence/review-coverage-2026-07-25-phase-9.md)): conversation thread list pages 50/200, active branches per thread 16/64, replay/export page 100/500 events; artifact revisions per artifact 32/128, artifacts per thread 64/256, metadata record 8/64 KiB, preview 16/64 KiB, citations 32/128 (2/8 KiB each), delivery-link TTL 5 min/24 h, delivery token 4/16 KiB, compare exactly 2 revisions; memory retention batch 500/5000; proactive capability TTL 24 h/31 d, capability token record 16 KiB; browser checkpoint URL 8 KiB/16 KiB, domain-state hash 256 B/1 KiB, host-data ref 2 KiB/8 KiB, 16/64 checkpoints per run; device stream chunk 1 MiB/8 MiB, concurrent device sessions per identity 1/4 (device wall/turns/tool calls consume shared `RunLimits`). All caps charge before persist/emit and fail closed on overflow. Benchmark placeholder: `node scripts/benchmark-0.0.14.mjs` (release Task 12) reports conversation replay, memory injection/consent, artifact revision/delivery, AG-UI co-work mapping, and connector refresh overhead against these budgets.
319
329
 
320
330
  Current surfaces:
321
331
 
@@ -499,7 +509,7 @@ Repository size at the same commit, counted from `src/` and `packages/` while ex
499
509
 
500
510
  Prism has no project generator before Phase 5, so a generated-Prism-project install/build size is **not applicable** at this baseline. The closest current install figure is the 72 MiB development workspace; it is not a scaffold target. The comparison Mastra default scaffold measured during the review used 439 MB `node_modules`, 300 MB build output, and 427 installed packages. Phase 5 must establish a real generated Prism project baseline and keep unselected storage, telemetry, eval, memory, server, and workflow dependencies absent.
501
511
 
502
- See [Review coverage — 2026-07-15](review-coverage-2026-07-15.md) for scope, primitive, package, and threat-boundary ownership.
512
+ See [Review coverage — 2026-07-15](_evidence/review-coverage-2026-07-15.md) for scope, primitive, package, and threat-boundary ownership.
503
513
 
504
514
  ### 0.0.5 Phase 2 verification (2026-07-15)
505
515
 
@@ -300,4 +300,4 @@ See **Threat model summary** and **Performance notes** above. Cross-cutting rule
300
300
  - [Resource loading](resource-loading.md): `ResourceLoader` decode helpers
301
301
  - [Model registry](model-registry.md): `ModelCapabilities` metadata
302
302
  - [Provider conformance](provider-conformance.md): content preservation and secret leak checks
303
- - [Review coverage (2026-07-14)](review-coverage-2026-07-14.md): traceability for C-005, C-010, C-011
303
+ - [Review coverage (2026-07-14)](_evidence/review-coverage-2026-07-14.md): traceability for C-005, C-010, C-011
@@ -273,7 +273,7 @@ Every migrated provider must pass this shared matrix (implemented in Task 1 test
273
273
  - [OpenAI-compatible provider](providers/openai-compatible.md): reference adapter subpath
274
274
  - [Structured output](structured-output.md): artifact loop fallback
275
275
  - [Provider request policies](provider-request-policies.md): cache and request hooks
276
- - [Review coverage (2026-07-14)](review-coverage-2026-07-14.md): finding → plan traceability
276
+ - [Review coverage (2026-07-14)](_evidence/review-coverage-2026-07-14.md): finding → plan traceability
277
277
 
278
278
  ## Task ownership map
279
279
 
@@ -417,6 +417,8 @@ void credentials;
417
417
 
418
418
  ## Extension and configuration notes
419
419
 
420
+ **Internal file structure (0.1.4).** Since 0.1.4 the contract declarations are spread across sibling modules behind the `src/contracts.ts` barrel: `src/contracts-core.ts` (JSON/content/core agent contracts plus the `SESSION_ENTRY_KINDS`/`session-store` values), `src/contracts-run-state.ts` (run-state and agent-session contracts including `AgentRunStatus` through `AgentSession`, decision/steer constants, and the error classes), and `src/contracts-protocol.ts` (pure protocol-payload types). The 295-name public surface is unchanged; this page groups the frozen contract by functionality rather than source file.
421
+
420
422
  - Contracts are host-owned and package-friendly. External packages can implement `AIProvider`, `ToolDefinition`, `CommandDefinition`, `AgentDefinition`, `InputBuilder`, `PromptBuilder`, `Middleware`, `ContextProvider`, `Skill`, `Extension`, config providers, data-only manifests, compaction strategies, store factories, resource loaders, settings providers, and credential resolvers.
421
423
  - `ExtensionAPI` is implemented by the extension kernel. It exposes explicit registries, ordered middleware registration, ordered event subscription/emission, and registration methods for Phase 2 contribution categories.
422
424
  - `AgentConfig.provider` can hold a direct provider instance for simple host wiring. Hosts that need config-driven selection should use `ModelConfig.provider` with explicit `createProviderRegistry()` / `createModelRegistry()` objects; Prism does not create a hidden global provider registry.