@arnilo/prism 0.0.8 → 0.0.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -3
- package/dist/agent-loops.d.ts +1 -0
- package/dist/agent-loops.js +44 -5
- package/dist/agents.js +141 -7
- package/dist/context-budget.d.ts +63 -0
- package/dist/context-budget.js +235 -0
- package/dist/contracts.d.ts +107 -0
- package/dist/contracts.js +77 -0
- package/dist/index.d.ts +7 -5
- package/dist/index.js +5 -4
- package/dist/input.d.ts +3 -0
- package/dist/input.js +71 -28
- package/dist/node/session-store-jsonl.js +4 -1
- package/dist/provider-events.d.ts +2 -0
- package/dist/provider-events.js +21 -13
- package/dist/providers/openai-compatible.js +8 -5
- package/dist/providers/transport.d.ts +10 -1
- package/dist/providers/transport.js +24 -8
- package/dist/rpc.js +13 -2
- package/dist/session-stores.d.ts +7 -2
- package/dist/session-stores.js +174 -4
- package/dist/structured-output.d.ts +5 -1
- package/dist/structured-output.js +18 -0
- package/dist/testing/persistence-schema.d.ts +1 -1
- package/dist/testing/persistence-schema.js +8 -2
- package/dist/testing/session-store-conformance.d.ts +6 -0
- package/dist/testing/session-store-conformance.js +36 -1
- package/dist/tools.js +2 -0
- package/docs/agent-events.md +3 -2
- package/docs/agent-loops.md +10 -3
- package/docs/agent-session-runtime.md +5 -1
- package/docs/browser-automation.md +124 -0
- package/docs/cli-rpc.md +2 -1
- package/docs/coding-agent-tools.md +178 -14
- package/docs/coding-security.md +84 -11
- package/docs/evaluations.md +13 -2
- package/docs/guardrails.md +2 -1
- package/docs/host-security.md +4 -2
- package/docs/index.md +19 -15
- package/docs/input-and-prompt-assembly.md +4 -1
- package/docs/migration.md +84 -0
- package/docs/node-jsonl-session-store.md +1 -1
- package/docs/performance.md +46 -0
- package/docs/postgres-persistence.md +3 -3
- package/docs/provider-conformance.md +1 -1
- package/docs/provider-packages.md +1 -1
- package/docs/provider-primitives.md +7 -1
- package/docs/providers/anthropic.md +92 -0
- package/docs/providers/google.md +87 -0
- package/docs/public-contracts.md +4 -0
- package/docs/release-and-install.md +197 -64
- package/docs/review-coverage-2026-07-20-phase-4.md +175 -0
- package/docs/review-coverage-2026-07-21-phase-5.md +172 -0
- package/docs/review-coverage-2026-07-22-phase-6.md +209 -0
- package/docs/session-store-conformance.md +2 -0
- package/docs/session-stores.md +40 -1
- package/docs/sqlite-persistence.md +3 -3
- package/docs/structured-output.md +9 -3
- package/docs/tools.md +3 -0
- package/docs/web-tools.md +1 -1
- package/docs/workflows.md +3 -0
- package/package.json +6 -5
|
@@ -4,7 +4,7 @@ import { createHash } from "node:crypto";
|
|
|
4
4
|
// this module defines the shared table/index/pagination/migration expectations
|
|
5
5
|
// adapter authors implement and test against before shipping dialect-specific DDL.
|
|
6
6
|
/** Current shared persistence schema version for production database adapters. */
|
|
7
|
-
export const PERSISTENCE_SCHEMA_VERSION =
|
|
7
|
+
export const PERSISTENCE_SCHEMA_VERSION = 4;
|
|
8
8
|
/** Guidance adapters must follow: values are bound parameters, never interpolated. */
|
|
9
9
|
export const PARAMETERIZED_QUERY_GUIDANCE = "Bind every user-supplied value (session ids, idempotency keys, tenant ids, timestamps, JSON payloads) as a query parameter. Quote/validate schema and table identifiers only; never interpolate untrusted strings into SQL text.";
|
|
10
10
|
const TENANT_COLUMNS = [
|
|
@@ -315,7 +315,12 @@ function migrationStep(version, name, description) {
|
|
|
315
315
|
}
|
|
316
316
|
: version === 2
|
|
317
317
|
? { table: "prism_usage", columns: ["scope", "turn", "attempt"], indexes: ["prism_usage_session_scope_recorded_idx"] }
|
|
318
|
-
:
|
|
318
|
+
: version === 3
|
|
319
|
+
? { tables: ["prism_run_feedback"], indexes: model.indexes.filter((index) => index.name.startsWith("prism_run_feedback_")).map((index) => index.name) }
|
|
320
|
+
: version === 4
|
|
321
|
+
// Adapter-local FTS objects (SQLite FTS5 / Postgres tsvector) map to this canonical name.
|
|
322
|
+
? { search: ["prism_session_search"], indexes: ["prism_sessions_updated_id_idx"] }
|
|
323
|
+
: (() => { throw new Error(`Unknown migration version ${version}`); })();
|
|
319
324
|
return {
|
|
320
325
|
version,
|
|
321
326
|
name,
|
|
@@ -332,6 +337,7 @@ export function createPersistenceMigrationContract() {
|
|
|
332
337
|
migrationStep(1, "001_init", "Create core session, branch, entry, idempotency, run, ledger, and migration tables."),
|
|
333
338
|
migrationStep(2, "002_usage_scope", "Distinguish provider-turn usage from aggregate run totals."),
|
|
334
339
|
migrationStep(3, "003_run_feedback", "Add immutable ownership-scoped run/trace feedback and evaluation links."),
|
|
340
|
+
migrationStep(4, "004_session_search", "Add bounded session search indexes and adapter-local FTS objects."),
|
|
335
341
|
],
|
|
336
342
|
lockGuidance: "Acquire a dialect-specific migration lock before applying steps (PostgreSQL advisory lock; SQLite exclusive transaction). Only one process should migrate at a time.",
|
|
337
343
|
leastPrivilegeGuidance: "Run migrations with a DDL-capable role; use a separate least-privilege runtime role limited to INSERT/SELECT/UPDATE on adapter tables. Never grant migration credentials to the agent runtime.",
|
|
@@ -10,6 +10,12 @@ export interface SessionStoreConformanceOptions {
|
|
|
10
10
|
* when the store does not implement `readBranchPath`.
|
|
11
11
|
*/
|
|
12
12
|
readonly exerciseReadBranchPath?: boolean;
|
|
13
|
+
/**
|
|
14
|
+
* When true, exercises optional `searchSessions` (empty page, limit cap,
|
|
15
|
+
* invalid limit/query rejection via `resolveSessionSearchQuery` semantics).
|
|
16
|
+
* Skipped when the store does not implement `searchSessions`.
|
|
17
|
+
*/
|
|
18
|
+
readonly exerciseSearchSessions?: boolean;
|
|
13
19
|
/** When true, appends concurrent children of the same parent (fork allowed). */
|
|
14
20
|
readonly exerciseConcurrentParentAppend?: boolean;
|
|
15
21
|
/**
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
// stores already satisfy. Mirrors the assertion shape already repeated across
|
|
6
6
|
// src/__tests__/session-stores.test.ts and node-session-store-jsonl.test.ts so
|
|
7
7
|
// adapter authors do not re-derive them. Throws plain Error; no test runner.
|
|
8
|
-
import { isSessionAppendConflict } from "../contracts.js";
|
|
8
|
+
import { HARD_MAX_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, isSessionAppendConflict, resolveSessionSearchQuery, } from "../contracts.js";
|
|
9
9
|
/**
|
|
10
10
|
* Assert that a `SessionStore` implementation satisfies the core adapter
|
|
11
11
|
* contract: round-trip append/list, duplicate-entry-id rejection,
|
|
@@ -63,6 +63,9 @@ export async function assertSessionStoreConforms(store, options = {}) {
|
|
|
63
63
|
throw new Error(`readBranchPath must return the ancestor chain root→leaf in order; got ${JSON.stringify(ids)}`);
|
|
64
64
|
}
|
|
65
65
|
}
|
|
66
|
+
if (options.exerciseSearchSessions && typeof store.searchSessions === "function") {
|
|
67
|
+
await assertSessionStoreSearchSessions(store);
|
|
68
|
+
}
|
|
66
69
|
await assertSessionStoreBranchIsolation(store, options);
|
|
67
70
|
if (options.exerciseConcurrentParentAppend) {
|
|
68
71
|
await assertConcurrentParentAppendAllowed(store, sessionId, now, make);
|
|
@@ -104,6 +107,38 @@ export async function runSessionStoreConformance(factory, options = {}) {
|
|
|
104
107
|
label: "dup",
|
|
105
108
|
}, { idempotencyKey: "reopen-idem", expectedParentId: parent?.id }), (error) => isSessionAppendConflict(error) && error.conflict.idempotencyDuplicate === true, "Restarted store must still deduplicate an exact idempotency retry");
|
|
106
109
|
}
|
|
110
|
+
async function assertSessionStoreSearchSessions(store) {
|
|
111
|
+
const search = store.searchSessions;
|
|
112
|
+
await reject(() => search({ limit: 0 }), (error) => error instanceof TypeError, "searchSessions must reject non-positive limit");
|
|
113
|
+
await reject(() => search({ limit: Number.NaN }), (error) => error instanceof TypeError, "searchSessions must reject NaN limit");
|
|
114
|
+
await reject(() => search({ limit: HARD_MAX_SESSION_SEARCH_LIMIT + 1 }), (error) => error instanceof TypeError, "searchSessions must reject oversize limit");
|
|
115
|
+
await reject(() => search({ query: "x".repeat(HARD_MAX_SESSION_SEARCH_QUERY_BYTES + 1) }), (error) => error instanceof TypeError, "searchSessions must reject oversize query string");
|
|
116
|
+
const empty = await search(resolveSessionSearchQuery({
|
|
117
|
+
workspaceRoot: "__prism_conformance_empty__",
|
|
118
|
+
limit: 5,
|
|
119
|
+
}));
|
|
120
|
+
if (!Array.isArray(empty.items)) {
|
|
121
|
+
throw new Error("searchSessions must return a PersistencePage with an items array");
|
|
122
|
+
}
|
|
123
|
+
const resolved = resolveSessionSearchQuery({ limit: 1 });
|
|
124
|
+
const page = await search(resolved);
|
|
125
|
+
if (page.items.length > resolved.limit) {
|
|
126
|
+
throw new Error(`searchSessions must honor limit; got ${page.items.length} items for limit ${resolved.limit}`);
|
|
127
|
+
}
|
|
128
|
+
for (const hit of page.items) {
|
|
129
|
+
if (typeof hit.sessionId !== "string" || !hit.sessionId) {
|
|
130
|
+
throw new Error("searchSessions hits must include a non-empty sessionId");
|
|
131
|
+
}
|
|
132
|
+
if ("credential" in hit || "apiKey" in hit || "password" in hit) {
|
|
133
|
+
throw new Error("searchSessions hits must not include credential fields");
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
const unscoped = await search({ limit: 100 });
|
|
137
|
+
const owned = await search({ tenantId: "__missing_tenant__", limit: 100 });
|
|
138
|
+
if (owned.items.length > unscoped.items.length) {
|
|
139
|
+
throw new Error("searchSessions ownership filter must not return more hits than an unscoped search");
|
|
140
|
+
}
|
|
141
|
+
}
|
|
107
142
|
async function assertSessionStoreBranchIsolation(store, options) {
|
|
108
143
|
const sessionId = options.sessionId ?? "conformance";
|
|
109
144
|
const otherSessionId = options.otherSessionId ?? `${sessionId}-other`;
|
package/dist/tools.js
CHANGED
|
@@ -176,6 +176,8 @@ async function checkCall(call, options, startedAt) {
|
|
|
176
176
|
return blocked(call, context, "unknown_tool", { message: `Unknown tool: ${call.name}` }, options, startedAt);
|
|
177
177
|
if (filterTools([tool], options.filter).length === 0)
|
|
178
178
|
return blocked(call, context, "tool_denied", { message: `Tool denied: ${call.name}` }, options, startedAt);
|
|
179
|
+
if (call.argumentsError)
|
|
180
|
+
return blocked(call, context, "invalid_arguments", call.argumentsError, options, startedAt);
|
|
179
181
|
if (!isJsonObject(call.arguments))
|
|
180
182
|
return blocked(call, context, "invalid_arguments", { message: "Tool arguments must be a JSON object" }, options, startedAt);
|
|
181
183
|
return undefined;
|
package/docs/agent-events.md
CHANGED
|
@@ -67,7 +67,7 @@ Agent / turn / message events:
|
|
|
67
67
|
| `message_started` / `message_finished` | `sessionId`, `runId`, `message: Message` |
|
|
68
68
|
| `message_delta` | `sessionId`, `runId`, `content: ContentBlock` (`tool_call_delta` fragments may appear here for live UI streaming; stored messages use final `tool_call` blocks) |
|
|
69
69
|
|
|
70
|
-
`message_delta.content.type === "tool_call_delta"` carries `{ index, id?, name?, argumentsText? }`. Treat it as a streaming fragment. The runtime reconstructs and persists a final `tool_call` before executing tools.
|
|
70
|
+
`message_delta.content.type === "tool_call_delta"` carries `{ index, id?, name?, argumentsText? }`. Treat it as a streaming fragment. The runtime reconstructs and persists a final `tool_call` before executing tools. Deltas missing `id`/`name` at stream end fail the provider turn with `ErrorInfo.code: "incomplete_delta"` (typed `ProviderTransportError`); they never throw a bare `Error`. Malformed JSON with id+name present recovers as a blocked tool result (`invalid_json_arguments`) instead.
|
|
71
71
|
|
|
72
72
|
Tool execution events:
|
|
73
73
|
|
|
@@ -127,7 +127,8 @@ turn_started → message_started → message_delta* → message_finished → tur
|
|
|
127
127
|
With opt-in `toolCalls: "bounded"`, a provider turn containing calls emits its normal assistant envelope followed by existing `tool_execution_*` events and matching persisted tool results; it emits no validation event and the next provider turn consumes that transcript. A post-`maxToolRounds` call emits terminal `artifact_failed` directly after `turn_finished` and has no tool execution event.
|
|
128
128
|
|
|
129
129
|
- `attempt` is 1-indexed per call-free validation candidate. It can differ from provider `turn` when bounded tool calls occur.
|
|
130
|
-
-
|
|
130
|
+
- Empty/whitespace-only call-free text (including thinking-only content) emits `artifact_validation_*` with `metadata.reason: "parse_error"` before any host parser runs.
|
|
131
|
+
- Single-shot runs emit zero artifact events. Session runs with `generate-validate-revise` require `artifact_finished` to resolve `succeeded`.
|
|
131
132
|
- **Validation failure triggering a revision is recoverable and never an `error`.** Terminal candidate-budget or `tool_round_limit` exhaustion emits `artifact_failed`; real failures remain on the `error` channel.
|
|
132
133
|
|
|
133
134
|
## Request/response example
|
package/docs/agent-loops.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
Agent loops make the agent's per-run turn-control flow a replaceable strategy without forking the runtime. The runtime owns provider calls, retry, abort, store appends, redaction, and event emission; a loop only orchestrates those shared primitives through a `LoopContext`. The default `singleShotLoop` is the former inline turn loop extracted verbatim — assemble → generate → append assistant message → optional tool dispatch → next turn. `generateValidateReviseLoop` is the first alternative loop: generate → parse → validate → revise up to a budget.
|
|
5
|
+
Agent loops make the agent's per-run turn-control flow a replaceable strategy without forking the runtime. The runtime owns provider calls, retry, abort, store appends, redaction, and event emission; a loop only orchestrates those shared primitives through a `LoopContext`. The default `singleShotLoop` is the former inline turn loop extracted verbatim — optional drain pending steers → assemble → generate → append assistant message → optional tool dispatch → next turn (continue while steers remain even if the provider returned no tool calls). `generateValidateReviseLoop` is the first alternative loop: generate → parse → validate → revise up to a budget.
|
|
6
6
|
|
|
7
7
|
Loops are opt-in. When no `loop` is configured, the runtime runs `singleShotLoop` and behavior is bit-for-bit with the pre-loop runtime.
|
|
8
8
|
|
|
@@ -58,6 +58,7 @@ await session.run(input, {
|
|
|
58
58
|
repairer: hostRepairer, // optional; default stringifies validation.errors[].message
|
|
59
59
|
maxRevisions: 3, // optional; default 3
|
|
60
60
|
toolCalls: "bounded", // optional; default "disabled"; uses limits.maxToolRounds
|
|
61
|
+
structuredOutputTiming: "final-turn-only", // optional; default "every-turn"
|
|
61
62
|
},
|
|
62
63
|
});
|
|
63
64
|
|
|
@@ -82,6 +83,10 @@ type AgentLoopOptions =
|
|
|
82
83
|
readonly maxRevisions?: number;
|
|
83
84
|
/** Default "disabled". "bounded" dispatches sequentially up to limits.maxToolRounds. */
|
|
84
85
|
readonly toolCalls?: "disabled" | "bounded";
|
|
86
|
+
readonly structuredOutput?: StructuredOutputOptions;
|
|
87
|
+
readonly structuredOutputMode?: "native" | "artifact-loop";
|
|
88
|
+
/** Default "every-turn". "final-turn-only" omits schema while tools may run. */
|
|
89
|
+
readonly structuredOutputTiming?: "every-turn" | "final-turn-only";
|
|
85
90
|
};
|
|
86
91
|
```
|
|
87
92
|
|
|
@@ -89,13 +94,15 @@ Host callback contracts (all generic over host `T`):
|
|
|
89
94
|
|
|
90
95
|
| Contract | Shape |
|
|
91
96
|
| --- | --- |
|
|
92
|
-
| `ArtifactParser<T>` | `(text: string, ctx: ArtifactContext) => ArtifactParseResult<T> \| Promise<...>` — parse assistant text to a typed value. A parse failure (`ok: false` or missing `value`) consumes revision budget exactly like a validation failure: the repairer receives `value: undefined` plus a synthetic failure (`errors[0].message` = the parse error, `metadata.reason: "parse_error"`), and budget exhaustion ends with terminal `artifact_failed`. |
|
|
97
|
+
| `ArtifactParser<T>` | `(text: string, ctx: ArtifactContext) => ArtifactParseResult<T> \| Promise<...>` — parse assistant text to a typed value. Empty/whitespace-only call-free text is rejected before the parser (`metadata.reason: "parse_error"`, message `no artifact text in model output`) so thinking-only/reasoning-only turns cannot succeed via the identity parser. A parse failure (`ok: false` or missing `value`) consumes revision budget exactly like a validation failure: the repairer receives `value: undefined` plus a synthetic failure (`errors[0].message` = the parse error, `metadata.reason: "parse_error"`), and budget exhaustion ends with terminal `artifact_failed`. |
|
|
93
98
|
| `ArtifactValidator<T>` | `(value: T, ctx: ArtifactContext) => ArtifactValidation \| Promise<...>` — return `{ ok: true }` or `{ ok: false, errors }`. |
|
|
94
99
|
| `ArtifactRepairer<T>` | `(value: T \| undefined, failure: ArtifactValidation, ctx: ArtifactContext) => AgentInput \| Promise<...>` — build the revision follow-up input. |
|
|
95
100
|
| `ArtifactValidation` | `{ ok: boolean; errors?: readonly { path?: string; message: string }[]; metadata?: ... }`. |
|
|
96
101
|
| `ArtifactContext` | `{ sessionId, runId, turn, signal, metadata }` — passed to every callback. |
|
|
97
102
|
| `ArtifactParseResult<T>` | `{ ok: boolean; value?: T; error?: string }`. |
|
|
98
103
|
|
|
104
|
+
Optional steer hooks on `LoopContext` (0.0.11): `hasPendingSteers?()` / `applyPendingSteers?()`. Hosts/custom loops that omit them keep pre-steer behavior; built-in loops drain at turn start.
|
|
105
|
+
|
|
99
106
|
`LoopContext` (what the runtime builds for the loop each run):
|
|
100
107
|
|
|
101
108
|
| Field | Purpose |
|
|
@@ -119,7 +126,7 @@ Host callback contracts (all generic over host `T`):
|
|
|
119
126
|
|
|
120
127
|
Events during a loop run are the existing `AgentEvent`s (`turn_started`, `message_started`, `message_delta`, `message_finished`, `turn_finished`, tool-execution events when the loop dispatches tools, `error` on real failures). Both built-in loops emit `turn_started` before each provider turn, `message_finished` for every assistant draft, and `turn_finished` after the assistant draft is appended. First-turn input is appended to live history once, matching the already-persisted user message.
|
|
121
128
|
|
|
122
|
-
Validation-failure-triggering-a-revision is **not** an `error` event — it is recoverable, like `tool_execution_blocked`. In bounded artifact mode, a tool-calling provider response emits normal assistant/tool lifecycle events, skips artifact parsing/validation, then the next turn sees its persisted result. `generateValidateReviseLoop` emits artifact events only for call-free candidates: `artifact_validation_started` → `artifact_validation_finished` → (`artifact_revision_started`)* → `artifact_finished` | `artifact_failed`. A request beyond `maxToolRounds` executes nothing and emits terminal `artifact_failed` with `result.metadata.reason === "tool_round_limit"`; see [Agent events § Artifact event ordering](agent-events.md#artifact-event-ordering). `singleShotLoop` emits zero artifact events. Real failures stay on the `error` channel.
|
|
129
|
+
Validation-failure-triggering-a-revision is **not** an `error` event — it is recoverable, like `tool_execution_blocked`. In bounded artifact mode, a tool-calling provider response emits normal assistant/tool lifecycle events, skips artifact parsing/validation, then the next turn sees its persisted result. `generateValidateReviseLoop` emits artifact events only for call-free candidates: `artifact_validation_started` → `artifact_validation_finished` → (`artifact_revision_started`)* → `artifact_finished` | `artifact_failed`. A request beyond `maxToolRounds` executes nothing and emits terminal `artifact_failed` with `result.metadata.reason === "tool_round_limit"`; see [Agent events § Artifact event ordering](agent-events.md#artifact-event-ordering). `singleShotLoop` emits zero artifact events. Real failures stay on the `error` channel. Session runs using `generate-validate-revise` resolve `succeeded` only after `artifact_finished`; terminal `artifact_failed` (including empty/thinking-only parse exhaustion) fails the run with `AgentRunError` (`error.code` from `result.metadata.reason`, e.g. `parse_error`).
|
|
123
130
|
|
|
124
131
|
A loop has no path to credentials, provider objects, or unredacted secrets. `LoopContext.generate` receives the already-policy-applied, middleware-run, redacted request; `LoopContext.emit` runs through `redactAgentEvent` with the active `SecretRedactor`.
|
|
125
132
|
|
|
@@ -78,7 +78,11 @@ Provider `thinking`/`reasoning` content emitted during a turn is preserved as `t
|
|
|
78
78
|
|
|
79
79
|
Missing providers fail closed: `run()` emits `error` and rejects before calling any provider. Provider `error` events emit session `error` and reject unless configured retry handles a transient provider-turn failure before output. Unknown tools fail closed through the tool harness and do not execute. Tool exceptions emit `tool_execution_error`, return an error `ToolResult`, and may still continue to the next provider turn.
|
|
80
80
|
|
|
81
|
-
Only one `run()` may be active per session. Concurrent `run()` calls emit `error` and reject immediately; Prism does not queue
|
|
81
|
+
Only one `run()` may be active per session. Concurrent `run()` / `prompt` / `followUp` calls emit `error` and reject immediately; Prism does not queue second prompts. Manual `compact()` also rejects while a run is active.
|
|
82
|
+
|
|
83
|
+
### Mid-run steer (0.0.11)
|
|
84
|
+
|
|
85
|
+
`session.steer(input, options?)` enqueues user text into the **same** active run (fail closed when no run). Default: inject at the next turn boundary (after tool rounds / before next provider assemble). `options.softInterrupt: true` aborts only the current provider stream, then continues the same `runId` with steered text. Pending queue caps: **8** messages / **64 KiB** UTF-8 total (`DEFAULT_MAX_PENDING_STEERS` / `DEFAULT_MAX_PENDING_STEER_BYTES`); overflow throws. Steered messages pass input guardrails + normal session append/redaction. Loops drain via optional `LoopContext.hasPendingSteers` / `applyPendingSteers`.
|
|
82
86
|
|
|
83
87
|
`session.abort(reason)` aborts the active run. The abort signal is passed to input assembly, provider requests, and tool execution; if a tool/provider path aborts after a tool call, Prism does not start another provider turn.
|
|
84
88
|
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
# Browser automation
|
|
2
|
+
|
|
3
|
+
## What it does
|
|
4
|
+
|
|
5
|
+
`@arnilo/prism-browser` exposes exactly four exclusive model-facing tools—`browser_open`, `browser_snapshot`, `browser_act`, and `browser_close`—over a host-supplied Playwright `Browser`. Prism creates one non-persistent `BrowserContext` per run, serializes actions, returns bounded AI-mode accessibility snapshots with snapshot-scoped refs, enforces egress/side-effect/upload/download/screenshot policy, and closes context/pages/listeners/quarantined downloads on close, abort, or manager disposal.
|
|
6
|
+
|
|
7
|
+
## When to use it
|
|
8
|
+
|
|
9
|
+
Use when an agent must interact with JavaScript-heavy or authenticated pages that search/fetch cannot cover. Prefer `@arnilo/prism-web-tools` for ordinary public retrieval. Do not use this package as a browser launcher, MCP proxy, visual planner, CDP console, or persistent profile manager.
|
|
10
|
+
|
|
11
|
+
## Inputs / request
|
|
12
|
+
|
|
13
|
+
| Tool | Model-visible input | Host-only construction input |
|
|
14
|
+
| --- | --- | --- |
|
|
15
|
+
| `browser_open` | optional absolute `http(s)` `url` | host Playwright `Browser` or `BrowserManager`, limits, `ExecutionPolicy`, `networkPolicy`, uploads/downloads |
|
|
16
|
+
| `browser_snapshot` | optional `pageId` | same manager/context |
|
|
17
|
+
| `browser_act` | `action` plus action-specific fields (`target`, `snapshotId`, `url`, `text`, `values`, `paths`, `downloadId`, `dialogResponse`, `pageId`, `clip`, …) | policy checked before side effects |
|
|
18
|
+
| `browser_close` | none | closes only the run-owned context, never the host Browser process |
|
|
19
|
+
|
|
20
|
+
`createBrowserTools({ browser, executionPolicy?, limits?, networkPolicy?, uploads?, downloads?, beforeSideEffect? })` builds the four tools. `createBrowserManager(...)` exposes host lifecycle helpers `closeRun(runId)` / `close()` and `listDownloads(runId)`.
|
|
21
|
+
|
|
22
|
+
Targets accepted by `browser_act`: snapshot `ref`, `role`(+`name`), `label`, `testId`, or `text`. CSS, XPath, selector strings, `page.evaluate`, CDP/devtools, extensions, and persistent/local profiles are unsupported.
|
|
23
|
+
|
|
24
|
+
`browser_act` actions: `navigate`, `click`, `type`, `fill`, `select`, `check`, `uncheck`, `scroll`, `wait`, `dialog`, `select_page`, `upload`, `screenshot`, `download_release`.
|
|
25
|
+
|
|
26
|
+
## Outputs / response / events
|
|
27
|
+
|
|
28
|
+
`browser_open` returns run/page ids and URL. `browser_snapshot` returns `snapshotId`, URL/title, bounded AI-mode aria YAML (`ariaSnapshot({ mode: "ai" })`), ref count, and truncation metadata. Refs are valid only for that snapshot id and become stale after navigation or mutation. `browser_act` returns the action, active page id, and URL; `screenshot` also returns bounded `ImageContent`; `download_release` returns quarantine metadata after host approval. `browser_close` is idempotent. Results mark `trust: "untrusted_external"`; page text must never alter tools, permissions, credentials, or policy.
|
|
29
|
+
|
|
30
|
+
## Request/response example
|
|
31
|
+
|
|
32
|
+
```json
|
|
33
|
+
{
|
|
34
|
+
"tool": "browser_snapshot",
|
|
35
|
+
"arguments": {},
|
|
36
|
+
"result": {
|
|
37
|
+
"snapshotId": "snap_ab12…",
|
|
38
|
+
"pageId": "page_1",
|
|
39
|
+
"url": "https://example.com/",
|
|
40
|
+
"title": "Example",
|
|
41
|
+
"refCount": 12,
|
|
42
|
+
"ariaSnapshot": "- main [ref=e8]:\n - button \"Submit\" [ref=e12]"
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
```json
|
|
48
|
+
{
|
|
49
|
+
"tool": "browser_act",
|
|
50
|
+
"arguments": {
|
|
51
|
+
"action": "click",
|
|
52
|
+
"target": { "ref": "e12" },
|
|
53
|
+
"snapshotId": "snap_ab12…"
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## Implementation example
|
|
59
|
+
|
|
60
|
+
```ts
|
|
61
|
+
import { chromium } from "playwright-core";
|
|
62
|
+
import {
|
|
63
|
+
createBrowserManager,
|
|
64
|
+
createBrowserTools,
|
|
65
|
+
createSharedSandboxBrowserOptions,
|
|
66
|
+
} from "@arnilo/prism-browser";
|
|
67
|
+
import { assertBrowserSandboxNetwork } from "@arnilo/prism-coding-security";
|
|
68
|
+
|
|
69
|
+
assertBrowserSandboxNetwork({
|
|
70
|
+
mode: "custom",
|
|
71
|
+
name: "prism-egress",
|
|
72
|
+
browserEgress: { proxyEndpoint: "http://127.0.0.1:3128", denyDirectEgress: true },
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
const aligned = createSharedSandboxBrowserOptions({
|
|
76
|
+
workspaceRoot: "/workspace",
|
|
77
|
+
downloadsRoot: "/downloads",
|
|
78
|
+
containedProxyAttestation: {
|
|
79
|
+
proxyEndpoint: "http://127.0.0.1:3128",
|
|
80
|
+
denyDirectEgress: true,
|
|
81
|
+
},
|
|
82
|
+
approveDownloadRelease: async (meta) => meta.bytes < 1_000_000,
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
const browser = await chromium.launch({ headless: true });
|
|
86
|
+
const manager = createBrowserManager({
|
|
87
|
+
browser,
|
|
88
|
+
...aligned,
|
|
89
|
+
limits: { maxPages: 4, maxActions: 100, maxSnapshotBytes: 256 * 1024 },
|
|
90
|
+
});
|
|
91
|
+
const tools = createBrowserTools({ manager, executionPolicy });
|
|
92
|
+
|
|
93
|
+
// On run terminal / abort / cancel:
|
|
94
|
+
await manager.closeRun(runId);
|
|
95
|
+
await manager.close();
|
|
96
|
+
await browser.close();
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
## Extension and configuration notes
|
|
100
|
+
|
|
101
|
+
- Compatibility line: `playwright-core@1.61.0` optional peer. Hosts pin browser binaries/images; Prism package install downloads nothing.
|
|
102
|
+
- Default/hard caps: pages 4/16; actions 100/256; queued actions 16/64; snapshot refs 2k/10k; depth 30/100; snapshot bytes 256 KiB/2 MiB; navigation 30s/120s; action 10s/60s; wait 30s/120s; run wall 20min/30min; popups 4/16; dialogs 16/64; close grace 5s/30s; network requests 1k/10k; redirects/request 10/32; WebSockets 8/32; screenshots 16/64 with 16/64 megapixels and 10 MiB/32 MiB encoded; uploads 8/32 files, 16 MiB/64 MiB each, 64 MiB/256 MiB aggregate; downloads 8/32 files, 32 MiB/256 MiB each, 64 MiB/512 MiB aggregate.
|
|
103
|
+
- Contexts use `serviceWorkers: "block"` and install `BrowserContext.route()` for every visible HTTP(S)/WebSocket request. `acceptDownloads` is enabled only when `downloads` is configured.
|
|
104
|
+
- `networkPolicy` defaults to `requireContainedProxy: true` (fail closed). Hosts must supply `containedProxyAttestation: { proxyEndpoint, denyDirectEgress: true }`. Private/loopback/link-local, `file`/`data`/`blob`/`javascript`/`devtools` schemes are denied by default. Playwright routing is defense in depth — production DNS/private egress is a host firewall/proxy.
|
|
105
|
+
- Uploads require absolute paths under `uploads.roots` (realpath-contained; symlink escapes rejected). Downloads stream into `downloads.quarantine` with SHA-256/MIME/name metadata; `download_release` requires host `approveRelease`. Screenshots return bounded `ImageContent`.
|
|
106
|
+
- Observation (`snapshot`, `wait`, open-without-url, `close`) vs mutation/high-impact (`navigate`, click/form, dialog accept, upload, download release, popup select) is classified for `ExecutionPolicy` / `beforeSideEffect`.
|
|
107
|
+
- `createSharedSandboxBrowserOptions()` aligns browser uploads/downloads with Task 1 sandbox `/workspace` and `/downloads`. `assertBrowserSandboxNetwork()` in `@arnilo/prism-coding-security` fails closed for custom Docker networks without browser egress attestation.
|
|
108
|
+
- Raw CSS is absent from production defaults. Ref resolution uses Playwright’s built-in `aria-ref=` selector with a package-owned snapshot ref table for staleness checks.
|
|
109
|
+
|
|
110
|
+
## Security and performance notes
|
|
111
|
+
|
|
112
|
+
Import is inert. Construction fails clearly when neither `browser` nor `manager` is supplied. Browser installation, launch, version, and control endpoint are host-owned. Prism never exposes `page.evaluate`, init scripts, CDP, extensions, persistent profiles, or model-supplied Playwright launch options. Secrets and storage state must not appear in snapshots, tool results, logs, or checkpoints. Finite caps charge before context/page/action/queue/snapshot/network/artifact retention; snapshots retain no unbounded DOM, console, request, response, or trace history. Unreleased downloads are deleted on context close.
|
|
113
|
+
|
|
114
|
+
Default tests use fake Playwright APIs only. Protected live gate: `PRISM_LIVE_PLAYWRIGHT=1` (or `PRISM_TEST_PLAYWRIGHT=1`) `npm run test:live -w @arnilo/prism-browser` exercises a local loopback hostile HTML fixture for snapshot refs, stale-ref rejection, CSS denial, private/file deny, upload containment, screenshot bounds, and download quarantine/release. Missing browser binaries fail closed when the gate is enabled. Adversarial network-free fixtures live in `eval-fixtures.test.ts`; see [Evaluations](evaluations.md) and `examples/coding-browser-evaluation.ts`.
|
|
115
|
+
|
|
116
|
+
## Related APIs
|
|
117
|
+
|
|
118
|
+
- [Tools](tools.md): registry, exclusive dispatch, validation, and ledger.
|
|
119
|
+
- [Web search, fetch, and extraction](web-tools.md): preferred non-interactive retrieval path.
|
|
120
|
+
- [Guardrails](guardrails.md): untrusted external content handling.
|
|
121
|
+
- [Host security](host-security.md): browser endpoint, approval, egress proxy, and artifact trust boundaries.
|
|
122
|
+
- [Performance and resource limits](performance.md): browser ceilings and charging points.
|
|
123
|
+
- [Coding execution approval and sandboxing](coding-security.md): optional shared disposable sandbox for coding+browser.
|
|
124
|
+
- [Migration](migration.md): additive optional package activation.
|
package/docs/cli-rpc.md
CHANGED
|
@@ -106,7 +106,7 @@ Branch-aware session commands return live handle details:
|
|
|
106
106
|
|
|
107
107
|
`sessionId` identifies the durable session. `leafId` is the selected branch tip. `handleId` is the RPC map key used by `switchSession`; forks that share the same `sessionId` get stable ids like `session-1#2` so the parent handle is not overwritten.
|
|
108
108
|
|
|
109
|
-
Invalid CLI flags return exit code `2`. Invalid JSON, missing ids, unknown RPC commands,
|
|
109
|
+
Invalid CLI flags return exit code `2`. Invalid JSON, missing ids, unknown RPC commands, unknown command contributions, and runtime failures return `ok: false` response envelopes without executing unknown tools or commands. `steer` with no active run (or overflow) returns `ok: false`.
|
|
110
110
|
|
|
111
111
|
## Request/response example
|
|
112
112
|
|
|
@@ -128,6 +128,7 @@ Invalid CLI flags return exit code `2`. Invalid JSON, missing ids, unknown RPC c
|
|
|
128
128
|
- `state`, `messages`, `setModel`, `switchSession`, `forkSession`, `cloneSession`, `checkout`, and registered `command` requests are processed immediately.
|
|
129
129
|
- `compact` is fail-closed: if the current session has an active run, it returns `ok: false` because the session rejects compaction during a run.
|
|
130
130
|
- A second `prompt` or `followUp` for the same session while it already has an active run returns `ok: false` immediately instead of blocking the input loop.
|
|
131
|
+
- `steer` enqueues mid-run user text for the active session (`params.input`, optional `params.softInterrupt`). Fails closed when no active run or when the pending steer queue overflows (8 messages / 64 KiB). Soft interrupt aborts the current provider stream only; the run continues.
|
|
131
132
|
|
|
132
133
|
Events streamed during a run keep the original prompt request id, even when an `abort` with a different request id cancels the run. The completion or error response for the prompt also uses the original prompt request id.
|
|
133
134
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
`@arnilo/prism-coding-agent` is an optional first-party package that provides host shell/filesystem tools as Prism `ToolDefinition` objects. It ships
|
|
5
|
+
`@arnilo/prism-coding-agent` is an optional first-party package that provides host shell/filesystem/repository tools as Prism `ToolDefinition` objects. It ships six default coding tools — `shell`, `read`, `write`, `edit`, `repo_list`, `repo_search` — plus opt-in structured Git/check set (`createGitTools`), opt-in `createAskUserDecisionTool({ ask })`, and bounded coding-plan/checkpoint helpers. The tools are **inert** until a host imports them and registers them into a `ToolRegistry`. Hosts may register any subset, omit aggregators entirely, or mix first-party tools with host-owned `ToolDefinition`s. Behavior for shell/read/write/edit is a behavioral port of the pi coding agent's tools, adapted to Prism's `ToolDefinition` / `ToolResult` contracts (no `@earendil-works/*` or `typebox` dependencies; only `diff` plus the Node standard library). List/search/Git are native Prism tools with no glob/ripgrep/Git-library dependency.
|
|
6
6
|
|
|
7
7
|
| Export | Purpose |
|
|
8
8
|
| --- | --- |
|
|
@@ -10,13 +10,24 @@
|
|
|
10
10
|
| `createReadTool(cwd, options?)` | `read` tool: read a text or image file into `TextContent` / `ImageContent`. |
|
|
11
11
|
| `createWriteTool(cwd, options?)` | `write` tool: create or overwrite a file, creating parent directories. |
|
|
12
12
|
| `createEditTool(cwd, options?)` | `edit` tool: precise exact-then-fuzzy text replacement in an existing file. |
|
|
13
|
-
| `
|
|
14
|
-
| `
|
|
15
|
-
| `
|
|
13
|
+
| `createRepoListTool(cwd, options?)` | `repo_list` tool: bounded deterministic repository listing. |
|
|
14
|
+
| `createRepoSearchTool(cwd, options?)` | `repo_search` tool: bounded literal/regex text search. |
|
|
15
|
+
| `createCodingTools(cwd, options?)` | Default six tools (`shell`, `read`, `write`, `edit`, `repo_list`, `repo_search`). |
|
|
16
|
+
| `createReadOnlyTools(cwd, options?)` | Read-only subset: `read`, `repo_list`, `repo_search`. |
|
|
17
|
+
| `createAllTools(cwd, options?)` | Identical to `createCodingTools` (Git tools remain opt-in via `createGitTools`). |
|
|
18
|
+
| `createGitTools(cwd, options?)` | Opt-in Git tools (`git_status`/`git_diff`/`git_branch`/`git_worktree`/`git_apply`/`git_commit`/`git_pr_handoff`) plus optional `coding_check`. |
|
|
19
|
+
| `createCodingCheckTool(cwd, options)` | Named host-declared checks; model selects only a name. |
|
|
20
|
+
| `createAskUserDecisionTool(options)` | Opt-in user decision tool (`ask_user_decision`); host supplies `ask` callback. Not in default aggregators. |
|
|
21
|
+
| `createLocalRepositoryOperations(limits?)` | Default streaming Node filesystem backend for list/search. |
|
|
22
|
+
| `createGitOperations(options)` | Typed Git operations backend (argument arrays, safe config, finite output). |
|
|
23
|
+
| `buildCodingCheckpointMetadata` / `validateCodingCheckpointMetadata` / `assertCodingResumeAllowed` | Bounded durable coding-task metadata for workflow `state.coding` (no second runtime). |
|
|
24
|
+
| `writeCodingPlanFile` / `readCodingPlanFile` / `createCodingPlanMarkdown` / `parseCodingPlanTodos` | Workspace plan/todo Markdown helpers with finite byte/todo caps and hash verification. |
|
|
25
|
+
| `fingerprintJson` / `CODING_STATE_KEY` | Stable tool/policy fingerprints and the shared-state key for coding metadata. |
|
|
16
26
|
| `detectSupportedImageMimeType(buf)` / `detectSupportedImageMimeTypeFromFile(path)` | Magic-byte image MIME detection (PNG/JPEG/GIF/WebP/BMP) used by `read`. |
|
|
17
27
|
| `DEFAULT_MAX_IMAGE_BYTES` | Default `read` image size ceiling (10 MB). |
|
|
18
|
-
| `DEFAULT_*` / `HARD_*` coding limit constants | Published text-scan, image, write/edit, shell
|
|
28
|
+
| `DEFAULT_*` / `HARD_*` coding limit constants | Published text-scan, image, write/edit, shell, repository, Git, check, handoff, and plan/checkpoint ceilings. |
|
|
19
29
|
| `ReadTextOptions` / `ReadTextResult` | Bounded text-page contract required by custom `ReadOperations`. |
|
|
30
|
+
| `RepositoryOperations` / `RepositoryLimitOptions` | Pluggable list/search backend and finite caps. |
|
|
20
31
|
| `TransformImage` / `TransformImageInput` | Types for the optional `read` `transformImage` callback. |
|
|
21
32
|
| `withFileMutationQueue(path, fn)` | Per-path serialization primitive re-exported for hosts. |
|
|
22
33
|
|
|
@@ -55,6 +66,7 @@ const tools = createCodingTools(workspaceRoot, {
|
|
|
55
66
|
| `read` | `read` |
|
|
56
67
|
| `write` | `write` |
|
|
57
68
|
| `edit` | `edit` |
|
|
69
|
+
| `repo_list` / `repo_search` | _(native; no pi equivalent)_ |
|
|
58
70
|
|
|
59
71
|
## Inputs / request
|
|
60
72
|
|
|
@@ -159,6 +171,146 @@ Each `edits[].oldText` must match a unique, non-overlapping region of the origin
|
|
|
159
171
|
|
|
160
172
|
`edit` result `metadata`: `{ diff, patch, firstChangedLine }` — a display-oriented diff, a standard unified patch, and the first changed line in the new file. These are host-readable; the model only sees the short confirmation (keeps model context small).
|
|
161
173
|
|
|
174
|
+
### `repo_list`
|
|
175
|
+
|
|
176
|
+
List repository entries with deterministic relative paths. Uses Node `opendir`/`lstat` only — no glob dependency. Does not follow symlinks; rejects path escapes outside the workspace root. Hidden names and excluded basenames (default `.git`, `node_modules`, `dist`) are skipped unless `includeHidden` is set / host `exclude` is overridden.
|
|
177
|
+
|
|
178
|
+
**Inputs:**
|
|
179
|
+
|
|
180
|
+
| Field | Type | Purpose |
|
|
181
|
+
| --- | --- | --- |
|
|
182
|
+
| `path` | `string` | Workspace-relative directory or file to list (default root). |
|
|
183
|
+
| `includeHidden` | `boolean` | Include dot names (default false). |
|
|
184
|
+
| `maxDepth` | `number` | Directory depth cap (default 32, hard 128). |
|
|
185
|
+
| `maxResults` | `number` | Page size (default 1,000, hard 10,000). |
|
|
186
|
+
| `offset` | `number` | Entries to skip before retaining (default 0). |
|
|
187
|
+
|
|
188
|
+
**Outputs:** text lines `kind\trelative/path[\tsize]` plus metadata (`truncated`, `truncatedBy`, `nextOffset`, `entries`, scan counts). Continue with `offset=nextOffset` when truncated by results.
|
|
189
|
+
|
|
190
|
+
### `repo_search`
|
|
191
|
+
|
|
192
|
+
Search text files under the workspace. Default mode is literal substring match; `mode: "regex"` enables length-bounded regular expressions. Binary files (NUL in a bounded prefix) and oversize files are skipped. Aggregate scanned bytes, matches, line bytes, pattern bytes, and wall time are finite.
|
|
193
|
+
|
|
194
|
+
**Inputs:**
|
|
195
|
+
|
|
196
|
+
| Field | Type | Purpose |
|
|
197
|
+
| --- | --- | --- |
|
|
198
|
+
| `query` | `string` | Literal or regex pattern (required). |
|
|
199
|
+
| `path` | `string` | Workspace-relative start path. |
|
|
200
|
+
| `mode` | `"literal" \| "regex"` | Default `literal`. |
|
|
201
|
+
| `caseSensitive` | `boolean` | Default false. |
|
|
202
|
+
| `includeHidden` | `boolean` | Default false. |
|
|
203
|
+
| `context` | `number` | Context lines before/after each match (default 5, hard 20). |
|
|
204
|
+
| `maxMatches` | `number` | Match cap (default 1,000, hard 10,000). |
|
|
205
|
+
|
|
206
|
+
**Outputs:** ripgrep-like lines `path:line:column:text` with optional `path-` / `path+` context, plus metadata (`matches`, `truncated`, scan/skip counts).
|
|
207
|
+
|
|
208
|
+
### Structured Git tools (`createGitTools`)
|
|
209
|
+
|
|
210
|
+
Opt-in tools over a host-pinned Git executable (`gitPath`, default `/usr/bin/git`) or sandbox `execFile`. Every invocation uses argument arrays with safe config (`core.hooksPath=/dev/null`, empty credential helper, pager disabled, `GIT_TERMINAL_PROMPT=0`). Shell is never used internally. Git tools are **not** included in `createCodingTools()` / `createAllTools()`.
|
|
211
|
+
|
|
212
|
+
| Tool | Purpose |
|
|
213
|
+
| --- | --- |
|
|
214
|
+
| `git_status` | `status --porcelain=v2 -z --branch` → structured branch + entries + `dirty`. |
|
|
215
|
+
| `git_diff` | Bounded `--no-ext-diff --no-textconv` diff; oversized output may spill via `artifactWriter`. |
|
|
216
|
+
| `git_branch` | `validate` / `list` / `create` / `switch` with `git check-ref-format --branch`. Switch refuses unrelated dirty trees unless `createCheckpoint=true`. |
|
|
217
|
+
| `git_worktree` | `list` / `add` / `remove` within finite worktree caps. |
|
|
218
|
+
| `git_apply` | `check` / `apply` / `reverse`; always `--check` before mutating apply. Apply requires clean/checkpoint; failures restore. |
|
|
219
|
+
| `git_commit` | Explicit-path `add` + `commit --no-verify -F <tempfile>`; requires host `commitIdentity`. Allows dirty entries that are exactly the requested paths; unrelated dirt requires checkpoint. Never pushes. |
|
|
220
|
+
| `git_pr_handoff` | Bounded `{ base, head, commits, changedPaths, diffstat, checks, artifact? }` for host PR creation. Never authenticates or opens a PR. |
|
|
221
|
+
| `coding_check` | Included when `checks` are declared: model selects only a name; executable/args/env are host-fixed. |
|
|
222
|
+
|
|
223
|
+
```ts
|
|
224
|
+
import { createGitTools } from "@arnilo/prism-coding-agent";
|
|
225
|
+
|
|
226
|
+
const gitTools = createGitTools(workspaceRoot, {
|
|
227
|
+
gitPath: "/usr/bin/git",
|
|
228
|
+
commitIdentity: { name: "Prism Bot", email: "bot@example.com" },
|
|
229
|
+
checks: {
|
|
230
|
+
test: { file: "/usr/bin/npm", args: ["test"] },
|
|
231
|
+
},
|
|
232
|
+
});
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
### Ask-user decision (`createAskUserDecisionTool`)
|
|
236
|
+
|
|
237
|
+
Opt-in `ask_user_decision` for ambiguous, high-impact direction choices. Model must pass a question plus 2+ options, each with **exactly 3 pros and 3 cons**. Host supplies `ask` (blocks until the user picks). Not in `createCodingTools` / `createAllTools` / `createReadOnlyTools`.
|
|
238
|
+
|
|
239
|
+
| Mode | How |
|
|
240
|
+
| --- | --- |
|
|
241
|
+
| Single (default) | `selectionMode: "single"` → host returns `{ selectedId }` (or length-1 `selectedIds`) |
|
|
242
|
+
| Multi | `selectionMode: "multiple"` → `{ selectedIds: [...] }` (non-empty, known ids) |
|
|
243
|
+
| Free-text | `allowCustom: true` → host may return `{ customText }` **XOR** selection (never both) |
|
|
244
|
+
| Blocking tool | `createAskUserDecisionTool({ ask })` — in-process UI callback |
|
|
245
|
+
| Durable workflow | `suspendAskUserDecision(request)` + `createAskUserDecisionResumeValidator()` / `validateAskUserDecisionResume` on `resumeWorkflow` |
|
|
246
|
+
| Agent durable adapter | `validateAskUserDecisionAgentResume({ request, answer })` — same validation; **no** new `AgentRunInterruption` kinds in 0.0.11 |
|
|
247
|
+
|
|
248
|
+
Custom-text caps match question defaults (2 KiB / hard 8 KiB). Options default max 6 (hard 16).
|
|
249
|
+
|
|
250
|
+
```ts
|
|
251
|
+
import { createToolRegistry } from "@arnilo/prism";
|
|
252
|
+
import {
|
|
253
|
+
createAskUserDecisionTool,
|
|
254
|
+
createCodingTools,
|
|
255
|
+
suspendAskUserDecision,
|
|
256
|
+
createAskUserDecisionResumeValidator,
|
|
257
|
+
} from "@arnilo/prism-coding-agent";
|
|
258
|
+
|
|
259
|
+
const tools = createToolRegistry([
|
|
260
|
+
...createCodingTools(workspaceRoot),
|
|
261
|
+
createAskUserDecisionTool({
|
|
262
|
+
ask: async ({ question, options, selectionMode, allowCustom }) =>
|
|
263
|
+
ui.ask({ question, options, selectionMode, allowCustom }),
|
|
264
|
+
}),
|
|
265
|
+
]);
|
|
266
|
+
|
|
267
|
+
// Workflow node:
|
|
268
|
+
return suspendAskUserDecision({
|
|
269
|
+
question: "Ship sqlite or postgres?",
|
|
270
|
+
options: [/* ≥2 with 3 pros + 3 cons each */],
|
|
271
|
+
selectionMode: "single",
|
|
272
|
+
allowCustom: false,
|
|
273
|
+
});
|
|
274
|
+
// resumeWorkflow(..., { validateResume: createAskUserDecisionResumeValidator() })
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
### Goal → verify helper (`runCodingGoalVerify`)
|
|
278
|
+
|
|
279
|
+
Thin composition over existing plan Markdown, named checks, workflow `suspend`/`resumeWorkflow`, and bounded PR handoff. **No Goal table / second runtime.** Peer `@arnilo/prism-workflows`. Example: `examples/coding-goal-verify.ts`.
|
|
280
|
+
|
|
281
|
+
```ts
|
|
282
|
+
import { runCodingGoalVerify } from "@arnilo/prism-coding-agent";
|
|
283
|
+
|
|
284
|
+
const result = await runCodingGoalVerify({
|
|
285
|
+
goal: "Fix the flake",
|
|
286
|
+
cwd: process.cwd(),
|
|
287
|
+
taskId: "flake-1",
|
|
288
|
+
baseBranch: "main",
|
|
289
|
+
branch: "fix/flake",
|
|
290
|
+
checkNames: ["test"],
|
|
291
|
+
checkDefinitions: { test: { file: "/usr/bin/npm", args: ["test"] } },
|
|
292
|
+
runCheck: hostRunCheck,
|
|
293
|
+
buildHandoff: hostBuildHandoff,
|
|
294
|
+
approval: { validateResume: hostValidate },
|
|
295
|
+
checkpoints,
|
|
296
|
+
ownership,
|
|
297
|
+
redactor,
|
|
298
|
+
});
|
|
299
|
+
```
|
|
300
|
+
|
|
301
|
+
### Durable coding plans and checkpoints
|
|
302
|
+
|
|
303
|
+
There is no `CodingRun`, todo database, or second approval engine. Persist executable plan/todos as ordinary workspace Markdown (for example `plans/<task>.md`) and store only bounded metadata under workflow `state.coding`:
|
|
304
|
+
|
|
305
|
+
| Field group | Stored in checkpoint | Not stored |
|
|
306
|
+
| --- | --- | --- |
|
|
307
|
+
| Plan / workspace export / patch artifacts | URI + SHA-256 + byte count | File contents, credentials, raw command output |
|
|
308
|
+
| Branch / worktree / base | Paths and ref names | Full diffs |
|
|
309
|
+
| Named checks | Name + exit code + short summary | Full stdout/stderr |
|
|
310
|
+
| Fingerprints | Workflow revision, definition hash, tool/policy fingerprints, optional image digest | Browser storage state, secrets, env |
|
|
311
|
+
|
|
312
|
+
Use `writeCodingPlanFile` / `readCodingPlanFile` for the workspace artifact, `buildCodingCheckpointMetadata` before `ctx.updateState({ coding })`, and `assertCodingResumeAllowed` before import/resume. Wrong owner/revision/hash/fingerprint fails closed. See `examples/durable-coding-workflow.ts` for a network-free plan → branch → edit → check → approval → handoff composition over `runWorkflow` / `resumeWorkflow` / `startWorkflowBackground`.
|
|
313
|
+
|
|
162
314
|
## Outputs / response / events
|
|
163
315
|
|
|
164
316
|
Every tool returns a `ToolResult` with `toolCallId`, `name`, `content` (`readonly ContentBlock[]`), optional `error`, and optional `metadata`. `write` and `edit` serialize per realpath through `withFileMutationQueue` so concurrent calls targeting one file do not interleave. `shell` is marked `exclusive`; tool dispatch serializes it at the turn level. The package emits no events of its own; hosts observe tool execution through the normal Prism `AgentEvent` stream via `dispatchToolCall`.
|
|
@@ -197,10 +349,10 @@ Minimal drop-in for any Prism app:
|
|
|
197
349
|
import { createToolRegistry } from "@arnilo/prism";
|
|
198
350
|
import { createCodingTools, createReadOnlyTools } from "@arnilo/prism-coding-agent";
|
|
199
351
|
|
|
200
|
-
// Full coding set (shell + read + write + edit) against the project root:
|
|
352
|
+
// Full coding set (shell + read + write + edit + repo_list + repo_search) against the project root:
|
|
201
353
|
const tools = createToolRegistry(createCodingTools(process.cwd()));
|
|
202
354
|
|
|
203
|
-
// Or a read-only set for inspection-only agents:
|
|
355
|
+
// Or a read-only set for inspection-only agents (read + repo_list + repo_search):
|
|
204
356
|
const ro = createToolRegistry(createReadOnlyTools(process.cwd()));
|
|
205
357
|
```
|
|
206
358
|
|
|
@@ -227,17 +379,18 @@ const remoteWrite = createWriteTool("/repo", {
|
|
|
227
379
|
|
|
228
380
|
## Extension and configuration notes
|
|
229
381
|
|
|
230
|
-
- **Pluggable operation backends.** Every tool accepts an `operations` seam. Custom `ReadOperations` must implement bounded `readText` plus `statFile`; custom `EditOperations` must implement `statFile`; read/write methods receive caps/signals. `BashOperations` must stream through `onData` and honor `signal`/`timeout`. A hostile custom backend can still violate its host-owned contract, so isolate it separately.
|
|
231
|
-
- **Per-tool options.** `ShellToolOptions` adds `timeout` and `maxTotalOutputBytes`; `ReadToolOptions` adds `maxScanBytes`; `WriteToolOptions` adds `maxInputBytes`; `EditToolOptions` adds `maxFileBytes`, `maxInputBytes`, and `maxEdits
|
|
232
|
-
- **Aggregator options.** `ToolsOptions` (`{ executionPolicy?, shell?, read?, write?, edit? }`) threads each sub-object to the matching tool. `createCodingTools()`, `createAllTools()`, and `createReadOnlyTools()` apply the shared policy unless that tool has an explicit per-tool override.
|
|
382
|
+
- **Pluggable operation backends.** Every tool accepts an `operations` seam. Custom `ReadOperations` must implement bounded `readText` plus `statFile`; custom `EditOperations` must implement `statFile`; read/write methods receive caps/signals. `BashOperations` must stream through `onData` and honor `signal`/`timeout`. Custom `RepositoryOperations` must honor depth/entry/file/match/scan/time caps and abort. A hostile custom backend can still violate its host-owned contract, so isolate it separately.
|
|
383
|
+
- **Per-tool options.** `ShellToolOptions` adds `timeout` and `maxTotalOutputBytes`; `ReadToolOptions` adds `maxScanBytes`; `WriteToolOptions` adds `maxInputBytes`; `EditToolOptions` adds `maxFileBytes`, `maxInputBytes`, and `maxEdits`; list/search accept `repository` limits and shared aggregator `ToolsOptions.repository`.
|
|
384
|
+
- **Aggregator options.** `ToolsOptions` (`{ executionPolicy?, shell?, read?, write?, edit?, list?, search?, repository? }`) threads each sub-object to the matching tool. `createCodingTools()`, `createAllTools()`, and `createReadOnlyTools()` apply the shared policy unless that tool has an explicit per-tool override. Read-only membership is deliberately `read` + `repo_list` + `repo_search` (0.0.9 behavior change).
|
|
385
|
+
- **Sandbox composition.** Prefer `@arnilo/prism-coding-security` `createSandboxCodingComposition(cwd, { workspaceMode, sandbox, ... })` (or tools-only wrappers). `workspaceMode` is required: `"sandbox"` keeps shell/read/write/edit/list/search on one disposable tree; `"host"` runs against host cwd and never claims containment. Mixed sandbox-shell + host-FS wiring throws unless `allowMixedWorkspaceWiring: true`. Same-tree Git: `createGitTools(composition.workspaceRoot, { execFile: sandbox.execFile, commitIdentity })`.
|
|
233
386
|
- **`ToolsOptions`** and the per-tool option types are exported from the package barrel for host configuration.
|
|
234
387
|
- No auto-discovery or manifest registration: import and register explicitly. This package registers no extensions and owns no globals (the mutation queue is a process-wide per-path map — see `ponytail:` note in the source).
|
|
235
388
|
|
|
236
389
|
## Security and performance notes
|
|
237
390
|
|
|
238
|
-
- **Host shell/filesystem access.** These tools run real commands and read/write real files. They provide **no sandbox**. Gate them with Prism `PermissionPolicy` / `ToolValidator` / trust policies before registering them for any provider turn. Shared `executionPolicy` applies to both full and read-only aggregators before filesystem/process side effects. See [Host security guide](host-security.md) and [Security/auth/trust](settings-auth-trust-security.md).
|
|
391
|
+
- **Host shell/filesystem access.** These tools run real commands and read/write/list/search real files. They provide **no sandbox**. Gate them with Prism `PermissionPolicy` / `ToolValidator` / trust policies before registering them for any provider turn. Shared `executionPolicy` applies to both full and read-only aggregators before filesystem/process side effects. See [Host security guide](host-security.md) and [Security/auth/trust](settings-auth-trust-security.md).
|
|
239
392
|
- **Non-zero exit is not an error.** A failing command is a normal `shell` result (exit code in metadata); only timeout/abort/spawn failures are error results. Do not assume `error == undefined` means the command succeeded.
|
|
240
|
-
- **Bounded I/O.** `read` streams one page and bounds scan bytes; image/edit reads use stat plus a shared cap-enforcing reader; write/edit inputs are measured before mutation. `shell` retains only a rolling display tail and synchronously spills accepted raw chunks so stream backpressure cannot grow heap; wall time and total raw output remain finite.
|
|
393
|
+
- **Bounded I/O.** `read` streams one page and bounds scan bytes; image/edit reads use stat plus a shared cap-enforcing reader; write/edit inputs are measured before mutation. `repo_list`/`repo_search` stream walks and charge depth/entry/file/match/scan/time before retention. Structured Git tools use argument arrays with finite output/path/ref/message/patch caps, disable hooks/credential prompts/external diff by default, and never push or open PRs. `shell` retains only a rolling display tail and synchronously spills accepted raw chunks so stream backpressure cannot grow heap; wall time and total raw output remain finite.
|
|
241
394
|
- **Per-path serialization.** Concurrent mutations to the same file serialize; concurrent mutations to different files do not block each other. The queue is a process-wide map — across sessions in one process, same-path writes still serialize (upgrade path: scope per registry if throughput matters).
|
|
242
395
|
- **Bounded image reads.** `read` rejects images over `maxImageBytes` (default 10 MB) by `stat` before read when possible; MIME is detected from magic bytes only. Optional `transformImage` is host-owned — the base package has no image-processing dependency.
|
|
243
396
|
|
|
@@ -252,8 +405,19 @@ const remoteWrite = createWriteTool("/repo", {
|
|
|
252
405
|
| Edit target / input / count | 8 MiB / 2 MiB / 100 | 64 MiB / 16 MiB / 1,000 | before target read/matching/write |
|
|
253
406
|
| Shell wall time | 600 seconds | 3,600 seconds | process-tree kill |
|
|
254
407
|
| Shell total stdout+stderr | 64 MiB | 1 GiB | process-tree kill; spill removal |
|
|
255
|
-
|
|
256
|
-
|
|
408
|
+
| Repo depth / entries / files / page | 32 / 10,000 / 10,000 / 1,000 | 128 / 100,000 / 100,000 / 10,000 | before descending/retaining next entry |
|
|
409
|
+
| Search scan / file / matches | 64 MiB / 8 MiB / 1,000 | 1 GiB / 64 MiB / 10,000 | before next file/match retention |
|
|
410
|
+
| Search pattern / line / context / time | 512 B / 50 KiB / 5 / 30 s | 4 KiB / 1 MiB / 20 / 300 s | before regex compile / line retain / deadline |
|
|
411
|
+
| Git paths / refs / message | 1,000 / 1 KiB / 64 KiB | 10,000 / 4 KiB / 256 KiB | before process/temp-file creation |
|
|
412
|
+
| Git output / diff lines / changed files / patch | 4 MiB / 10,000 / 1,000 / 16 MiB | 64 MiB / 100,000 / 10,000 / 64 MiB | stream before retain; artifact spill optional |
|
|
413
|
+
| Worktrees | 4 | 16 | before add |
|
|
414
|
+
| Named checks (names / concurrency / time / lines / output) | 8 / 1 / 10 min / 2,000 / 4 MiB | 32 / 4 / 60 min / 100,000 / 64 MiB | construction / before start / line retention |
|
|
415
|
+
| PR handoff JSON / commits | 256 KiB / 100 | 1 MiB / 1,000 | before result exposure |
|
|
416
|
+
| Plan markdown / todos / todo text | 256 KiB / 1,000 / 512 B | 1 MiB / 10,000 / 4 KiB | before write/parse/checkpoint |
|
|
417
|
+
| Coding checkpoint metadata / artifact refs / artifact bytes | 64 KiB / 16 / 256 MiB | 512 KiB / 64 / 2 GiB | before state save / resume verify |
|
|
418
|
+
| Check summary text | 1 KiB | 8 KiB | before checkpoint retention |
|
|
419
|
+
|
|
420
|
+
Every configurable value is a positive safe integer (context may be zero); Prism rejects rather than clamps invalid values. Limits control resources, not authority: they do not replace root containment, approval, validation, or a sandbox.
|
|
257
421
|
|
|
258
422
|
## Related APIs
|
|
259
423
|
|