@arnilo/prism 0.1.6 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/dist/agent-approval.d.ts +10 -1
- package/dist/agent-approval.js +81 -0
- package/dist/agent-run-lifecycle.js +6 -8
- package/dist/cache-telemetry.d.ts +58 -0
- package/dist/cache-telemetry.js +102 -0
- package/dist/cli-provider-add.d.ts +37 -0
- package/dist/cli-provider-add.js +293 -0
- package/dist/cli-runner.d.ts +5 -1
- package/dist/cli-runner.js +13 -1
- package/dist/index.d.ts +3 -1
- package/dist/index.js +2 -1
- package/docs/agent-session-runtime.md +2 -0
- package/docs/cli-rpc.md +33 -0
- package/docs/coding-security.md +27 -6
- package/docs/host-security.md +1 -1
- package/docs/index.md +7 -7
- package/docs/migration.md +25 -0
- package/docs/model-routing.md +45 -0
- package/docs/provider-caching.md +63 -0
- package/docs/provider-packages.md +2 -0
- package/docs/public-contracts.md +1 -1
- package/docs/release-and-install.md +43 -0
- package/docs/work-tools.md +24 -1
- package/package.json +3 -3
- package/templates/provider/CHANGELOG.md.tmpl +5 -0
- package/templates/provider/README.md.tmpl +41 -0
- package/templates/provider/docs/providers/NAME.md.tmpl +61 -0
- package/templates/provider/package.json.tmpl +49 -0
- package/templates/provider/src/cache.ts.tmpl +20 -0
- package/templates/provider/src/index.ts.tmpl +33 -0
- package/templates/provider/src/models.ts.tmpl +16 -0
- package/templates/provider/src/provider.ts.tmpl +23 -0
- package/templates/provider/src/tests/provider.test.ts.tmpl +104 -0
- package/templates/provider/tsconfig.json.tmpl +16 -0
package/dist/cli-runner.d.ts
CHANGED
|
@@ -60,10 +60,14 @@ export interface CliRuntime {
|
|
|
60
60
|
readonly initTemplatesRoot?: string;
|
|
61
61
|
/** Override version stamped by `prism init` (tests). */
|
|
62
62
|
readonly initPackageVersion?: string;
|
|
63
|
+
/** Override provider-scaffold template root (tests). */
|
|
64
|
+
readonly providerTemplatesRoot?: string;
|
|
65
|
+
/** Override version stamped by `prism providers add` (tests). */
|
|
66
|
+
readonly providerPackageVersion?: string;
|
|
63
67
|
/** Working directory for relative `prism init` destinations (tests). */
|
|
64
68
|
readonly cwd?: string;
|
|
65
69
|
}
|
|
66
|
-
export declare const usage = "Usage: prism [--mode print|json|rpc] [-p prompt] [options]\n prism init <dir> [--provider <name>] [--with-workflows] [--with-evals] [--force]\n\nOptions:\n -p, --prompt <text> Prompt to run in print/json mode\n --provider <name> Explicit provider id (mock is built in for smoke tests)\n --model <name> Explicit model name\n --session <id> Session id\n --system <text> System instructions\n --context <text> Context text\n --compact <entries> Auto-compaction threshold\n --max-tool-rounds <n> Maximum tool rounds\n --discover Enable workspace contribution discovery (opt-in)\n --discover-kinds <csv> Kinds to discover (default: skill; skill,tool,context,instructions)\n --no-discovery Disable discovery even if --discover is set\n --agents-config <path> App config root holding agents/<name>/AGENT.md bundles (opt-in)\n --no-agents-md Skip auto-loading <workspaceRoot>/AGENTS.md\n --no-system-md Skip auto-loading the global SYSTEM.md layer\n --agents-md-file <path> Read AGENTS.md from <path> instead (trust-gated, source: app)\n --system-md-file <path> Read SYSTEM.md from <path> instead (source: user)\n -h, --help Show this help\n";
|
|
70
|
+
export declare const usage = "Usage: prism [--mode print|json|rpc] [-p prompt] [options]\n prism init <dir> [--provider <name>] [--with-workflows] [--with-evals] [--force]\n prism providers add <name> [--base-url <url>] [--env-key <name>] [--model <id>] [--force]\n\nOptions:\n -p, --prompt <text> Prompt to run in print/json mode\n --provider <name> Explicit provider id (mock is built in for smoke tests)\n --model <name> Explicit model name\n --session <id> Session id\n --system <text> System instructions\n --context <text> Context text\n --compact <entries> Auto-compaction threshold\n --max-tool-rounds <n> Maximum tool rounds\n --discover Enable workspace contribution discovery (opt-in)\n --discover-kinds <csv> Kinds to discover (default: skill; skill,tool,context,instructions)\n --no-discovery Disable discovery even if --discover is set\n --agents-config <path> App config root holding agents/<name>/AGENT.md bundles (opt-in)\n --no-agents-md Skip auto-loading <workspaceRoot>/AGENTS.md\n --no-system-md Skip auto-loading the global SYSTEM.md layer\n --agents-md-file <path> Read AGENTS.md from <path> instead (trust-gated, source: app)\n --system-md-file <path> Read SYSTEM.md from <path> instead (source: user)\n -h, --help Show this help\n";
|
|
67
71
|
export declare function parseCliArgs(argv: readonly string[]): CliOptions;
|
|
68
72
|
export declare function runCli(argv: readonly string[], runtime: CliRuntime): Promise<number>;
|
|
69
73
|
export declare function runPromptMode(session: AgentSession, options: CliOptions, stdout: Writable, mode: "print" | "json"): Promise<void>;
|
package/dist/cli-runner.js
CHANGED
|
@@ -2,6 +2,7 @@ import { readFile } from "node:fs/promises";
|
|
|
2
2
|
import { basename, dirname } from "node:path";
|
|
3
3
|
import process from "node:process";
|
|
4
4
|
import { initUsage, runInitCommand } from "./cli-init.js";
|
|
5
|
+
import { providerAddUsage, runProviderAddCommand } from "./cli-provider-add.js";
|
|
5
6
|
import { createContributionRegistries, registerDiscoveredContributions } from "./contributions.js";
|
|
6
7
|
import { createAgent, createContributionRegistry, createMockProvider, providerDone, providerTextDelta, resolveInstructionInjectors, } from "./index.js";
|
|
7
8
|
import { discoverAgentBundles } from "./node/agent-definitions.js";
|
|
@@ -13,6 +14,7 @@ import { runRpcServer } from "./rpc.js";
|
|
|
13
14
|
import { createSkillRegistry } from "./skills.js";
|
|
14
15
|
export const usage = `Usage: prism [--mode print|json|rpc] [-p prompt] [options]
|
|
15
16
|
prism init <dir> [--provider <name>] [--with-workflows] [--with-evals] [--force]
|
|
17
|
+
prism providers add <name> [--base-url <url>] [--env-key <name>] [--model <id>] [--force]
|
|
16
18
|
|
|
17
19
|
Options:
|
|
18
20
|
-p, --prompt <text> Prompt to run in print/json mode
|
|
@@ -209,6 +211,16 @@ export async function runCli(argv, runtime) {
|
|
|
209
211
|
};
|
|
210
212
|
return runInitCommand(argv.slice(1), initRuntime);
|
|
211
213
|
}
|
|
214
|
+
if (argv[0] === "providers" && argv[1] === "add") {
|
|
215
|
+
const providerRuntime = {
|
|
216
|
+
stdout: runtime.stdout,
|
|
217
|
+
stderr: runtime.stderr,
|
|
218
|
+
...(runtime.providerTemplatesRoot !== undefined ? { templatesRoot: runtime.providerTemplatesRoot } : {}),
|
|
219
|
+
...(runtime.providerPackageVersion !== undefined ? { packageVersion: runtime.providerPackageVersion } : {}),
|
|
220
|
+
...(runtime.cwd !== undefined ? { cwd: runtime.cwd } : {}),
|
|
221
|
+
};
|
|
222
|
+
return runProviderAddCommand(argv.slice(2), providerRuntime);
|
|
223
|
+
}
|
|
212
224
|
let options;
|
|
213
225
|
try {
|
|
214
226
|
options = parseCliArgs(argv);
|
|
@@ -218,7 +230,7 @@ export async function runCli(argv, runtime) {
|
|
|
218
230
|
return 2;
|
|
219
231
|
}
|
|
220
232
|
if (options.help) {
|
|
221
|
-
write(runtime.stdout, `${usage}\n${initUsage}`);
|
|
233
|
+
write(runtime.stdout, `${usage}\n${initUsage}\n${providerAddUsage}`);
|
|
222
234
|
return 0;
|
|
223
235
|
}
|
|
224
236
|
try {
|
package/dist/index.d.ts
CHANGED
|
@@ -11,6 +11,8 @@ export type { ArtifactApproval, ArtifactApprovalState, ArtifactBodyErrorCode, Ar
|
|
|
11
11
|
export { ARTIFACT_BODY_ERROR_CODES, ARTIFACT_CHECKPOINT_NAMESPACE, ArtifactBodyStoreError, ArtifactError, artifactApprovalState, artifactCheckpointKey, } from "./artifacts.js";
|
|
12
12
|
export type { ApplyCacheControlOptions, CacheControlledContentBlock, CacheControlledMessage, CacheControlValue, CacheUsageReport, } from "./cache-helpers.js";
|
|
13
13
|
export { applyCacheControl, cacheHitRate, cacheSavings, cacheUsageReport, mapCacheRetention, sanitizeCacheKey } from "./cache-helpers.js";
|
|
14
|
+
export type { CacheTelemetry, CacheTelemetryOptions, CacheTelemetryReport, CacheTelemetrySample, } from "./cache-telemetry.js";
|
|
15
|
+
export { CACHE_TELEMETRY_OVERFLOW_KEY, CacheTelemetryError, DEFAULT_CACHE_TELEMETRY_CAP, createCacheTelemetry, } from "./cache-telemetry.js";
|
|
14
16
|
export type { MemoryCheckpointStoreOptions } from "./checkpoints.js";
|
|
15
17
|
export { CHECKPOINT_CONFLICT_CODE, CheckpointConflictError, createMemoryCheckpointStore } from "./checkpoints.js";
|
|
16
18
|
export type { DefaultCompactionStrategyOptions } from "./compaction.js";
|
|
@@ -105,5 +107,5 @@ export { createToolParameterValidator, createToolRegistry, dispatchToolCall, fil
|
|
|
105
107
|
export type { ResolvedUseCaseModel, ResolveUseCaseModelInput, UseCaseModelBinding, } from "./use-case-model.js";
|
|
106
108
|
export { resolveUseCaseModel, resolveUseCaseModelBinding, useCaseCredentialProviderId, } from "./use-case-model.js";
|
|
107
109
|
export declare const name = "prism";
|
|
108
|
-
export declare const version = "0.
|
|
110
|
+
export declare const version = "0.2.0";
|
|
109
111
|
export declare const description = "Agent harness for AI providers, agents, sessions, and tools.";
|
package/dist/index.js
CHANGED
|
@@ -6,6 +6,7 @@ export { AGENT_RUN_STATE_NAMESPACE, AGENT_RUN_STATE_SCHEMA_VERSION, agentFingerp
|
|
|
6
6
|
export { createAgent, createAgentSession, resumeAgentRun, resumeAgentRunStream } from "./agents.js";
|
|
7
7
|
export { ARTIFACT_BODY_ERROR_CODES, ARTIFACT_CHECKPOINT_NAMESPACE, ArtifactBodyStoreError, ArtifactError, artifactApprovalState, artifactCheckpointKey, } from "./artifacts.js";
|
|
8
8
|
export { applyCacheControl, cacheHitRate, cacheSavings, cacheUsageReport, mapCacheRetention, sanitizeCacheKey } from "./cache-helpers.js";
|
|
9
|
+
export { CACHE_TELEMETRY_OVERFLOW_KEY, CacheTelemetryError, DEFAULT_CACHE_TELEMETRY_CAP, createCacheTelemetry, } from "./cache-telemetry.js";
|
|
9
10
|
export { CHECKPOINT_CONFLICT_CODE, CheckpointConflictError, createMemoryCheckpointStore } from "./checkpoints.js";
|
|
10
11
|
export { createDefaultCompactionStrategy, isCompactionEntryData } from "./compaction.js";
|
|
11
12
|
export { assertJsonObject, isJsonObject, loadConfigLayers, mergeConfigLayers } from "./config.js";
|
|
@@ -57,6 +58,6 @@ export { DEFAULT_TOOL_RESULT_FOLD_MAX_SUMMARY_BYTES, DEFAULT_TOOL_RESULT_FOLD_MI
|
|
|
57
58
|
export { createToolParameterValidator, createToolRegistry, dispatchToolCall, filterTools } from "./tools.js";
|
|
58
59
|
export { resolveUseCaseModel, resolveUseCaseModelBinding, useCaseCredentialProviderId, } from "./use-case-model.js";
|
|
59
60
|
export const name = "prism";
|
|
60
|
-
export const version = "0.
|
|
61
|
+
export const version = "0.2.0";
|
|
61
62
|
export const description = "Agent harness for AI providers, agents, sessions, and tools.";
|
|
62
63
|
//# sourceMappingURL=index.js.map
|
|
@@ -187,6 +187,8 @@ Set `runState` with a host-owned `CheckpointStore`, stable `definitionRevision`,
|
|
|
187
187
|
|
|
188
188
|
`*_for_run` outcomes append a `StickyDecision` to the durable run state: later calls in the same run matching the scope exactly (all recorded fields) proceed or are blocked without a new suspension, policy still enforced at dispatch. Sticky decisions expire when the run reaches any terminal status. Caps: 32 pending decisions per run (hard 128), 64 sticky decisions (hard 256), 2 KB decision reasons, 16 KB elicitation payloads.
|
|
189
189
|
|
|
190
|
+
**Runtime input validation (0.2.0, plan 020 Task 2).** Every public resume entrypoint (`resumeAgentRun`, `resumeAgentRunStream`, `AgentRunLifecycle.resume()`/`resumeStream()`) validates the complete resume input in core before any checkpoint read/write, agent resolution, subscription, or tool execution: a non-null object, positive safe-integer `expectedVersion`, exactly one of `decision`/`decisions`, legacy `decision` exactly `approve`/`deny`, and a non-empty batch ≤ 128 entries whose entries are objects with a bounded non-empty `approvalId`, a whitelisted outcome, an optional string `reason` within the 2 KB limit, and JSON-object `modifiedArguments`/`elicitation` within the 16 KB limit. Unknown legacy decisions (e.g. `"sideways"`) and malformed untyped batches fail closed with `AgentDecisionError` (`ERR_PRISM_DECISION_INVALID`/`..._LIMIT`/`..._DUPLICATE`) under a **no-side-effect guarantee**: zero checkpoint writes/CAS changes, zero tool calls, zero resumed events. This holds for plain-JavaScript and `as any` callers; the server's transport parser is defense in depth, not the security boundary. State-dependent checks (foreign/stale approval ids, scope, schema, policy) still run in the atomic batch resolver.
|
|
191
|
+
|
|
190
192
|
```ts
|
|
191
193
|
const result = await session.run("Publish draft", {
|
|
192
194
|
runState: { checkpoints, definitionRevision: "2026-07-20.1", interruptBeforeTool: true },
|
package/docs/cli-rpc.md
CHANGED
|
@@ -36,6 +36,38 @@ prism init <dir> [--provider <name>] [--with-workflows] [--with-evals] [--force]
|
|
|
36
36
|
|
|
37
37
|
Default generation installs only `@arnilo/prism` (mock provider). Selecting a real provider adds exactly one `@arnilo/prism-provider-*` package. Storage, telemetry, memory, and server packages are never added unless a later phase introduces an explicit flag for them. Rerunning without `--force` refuses non-empty destinations and existing generated files. `.env.example` contains placeholders only; `.gitignore` excludes `.env` and local stores.
|
|
38
38
|
|
|
39
|
+
### `prism providers add` (0.1.7)
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
prism providers add <name> [--base-url <url>] [--env-key <name>] [--model <id>] [--force]
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
Scaffolds an OpenAI-compatible provider package into `./<name>`: `package.json`
|
|
46
|
+
(peer dep on `@arnilo/prism`, `sideEffects: false`, publish metadata mirroring
|
|
47
|
+
first-party providers), `tsconfig.json`, `README.md`, `CHANGELOG.md`,
|
|
48
|
+
`src/index.ts` (`defineProviderPackage` + auth-method registration),
|
|
49
|
+
`src/provider.ts` (built on `createOpenAICompatibleProvider`),
|
|
50
|
+
`src/models.ts` (starter `ModelConfig` list), `src/cache.ts` (cache-hint
|
|
51
|
+
mapping helpers via the shared core helpers), `src/__tests__/provider.test.ts`
|
|
52
|
+
(wired to `@arnilo/prism/testing/provider-conformance`), and a
|
|
53
|
+
`docs/providers/<name>.md` stub.
|
|
54
|
+
|
|
55
|
+
| Flag / arg | Purpose |
|
|
56
|
+
| --- | --- |
|
|
57
|
+
| `<name>` | npm-validated provider/package name (lowercase); also the target directory. |
|
|
58
|
+
| `--base-url <url>` | Default Chat Completions base URL (default `https://api.example.com/v1`). |
|
|
59
|
+
| `--env-key <name>` | Credential environment-var identifier, e.g. `ACME_API_KEY` (default `<NAME>_API_KEY`). |
|
|
60
|
+
| `--model <id>` | Starter model id (default `<name>-large`). |
|
|
61
|
+
| `--force` | Overwrite generated files when the destination already exists. |
|
|
62
|
+
| `-h`, `--help` | Print providers-add usage. |
|
|
63
|
+
|
|
64
|
+
Scaffold output is host-chosen: it is never auto-registered into repo
|
|
65
|
+
workspaces, umbrellas, or any resolver. The generated conformance test is
|
|
66
|
+
offline (mock fetch) and proves stream shape, tool-call delta reconstruction,
|
|
67
|
+
header ownership, secret-leak redaction, and serialized-content coverage
|
|
68
|
+
against the base provider. Replace the starter model metadata and the docs
|
|
69
|
+
stub with docs-verified values before publishing.
|
|
70
|
+
|
|
39
71
|
### Run/RPC CLI flags
|
|
40
72
|
|
|
41
73
|
| Flag | Purpose |
|
|
@@ -181,6 +213,7 @@ Suspended workflow resume parameters are `{ workflowId, runId, decision: "approv
|
|
|
181
213
|
- JSONL is processed line by line with Node stdlib; no parser dependency, worker, watcher, or queue is added.
|
|
182
214
|
- Unknown or malformed CLI/RPC input fails closed. Workflow resume validates decision and positive `expectedVersion`; ownership remains host-selected and checkpoint-enforced.
|
|
183
215
|
- `prism init` refuses non-empty destinations without `--force`, keeps writes inside the destination root, and never executes downloaded code beyond the user's later `npm install`.
|
|
216
|
+
- `prism providers add` validates the name against npm package-name rules (lowercase, no separators or `..`), refuses path traversal and symlinked directories escaping the destination, validates `--base-url` as an http(s) URL and `--env-key` as a shell-safe identifier, and writes placeholders only — generated code never contains secrets.
|
|
184
217
|
- Generated `.env.example` values are placeholders only; `.gitignore` excludes `.env` and local store files.
|
|
185
218
|
- Default generated install stays small (~27 MB with TypeScript tooling in a clean consumer install versus Mastra's measured 439 MB scaffold); unselected storage/telemetry/eval/workflow packages are omitted.
|
|
186
219
|
- Branch handles (`handleId`, `sessionId`, `leafId`) are identifiers only; do not encode credentials, tokens, provider objects, or secrets into them.
|
package/docs/coding-security.md
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
| `createSandboxCodingTools` / `createSandboxReadOnlyTools` | Thin wrappers that return `tools` only (compat); still require `workspaceMode`. |
|
|
14
14
|
| `createSandboxFilesystemOperations` / `createSandboxRepositoryOperations` | Optional execFile-backed FS/list/search backends for a disposable sandbox tree. |
|
|
15
15
|
| `createDockerSandbox(options)` | Creates one disposable non-root Docker container with read-only root/source, bounded tmpfs workspace, typed `execFile`, import/export, and stop/kill/cleanup. |
|
|
16
|
-
| `createNativeSandbox(options)` | Linux-only network-free backend: every command runs in a fresh network namespace (`unshare`), POSIX `ulimit` hard caps, cwd-in-root containment; fails closed at creation on platforms/privileges that cannot deny egress. Docker remains the stronger, documented reference backend. |
|
|
16
|
+
| `createNativeSandbox(options)` | Linux-only network-free backend: every command runs in a fresh network namespace (`unshare`), POSIX `ulimit` hard caps, cwd-in-root containment; fails closed at creation on platforms/privileges that cannot deny egress. Reports truthful capability metadata (`networkIsolated`/`egressRestricted` true, `filesystemIsolated`/`processIsolated`/`privilegeIsolated` false). Docker remains the stronger, documented reference backend. |
|
|
17
17
|
| `SandboxProcessHandle` | Optional long-running process handle (`write`/`signal`/`kill`/`release`/`wait`) returned by `DisposableSandbox.startProcess?`. |
|
|
18
18
|
| `createEgressPolicy(options)` | Deny-all allow-list policy: exact host/port/protocol rules plus frozen `npm-registry` / `github` presets; SHA-256 fingerprint. |
|
|
19
19
|
| `createAllowListEgressProxy(options)` | HTTP forward proxy + CONNECT tunnel enforcing the policy: pinned DNS (rebinding defense), private/metadata IP denial, redirect re-validation + hop cap, byte/time caps, per-decision audit, attestation for sandbox composition. |
|
|
@@ -34,7 +34,7 @@ Use this package when coding tools need path scoping, human approval, command ru
|
|
|
34
34
|
|
|
35
35
|
Use `createDockerSandbox()` when the host wants a production-reference containment boundary. Prism does **not** claim OS-level isolation unless the host constructs this adapter (or supplies an equivalent custom `DisposableSandbox`). Default policy denies shell/write/edit/delete/move without an `approve` callback and rejects paths outside configured roots. Coding shell definitions are marked `exclusive: true`, matching the approval policy's shell decision, so a single-shot turn containing shell work runs sequentially even when `toolConcurrency > 1`. Non-shell turns retain configured parallelism.
|
|
36
36
|
|
|
37
|
-
Use `createNativeSandbox()` when the host has no container runtime and needs network-free containment (0.1.6, plan 018 closeout `native-sandbox`). Linux only; creation fails closed with a documented error on other platforms or when the OS cannot create a network namespace (no root/CAP_SYS_ADMIN and no unprivileged user namespaces). Every command runs in a fresh netns — **loopback is down**, so even localhost connections fail; hosts that need loopback keep the Docker backend. Containment is egress denial + `ulimit` hard caps (address space from `memoryBytes`, CPU-time wall backstop, fd count from `maxFds`) + cwd-inside-root (`assertPathInsideRoots`, symlink-aware). The native backend does **not** isolate the filesystem: commands run as the invoking OS user with full host-tree access, so pair it with `createSandboxCodingComposition`/`createSandboxFilesystemOperations` (per-op `assertSandboxPath`) and the approval policy, exactly as with any custom `DisposableSandbox`. Host env is never inherited; `env` is an exact allow-list (PATH only by default). No `startProcess` (ProcessSessions fails closed with `ERR_PRISM_PROCESS_UNSUPPORTED`), no CPU-rate/pids/fs-size caps (cgroup-only). Secrets passed as `secrets` are redacted from surfaced errors. See `docs/_evidence/phase18-primitive-review.md` for the full threat model.
|
|
37
|
+
Use `createNativeSandbox()` when the host has no container runtime and needs network-free containment (0.1.6, plan 018 closeout `native-sandbox`). Linux only; creation fails closed with a documented error on other platforms or when the OS cannot create a network namespace (no root/CAP_SYS_ADMIN and no unprivileged user namespaces). Every command runs in a fresh netns — **loopback is down**, so even localhost connections fail; hosts that need loopback keep the Docker backend. Containment is egress denial + `ulimit` hard caps (address space from `memoryBytes`, CPU-time wall backstop, fd count from `maxFds`) + cwd-inside-root (`assertPathInsideRoots`, symlink-aware). The native backend does **not** isolate the filesystem: commands run as the invoking OS user with full host-tree access, so pair it with `createSandboxCodingComposition`/`createSandboxFilesystemOperations` (per-op `assertSandboxPath`) and the approval policy, exactly as with any custom `DisposableSandbox`. Its `capabilities` report `networkIsolated: true` and `egressRestricted: true` but `filesystemIsolated`/`processIsolated`/`privilegeIsolated: false` — the native backend is never a containment boundary for untrusted code (see [Sandbox capabilities](#sandbox-capabilities-020-plan-020-task-4)). Host env is never inherited; `env` is an exact allow-list (PATH only by default). No `startProcess` (ProcessSessions fails closed with `ERR_PRISM_PROCESS_UNSUPPORTED`), no CPU-rate/pids/fs-size caps (cgroup-only). Secrets passed as `secrets` are redacted from surfaced errors. See `docs/_evidence/phase18-primitive-review.md` for the full threat model.
|
|
38
38
|
|
|
39
39
|
Use `createEgressPolicy()` + `createAllowListEgressProxy()` when a coding agent needs outbound network access under an explicit allow list: package installs, forge API calls, or source fetches — never unrestricted egress. The proxy is inert until `start()`; nothing binds or resolves on import or construction.
|
|
40
40
|
|
|
@@ -104,7 +104,27 @@ const sandbox = await createDockerSandbox({ docker, image, sourceRoot, user, net
|
|
|
104
104
|
|
|
105
105
|
`createCodingApprovalPolicy()` returns an `ExecutionPolicy`. Allowed checks return `ExecutionDecision { allowed: true }`; denied checks include a stable reason; shell decisions set `exclusive: true`. Sandbox adapters return coding-agent-compatible `BashOperations`, receive `onData(Buffer)` for ordered stdout/stderr forwarding through the shell tool's existing bounded accumulator, and never grant policy approval themselves.
|
|
106
106
|
|
|
107
|
-
`createSandboxCodingComposition()` returns `{ tools, composition }` where `SandboxCodingComposition` carries `workspaceMode`, `
|
|
107
|
+
`createSandboxCodingComposition()` returns `{ tools, composition }` where `SandboxCodingComposition` carries `workspaceMode`, `capabilities`, `containmentClaim` (deprecated), `mixedWiringAllowed`, `warnings`, `workspaceRoot`, and optional `treeIdentity` (from `importIdentity` / `lastExportIdentity`). `capabilities` is a complete, frozen `SandboxCapabilities` object: `workspaceCoherent` derives from actual shell/filesystem/repository wiring; the isolation fields derive only from validated adapter capability metadata, never from `execFile`/`close` duck typing or custom operations being present. Host mode and escape-hatch mixed wiring always report no isolation capability — never treat host mode as contained execution.
|
|
108
|
+
|
|
109
|
+
### Sandbox capabilities (0.2.0, plan 020 Task 4)
|
|
110
|
+
|
|
111
|
+
Every sandbox backend and every composition reports a frozen, complete capability object:
|
|
112
|
+
|
|
113
|
+
| Capability | Meaning | Docker | Native | Custom / omitted |
|
|
114
|
+
| --- | --- | --- | --- | --- |
|
|
115
|
+
| `workspaceCoherent` | Shell, filesystem, and repository tools observe one workspace tree. | true | true | true only when tree backends are bound and mixed wiring denied |
|
|
116
|
+
| `filesystemIsolated` | Sandbox processes cannot touch the host filesystem. | true | **false** (full host-tree access by design) | false unless host attests |
|
|
117
|
+
| `networkIsolated` | No reachable network. | true only for `network: { mode: "none" }` | true (fresh netns per command, egress denied, loopback down) | false unless host attests |
|
|
118
|
+
| `processIsolated` | Sandbox processes run in a separate process namespace. | true | **false** | false unless host attests |
|
|
119
|
+
| `privilegeIsolated` | Sandbox processes cannot obtain host privileges. | **false** (root-in-container without user namespaces is not reliable) | **false** | false unless host attests |
|
|
120
|
+
| `egressRestricted` | Any egress is forced through a controlled proxy/firewall. | true for mode `none`, or a custom network carrying a validated `EgressAttestation` | true (no egress at all) | false unless host attests |
|
|
121
|
+
|
|
122
|
+
Rules:
|
|
123
|
+
|
|
124
|
+
- **Omission is false.** A `SandboxAdapter` without `capabilities` metadata — or with malformed metadata (non-object, missing fields, non-boolean values, unknown keys) — resolves every isolation field false. `resolveSandboxCapabilities()` validates, copies, and freezes explicit metadata; a backend can never gain a capability by omission, interface shape, or mixed wiring.
|
|
125
|
+
- **Explicit metadata is host attestation.** Prism validates shape and freezes the object; the host is responsible for the underlying controls being real.
|
|
126
|
+
- **`containmentClaim` is deprecated (0.2.0).** Retained for 0.1.7 compatibility as the conservative projection `workspaceCoherent && filesystemIsolated && networkIsolated && processIsolated` (privilege isolation excluded). It can only be `true` when every required capability is true — never authorize a security-sensitive action from this boolean alone; use `composition.capabilities`.
|
|
127
|
+
- **Capability construction is O(1)** — one small frozen object per sandbox/composition; no command, filesystem, Docker, DNS, or network operation.
|
|
108
128
|
|
|
109
129
|
`createDockerSandbox()` returns a `DisposableSandbox`: typed `execFile(file, args)`, shell-compatible `exec`, `status`, cooperative `stop`, forced `kill`, and idempotent `close`. Import may surface `importIdentity`; successful export updates `lastExportIdentity`. `close({ export })` can stream a bounded workspace tar plus SHA-256/entry/byte metadata through a host callback; checkpoints should retain only host artifact references/hashes, never whole workspaces. Optional `startProcess?(SandboxExecFileRequest)` returns a `SandboxProcessHandle` for long-running work consumed by coding-agent `createProcessSessions({ sandbox })`; absence means one-shot-only — ProcessSessions fails closed with `ERR_PRISM_PROCESS_UNSUPPORTED` (no native fallback). The Docker reference adapter does not implement `startProcess` yet; capability is detected, never assumed. See [Process sessions](process-sessions.md).
|
|
110
130
|
|
|
@@ -151,7 +171,8 @@ const { tools, composition } = createSandboxCodingComposition("/srv/jobs/task-1/
|
|
|
151
171
|
executionPolicy: policy,
|
|
152
172
|
repository: { exclude: [".git", "node_modules", "dist"] },
|
|
153
173
|
});
|
|
154
|
-
// composition.
|
|
174
|
+
// composition.capabilities.filesystemIsolated === true for the Docker adapter with network: none
|
|
175
|
+
// composition.containmentClaim is deprecated compatibility metadata only
|
|
155
176
|
|
|
156
177
|
// Same-tree Git/check (opt-in; not folded into coding tools):
|
|
157
178
|
const gitTools = createGitTools(composition.workspaceRoot, {
|
|
@@ -161,7 +182,7 @@ const gitTools = createGitTools(composition.workspaceRoot, {
|
|
|
161
182
|
|
|
162
183
|
// Host mode (explicit non-contained): omit sandbox; never claim containment.
|
|
163
184
|
const host = createSandboxCodingComposition(hostCwd, { workspaceMode: "host", executionPolicy: policy });
|
|
164
|
-
// host.composition.
|
|
185
|
+
// host.composition.capabilities: workspaceCoherent only; every isolation field false; containmentClaim false
|
|
165
186
|
|
|
166
187
|
await sandbox.execFile({ file: "npm", args: ["test"], cwd: "/workspace" });
|
|
167
188
|
await sandbox.close({
|
|
@@ -173,7 +194,7 @@ await sandbox.close({
|
|
|
173
194
|
|
|
174
195
|
Policies are ordinary host values: attach one globally through `createCodingTools()`/`createReadOnlyTools()`/`createSandboxCodingComposition()` or per tool. A per-tool policy overrides the shared policy. `SandboxAdapter` / `DisposableSandbox` are replaceable and host-owned; approval policy and sandboxing are separate layers. Custom remote sandboxes can implement `DisposableSandbox` without using Docker.
|
|
175
196
|
|
|
176
|
-
`createSandboxCodingComposition()` requires `workspaceMode`. Sandbox mode auto-wires FS/list/search through `DisposableSandbox.execFile` (or host-supplied custom operations) so mutations stay on the disposable tree until export. Host mode runs every coding tool against the host cwd and
|
|
197
|
+
`createSandboxCodingComposition()` requires `workspaceMode`. Sandbox mode auto-wires FS/list/search through `DisposableSandbox.execFile` (or host-supplied custom operations) so mutations stay on the disposable tree until export. Host mode runs every coding tool against the host cwd and reports no isolation capability (`containmentClaim` deprecated false). Sandbox shell + host FS throws unless `allowMixedWorkspaceWiring: true` (warnings + all capabilities false). Opt-in structured Git tools (`createGitTools(composition.workspaceRoot, { execFile: sandbox.execFile, commitIdentity })`) share the same tree/cwd; Prism still never pushes or opens PRs. Optional `@arnilo/prism-browser` can share the same disposable boundary: use `assertBrowserSandboxNetwork()` before browse-ready custom networks, and `createSharedSandboxBrowserOptions({ workspaceRoot, downloadsRoot, containedProxyAttestation })` so uploads/downloads align with `/workspace` and `/downloads`. Close the browser context before disposing the sandbox.
|
|
177
198
|
|
|
178
199
|
The Docker reference adapter starts by recorded container ID/label, uses argument arrays only, mounts source read-only, populates a size-bounded tmpfs `/workspace`, drops all capabilities, enables `no-new-privileges`, runs with `--init`, and never exposes the Docker socket, privileged mode, or host PID/IPC namespaces. Image pull/build/update stays outside Prism. Protected real-Docker checks are opt-in via `PRISM_TEST_DOCKER_SANDBOX=1` with host-supplied `PRISM_TEST_DOCKER_BIN` and digest-pinned `PRISM_TEST_DOCKER_IMAGE`.
|
|
179
200
|
|
package/docs/host-security.md
CHANGED
|
@@ -147,7 +147,7 @@ Wire those values where they matter: provider adapters receive the resolved cred
|
|
|
147
147
|
- AG-UI MCP Apps requires negotiated `mcpApps`, exact proxy origin/auth, owned-run context, approval, one bridge, separate-origin sandbox (`allow-scripts allow-same-origin`), and no-wider CSP. Never execute HTML in host origin or retry a UI mutation; Task 4 adds recovery.
|
|
148
148
|
- AG-UI A2A requires exact-origin verified client, host-owned task selection/correlation, explicit data/tool/A2UI projection, and reauthorized follow/cancel.
|
|
149
149
|
- `@arnilo/prism-server` exposes no agent/workflow by default and requires `authorize()` for every matched operation. Derive complete tenant/account/user ownership from validated host identity, never request JSON. Workflow active identity and cancellation compare exact ownership; a tenant-only scope intentionally cannot cancel a checkpoint/run carrying account or user identity. The artifact review service (`createArtifactService`) requires authenticated identity + thread ownership on every attach/revise/compare/approve/reject/download, resolves concurrent reviewers via checkpoint CAS (no lost approvals), rejects local filesystem paths in `uri`/citations, redacts records before persist and on response, and serves downloads only through signed expiring links that are reauthorized against the token's ownership per request. When a blob store is wired (`bodies: ArtifactBodyStore`, 0.0.28), delivery links additionally resolve through `bodies.presign`; the reference `createS3ArtifactBodyStore` verifies ownership on every operation, verifies size/SHA-256/MIME on put and get (fail closed), refuses delete under legal hold (host `isHeld` callback), keeps credentials host-resolved, and never discloses bucket/path/key in errors, telemetry, or artifact records. Pass the current explicitly revised workflow definition so recursive hash mismatch fails before abort or durable mutation. Configure exact host/origin allow-lists where needed, wire redaction before execution, retain tool/workflow policy checks, and adapt the Web handler behind host TLS/rate limits. Disconnect abort is default; persistent reconnect/status belongs to durable workflow checkpoints, not an invented in-memory agent result cache.
|
|
150
|
-
- Coding tools from `@arnilo/prism-coding-agent` accept an optional `ExecutionPolicy` checked inside each tool before side effects; shared policy propagation includes `createReadOnlyTools()`. They enforce finite text-scan/image/edit/write/shell limits, repository list/search depth/entry/match/scan/time caps, structured Git path/ref/message/output/patch/worktree caps, named-check concurrency/output caps, a 600-second default shell wall time, and a 64 MiB default total-output ceiling. Opt-in `createGitTools()` uses argument arrays with hooks/credential prompts/external diff disabled, requires host `commitIdentity` for commits, and never pushes or opens PRs. Successful truncated shell output leaves a host-owned exclusive `0600` temp file; delete `metadata.fullOutputPath` after use. Error/abort/timeout/overflow removes unpublished spills. Custom read/edit/shell/repository backends must honor supplied caps/signals. Use `@arnilo/prism-coding-security` for path roots, command rules, identity-scoped approval caching, required `workspaceMode` on `createSandboxCodingComposition()` / `createSandboxCodingTools()`, and the optional `createDockerSandbox()` reference adapter. **Host mode is never contained execution** (
|
|
150
|
+
- Coding tools from `@arnilo/prism-coding-agent` accept an optional `ExecutionPolicy` checked inside each tool before side effects; shared policy propagation includes `createReadOnlyTools()`. They enforce finite text-scan/image/edit/write/shell limits, repository list/search depth/entry/match/scan/time caps, structured Git path/ref/message/output/patch/worktree caps, named-check concurrency/output caps, a 600-second default shell wall time, and a 64 MiB default total-output ceiling. Opt-in `createGitTools()` uses argument arrays with hooks/credential prompts/external diff disabled, requires host `commitIdentity` for commits, and never pushes or opens PRs. Successful truncated shell output leaves a host-owned exclusive `0600` temp file; delete `metadata.fullOutputPath` after use. Error/abort/timeout/overflow removes unpublished spills. Custom read/edit/shell/repository backends must honor supplied caps/signals. Use `@arnilo/prism-coding-security` for path roots, command rules, identity-scoped approval caching, required `workspaceMode` on `createSandboxCodingComposition()` / `createSandboxCodingTools()`, and the optional `createDockerSandbox()` reference adapter. **Host mode is never contained execution** (every isolation capability false). Sandbox mode reports isolation only from validated adapter capability metadata: `composition.capabilities` carries the frozen `SandboxCapabilities` object (`workspaceCoherent`, `filesystemIsolated`, `networkIsolated`, `processIsolated`, `privilegeIsolated`, `egressRestricted`); the deprecated `containmentClaim` is a conservative projection and must never be used alone. Authorize security-sensitive actions from the individual capabilities the policy actually needs — e.g. require `filesystemIsolated` before hosting untrusted coding tasks, and `egressRestricted` before any network-capable run. Mixed wiring requires `allowMixedWorkspaceWiring` and still reports no isolation. Limits alone are not containment: construct the Docker adapter (absolute CLI, digest-pinned image, network none by default) or an equivalent host sandbox before treating coding execution as production-safe. Docker daemon/image trust, egress firewall/proxy, and artifact retention remain host-owned.
|
|
151
151
|
- Allow-list egress (0.0.26, `@arnilo/prism-coding-security`): `createEgressPolicy()` is deny-all with exact host/port/protocol rules and frozen `npm-registry`/`github` presets; `createAllowListEgressProxy()` is an HTTP forward proxy + CONNECT tunnel that pins DNS answers and verifies the connected address (rebinding defense), denies private/link-local/metadata ranges unless a rule opts in, re-validates every redirect hop against policy, and cuts oversized/slow transfers at frozen byte/time caps. TLS passes through without interception. Every allow/deny writes an audit record with no secrets. The proxy is inert until `start()`; `reloadPolicy()` is the only rule change path. `composeEgressSandboxNetwork(proxy.attestation(), name)` records validated attestation as `prism.egress.*` container labels — evidence, not enforcement: the host must restrict the Docker network so the proxy is the only reachable path, and `denyDirectEgress: true` is a claim the host makes true by topology. The proxy is not a firewall and cannot stop a container whose network reaches the internet directly.
|
|
152
152
|
- Optional `@arnilo/prism-browser` requires a host-supplied Playwright Browser (`playwright-core@1.61.0` peer). Import is inert. One non-persistent context belongs to one run; actions serialize; refs are snapshot-scoped; CSS/evaluate/CDP/persistent profiles are denied. Context routing + `serviceWorkers: "block"` deny file/data/blob/devtools/private/loopback by default and require contained-proxy attestation for external egress (Playwright routing is defense in depth, not DNS containment). Uploads are realpath-rooted; downloads quarantine with hash/MIME until host `approveRelease`; screenshots return bounded `ImageContent`. Observation vs mutation/high-impact actions map to `ExecutionPolicy`. Treat snapshot/page text as untrusted external content. Close contexts with `browser_close` or `manager.closeRun(runId)` on abort/terminal. Browser control endpoint, binary/image pin, and real egress firewall/proxy remain host-owned. Shared sandbox: `createSharedSandboxBrowserOptions()` + `assertBrowserSandboxNetwork()`.
|
|
153
153
|
- Browser verified-state checkpoints (0.0.14, `createBrowserCheckpointLedger()`) store URL + domain-state hash + host data refs only — never serialized browser internals (cookies/storage/contexts). After any resume/interruption the ledger fails closed (`assertVerifiedBeforeSideEffect`) until the host reloads + verifies, so side effects never replay on stale state.
|
package/docs/index.md
CHANGED
|
@@ -8,10 +8,10 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
8
8
|
## Identity and governance
|
|
9
9
|
- [Agent identity](agent-identity.md): host-verified `Principal` / `AgentIdentity`, delegation narrowing, ownership projection, and redacted telemetry refs for enterprise runs/tools/MCP/A2A/workflows; optional OIDC/JWKS verifier adapter (`@arnilo/prism-credentials-node/oidc` — pinned issuer/audience/JWKS, bounded claims, fail closed).
|
|
10
10
|
- [Policy and audit](policy-and-audit.md): optional `@arnilo/prism-policy` decision ledger (allow/deny/modify/approval), evidence refs only, cursor-paginated WORM export, and durable PostgreSQL composition; 0.0.28 adds the OPA REST evaluator (`@arnilo/prism-policy/opa` — pinned SSRF-checked endpoint, redacted input, fail-closed deny, optional bundle-revision pin).
|
|
11
|
-
- [Model routing](model-routing.md): optional `@arnilo/prism-model-router` allow-list/residency/budget/rate/circuit/fallback governance with redacted diagnostics; durable state requires awaited identity-scoped calls.
|
|
11
|
+
- [Model routing](model-routing.md): optional `@arnilo/prism-model-router` allow-list/residency/budget/rate/circuit/fallback governance with redacted diagnostics; durable state requires awaited identity-scoped calls; host-configurable selection policies (reference cost/latency policy ranks by `ModelCost` then in-memory latency EMA fed from `recordOutcome`).
|
|
12
12
|
|
|
13
13
|
## Agent/session runtime
|
|
14
|
-
- [Agent/session runtime](agent-session-runtime.md): create explicit or opt-in secure agents/sessions, get direct `AgentRunResult` values from `run`/`prompt`, mid-run `steer` (turn-boundary or softInterrupt), use integrated `stream()`/`resumeAgentRunStream()`, shared batch pending-decisions / sticky run-scope approvals, subscribe to normalized events, and expose opted-in durable lifecycle capabilities (0.1.6 plan 018 closeout `checkpoint-bodies`: optional `includeSkillBodies` persists the exact loaded-skill instructions with the names-only `persistSessionState`, so resume re-renders bodies registry-independently; ≤64 bodies, `maxStateBytes` refuses oversize).
|
|
14
|
+
- [Agent/session runtime](agent-session-runtime.md): create explicit or opt-in secure agents/sessions, get direct `AgentRunResult` values from `run`/`prompt`, mid-run `steer` (turn-boundary or softInterrupt), use integrated `stream()`/`resumeAgentRunStream()`, shared batch pending-decisions / sticky run-scope approvals, subscribe to normalized events, and expose opted-in durable lifecycle capabilities (0.1.6 plan 018 closeout `checkpoint-bodies`: optional `includeSkillBodies` persists the exact loaded-skill instructions with the names-only `persistSessionState`, so resume re-renders bodies registry-independently; ≤64 bodies, `maxStateBytes` refuses oversize; 0.2.0 plan 020 Task 2: durable resume input is validated in core before any side effect — fail-closed durable resume rejects unknown legacy decisions and malformed batches with zero checkpoint writes, tool calls, or resumed events).
|
|
15
15
|
- [Agent definitions](agent-definitions.md): resolve declarative `AgentDefinition` values via `resolveAgentDefinition`, and turn app-config `<configRoot>/agents/<name>/AGENT.md` bundles into runnable agents via `discoverAgentBundles` / `resolveAgentBundle` (explicit tool/skill activation by name, fail-closed omitted capabilities, migration-only `activateAllCapabilities`, strict duplicate scope checks, configurable prompt layers, no auto-discovery).
|
|
16
16
|
- [Agent loops](agent-loops.md): replaceable per-run control loops — `singleShotLoop` default, opt-in bounded artifact-loop tool rounds, and durable custom-loop `revision`/`snapshot`/`restore` hooks with fail-closed resume.
|
|
17
17
|
- [Guardrails](guardrails.md): typed fail-closed input/output/tool checks with buffered provider output and redacted decision records.
|
|
@@ -43,7 +43,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
43
43
|
- [Provider primitives](provider-primitives.md): shared bounded transport and OpenAI serialization helpers — migrated across first-party providers; native structured-output and observability contracts.
|
|
44
44
|
- [Provider layer](provider-layer.md): register and resolve host-owned providers/models, choose replace-or-error duplicate policy, create provider events, stream/reconstruct tool-call deltas, use generic provider request options, and test with the mock provider; deprecated provider-level timeout/retry hints point to runtime abort/retry.
|
|
45
45
|
- [Model registry](model-registry.md): register and resolve `ModelConfig` records with capabilities, limits, cost, cache support metadata, compat data, and duplicate policy.
|
|
46
|
-
- [Provider caching](provider-caching.md): use `PromptCacheHints`, `PromptCacheBreakpoint`, `ModelCacheCapabilities`, cache-aware stable-prefix guidance, and shared cache diagnostics helpers; includes the complete per-provider explicit/implicit cache matrix plus no-Prism-cache entries (including Anthropic, Google, Alibaba, Ollama, cloud adapters, and host-owned AI SDK); cache hints are best-effort and cache keys are never secrets.
|
|
46
|
+
- [Provider caching](provider-caching.md): use `PromptCacheHints`, `PromptCacheBreakpoint`, `ModelCacheCapabilities`, cache-aware stable-prefix guidance, and shared cache diagnostics helpers; includes the complete per-provider explicit/implicit cache matrix plus no-Prism-cache entries (including Anthropic, Google, Alibaba, Ollama, cloud adapters, and host-owned AI SDK); cache hints are best-effort and cache keys are never secrets; `createCacheTelemetry()` aggregates per-provider/model hit rate and cache-token totals from the `usage` event stream for tuning the `cache_aware` layout.
|
|
47
47
|
- [Thinking and reasoning](thinking-and-reasoning.md): portable `ThinkingLevel` helpers (`applyThinkingLevel` / `thinkingCompatFor`) map per-turn effort into provider `compat` fields; model defaults stay on `ModelConfig.compat`; no second options tree.
|
|
48
48
|
- [Use-case model selection](use-case-model-selection.md): bind `{ model?, provider?, thinkingLevel? }` for observational memory, LLM compaction, and other non-session LLM jobs with explicit session-model fallback via `resolveUseCaseModel`.
|
|
49
49
|
- [Provider request policies](provider-request-policies.md): chain `ProviderRequestPolicy` hooks, use `createSessionCachePolicy`, and merge legacy/structured cache options safely.
|
|
@@ -70,7 +70,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
70
70
|
- [Tool validator JSON Schema package](../packages/tool-validator-json-schema/README.md): optional `@arnilo/prism-tool-validator-json-schema` adapter for `tool.parameters`.
|
|
71
71
|
- [MCP client bridge and server exposure](mcp-tools.md): SDK-1.30.0 bounded tools/resources/prompts, host-owned roots/sampling/elicitation, exact-origin DNS-pinned client transport, and principal-bound opt-in Streamable HTTP sessions. 0.0.28 adds MCP OAuth: `createMcpOAuthTransport`/`createMcpOAuthFetch`/`createMcpClientAuth` (RFC 9728/8414 discovery, PKCE, RFC 8707 audience binding, RFC 7009 revocation, host-owned bounded state) and server `protectedResource` metadata + `WWW-Authenticate` challenges.
|
|
72
72
|
- [Web search, fetch, and extraction](web-tools.md): optional host-selected Brave/Exa discovery and Firecrawl Markdown/schema tools with native fetch, stable citations, late credentials, finite limits, and explicit untrusted-content boundaries.
|
|
73
|
-
- [Work tools](work-tools.md): optional `@arnilo/prism-work-tools` identity-scoped M365 + GWS connectors (hard-coded CLI argv, draft-then-approve, state-machine idempotency, shared result shapes); 0.0.14 adds a late-bound per-identity `tokenProvider` (env-only, fail-closed).
|
|
73
|
+
- [Work tools](work-tools.md): optional `@arnilo/prism-work-tools` identity-scoped M365 + GWS connectors (hard-coded CLI argv, draft-then-approve, state-machine idempotency, shared result shapes); 0.0.14 adds a late-bound per-identity `tokenProvider` (env-only, fail-closed); 0.2.0 plan 020 Task 3 provides an isolated subprocess environment (fixed allow-listed base + explicit env + late-bound token env, forced `HOME`/telemetry controls, 64-name/64-KiB caps) and requires host-pinned **absolute** binary/configDir paths.
|
|
74
74
|
- [Work connectors](work-connectors.md): connector principles, capability gates, scoped OAuth establishment (0.0.14), and out-of-scope boundaries (Slack/Teams channels not shipped) for Microsoft 365 / Google Workspace.
|
|
75
75
|
- [Browser automation](browser-automation.md): optional `@arnilo/prism-browser` with host-supplied Playwright contexts, AI-mode snapshots/refs, ordered `browser_open`/`browser_snapshot`/`browser_act`/`browser_close` plus (0.1.4) `browser_evaluate`/`browser_observe` and CDP `block_urls`/`unblock_urls`/`throttle`/`emulate` act actions on Chromium hosts, egress/side-effect/upload/download/screenshot policy, finite page/action/snapshot/network/artifact caps, and 0.0.14 verified-state checkpoints with reload/verify-before-side-effect.
|
|
76
76
|
- [Device adapters](device-adapters.md): deny-by-default realtime voice / desktop-control contract + conformance (0.0.14); no vendor package — admission fails closed without explicit consent+sandbox+approval, stream bounds, shared `RunLimits`, redacted telemetry.
|
|
@@ -78,7 +78,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
78
78
|
- [Language intelligence](language-intelligence.md): optional host-activated `createLanguageIntelligence` — bounded in-package LSP 3.17 JSON-RPC client (Content-Length framing), host-selected server command/args per language, workspace symbols/definitions/references/diagnostics/hover/rename; lazy spawn; URI root confinement; rename gated by `ExecutionPolicy` + atomic write/mutation queue; frozen message/diagnostic/pending/result/timeout/server caps. No `vscode-languageserver-protocol` dependency.
|
|
79
79
|
- [Process sessions](process-sessions.md): optional host-activated `createProcessSessions` — long-running process registry (start/cursor-paged output/input/wait/signal/kill/release), native or sandbox `startProcess` backend (fail closed when absent), ownership/identity + expiry sweep on access, `reconcile` / sandbox-loss → `unknown` (never fabricates exitCode), durable command fingerprint metadata, `CodingProcessEvent` host sink, `ExecutionPolicy` before spawn and on mutate, frozen session/input/lifetime/output caps; PTY fails closed as unsupported.
|
|
80
80
|
- [Forge integration](forge-integration.md): optional host-activated `createGitHubForge` — reference GitHub adapter (issue context, authenticated push via `BoundGitRunner` + `GIT_CONFIG_*` credential injection, PR create/update, review comments, check/status retrieval, bounded `reconcileHandoff`), every mutation gated by `ExecutionPolicy` and recorded in `ToolEffectStore` (retry never duplicates PRs/comments), typed `ForgeError` codes (auth/API/stale/rate-limit/limit/ownership), frozen page/payload/comment/concurrency/timeout caps, no octokit dependency, tokens never in argv/logs/events.
|
|
81
|
-
- [Coding execution approval and sandboxing](coding-security.md): path/command approval, identity-scoped caching, shell-turn exclusivity, required `workspaceMode` (`host`/`sandbox`) with fail-closed mixed wiring, `createSandboxCodingComposition()`
|
|
81
|
+
- [Coding execution approval and sandboxing](coding-security.md): path/command approval, identity-scoped caching, shell-turn exclusivity, required `workspaceMode` (`host`/`sandbox`) with fail-closed mixed wiring, `createSandboxCodingComposition()` sandbox capability metadata — 0.2.0 plan 020 Task 4 ships explicit `SandboxCapabilities` (`workspaceCoherent`/`filesystemIsolated`/`networkIsolated`/`processIsolated`/`privilegeIsolated`/`egressRestricted`) with omission resolving false, truthful Docker/native metadata, and `containmentClaim` retained only as a deprecated conservative projection — disposable Docker/OCI sandbox reference with bounded workspace import/export (0.1.6 adds the Linux-only network-free `createNativeSandbox` backend — fresh netns per command via `unshare`, `ulimit` hard caps, cwd containment, fails closed where egress denial is impossible), optional `DisposableSandbox.startProcess` / `SandboxProcessHandle` for process-session backends, and allow-list egress (`createEgressPolicy` deny-all exact rules + frozen presets, `createAllowListEgressProxy` HTTP/CONNECT proxy with pinned-DNS rebinding defense, private/metadata IP denial, redirect re-validation + hop cap, byte/time caps, per-decision audit, `composeEgressSandboxNetwork` attestation recorded as `prism.egress.*` labels; TLS pass-through, no interception).
|
|
82
82
|
|
|
83
83
|
## Extensions/plugins
|
|
84
84
|
- [Contribution discovery (workspace)](contribution-discovery.md): opt-in, realpath-contained directory scanner turning `SKILL.md`/`manifest.json` into inert `DiscoveredContribution` envelopes the host registers — no `import()`, no auto-activate, no provider scanning. Per-agent bundles remain app-controlled and are documented under Agent/session runtime.
|
|
@@ -103,7 +103,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
103
103
|
- [AG-UI adoption evaluation](ag-ui-adoption.md): official 0.0.57 input/event/capability matrix and shipped hardened MCP/MCP Apps/A2A handshake boundaries.
|
|
104
104
|
|
|
105
105
|
## CLI/RPC
|
|
106
|
-
- [CLI/RPC](cli-rpc.md): Run print/json modes and LF-delimited RPC over the public AgentSession runtime, including mid-run `steer`, branch-handle results, fixed `forkSession`, and `checkout`. `prism init` scaffolds a tiny TypeScript project with one selected provider and an offline mock test.
|
|
106
|
+
- [CLI/RPC](cli-rpc.md): Run print/json modes and LF-delimited RPC over the public AgentSession runtime, including mid-run `steer`, branch-handle results, fixed `forkSession`, and `checkout`. `prism init` scaffolds a tiny TypeScript project with one selected provider and an offline mock test; `prism providers add <name>` scaffolds an OpenAI-compatible provider package (manifest, provider, models, cache helpers, conformance test, docs stub).
|
|
107
107
|
- [Workflows](workflows.md): optional `@arnilo/prism-workflows` typed bounded DAG orchestration — explicit recursive definition revisions, exact-owner cancellation/active identity, finite hard limits, durable human suspend/resume, schedules/background execution, revocable proactive schedule capability tokens, nested workflows, replay, coordination, events, and optional RPC/Web bindings. Compose coding plans/checkpoints via workspace Markdown + `state.coding` without a second runtime. Interactive TUI (C-012) deferred.
|
|
108
108
|
- [Workflow orchestration primitives](workflow-orchestration-primitives.md): architecture inventory — workflow adapters consume core `CheckpointStore`, `LeaseStore`, and bounded `EventMultiplexer`; run control and optional RPC commands stay package-local.
|
|
109
109
|
- [Workflow/TUI scope](workflow-tui-primitives.md): records why 0.0.5 ships workflow APIs/RPC control but no interactive terminal UI.
|
|
@@ -129,7 +129,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
129
129
|
- [Ponytail behavior integration](ponytail.md): optional `@arnilo/prism-ponytail` — upstream Ponytail skills/commands, `ponytail-mode` injector, session `ponytail-mode` persistence; resolves peer `@dietrichgebert/ponytail` or `upstreamPath`; opt-in (not in code/sdk profiles).
|
|
130
130
|
|
|
131
131
|
## Release and install
|
|
132
|
-
- [Release and install](release-and-install.md): current **0.
|
|
132
|
+
- [Release and install](release-and-install.md): current **0.2.0** 50-package graph (root + 49 workspace packages) — plan 020 the fail-closed runtime-and-sandbox-security cut on the 0.2.x review-remediation line: durable-resume decision validation in core (`assertValidAgentRunResume` — unknown decisions/malformed batches fail closed with `ERR_PRISM_DECISION_*` before any state claim, checkpoint write, or tool execution; server parser remains defense in depth), isolated work-tool subprocess environments (`@arnilo/prism-work-tools` — fixed base allow-list + explicit env + forced HOME/telemetry + late-bound per-identity tokens, 64-name/64-KiB caps, absolute binary/configDir, linear output capture), and explicit sandbox capabilities (`@arnilo/prism-coding-security` — `SandboxAdapter.capabilities` with omission-is-false fail-closed resolution, `SandboxCodingComposition.capabilities` from verified wiring, `containmentClaim` deprecated as the conservative projection; Docker reports only verified controls, native reports filesystem/process/privilege `false`); public-entrypoint security conformance (`scripts/phase20-security.test.mjs`, wired into `security:threat-suites`), packed plain-JS consumer regressions, and the sandbox-browser workflow's fail-loud Docker/native capability evidence gate — 0.2.0 never ships while a blocker is skipped; migration and rollback notes in `docs/migration.md` `0.1.7 → 0.2.0`, store-compatible with 0.1.7 in both directions; 0.1.7 was the performance-and-DX patch — dependency-free `createCacheTelemetry()` per-provider/model cache hit/miss aggregator (bounded cardinality with `__overflow__`, token counters/rates only, host-activated), host-configurable `ModelRouterSelectionPolicy` on `createModelRouter` with the reference `createCostLatencySelection` (ModelCost rank then in-memory latency EMA, default ordered behavior byte-identical), `prism providers add <name>` OpenAI-compatible provider scaffold (manifest/provider/models/cache/conformance test/docs stub, npm-name + traversal + symlink-escape validation, placeholders only), and the async `AgUiProjection` verification closeout (plan 009 Task 15 evidence recorded, no new code); plan 017 the documented breaking cut — deprecated-option removal with `docs/migration.md` `0.1.4 → 0.1.5` section and reviewed compat-baseline regeneration via `--allow-break` then `--update-baseline`: the inert provider request knobs, `RunOptions.maxToolRounds`, observational-memory flat settings keys + top-level worker aliases, `ReadToolOptions.autoResizeImages`, `INIT_PROVIDERS`; all removals fail closed naming their replacement; plan 016 internal god-module split — `agents.ts`/`contracts.ts` reorganized behind barrel re-exports with a byte-identical public entry surface, measured tree-shaking improvement in `scripts/phase16-baseline.json`, and additive `@arnilo/prism-browser` Chrome DevTools Protocol capabilities — `browser_evaluate`/`browser_observe` and `block_urls`/`unblock_urls`/`throttle`/`emulate` act actions; plan 015 dead-code and deprecation hygiene on the frozen 0.1.x line — parameterized benchmark runner `scripts/benchmark.mjs` absorbing the per-version runners, archived review-coverage evidence in `docs/_evidence/`, non-blocking unused-code sweep `npm run sweep:unused`, opt-in checkpoint persistence for loaded-skill names and read-path sets; plan 014 Alibaba provider enrichment — embeddings, video input, verified compatible-mode surface decision table; plan 013 post-release hardening — build single-flight, MCP SSE relay test, combined coverage summary, canonical manifest-count narrative, ACP modes/config persistence guidance; Phase 12 release-candidate hardening; plan 012 — freeze manifest, compatibility matrix, upgrade matrix, packed-install e2e journeys, restart-recovery evidence, capacity envelopes, security policy), exact-peer/install/tarball rules, deterministic resumable publication and publish dry-run, frozen 0.1.x compatibility and support matrix (Node/PostgreSQL/platform/provider/protocol pins and unsupported combinations, machine-checked against `scripts/phase12-freeze-manifest.json`), protected PostgreSQL gate, pinned supply-chain gates, offline tests, the 0.0.15 provider/AI-SDK/RAG/memory protected live-canary matrix, and sandbox-browser Docker/Playwright gates.
|
|
133
133
|
- [0.1.0 / 1.0 readiness gates](0.1.0-readiness.md): command-per-gate 1.0 readiness table — frozen API surface + compat gate, migration/docs tripwires, budget table, live-suite matrix, security matrix, current-line status (**0.0.23** published target), signed-publication/live-canary prerequisites for 1.0, and Phase 12 demand-evidence entry criteria.
|
|
134
134
|
- [Review coverage archive](_evidence/): per-phase evidence freezes (plans 067–079, releases 0.0.4–0.0.16) — traceability matrices, provider validation, capability/primitive/limit matrices, benchmark budgets, and artifact-diet findings; tarball-excluded, kept in-repo for audit.
|
|
135
135
|
|
package/docs/migration.md
CHANGED
|
@@ -1,5 +1,30 @@
|
|
|
1
1
|
# Migration guide
|
|
2
2
|
|
|
3
|
+
## 0.1.7 → 0.2.0 fail-closed runtime and sandbox security (plan 020)
|
|
4
|
+
|
|
5
|
+
Release **0.2.0** (plan 020) is the first cut of the 0.2.x review-remediation line: it closes the three security blockers found in the 2026-08-12 comprehensive review. The API surface is **additive-only** (plain compat gate at 0.2.0 shows zero removed/changed declarations; no `--allow-break`), but three behaviors are deliberately tightened for security, so untyped/legacy callers may now fail where 0.1.7 silently proceeded:
|
|
6
|
+
|
|
7
|
+
1. **Durable-resume decision validation (core).** `resumeAgentRun`/`resumeAgentRunStream` (and the lifecycle/resume-stream entrypoints behind them) now validate the resume payload **before any state claim, checkpoint write, or tool execution**. Unknown legacy decisions (anything other than `approve`/`deny`), malformed decision batches, oversized reasons/elicitation, and duplicate approval ids fail closed with a stable `AgentDecisionError` (`ERR_PRISM_DECISION_INVALID`/`ERR_PRISM_DECISION_LIMIT`/`ERR_PRISM_DECISION_DUPLICATE`), leave the checkpoint version untouched, and execute no tool. In 0.1.7 an unknown decision string (e.g. `"sideways"`) was accepted, the checkpoint was CAS-claimed to `running`, and the suspended tool executed. The HTTP server parser (`readAgentDecisions`) is unchanged — it remains defense in depth, not the security boundary.
|
|
8
|
+
|
|
9
|
+
```js
|
|
10
|
+
// 0.2.0: fails closed, no side effect, version untouched
|
|
11
|
+
try {
|
|
12
|
+
await resumeAgentRun(agent, ref, { expectedVersion: v, decision: "sideways" }, opts);
|
|
13
|
+
} catch (error) {
|
|
14
|
+
error.code; // "ERR_PRISM_DECISION_INVALID"
|
|
15
|
+
}
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
2. **Work-tool subprocess environments (`@arnilo/prism-work-tools`).** `createCliRunner` no longer inherits the full host `process.env`. The child environment is now: fixed base allow-list (`PATH`, `LANG`, `LC_ALL`, `TZ`; Windows adds `SYSTEMROOT`/`SystemRoot`/`TEMP`/`TMP`/`PATHEXT`/`COMSPEC`), then explicit validated `options.env`, then forced controls (`HOME` = `configDir`, `CLIMICROSOFT365_DISABLETELEMETRY=1`), then the late-bound per-identity token layer (`M365_ACCESSTOKEN`/`GOOGLE_ACCESS_TOKEN` style). Caps: 64 names / 64 KiB total (`ERR_PRISM_WORK_ENV`). `binary` and `configDir` must now be **absolute paths** (`path.isAbsolute`), and output capture is linear (single final `Buffer.concat`, capped at `maxStdoutBytes`/`maxStderrBytes`). In 0.1.7 the child inherited every ambient host variable.
|
|
19
|
+
|
|
20
|
+
3. **Explicit sandbox capabilities (`@arnilo/prism-coding-security`).** `SandboxAdapter` gains the optional `capabilities` field — `workspaceCoherent`/`filesystemIsolated`/`networkIsolated`/`processIsolated`/`privilegeIsolated`/`egressRestricted` (immutable booleans). Omission or malformed metadata resolves every isolation field `false` (fail-closed). `SandboxCodingComposition` now carries a resolved `capabilities` object; the old boolean `containmentClaim` is **deprecated** and is the conservative projection `workspaceCoherent && filesystemIsolated && networkIsolated && processIsolated`. Built-ins: Docker reports `filesystemIsolated: true`/`processIsolated: true`/`networkIsolated: true` only for `--network=none`/attested networks, `privilegeIsolated: false` by default; native sandbox reports `networkIsolated: true`/`egressRestricted: true` but **never** filesystem/process/privilege isolation. In 0.1.7 any `DisposableSandbox`-shaped adapter could make `containmentClaim` report `true` with no isolation-capability inspection; in 0.2.0 an un-attested adapter claims `workspaceCoherent` at most. Authorization should read the individual capabilities, never the deprecated boolean.
|
|
21
|
+
|
|
22
|
+
**Store compatibility:** 0.2.0 is store-compatible with 0.1.7 in both directions — no persisted-shape change, no migration step. Checkpoint, session-store, approval, and registry payloads are byte-identical; only the resume *input* validation is new.
|
|
23
|
+
|
|
24
|
+
**Rollout:** upgrade core first (resume validation applies immediately to all hosts), then `@arnilo/prism-work-tools` (pass absolute `binary`/`configDir` and any ambient keys your connector needs via `options.env` — the allow-list is deny-by-default by design), then `@arnilo/prism-coding-security` (capability-aware policy code; the deprecated `containmentClaim` keeps working with the stricter semantics).
|
|
25
|
+
|
|
26
|
+
**Rollback risk:** restoring 0.1.7 restores all three defects — rollback is **not** a mitigation. Hosts that must roll back should disable resume side effects and work-tool execution at their own boundary until they can return to 0.2.0.
|
|
27
|
+
|
|
3
28
|
## 0.1.4 → 0.1.5 deprecated-option removal (documented breaking cut)
|
|
4
29
|
|
|
5
30
|
Release **0.1.5** (plan 017) removes the deprecated compatibility surface that 0.1.x kept after 0.0.19: the inert provider timeout/retry knobs, the `maxToolRounds` run-option alias, the pre-0.0.19 observational-memory flat keys and worker aliases, the read-tool `autoResizeImages` flag, and the `INIT_PROVIDERS` constant. This is the **documented breaking cut** announced in the 0.1.4 migration section; every other 0.1.x release keeps the compat baseline green. Three roadmap labels from the original 0.1.5 task were corrected during planning and are honored here:
|
package/docs/model-routing.md
CHANGED
|
@@ -91,6 +91,51 @@ await enterprise.close();
|
|
|
91
91
|
|
|
92
92
|
Router is optional. Chain returned `providerRequestPolicy` with other `ProviderRequestPolicy` values. Wire `onDiagnostics` to `@arnilo/prism-policy` when audit export is required. OpenRouter package behavior is unchanged; routing metadata participates only when this gate allows it.
|
|
93
93
|
|
|
94
|
+
### Selection policies (0.1.7)
|
|
95
|
+
|
|
96
|
+
By default the router tries candidates in input order: the primary model, then
|
|
97
|
+
`fallbacks` in order. A host can instead supply a `selection` policy on
|
|
98
|
+
`createModelRouter` to rank the candidates before the governance checks run:
|
|
99
|
+
|
|
100
|
+
```ts
|
|
101
|
+
import { createCostLatencySelection, createModelRouter } from "@arnilo/prism-model-router";
|
|
102
|
+
|
|
103
|
+
const router = createModelRouter({
|
|
104
|
+
resolver,
|
|
105
|
+
selection: createCostLatencySelection({ latencyWeight: 0.5 }),
|
|
106
|
+
fallbacks: [cheaperModel],
|
|
107
|
+
});
|
|
108
|
+
|
|
109
|
+
// host-measured provider-call latency feeds the policy's EMA:
|
|
110
|
+
await router.recordOutcome({ identity, provider, model, success: true, latencyMs: 412 });
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
`ModelRouterSelectionPolicy` is `{ name, rank(candidates, request), observe? }`:
|
|
114
|
+
|
|
115
|
+
- `rank` must return a **permutation** of the input candidates. Any other
|
|
116
|
+
result (added, dropped, or duplicated candidates) fails closed with
|
|
117
|
+
`ERR_PRISM_MODEL_ROUTER_POLICY` — a policy can never widen the allow-list,
|
|
118
|
+
residency, or budget decisions, because those checks still run per candidate
|
|
119
|
+
after ranking.
|
|
120
|
+
- `observe` receives outcome feedback from `router.recordOutcome`, including
|
|
121
|
+
the host-supplied `latencyMs` (validated finite non-negative).
|
|
122
|
+
- The policy name is recorded in selection diagnostics (the redaction cap
|
|
123
|
+
still applies). Absent `selection`, behavior is identical to 0.1.6.
|
|
124
|
+
|
|
125
|
+
`createCostLatencySelection()` is the reference policy:
|
|
126
|
+
|
|
127
|
+
- Ranks by unit price first: `ModelCost.input` + `output` + `cacheRead`,
|
|
128
|
+
normalized by the cost unit (`per_million_tokens` vs per-token). Models
|
|
129
|
+
without valid cost metadata rank after all priced models, preserving their
|
|
130
|
+
relative input order.
|
|
131
|
+
- Breaks cost ties by recent measured latency — an in-memory per-
|
|
132
|
+
provider/model EMA fed from `recordOutcome` `latencyMs`. Cold start (no
|
|
133
|
+
samples) is pure cost order.
|
|
134
|
+
- `latencyWeight` (default 0.5) is the EMA smoothing factor: 0 keeps the
|
|
135
|
+
first sample, 1 tracks only the latest. `ponytail:` the EMA is in-memory and
|
|
136
|
+
process-local; durable latency statistics would require a
|
|
137
|
+
`ModelRouterStateStore` contract change and are demand-gated.
|
|
138
|
+
|
|
94
139
|
## Security and performance notes
|
|
95
140
|
|
|
96
141
|
- Allow-list and residency denies never call the underlying resolver.
|
package/docs/provider-caching.md
CHANGED
|
@@ -220,6 +220,69 @@ Static featured catalogs remain offline bootstrap and must **not** invent pricin
|
|
|
220
220
|
- Cache usage reports contain only usage counts and optional pricing/currency; they do not include prompt text, cache keys, headers, credentials, or provider payloads.
|
|
221
221
|
- `applyCacheControl()` returns new message objects for stamped anchors and does not mutate input messages.
|
|
222
222
|
|
|
223
|
+
## Cache telemetry
|
|
224
|
+
|
|
225
|
+
### What it does
|
|
226
|
+
|
|
227
|
+
`createCacheTelemetry()` is a dependency-free aggregator that turns the per-call
|
|
228
|
+
`Usage.cacheReadTokens`/`cacheWriteTokens` counters into per-provider/model
|
|
229
|
+
statistics hosts can use to tune the `cache_aware` input layout: request count,
|
|
230
|
+
cache-read/write token totals, hit rate, and an estimated read-token savings
|
|
231
|
+
when the model carries cost metadata.
|
|
232
|
+
|
|
233
|
+
### When to use it
|
|
234
|
+
|
|
235
|
+
Use it when you want to observe cache effectiveness per provider/model over a
|
|
236
|
+
session, a day, or a run ledger. It is opt-in by construction: importing the
|
|
237
|
+
module never collects anything — the host explicitly wires `record()` to its
|
|
238
|
+
`usage` `ProviderEvent` stream or to run-ledger usage records.
|
|
239
|
+
|
|
240
|
+
### Inputs / request
|
|
241
|
+
|
|
242
|
+
| Input | Meaning |
|
|
243
|
+
| --- | --- |
|
|
244
|
+
| `usage` (`Usage`) | One usage record: `cacheReadTokens`, `cacheWriteTokens`, `inputTokens` are validated (non-negative safe integers; a violation throws `CacheTelemetryError` and mutates nothing). |
|
|
245
|
+
| `model` (`ModelConfig?`) | Attribution key (`provider` + `model`). Omit it for provider-only aggregation into the `unknown` bucket. Cost metadata (`ModelCost.input`/`cacheRead`) enables `estimatedSavings`. |
|
|
246
|
+
| `options.maxKeys` | Distinct provider/model keys before excess keys collapse into the `__overflow__` bucket (default `DEFAULT_CACHE_TELEMETRY_CAP` = 256). |
|
|
247
|
+
|
|
248
|
+
### Outputs / response / events
|
|
249
|
+
|
|
250
|
+
`report()` returns `{ samples, overflowed, totalRequests, totalCacheReadTokens,
|
|
251
|
+
totalCacheWriteTokens }`. Each sample carries `provider`, `model`, `requests`,
|
|
252
|
+
`cacheReadTokens`, `cacheWriteTokens`, `inputTokens`, `hitRate` (total reads /
|
|
253
|
+
total input — the same math as `cacheHitRate()`), and `estimatedSavings` with
|
|
254
|
+
`currency` only when the model has cost metadata. Samples are sorted by
|
|
255
|
+
provider then model. `reset()` clears all samples; `size` reports the current
|
|
256
|
+
distinct-key count.
|
|
257
|
+
|
|
258
|
+
### Request/response example
|
|
259
|
+
|
|
260
|
+
```ts
|
|
261
|
+
import { createCacheTelemetry } from "@arnilo/prism";
|
|
262
|
+
|
|
263
|
+
const telemetry = createCacheTelemetry();
|
|
264
|
+
for await (const event of provider.generate(request)) {
|
|
265
|
+
if (event.type === "usage") telemetry.record(event.usage, request.model);
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
const report = telemetry.report();
|
|
269
|
+
for (const sample of report.samples) {
|
|
270
|
+
console.log(sample.provider, sample.model, sample.hitRate, sample.cacheReadTokens);
|
|
271
|
+
}
|
|
272
|
+
```
|
|
273
|
+
|
|
274
|
+
### Security and performance notes
|
|
275
|
+
|
|
276
|
+
- Reports carry token counters, rates, currency, and provider/model names only
|
|
277
|
+
— never prompt content, cache keys, headers, credentials, or identity fields
|
|
278
|
+
(redaction-safe by construction).
|
|
279
|
+
- Cardinality is bounded: beyond `maxKeys` distinct provider/model keys, excess
|
|
280
|
+
keys accumulate in a single `__overflow__` bucket; memory cannot grow with
|
|
281
|
+
hostile model names (`ponytail:` ceiling — upgrade to host-configurable caps
|
|
282
|
+
or LRU eviction only if a real deployment exceeds it).
|
|
283
|
+
- `record()` is O(1) per usage event; `report()` is O(keys). No secrets or
|
|
284
|
+
cache keys are accepted or stored.
|
|
285
|
+
|
|
223
286
|
## Related APIs
|
|
224
287
|
|
|
225
288
|
- [Input and prompt assembly](input-and-prompt-assembly.md): opt-in cache-aware ordering for stable provider payload prefixes.
|
|
@@ -74,6 +74,8 @@ First-party providers map generic `ModelConfig.parameters.maxTokens` to real out
|
|
|
74
74
|
|
|
75
75
|
## First-party provider package skeletons
|
|
76
76
|
|
|
77
|
+
Scaffold new OpenAI-compatible provider packages with `prism providers add <name>` (see [CLI/RPC](cli-rpc.md#prism-providers-add-017)): it generates the manifest, provider (`createOpenAICompatibleProvider`), starter models, cache-hint helpers, an offline conformance test, and a docs stub — mirroring the first-party skeleton conventions below. Scaffold output is host-chosen and never auto-registered.
|
|
78
|
+
|
|
77
79
|
Phase 12 adds explicit npm workspaces for [`@arnilo/prism-provider-openai`](providers/openai.md), [`@arnilo/prism-provider-opencode-go`](providers/opencode-go.md), [`@arnilo/prism-provider-openrouter`](providers/openrouter.md), [`@arnilo/prism-provider-zai`](providers/zai.md), [`@arnilo/prism-provider-kimi`](providers/kimi.md), and [`@arnilo/prism-provider-neuralwatt`](providers/neuralwatt.md). Each package starts with a side-effect-free `create*ProviderPackage()` export, README, TypeScript build, network-free default tests, and real opt-in live smoke tests.
|
|
78
80
|
|
|
79
81
|
Phase 6 also adds optional [`@arnilo/prism-provider-ai-sdk`](providers/ai-sdk.md), which adapts a host-owned AI SDK `LanguageModelV4` to Prism's `AIProvider`. It joins `@arnilo/prism-providers` as the seventh adapter while remaining independent from the six HTTP implementations.
|
package/docs/public-contracts.md
CHANGED
|
@@ -133,7 +133,7 @@ Important request shapes:
|
|
|
133
133
|
| `AgentSessionConfig` | Session creation input: optional id, agent, store, leaf id, and metadata. |
|
|
134
134
|
| `RunOptions` | Per-run overrides: optional abort signal, model, input layout, run limits (incl. `limits.maxToolRounds`), provider options/request policies, system prompt layers, compaction, retry, metadata, skill selection, validate, redactor, and loop. |
|
|
135
135
|
| `SubscribeOptions` / `SubscriberOverflowPolicy` | Live `AgentEvent` subscriber queue limit and overflow policy: `maxQueuedEvents`, `overflow: "close" \| "drop_oldest" \| "drop_newest"`. |
|
|
136
|
-
| `resumeAgentRunStream` / `AgentRunResumeStreamOptions` | One durable-run event stream: existing checkpoint/resume options plus `signal` and bounded subscriber options. `AgentRunLifecycle.resumeStream()` adds host capability resolution; no protocol types enter core. |
|
|
136
|
+
| `resumeAgentRunStream` / `AgentRunResumeStreamOptions` | One durable-run event stream: existing checkpoint/resume options plus `signal` and bounded subscriber options. `AgentRunLifecycle.resumeStream()` adds host capability resolution; no protocol types enter core. Runtime resume validation (0.2.0): every resume entrypoint validates the full input in core before any checkpoint write, tool call, or event — unknown legacy decisions and malformed batches fail closed with `AgentDecisionError` and no side effect; see [Agent/session runtime § Durable interruption](agent-session-runtime.md#durable-interruption). |
|
|
137
137
|
| `AgentConfig.loop` / `RunOptions.loop` | Replaceable per-run control loop: `singleShotLoop` default, `generate-validate-revise` options, or a custom `AgentLoopStrategy`. `RunOptions.loop` wins. See [Agent loops](agent-loops.md). |
|
|
138
138
|
| `AgentLoopStrategy` | `{ name; run(ctx: LoopContext): Promise<Usage \| undefined> }` — orchestrates shared runtime primitives via `LoopContext`. |
|
|
139
139
|
| `LoopContext` | Loop-facing surface: run ids, signal, live `history`, `input`/`inputMessages`/`maxToolRounds`, and bound `assemble`/`generate`/`dispatchToolCall`/`appendMessage`/`emit` primitives. |
|