@arnilo/prism 0.0.15 → 0.0.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/CHANGELOG.md +31 -0
  2. package/dist/agent-definitions.js +2 -3
  3. package/dist/agent-loops.js +12 -7
  4. package/dist/agent-run-lifecycle.d.ts +1 -2
  5. package/dist/agent-run-lifecycle.js +1 -1
  6. package/dist/agent-run-state.js +29 -4
  7. package/dist/agents.d.ts +1 -1
  8. package/dist/agents.js +163 -61
  9. package/dist/cache-helpers.js +18 -9
  10. package/dist/checkpoints.d.ts +4 -0
  11. package/dist/checkpoints.js +17 -9
  12. package/dist/cli-init.js +3 -7
  13. package/dist/cli-runner.d.ts +2 -6
  14. package/dist/cli-runner.js +71 -33
  15. package/dist/compaction.js +5 -4
  16. package/dist/config.js +7 -4
  17. package/dist/content.js +26 -24
  18. package/dist/context-budget.js +18 -11
  19. package/dist/contracts.d.ts +16 -3
  20. package/dist/contracts.js +4 -1
  21. package/dist/contribution-parsing.js +6 -2
  22. package/dist/contributions.d.ts +2 -0
  23. package/dist/contributions.js +3 -0
  24. package/dist/conversations.js +2 -1
  25. package/dist/credentials.d.ts +8 -2
  26. package/dist/credentials.js +9 -3
  27. package/dist/event-multiplexer.js +18 -4
  28. package/dist/extensions.d.ts +7 -1
  29. package/dist/extensions.js +64 -6
  30. package/dist/feedback.js +12 -10
  31. package/dist/guardrails.d.ts +1 -1
  32. package/dist/guardrails.js +26 -17
  33. package/dist/identity.js +10 -2
  34. package/dist/index.d.ts +82 -83
  35. package/dist/index.js +42 -42
  36. package/dist/input.d.ts +2 -2
  37. package/dist/input.js +50 -27
  38. package/dist/instruction-injection.d.ts +1 -1
  39. package/dist/middleware.js +9 -1
  40. package/dist/models.d.ts +2 -0
  41. package/dist/models.js +3 -0
  42. package/dist/node/agent-definitions.js +16 -8
  43. package/dist/node/contribution-discovery.d.ts +1 -2
  44. package/dist/node/contribution-discovery.js +3 -3
  45. package/dist/node/session-store-jsonl.js +10 -7
  46. package/dist/node/settings.d.ts +1 -1
  47. package/dist/node/settings.js +1 -1
  48. package/dist/node/system-project-prompts.js +2 -4
  49. package/dist/node/trust.js +1 -1
  50. package/dist/persistence-lifecycle.js +1 -3
  51. package/dist/provider-events.js +3 -1
  52. package/dist/provider-request-policy.js +3 -4
  53. package/dist/providers/media.d.ts +1 -1
  54. package/dist/providers/openai-compatible.d.ts +42 -1
  55. package/dist/providers/openai-compatible.js +110 -49
  56. package/dist/providers/openai-primitives.js +7 -7
  57. package/dist/providers/transport.d.ts +6 -0
  58. package/dist/providers/transport.js +21 -0
  59. package/dist/providers.d.ts +2 -0
  60. package/dist/providers.js +3 -0
  61. package/dist/redaction.d.ts +1 -0
  62. package/dist/redaction.js +26 -9
  63. package/dist/resources.d.ts +2 -2
  64. package/dist/resources.js +2 -2
  65. package/dist/retry.d.ts +5 -0
  66. package/dist/retry.js +8 -1
  67. package/dist/rpc.js +42 -9
  68. package/dist/run-ledger.d.ts +6 -0
  69. package/dist/run-ledger.js +16 -13
  70. package/dist/run-limits.js +49 -10
  71. package/dist/secure-agent.js +1 -1
  72. package/dist/security.js +7 -2
  73. package/dist/session-stores.d.ts +1 -1
  74. package/dist/session-stores.js +28 -24
  75. package/dist/structured-output.js +2 -2
  76. package/dist/system-prompts.js +7 -2
  77. package/dist/testing/compaction-conformance.js +5 -1
  78. package/dist/testing/extension-conformance.js +15 -3
  79. package/dist/testing/feedback.d.ts +1 -3
  80. package/dist/testing/feedback.js +1 -1
  81. package/dist/testing/persistence-schema.js +206 -37
  82. package/dist/testing/provider-conformance.js +3 -3
  83. package/dist/testing/run-ledger-conformance.js +1 -1
  84. package/dist/testing/session-store-conformance.js +1 -1
  85. package/dist/testing/tool-conformance.js +30 -5
  86. package/dist/thinking.js +4 -1
  87. package/dist/tools.d.ts +2 -2
  88. package/dist/tools.js +24 -5
  89. package/docs/0.1.0-readiness.md +139 -0
  90. package/docs/agent-events.md +2 -1
  91. package/docs/agent-session-runtime.md +2 -2
  92. package/docs/cli-rpc.md +1 -5
  93. package/docs/coding-agent-tools.md +2 -0
  94. package/docs/compaction-and-retry.md +3 -1
  95. package/docs/contribution-registries.md +1 -0
  96. package/docs/credentials-and-redaction.md +1 -1
  97. package/docs/extensions.md +1 -1
  98. package/docs/guardrails.md +13 -2
  99. package/docs/index.md +4 -2
  100. package/docs/input-and-prompt-assembly.md +3 -3
  101. package/docs/middleware-hooks.md +2 -2
  102. package/docs/migration.md +40 -0
  103. package/docs/performance.md +33 -0
  104. package/docs/providers/openai-compatible.md +28 -1
  105. package/docs/public-contracts.md +2 -2
  106. package/docs/release-and-install.md +127 -17
  107. package/docs/session-stores.md +1 -1
  108. package/package.json +14 -6
  109. package/docs/review-coverage-2026-07-14.md +0 -260
  110. package/docs/review-coverage-2026-07-15.md +0 -193
  111. package/docs/review-coverage-2026-07-17-provider-validation.md +0 -192
  112. package/docs/review-coverage-2026-07-19-phase-3.md +0 -174
  113. package/docs/review-coverage-2026-07-20-phase-4.md +0 -175
  114. package/docs/review-coverage-2026-07-21-phase-5.md +0 -172
  115. package/docs/review-coverage-2026-07-22-phase-6.md +0 -209
  116. package/docs/review-coverage-2026-07-22-phase-7.md +0 -173
  117. package/docs/review-coverage-2026-07-23-phase-8.md +0 -245
  118. package/docs/review-coverage-2026-07-25-phase-9.md +0 -256
  119. package/docs/review-coverage-2026-07-26-phase-10.md +0 -132
@@ -0,0 +1,139 @@
1
+ # 0.1.0 / 1.0 Readiness Gates
2
+
3
+ Status: **0.0.16 is a 1.0 readiness review, not an automatic 1.0 release.**
4
+ This page distills the Phase 11 (0.0.16) gates into one command-per-gate table
5
+ so readiness is checkable, not prose. Every gate below is a runnable command
6
+ with last evidence captured on 2026-07-26 (release 0.0.16, Node v24.18.0,
7
+ Linux x86_64). The decision to cut 1.0 stays with the operator after the
8
+ operator-gated legs run in a protected environment and the Phase 12 demand
9
+ evidence exists.
10
+
11
+ Evidence trail: [`docs/review-coverage-2026-07-26-phase-11.md`](./review-coverage-2026-07-26-phase-11.md)
12
+ (addenda 0–9), [`docs/release-and-install.md`](./release-and-install.md),
13
+ [`docs/migration.md`](./migration.md), [`docs/performance.md`](./performance.md).
14
+
15
+ ## Gate table
16
+
17
+ | Gate | Command | Last evidence (2026-07-26) | Owner |
18
+ |---|---|---|---|
19
+ | Full quality gate | `npm run sdk:ready` | RC=0: typecheck (+examples), lint 0, format clean, full test, coverage, pack, release:gate | CI |
20
+ | Exact version graph | `node scripts/release.mjs check --version <v>` | 0.0.16 pass: exact versions/ranges/lockfile/access + registry-collision check, 44 manifests | CI + operator |
21
+ | Frozen public API surface + compat gate | `node scripts/release.mjs gate` | 0 breaks / 0 errors vs 44 checked-in baselines (`scripts/compat-baseline/`); only additive delta is `resolveRedactor` | CI |
22
+ | Migration coverage 0.0.5→0.0.16 | `node --test dist/__tests__/docs.test.js` | 112/112; `docs/migration.md` has one section per release 0.0.5→0.0.16, each tripwired | Maintainer |
23
+ | Deterministic artifact budget | `node --test dist/__tests__/budget-gate.test.mjs` | root 579.2 kB / 2.1 MB / 270 files within +5% of baseline; startup 38 ms < 250 ms ceiling | CI (in `npm test`) |
24
+ | Performance benchmark medians | `node scripts/benchmark-0.0.16.mjs` | 6 network-free scenarios within ±25%; 0 backpressure / 0 resource-limit signals | On-demand release evidence |
25
+ | Secret scan | `node scripts/scan-secrets.mjs` | 3095 files / 0 findings | CI |
26
+ | License / SBOM | `node scripts/verify-sbom.mjs` | 188 packages / 8 licenses, all allow-listed | CI |
27
+ | Dependency audit | `npm audit --audit-level=high` | rc=0 (2 moderate, 0 high) | CI |
28
+ | Whitespace hygiene | `git diff --check` | clean | CI |
29
+ | Publish order + tarball validation | `node scripts/release.mjs publish --version <v> --dry-run --allow-dirty --allow-untagged` | 44/44 packages `dry-run`, deterministic dependency order, no failures | Operator (dry-run), CI |
30
+ | Node 20 compatibility | CI `node20-compat` (build + public-import smoke) | all 21 root exports import cleanly on Node 20.20.2 | CI |
31
+ | PostgreSQL suite | `npm run test:postgres` | **operator-gated** (requires live PostgreSQL) | Operator |
32
+ | Keychain / live-provider suites | `npm run test:live` (protected) | **operator-gated** (requires credentials) | Operator |
33
+ | SAST | GitHub CodeQL | **operator-gated** (runs in CI workflow) | CI |
34
+ | Signed, provenance publication | `npm run release:publish` (clean tagged tree, OIDC) | **operator-gated** (see "Remaining for 1.0") | Operator |
35
+
36
+ ## Frozen public API surface
37
+
38
+ The compat gate diffs every package's generated `.d.ts` export surface against
39
+ checked-in baselines in `scripts/compat-baseline/` (one file per package,
40
+ regenerated at 0.0.16). It fails on any **removed** export or changed
41
+ declaration; additive exports are allowed. `scripts/release-gates.mjs` also
42
+ enforces a tarball deny list (no `docs/review-coverage-*` in published
43
+ artifacts) and exact version-range drift. A genuine break requires
44
+ `--allow-break` **and** a `docs/migration.md` entry mentioning the version.
45
+
46
+ **Baseline maintenance:** `scripts/compat-baseline/` must stay committed.
47
+ Regenerate only after review with `node scripts/release.mjs gate --update-baseline`,
48
+ having first confirmed zero removed exports (the gate's order-sensitive
49
+ signatures can drift on a TypeScript bump without any real API change).
50
+
51
+ ## Migration coverage 0.0.5 → 0.0.16
52
+
53
+ `docs/migration.md` carries one section per release from 0.0.5 through 0.0.16;
54
+ `docs.test.ts` tripwires each section heading and key phrase, so a missing or
55
+ gutted migration section fails the suite. 0.0.16's only user-facing change is
56
+ the additive `resolveRedactor` export (no breaking changes).
57
+
58
+ ## Budget table
59
+
60
+ Deterministic budgets (CI gate, `scripts/budget-gate.test.mjs`):
61
+
62
+ | Metric | Baseline | Tolerance | 0.0.16 measured |
63
+ |---|---|---|---|
64
+ | Root packed bytes | 575,680 | +5% | 579.2 kB (within) |
65
+ | Root unpacked bytes | 2,043,402 | +5% | 2.1 MB (within) |
66
+ | Root file count | 270 | +5% | 270 |
67
+ | Cold-startup import | 38 ms | ceiling 250 ms | ~38 ms |
68
+ | Aggregate packed (44 manifests, reference only) | 1,217,694 | +10% | not gated in fast test |
69
+
70
+ Benchmark medians (on-demand evidence, `scripts/benchmark-0.0.16.mjs`, ±25%):
71
+
72
+ | Scenario | throughput/s | p50 ms | p95 ms |
73
+ |---|---|---|---|
74
+ | openai-hosted-continuation | 5,386.0 | 0.1317 | 0.2907 |
75
+ | openai-realtime-envelope | 880.3 | 1.1293 | 1.2399 |
76
+ | ai-sdk-v4-stream-mapping | 22,403.0 | 0.0230 | 0.0734 |
77
+ | provider-package-metadata | 55,829.2 | 0.0065 | 0.0394 |
78
+ | rag-parse-replace-rerank-retrieve | 4,800.3 | 0.1432 | 0.4064 |
79
+ | memory-retention-export-rebuild | 8,763.2 | 0.0686 | 0.1892 |
80
+
81
+ Baselines are a 2026-07-26 snapshot (`scripts/budgets.json`); raise them only
82
+ after a deliberate reviewed performance change.
83
+
84
+ ## Live-suite matrix (operator-gated)
85
+
86
+ | Suite | Command | Environment |
87
+ |---|---|---|
88
+ | PostgreSQL persistence | `npm run test:postgres` | live PostgreSQL |
89
+ | Keychain credentials | protected live suite | OS keychain |
90
+ | Provider live-canary (OpenAI/Anthropic/Google/Kimi/Ollama/…) | protected live-canary matrix | vendor credentials |
91
+ | RAG / memory / workflows live journeys | protected live-canary matrix | vendor + DB credentials |
92
+
93
+ These do not run on a contributor machine; their evidence is recorded in the
94
+ protected environment, never faked.
95
+
96
+ ## Security matrix
97
+
98
+ | Control | Command / source | 0.0.16 status |
99
+ |---|---|---|
100
+ | Secret scan | `node scripts/scan-secrets.mjs` | 3095 files / 0 findings |
101
+ | License / SBOM | `node scripts/verify-sbom.mjs` | 188 packages / 8 licenses, allow-listed |
102
+ | Dependency audit | `npm audit --audit-level=high` | 0 high (2 moderate) |
103
+ | SAST | GitHub CodeQL workflow | CI-gated |
104
+ | Sandbox / protocol / tenant threat suites | `npm test` (coding-security, MCP, policy, guardrail suites) | green |
105
+ | Signed deterministic publication | `release.mjs publish` on clean tagged tree | operator-gated |
106
+
107
+ ## Remaining for 1.0 (operator / protected environment)
108
+
109
+ Exact prerequisites that must be satisfied before cutting 1.0:
110
+
111
+ 1. **Signed tag + commits:** create and sign `v0.1.0` (or `v1.0.0`) on a clean
112
+ tree; `release.mjs publish` refuses real publication with `--allow-dirty`
113
+ or `--allow-untagged`.
114
+ 2. **npm authentication + OIDC provenance/attestation:** publish with
115
+ `--provenance` and `--access public` from the protected registry identity.
116
+ 3. **Protected live-canary matrix green:** provider/RAG/memory/workflows live
117
+ journeys pass with real credentials.
118
+ 4. **PostgreSQL + keychain protected suites green.**
119
+ 5. **CodeQL SAST green** on the release commit.
120
+ 6. **`scripts/compat-baseline/` committed** so CI's compat gate has a
121
+ checked-in baseline.
122
+ 7. **Phase 12 demand evidence** (below) recorded for any capability that 1.0
123
+ is expected to anchor.
124
+
125
+ ## Phase 12 demand-evidence entry criteria
126
+
127
+ Phase 12 (demand-gated 0.1.x) promotes no capability on comparison-table parity
128
+ alone. Each candidate must present, before it becomes a numbered plan:
129
+
130
+ - **Named user** (a concrete person/team who will use it),
131
+ - **Concrete integration** (the real system it connects to),
132
+ - **Operational owner** (who runs and pages for it),
133
+ - **Measurable acceptance criteria** (scale/cost/latency/storage budgets that
134
+ do not expand default core/install/runtime cost),
135
+ - then the pipeline: demand evidence → primitive review → threat model →
136
+ optional package/service → conformance → release gate.
137
+
138
+ The 0.0.16 readiness gates above are the stable API/compat/budget/security
139
+ floor that Phase 12 capabilities must consume and must not regress.
@@ -44,7 +44,7 @@ The `AgentEvent` union (grouped by concern):
44
44
  | Assistant messages | `message_started`, `message_delta`, `message_finished` |
45
45
  | Tool execution | `tool_execution_started`, `tool_execution_progress`, `tool_execution_finished`, `tool_execution_error`, `tool_execution_blocked` |
46
46
  | Guardrails | `guardrail_decision` |
47
- | Queue/subscribers | `queue_updated`, `event_subscriber_overflow` |
47
+ | Queue/subscribers | `queue_updated`, `event_subscriber_overflow`, `steer_rejected` |
48
48
  | Compaction | `compaction_started`, `compaction_finished` |
49
49
  | Retry | `retry_scheduled` |
50
50
  | Artifacts | `artifact_validation_started`, `artifact_validation_finished`, `artifact_revision_started`, `artifact_finished`, `artifact_failed` |
@@ -92,6 +92,7 @@ Queue / subscriber / compaction / retry / provider events:
92
92
  | Variant | Fields |
93
93
  | --- | --- |
94
94
  | `queue_updated` | `sessionId`, `runId`, `size: number` |
95
+ | `steer_rejected` | `sessionId`, `runId`, `message: Message` (redacted), `record: GuardrailRecord` — a steered message dropped by a terminal input guardrail; the run continues without it |
95
96
  | `event_subscriber_overflow` | `sessionId`, `droppedEvents: number`, `maxQueuedEvents: number`, `overflow: "close" \| "drop_oldest" \| "drop_newest"` |
96
97
  | `compaction_started` | `sessionId`, `runId?` |
97
98
  | `compaction_finished` | `sessionId`, `runId?`, `summary: string` |
@@ -85,7 +85,7 @@ Only one `run()` may be active per session. Concurrent `run()` / `prompt` / `fol
85
85
 
86
86
  ### Mid-run steer (0.0.11)
87
87
 
88
- `session.steer(input, options?)` enqueues user text into the **same** active run (fail closed when no run). Default: inject at the next turn boundary (after tool rounds / before next provider assemble). `options.softInterrupt: true` aborts only the current provider stream, then continues the same `runId` with steered text. Pending queue caps: **8** messages / **64 KiB** UTF-8 total (`DEFAULT_MAX_PENDING_STEERS` / `DEFAULT_MAX_PENDING_STEER_BYTES`); overflow throws. Steered messages pass input guardrails + normal session append/redaction. Loops drain via optional `LoopContext.hasPendingSteers` / `applyPendingSteers`.
88
+ `session.steer(input, options?)` enqueues user text into the **same** active run (fail closed when no run). Default: inject at the next turn boundary (after tool rounds / before next provider assemble). `options.softInterrupt: true` aborts only the current provider stream, then continues the same `runId` with steered text. Pending queue caps: **8** messages / **64 KiB** UTF-8 total (`DEFAULT_MAX_PENDING_STEERS` / `DEFAULT_MAX_PENDING_STEER_BYTES`); overflow throws. Steered messages pass input guardrails + normal session append/redaction. A `block`/`tripwire` on a steered message drops just that message: Prism emits `guardrail_decision` plus a `steer_rejected` event (redacted message + `GuardrailRecord`) and the run continues; the message never enters history or the session store. Run-start input blocking still fails the run. `interrupt` on a steered message fails closed (durable suspension is only for run-start input). Loops drain via optional `LoopContext.hasPendingSteers` / `applyPendingSteers`.
89
89
 
90
90
  `session.abort(reason)` aborts the active run. The abort signal is passed to input assembly, provider requests, and tool execution; if a tool/provider path aborts after a tool call, Prism does not start another provider turn.
91
91
 
@@ -187,7 +187,7 @@ if (result.status === "suspended") {
187
187
  }
188
188
  ```
189
189
 
190
- Resume requires exact checkpoint ownership, version, agent fingerprint, and revision. Prism CAS-claims approval before work, rechecks normal guardrail/permission/validation/limit paths, and marks a pending tool dispatched before its side effect. `createAgentRunLifecycle()` wraps the same core path for server/MCP hosts: adapters pass only authorized ownership, status returns only `{ state, version }`, and `resolveAgent()` supplies current agent/revision. `resumeStream()` uses that same claim path and bounded subscriber, so adapters do not poll or duplicate resume logic. Remote restart requires both checkpoint and session stores to be durable. A crash after that mark is ambiguous and is never replayed automatically; use host tool idempotency keyed by `runId`/`toolCallId` or resolve it manually. Checkpoints contain bounded redacted state plus session/leaf references, never provider objects, callbacks, signals, credentials, or raw secrets. Only built-in loop options are durable; custom `AgentLoopStrategy` rejects before provider work.
190
+ Resume requires exact checkpoint ownership, version, agent fingerprint, and revision. The fingerprint hashes the agent id/name, `definitionRevision`, model, instructions, system-prompt contributions, skills (name/instructions/tool names), tool definitions (name/parameters/exclusive), guardrail definitions (name/stage/revision), and loop strategy — changing any of them without bumping `definitionRevision` fails resume closed instead of silently continuing with different agent semantics. Prism CAS-claims approval before work, rechecks normal guardrail/permission/validation/limit paths, and marks a pending tool dispatched before its side effect. `createAgentRunLifecycle()` wraps the same core path for server/MCP hosts: adapters pass only authorized ownership, status returns only `{ state, version }`, and `resolveAgent()` supplies current agent/revision. `resumeStream()` uses that same claim path and bounded subscriber, so adapters do not poll or duplicate resume logic. Remote restart requires both checkpoint and session stores to be durable. A crash after that mark is ambiguous and is never replayed automatically; use host tool idempotency keyed by `runId`/`toolCallId` or resolve it manually. Checkpoints contain bounded redacted state plus session/leaf references, never provider objects, callbacks, signals, credentials, or raw secrets. State is bounded at save by `runState.maxStateBytes` (default 256 KB, at most the 1 MB hard cap); load bounds against the 1 MB hard cap only, so state saved with a raised limit stays resumable while oversized records are still rejected. Only built-in loop options are durable; custom `AgentLoopStrategy` rejects before provider work.
191
191
 
192
192
  ## Secure composition
193
193
 
package/docs/cli-rpc.md CHANGED
@@ -45,10 +45,6 @@ Default generation installs only `@arnilo/prism` (mock provider). Selecting a re
45
45
  | `--provider <name>` | Explicit provider id. The built-in `mock` id is only a smoke-test provider. |
46
46
  | `--model <name>` | Explicit model name. |
47
47
  | `--session <id>` | Session id. |
48
- | `--config <path>` | Explicit config path recorded by the adapter; not auto-loaded. |
49
- | `--resource <uri>` | Explicit resource URI recorded by the adapter; not auto-loaded. |
50
- | `--extension <name>` | Explicit extension name recorded by the adapter; not auto-loaded/imported. |
51
- | `--tool <name>` | Explicit tool name recorded by the adapter; not auto-enabled. |
52
48
  | `--system <text>` | System instructions. |
53
49
  | `--context <text>` | Context text reserved for host adapters. |
54
50
  | `--compact <entries>` | Auto-compaction threshold for the run. |
@@ -202,6 +198,6 @@ Suspended workflow resume parameters are `{ workflowId, runId, decision: "approv
202
198
  - [Observational memory compaction package](compaction-observational-memory.md): optional `om:status` and `om:view` command factories for explicitly wired hosts.
203
199
  - [Workflows](workflows.md): optional `createWorkflowCommands()` for direct/background/replay/status/cancel/resume and selected schedule control over the same RPC `command` seam.
204
200
 
205
- The CLI records flags but does not auto-load project-local resources, extensions, tools, or config. The two system/project prompt files are the exception: in print/json modes the CLI auto-loads `<workspaceRoot>/AGENTS.md` (trust-gated) and an app-supplied `SYSTEM.md` layer as `AgentConfig.systemPrompt` layers composed with `--system` (base); `--no-agents-md` / `--no-system-md` skip them and `--agents-md-file` / `--system-md-file` override the paths. The CLI does not default `globalRoot` to the user's home directory — pass it from a host adapter or use `--agents-config <path>` for the app-config bundle layout. RPC mode does not auto-read these files (the host owns the session factory). Hosts must make explicit trust and permission decisions before wiring any other local loading.
201
+ The CLI records flags but does not auto-load project-local resources, extensions, tools, or config. `--config`, `--resource`, `--extension`, and `--tool` were parsed-and-recorded in earlier builds without any effect; they are now rejected loudly (`<flag> is not supported in this build`) until a CLI-harness plan wires them. The two system/project prompt files are the exception: in print/json modes the CLI auto-loads `<workspaceRoot>/AGENTS.md` (trust-gated) and an app-supplied `SYSTEM.md` layer as `AgentConfig.systemPrompt` layers composed with `--system` (base); `--no-agents-md` / `--no-system-md` skip them and `--agents-md-file` / `--system-md-file` override the paths. The CLI does not default `globalRoot` to the user's home directory — pass it from a host adapter or use `--agents-config <path>` for the app-config bundle layout. RPC mode does not auto-read these files (the host owns the session factory). Hosts must make explicit trust and permission decisions before wiring any other local loading.
206
202
 
207
203
  For app-controlled agent bundles under `<configRoot>/agents/<name>/AGENT.md` (including the three-layer `SYSTEM.md` → `AGENT.md` body → repo `AGENTS.md` prompt append and the union skill/tool scopes), see [Agent definitions](agent-definitions.md).
@@ -367,6 +367,8 @@ const shell = createShellTool("/repo", {
367
367
  maxLines: 500,
368
368
  timeout: 600,
369
369
  maxTotalOutputBytes: 64 * 1024 * 1024,
370
+ // Optional: scrub the environment the spawn hook and child process see (default: full process.env clone).
371
+ envAllowlist: ["PATH", "HOME", "LANG"],
370
372
  });
371
373
 
372
374
  const remoteWrite = createWriteTool("/repo", {
@@ -68,10 +68,12 @@ createDefaultRetryPolicy(options?: DefaultRetryPolicyOptions): RetryPolicy
68
68
  | `maxAttempts` | Total provider-turn attempts; defaults to `3`. |
69
69
  | `baseDelayMs` | First retry delay; defaults to `100`. |
70
70
  | `maxDelayMs` | Backoff cap; defaults to `1000`. |
71
+ | `jitter` | Symmetric jitter fraction on computed delays; defaults to `0.25` (±25%). Set `0` for exact delays. |
72
+ | `random` | Random source for jitter (tests); defaults to `Math.random`. |
71
73
  | `secrets` | Exact known secret strings to redact from retry errors/events. |
72
74
  | `metadata` | Explicit host metadata for retry policy context. |
73
75
 
74
- `RunOptions.retry: false` disables configured retry for that run. Default classification retries generic transient codes/messages such as `ETIMEDOUT`, `ECONNRESET`, `429`, `500`, `502`, `503`, `504`, `timeout`, `rate_limit`, and `temporarily_unavailable`; aborts and non-transient errors fail closed.
76
+ `RunOptions.retry: false` disables configured retry for that run. Default classification retries generic transient codes/messages such as `ETIMEDOUT`, `ECONNRESET`, `429`, `500`, `502`, `503`, `504`, `timeout`, `rate_limit`, and `temporarily_unavailable`; aborts and non-transient errors fail closed. Delays are exponential (`baseDelayMs * 2^(attempt-1)`, capped at `maxDelayMs`) with symmetric jitter, so concurrent sessions do not retry in lockstep during a shared outage. When `ErrorInfo.retryAfterMs` is set — first-party HTTP providers populate it from the `Retry-After` response header via `httpStatusError()` — the hint wins over computed backoff, jitter still applies, and the result is always capped at `maxDelayMs` so a hostile or huge hint cannot pin a run.
75
77
 
76
78
  `CompactionEntryData` is stored in `SessionEntry.data` for compaction entries:
77
79
 
@@ -27,6 +27,7 @@ createContributionRegistries(options?: { duplicate?: "replace" | "error" }): Con
27
27
  | Method | Input | Result |
28
28
  | --- | --- | --- |
29
29
  | `register(key, contribution)` | string key and contribution | Stores/replaces the contribution for that key; throws `Duplicate <label>: <key>` when `duplicate: "error"`. |
30
+ | `unregister(key)` | string key | Removes the contribution; returns `false` when the key was not registered. `providers.unregister(id)` and `models.unregister(provider, model)` mirror this on the specialized registries. |
30
31
  | `get(key)` | string key | Returns the contribution or `undefined`. |
31
32
  | `resolve(key)` | string key | Returns the contribution or throws `Unknown <label>: <key>`. |
32
33
  | `list()` | none | Returns contributions in insertion order. |
@@ -132,4 +132,4 @@ A future provider-local OAuth adapter needs published permission for third-party
132
132
  - [LLM compaction package](compaction-llm.md): resolves optional summary-provider credentials per compaction call and redacts exact known values.
133
133
  - [OpenAI-compatible provider](providers/openai-compatible.md): resolves API keys per request and redacts known values from adapter errors.
134
134
 
135
- Phase 10 added `createMemoryCredentialStore()`, `createChainedCredentialResolver()`, and `createSecretRedactor()` for opt-in in-memory auth and runtime redaction. Phase 11 adds OAuth/API-key contracts plus explicit resolver order helpers. Core still has no persistent secret store and does not read environment variables or files for credentials. For durable storage, use [`@arnilo/prism-credentials-node`](credential-storage.md) encrypted-file or keychain backends. See [Security/auth/trust](settings-auth-trust-security.md).
135
+ Phase 10 added `createMemoryCredentialStore()`, `createChainedCredentialResolver()`, and `createSecretRedactor()` for opt-in in-memory auth and runtime redaction. By default the memory store serves a providerless record for a provider-scoped request of the same name — that record is then shared across every provider; pass `{ allowProviderFallback: false }` for exact-match-only resolution (strict provider scoping). Phase 11 adds OAuth/API-key contracts plus explicit resolver order helpers. Core still has no persistent secret store and does not read environment variables or files for credentials. For durable storage, use [`@arnilo/prism-credentials-node`](credential-storage.md) encrypted-file or keychain backends. See [Security/auth/trust](settings-auth-trust-security.md).
@@ -37,7 +37,7 @@ createExtensionEventBus(options?: { errorPolicy?: "event" | "throw"; secrets?: r
37
37
 
38
38
  ## Outputs / response / events
39
39
 
40
- - `kernel.load(extensions)` calls each extension's `setup(api)` in host-provided order.
40
+ - `kernel.load(extensions)` calls each extension's `setup(api)` in host-provided order and resolves to `LoadedExtension[]` (`{ name, dispose() }`). A failed `setup` unwinds that extension's partial registrations (no orphaned half-loads). `dispose()` removes the extension's registry contributions and middleware/event subscriptions via each registry's `unregister(key)` — best-effort, idempotent, and limited to registries/subscriptions: side effects outside the registries (files, network, spawned work) are not unwound, and load-order/dependency graphs between extensions are out of scope.
41
41
  - `kernel.registries` exposes the explicit contribution registry bundle.
42
42
  - `kernel.events.on(type, handler)` registers ordered event handlers and returns an unsubscribe function.
43
43
  - `kernel.events.emit(event)` calls matching handlers in registration order.
@@ -26,11 +26,22 @@ const guardrails: Guardrails = { input: [pii], maxConcurrency: 1 };
26
26
 
27
27
  Set `AgentConfig.guardrails` for every session run or `RunOptions.guardrails` to append checks for one run. `DispatchToolCallOptions.guardrails`, workflow `RunWorkflowOptions.guardrails`, and MCP server `CreatePrismMcpServerOptions.guardrails` apply tool stages to direct calls. A stage has `Guardrail<"input" | "output" | "tool_input" | "tool_output">`, a name, optional revision, and `evaluate(context)` result.
28
28
 
29
- Decisions are `allow`, `block`, `tripwire`, or `interrupt`. Evaluation defaults to declaration-order sequential. `maxConcurrency` may be 1–16; records are emitted in declaration order. Thrown or malformed decisions become a fail-closed tripwire. Decision reasons are capped at 4 KiB and metadata at 16 KiB after JSON normalization and optional redaction.
29
+ Decisions are `allow`, `block`, `tripwire`, or `interrupt`. Evaluation defaults to declaration-order sequential. `maxConcurrency` may be 1–16; records are emitted in declaration order. Thrown or malformed decisions become a fail-closed tripwire. A throwing guardrail produces a `guardrail_failed` record whose `metadata.error` carries the underlying error message — redacted and bounded to 4 KiB — so failures stay diagnosable without leaking internals. Decision reasons are capped at 4 KiB and metadata at 16 KiB after JSON normalization and optional redaction.
30
30
 
31
31
  ## Outputs / response / events
32
32
 
33
- Every evaluated guard produces a redacted `guardrail_decision` `AgentEvent` with a bounded `GuardrailRecord`. Optional OpenTelemetry instrumentation records only controlled stage/action on a short run-child span; guardrail name, reason, and metadata are excluded. An input or output terminal decision rejects the run with `GuardrailError`; `tripwire` stops remaining evaluation. A tool-input or tool-output `block` returns a redacted blocked `ToolResult`; a `tripwire` rejects the enclosing run. `interrupt` is reserved for durable runs and currently fails closed with `ERR_PRISM_GUARDRAIL_INTERRUPT_UNAVAILABLE`.
33
+ Every evaluated guard produces a redacted `guardrail_decision` `AgentEvent` with a bounded `GuardrailRecord`. Optional OpenTelemetry instrumentation records only controlled stage/action on a short run-child span; guardrail name, reason, and metadata are excluded. An input or output terminal decision rejects the run with `GuardrailError`; `tripwire` stops remaining evaluation. A tool-input or tool-output `block` returns a redacted blocked `ToolResult`; a `tripwire` rejects the enclosing run. `interrupt` is reserved for durable runs: at the input stage of a fresh durable run it suspends the run for approval (persisted `input_guardrail` interruption); anywhere else it currently fails closed with `ERR_PRISM_GUARDRAIL_INTERRUPT_UNAVAILABLE`. Resuming a suspended durable run re-evaluates input guardrails on the stored input, and the resume decision itself counts as the approval: a repeated input-stage `interrupt` does not re-suspend or fail the resumed run, while `block`/`tripwire` still reject it.
34
+
35
+ Action outcome by stage:
36
+
37
+ | Stage | `block` | `tripwire` | `interrupt` |
38
+ | --- | --- | --- | --- |
39
+ | `input` | run rejected (`GuardrailError`); steered message: dropped + `steer_rejected`, run continues | run rejected; steered message: dropped + `steer_rejected`, run continues | fresh durable run: suspends for approval; otherwise fails closed |
40
+ | `output` | run rejected | run rejected | fails closed (`ERR_PRISM_GUARDRAIL_INTERRUPT_UNAVAILABLE`) |
41
+ | `tool_input` | blocked `ToolResult`, run continues | run rejected | fails closed |
42
+ | `tool_output` | blocked `ToolResult`, run continues | run rejected | fails closed |
43
+
44
+ The `GuardrailError` message names the stage so unsupported `interrupt` placements are diagnosable without reading core source.
34
45
 
35
46
  Ordering is fixed:
36
47
 
package/docs/index.md CHANGED
@@ -50,7 +50,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
50
50
  - Phase 12 package workspaces: [`@arnilo/prism-provider-openai`](providers/openai.md) (Responses hosted-tool attribution, bounded continuation, Realtime session seam), [`@arnilo/prism-provider-anthropic`](providers/anthropic.md) (native Messages, `cache_control`, thinking, caller-gated `listAnthropicModels`), [`@arnilo/prism-provider-google`](providers/google.md) (native Gemini `generateContent` SSE, caller-gated `listGoogleModels`), [`@arnilo/prism-provider-opencode-go`](providers/opencode-go.md) (official Go open models, dual-route Anthropic/OpenAI, caller-gated `listOpenCodeGoModels`, `reasoning_content`/thinking preserve), [`@arnilo/prism-provider-openrouter`](providers/openrouter.md) (app-controlled catalog, caller-gated `listOpenRouterModels`, `reasoning` merge/preserve, `cache_control` + sticky `session_id`), [`@arnilo/prism-provider-zai`](providers/zai.md) (official `thinking`/`reasoning_effort`/`tool_stream`, implicit cache, caller-gated `listZaiModels`), [`@arnilo/prism-provider-kimi`](providers/kimi.md), [`@arnilo/prism-provider-alibaba`](providers/alibaba.md) (Alibaba Cloud Model Studio / DashScope + Coding Plan, OpenAI-compatible, caller-gated `listAlibabaModels`, implicit + explicit `cache_control` caching, Qwen `enable_thinking`), [`@arnilo/prism-provider-ollama`](providers/ollama.md) (Ollama Cloud + local, OpenAI-compatible, caller-gated `listOllamaModels`, implicit-only caching, `reasoning_effort`), and [`@arnilo/prism-provider-neuralwatt`](providers/neuralwatt.md) with implicit vLLM prefix caching, reasoning controls (`reasoning_effort`/`thinking_token_budget`/`enable_thinking`/`preserve_thinking`/`clear_thinking`), reasoning preservation, OpenAI-style tool-call loop, quota, telemetry, and retry classification helpers.
51
51
  - Phase 8 enterprise cloud (workload identity; separate from consumer Anthropic/Google): [`@arnilo/prism-provider-azure`](providers/azure.md) (Entra / Foundry), [`@arnilo/prism-provider-bedrock`](providers/bedrock.md) (IAM/IRSA + region/PrivateLink), [`@arnilo/prism-provider-vertex`](providers/vertex.md) (ADC / Vertex OpenAPI).
52
52
  - Optional AI SDK adapter: [`@arnilo/prism-provider-ai-sdk`](providers/ai-sdk.md) maps host-owned pinned `LanguageModelV4` models onto Prism `AIProvider` streams (offline-tested `@ai-sdk/provider` version matrix; no Prism catalog; maps metadata/tool authority/`finish.usage` cache tokens; reasoning is host-model-owned).
53
- - [OpenAI-compatible provider](providers/openai-compatible.md): optional provider subpath using native or injected `fetch` for Chat Completions streaming (`chatCompletionsUrl` / `authStyle` overrides for enterprise adapters).
53
+ - [OpenAI-compatible provider](providers/openai-compatible.md): optional provider subpath using native or injected `fetch` for Chat Completions streaming (`chatCompletionsUrl` / `authStyle` overrides for enterprise adapters; `buildBodyExtra` / `mapMessages` / `mapUsage` / `extraHeaders` hooks for vendor variants).
54
54
 
55
55
  ## Input, prompt, and context assembly
56
56
  - [SDK customization guide](customization.md): map provider resolution, middleware, context, builders, injectors, loops, compaction, retry, stores, and skills to explicit host-wired APIs.
@@ -117,7 +117,9 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
117
117
  - `examples/`: compile-checked typed examples and runnable mock demos (SDK basics, provider registration, auth, tools, [`examples/ag-ui-server.ts`](../examples/ag-ui-server.ts), [`examples/enterprise-identity.ts`](../examples/enterprise-identity.ts), [`examples/enterprise-policy-audit.ts`](../examples/enterprise-policy-audit.ts), [`examples/enterprise-work-connectors.ts`](../examples/enterprise-work-connectors.ts), [`examples/conversation-durable-replay.ts`](../examples/conversation-durable-replay.ts), [`examples/artifact-review-delivery.ts`](../examples/artifact-review-delivery.ts), [`examples/server-deployment-seams.ts`](../examples/server-deployment-seams.ts), cache-aware prompt assembly, NeuralWatt agent run ([`examples/neuralwatt-agent-run.ts`](../examples/neuralwatt-agent-run.ts)), [`examples/coding-compaction.ts`](../examples/coding-compaction.ts), stores/branching, structured-output/artifact-loop, CLI, RPC, workflow orchestration).
118
118
 
119
119
  ## Release and install
120
- - [Release and install](release-and-install.md): current **0.0.15** 43-package graph (Phase 10 provider/AI-SDK/RAG/memory parity; no new package), exact-peer/install/tarball rules, deterministic resumable publication and publish dry-run, pinned supply-chain gates, offline tests, the 0.0.15 provider/AI-SDK/RAG/memory protected live-canary matrix, and sandbox-browser Docker/Playwright gates.
120
+ - [Release and install](release-and-install.md): current **0.0.17** 44-package graph (code-review hardening release; plan 081), exact-peer/install/tarball rules, deterministic resumable publication and publish dry-run, pinned supply-chain gates, offline tests, the 0.0.15 provider/AI-SDK/RAG/memory protected live-canary matrix (still standing for 0.0.16), and sandbox-browser Docker/Playwright gates.
121
+ - [0.1.0 / 1.0 readiness gates](0.1.0-readiness.md): command-per-gate 1.0 readiness table — frozen API surface + compat gate, migration coverage 0.0.5→0.0.16, budget table, live-suite matrix, security matrix, the signed-publication/live-canary prerequisites remaining for 1.0, and the Phase 12 demand-evidence entry criteria.
122
+ - [Review coverage (2026-07-26 Phase 11)](review-coverage-2026-07-26-phase-11.md): Plan 079 evidence freeze — baseline size/startup/benchmark budgets, hotspot domain extraction table, confirmed duplication survivors (redactor/cleanJson/row-codecs/checkpoints/exec-runner/approval/ownership), profile adoption recommendations, and tarball artifact-diet findings for 0.0.16.
121
123
  - [Review coverage (2026-07-26 Phase 10)](review-coverage-2026-07-26-phase-10.md): Plan 078 evidence freeze — OpenAI hosted tools/continuation/realtime, AI SDK version matrix, remaining provider metadata parity, RAG replaceSource/loaders/parsers/reranker/provenance/ingestion-status, memory export/rebuild/conformance, and 0.0.15 (43 → 43 manifests; no new package) release gates.
122
124
  - [Review coverage (2026-07-25 Phase 9)](review-coverage-2026-07-25-phase-9.md): Plan 077 evidence freeze — conversation service, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth, browser checkpoint composition, and deny-by-default device contracts for 0.0.14 (41 → 43 manifests; only the two provider packages are new).
123
125
  - [Review coverage (2026-07-23 Phase 8)](review-coverage-2026-07-23-phase-8.md): Plan 076 evidence freeze — enterprise identity/policy/router packages, Azure/Bedrock/Vertex adapters, server deployment seams, persistence lifecycle hooks, and M365/GWS work-connector bounds for 0.0.13.
@@ -2,9 +2,9 @@
2
2
 
3
3
  ## What it does
4
4
 
5
- `createDefaultInputBuilder()` turns common host input into Prism `Message[]` without starting an agent loop or calling a provider. It accepts strings, `Message`, or `Message[]`, and can add host-supplied instructions, history, summaries, attachments, explicit text resources, tool results, metadata, and optional `input_assembly` middleware.
5
+ `createDefaultInputBuilder()` turns common host input into Prism `Message[]` without starting an agent loop or calling a provider. It accepts strings, `Message`, or `Message[]`, and can add host-supplied instructions, history, summaries, attachments, explicit text resources, tool results, and metadata. It never applies `input_assembly` middleware itself: `assembleProviderInput()` owns that hook and runs it exactly once after whichever `InputBuilder` is installed returns, so a custom builder cannot bypass it.
6
6
 
7
- `createDefaultPromptBuilder()` composes messages, context blocks, selected skills, and host-supplied active tools into provider-ready messages. `assembleProviderInput()` wires input assembly, ordered context resolution, prompt middleware, and prompt composition into a `ProviderRequest` without calling a provider. Layered system prompts are composed before this helper and passed as `systemInstructions`. `renderPromptTemplate()` expands tiny `{{name}}` variables for CLI/RPC prompt strings before input assembly.
7
+ `createDefaultPromptBuilder()` composes messages, context blocks, selected skills, and host-supplied active tools into provider-ready messages. The `Available tools:` text listing is emitted only for models without declared tool support (`model.capabilities.tools !== true` — unknown capability keeps it, fail-safe for text-only providers); tool-capable models receive schemas via `request.tools` and skip the duplicated text. `assembleProviderInput()` wires input assembly, ordered context resolution, prompt middleware, and prompt composition into a `ProviderRequest` without calling a provider. Layered system prompts are composed before this helper and passed as `systemInstructions`. `renderPromptTemplate()` expands tiny `{{name}}` variables for CLI/RPC prompt strings before input assembly.
8
8
 
9
9
  ## When to use it
10
10
 
@@ -161,7 +161,7 @@ const request = await assembleProviderInput({
161
161
  });
162
162
  ```
163
163
 
164
- `input_assembly`, `context`, and `prompt_build` middleware are not global. They run only for helper calls that receive a `MiddlewareRegistry`, in that assembly order. `assembleProviderInput()` keeps provider `tools` equal to the host-supplied active tool list after prompt middleware.
164
+ `input_assembly`, `context`, and `prompt_build` middleware are not global. They run only for helper calls that receive a `MiddlewareRegistry`, in that assembly order. Inside `assembleProviderInput()`, `input_assembly` always runs — for both the default and any custom `InputBuilder`, and on both the plain and context-budget paths. `assembleProviderInput()` keeps provider `tools` equal to the host-supplied active tool list after prompt middleware.
165
165
 
166
166
  ## Security and performance notes
167
167
 
@@ -43,11 +43,11 @@ Built-in hook names:
43
43
  | `run(hook, value)` | hook name and payload | Runs registered middleware and returns the final payload. |
44
44
  | `list(hook)` | hook name | Returns registered middleware for inspection. |
45
45
 
46
- `Middleware<T>` receives `(value, next)` and returns a value or promise. Calling `next(updatedValue)` passes an updated value to later middleware.
46
+ `Middleware<T>` receives `(value, next)` and returns a value or promise. Calling `next(updatedValue)` passes an updated value to later middleware. Two rules are enforced: call `next()` **at most once** — a second call throws (routed through the registry `errorPolicy`) naming hook and index; and **either** `return next(v)` **or** return a new value, never both — when `next(v)` was already called, a conflicting return is discarded and diagnosed via `onError` (the `next()` value wins).
47
47
 
48
48
  ## Outputs / response / events
49
49
 
50
- `run()` returns the transformed value. If no middleware is registered for a hook, `run()` returns the original value. `assembleProviderInput()` calls Phase 5 hooks in this order when middleware is supplied: `input_assembly`, then `context`, then `prompt_build`. The agent/session runtime applies configured provider request policies, then invokes `provider_request` once with the `ProviderRequest` before `AIProvider.generate()`, invokes `tool_call` and `tool_result` through `dispatchToolCall()` for complete provider tool calls, invokes `compaction` with `{ context, result }` after a compaction strategy returns and before the runtime appends its standard compaction entry, and invokes `retry` with `{ context, decision }` before scheduling a provider-turn retry. There is no `provider_response` hook; observing provider output belongs to the provider adapter or subscriber events.
50
+ `run()` returns the transformed value. If no middleware is registered for a hook, `run()` returns the original value. `assembleProviderInput()` calls Phase 5 hooks in this order when middleware is supplied: `input_assembly`, then `context`, then `prompt_build`. The `input_assembly` call is unconditional — it runs after whatever `InputBuilder` produced the messages, so host middleware at that hook cannot be skipped by a custom builder. The agent/session runtime applies configured provider request policies, then invokes `provider_request` once with the `ProviderRequest` before `AIProvider.generate()`, invokes `tool_call` and `tool_result` through `dispatchToolCall()` for complete provider tool calls, invokes `compaction` with `{ context, result }` after a compaction strategy returns and before the runtime appends its standard compaction entry, and invokes `retry` with `{ context, decision }` before scheduling a provider-turn retry. There is no `provider_response` hook; observing provider output belongs to the provider adapter or subscriber events.
51
51
 
52
52
  With default `errorPolicy: "event"`, middleware errors become `extension_error` events when `onError` is provided, and later middleware still runs with the current value. With `errorPolicy: "throw"`, `run()` rejects on the first middleware error.
53
53
 
package/docs/migration.md CHANGED
@@ -1,5 +1,16 @@
1
1
  # Migration guide
2
2
 
3
+ ## 0.0.16 → 0.0.17 code-review hardening (small intentional breaks)
4
+
5
+ Release **0.0.17** implements the 2026-07-29 full implementation review (plan 081): twenty fixes across durable runs, guardrails, retry, extension lifecycle, CLI, and provider plumbing. Most changes are additive or internal; four intentionally change existing behavior:
6
+
7
+ 1. **CLI: inert flags now rejected.** `--config`, `--resource`, `--extension`, and `--tool` were parsed-and-recorded without effect; `parseCliArgs` now throws `CliUsageError("<flag> is not supported in this build")`. The dead `config` / `resources` / `extensions` / `tools` fields were removed from `CliOptions`. Hosts passing those flags must drop them until a CLI-harness plan wires them.
8
+ 2. **`ExtensionKernel.load()` returns handles.** `load(extensions)` now resolves to `LoadedExtension[]` (`{ name, dispose() }`) instead of `void`; callers ignoring the return value are unaffected. A failed `setup` now unwinds that extension's partial registrations. Contribution registries, `ProviderRegistry`, and `ModelRegistry` gain `unregister(...)` (additive).
9
+ 3. **Default prompt builder omits the tool text list for tool-capable models.** When `model.capabilities.tools === true`, the `Available tools:` system message is no longer emitted (schemas already travel via `request.tools`); unknown/`false` capability keeps it. Saves duplicated tokens per turn; observable only in prompt text.
10
+ 4. **Default retry policy applies jitter and honors Retry-After.** `createDefaultRetryPolicy` now applies ±25% jitter (`jitter`/`random` options) and honors `error.retryAfterMs` (populated from provider `Retry-After` headers), capped by `maxDelayMs`. Delays are no longer deterministic unless `random` is injected.
11
+
12
+ Additive-only highlights: `MemoryCredentialStoreOptions.allowProviderFallback` (strict provider scoping opt-in), `createMemoryCheckpointStore` `maxRecords`/`maxValueBytes` bounds, `ShellToolOptions.envAllowlist`, guardrail `steer_rejected` event, `ErrorInfo.retryAfterMs`, agent fingerprint now covers instructions/system prompt/skills (existing durable runs resume or fail fingerprint exactly as before — the fingerprint only got stricter).
13
+
3
14
  ## What it does
4
15
 
5
16
  Prism 0.0.6 preserves documented 0.0.3 agent construction except for two intentional Phase 3 public-API cleanups:
@@ -7,6 +18,35 @@ Prism 0.0.6 preserves documented 0.0.3 agent construction except for two intenti
7
18
  1. **`session.run()` / `session.prompt()` return `AgentRunResult`** and `session.stream()` starts one owned run after subscribing. Callers that ignored the previous `Promise<void>` keep working; failed/aborted runs reject with `AgentRunError` (`.result` attached).
8
19
  2. **`AgentConfig.extensions` / `settings` / `credentials` are removed.** Wire extensions through `createExtensionKernel()`, read settings in the host, and pass credential resolvers to the provider edge.
9
20
 
21
+ ## 0.0.15 → 0.0.16 simplification, shared survivors, and release gates (additive, pre-release)
22
+
23
+ Release **0.0.16** is a simplification/readiness release: no runtime behavior changes, no package retired, and the only public-surface change is one additive export plus one internal package. The published root tarball is smaller and the release now runs offline pre-publish gates. See [Phase 11 evidence](review-coverage-2026-07-26-phase-11.md).
24
+
25
+ ### New shared export: `resolveRedactor` (additive)
26
+
27
+ `@arnilo/prism` now exports `resolveRedactor(redactor?, secrets?)` from `src/redaction.ts` — the single survivor of four private copies previously duplicated across `evals`, `memory`, `rag`, and `workflows`. Those packages now source it from core; no package previously exported it, so this is purely additive (added to the frozen value-export surface deliberately). Hosts that resolved a redactor by hand can use it directly:
28
+
29
+ ```ts
30
+ import { resolveRedactor } from "@arnilo/prism";
31
+ const redactor = resolveRedactor(undefined, [apiKey, process.env.SECRET]);
32
+ ```
33
+
34
+ Provider JSON cleanup (`cleanJson`) was deliberately **not** consolidated: the nine provider copies are private one-liners with real wire-shape variants (neuralwatt/openrouter also strip `null`), so they remain per-package. Checkpoint codecs were already consolidated in `workflows/src/checkpoint-core.ts`, and the executable `spawn` sites stay per-domain because each encodes distinct security invariants.
35
+
36
+ ### New internal package: `@arnilo/prism-session-store-codecs`
37
+
38
+ The two 409-line SQLite/Postgres row-mapper files (which differed only in the `redacted` boolean representation) were replaced by a shared `createSessionRowMappers<R>(codec)` factory in the new `@arnilo/prism-session-store-codecs` package (44th manifest). It is an internal implementation detail of the two session stores — not enrolled in `prism-all` or any profile family — so no install recipe or import changes for consumers.
39
+
40
+ Surface note: `@arnilo/prism-session-store-sqlite` and `@arnilo/prism-session-store-postgres` no longer re-export the individual row-mapper functions (`rowToSessionRecord`, `sessionEntryToRow`, `encodeEntryCursor`, `decodeEntryCursor`, `parentKey`, and the other `*ToRow`/`rowTo*` helpers). These were persistence internals; the supported entry points remain `createSqlitePersistence` / `createPostgresPersistence` and friends. If you imported a mapper directly, build the equivalent with `createSessionRowMappers(codec)` from `@arnilo/prism-session-store-codecs` (pass the SQLite INTEGER or Postgres BOOLEAN `redacted` codec).
41
+
42
+ ### Profiles: all six retained (no migration)
43
+
44
+ Adoption evidence (manifest dependents + docs/examples) froze all six profiles — `prism-all`, `prism-base`, `prism-code`, `prism-compaction`, `prism-providers`, `prism-sdk` — as **retain**; zero retirements. Task 0's "compaction/base zero dependents" was a measurement error (profiles are manifest-only and never imported in `src`). The profiles form a layered DAG (`all → {code, sdk, providers}`, `code/sdk → base → compaction`). Install recipes are unchanged except a new standalone `prism-compaction` recipe in [release-and-install.md](release-and-install.md). No profile migration is needed.
45
+
46
+ ### Smaller root tarball + offline release gates (no runtime impact)
47
+
48
+ The root package no longer ships the historical `docs/review-coverage-*.md` evidence (11 files, ~283 KB): the packed tarball dropped from 659,478 to ≈575,680 bytes (281 → 270 files). `npm run release:gate` now runs offline pre-publish gates (API-surface `.d.ts` diff vs `scripts/compat-baseline/`, tarball deny-list, exact version ranges) and is part of `npm run sdk:ready`. Performance budgets are recorded in `scripts/budgets.json` and enforced by `scripts/budget-gate.test.mjs` (in `npm test`) and `scripts/benchmark-0.0.16.mjs`; see [performance.md](performance.md). None of this changes SDK runtime behavior.
49
+
10
50
  ## 0.0.14 → 0.0.15 OpenAI hosted tools, continuation, and realtime (additive, pre-release)
11
51
 
12
52
  `@arnilo/prism-provider-openai` now distinguishes server-executed calls with `authority: "provider-hosted"`; host dispatchers must not execute or reply to them. Incomplete Responses streams self-resume with an opaque `previous_response_id` cursor (at most 4 KiB, at most eight hops) and surface `continuation_required`; cap or duplicate-cursor failure now ends with a provider error instead of a silent partial response.
@@ -6,6 +6,39 @@ Evaluation defaults are finite: 100 trace rows × 20 pages and 4 MiB aggregate t
6
6
 
7
7
  This page states Prism runtime limits that keep slow consumers and long sessions from becoming unbounded memory or latency problems.
8
8
 
9
+ ## Release 0.0.16 performance budgets and artifact diet
10
+
11
+ Release 0.0.16 is a simplification/readiness release: it added no performance-affecting code, so the six network-free scenario medians are held at the 0.0.15 baseline and the win is a smaller published artifact. Budgets live in `scripts/budgets.json` (measured baselines + tolerance) and are enforced two ways:
12
+
13
+ - **Fast gate (every `npm test`)** — `scripts/budget-gate.test.mjs` re-packs the root tarball (`npm pack --dry-run --json`) and fails if packed bytes, unpacked bytes, or file count exceed baseline + 5%, and fails if cold-process `import('./dist/index.js')` exceeds the 250 ms sanity ceiling. Negative fixtures prove an inflated/regressed value fails.
14
+ - **Release evidence runner** — `node scripts/benchmark-0.0.16.mjs` re-measures root pack + startup, spawns `benchmark-0.0.15.mjs` for the six scenario medians (reused unchanged), compares every value to `budgets.json` (throughput floor / latency ceiling at ±25%), prints the evidence report below, and exits non-zero on any regression.
15
+
16
+ **Artifact diet (the 0.0.16 finding).** The Task 1 tarball deny list dropped the historical `docs/review-coverage-*.md` (11 files, 283,022 bytes) from the root package: the root tarball went from **659,478 packed / 2,310,686 unpacked / 281 files** (0.0.15) to a budgeted **≈575,680 packed / 2,043,402 unpacked / 270 files**. The per-release `scripts/benchmark-0.0.*.mjs` history never shipped in artifacts (root `files` is `dist`/`docs`/`templates`/`CHANGELOG.md` only — zero `scripts/` entries packed), so no archive move was needed; `benchmark-0.0.16.mjs` consolidates the current evidence behind one budget-gating runner.
17
+
18
+ **Recorded budgets (`scripts/budgets.json`, measured 2026-07-26, Node v24.18.0, Linux x86_64):**
19
+
20
+ | Budget | Baseline | Tolerance |
21
+ | --- | --- | --- |
22
+ | Root packed bytes | 575,680 | +5% |
23
+ | Root unpacked bytes | 2,043,402 | +5% |
24
+ | Root file count | 270 | +5% |
25
+ | Aggregate packed bytes (44 manifests, reference only) | 1,217,694 | +10% |
26
+ | Startup `import('./dist/index.js')` | ~38 ms | ceiling 250 ms |
27
+ | Six scenario medians (below) | 0.0.15 baseline | ±25% |
28
+
29
+ **0.0.16 measured evidence** (`node scripts/benchmark-0.0.16.mjs`, 100 iterations each, network-free, 0 backpressure / 0 resource-limit signals; all 22 budget checks passed):
30
+
31
+ | Scenario | throughput/s | p50 ms | p95 ms |
32
+ | --- | --- | --- | --- |
33
+ | openai-hosted-continuation | 5,514.9 | 0.1305 | 0.2735 |
34
+ | openai-realtime-envelope | 900.5 | 1.1277 | 1.2002 |
35
+ | ai-sdk-v4-stream-mapping | 23,850.4 | 0.0225 | 0.0795 |
36
+ | provider-package-metadata | 54,097.0 | 0.0066 | 0.0386 |
37
+ | rag-parse-replace-rerank-retrieve | 5,176.6 | 0.1428 | 0.3671 |
38
+ | memory-retention-export-rebuild | 13,952.1 | 0.0470 | 0.1339 |
39
+
40
+ Root startup measured ≈37.7 ms (ceiling 250 ms). Timing is machine-dependent, so medians carry a wide ±25% band and are release evidence rather than tight cross-machine guarantees; the deterministic artifact-size gate is the hard CI tripwire. Raise the baselines in `scripts/budgets.json` after a deliberate, reviewed performance change.
41
+
9
42
  ## Release 0.0.15 provider, RAG, and memory evidence
10
43
 
11
44
  Run `node scripts/benchmark-0.0.15.mjs`; `PRISM_BENCH_ITERATIONS` accepts 10–100,000 (default 100). Schema/bounds test: `node --test scripts/benchmark-0.0.15.test.mjs`. Default mode is network-free: fake Responses SSE/WebSocket transports, a fake AI SDK v4 model, zero-fetch provider-package registration, hash embeddings, in-memory RAG replacement/reranking/retrieval/status, and in-memory memory retention/export/rebuild.
@@ -32,6 +32,22 @@ Options:
32
32
  | `fetch` | `typeof fetch` | Optional fetch implementation for tests or custom hosts. |
33
33
  | `chatCompletionsUrl` | `string \| ((request) => string)` | Optional full chat-completions URL override (Azure deployment paths). |
34
34
  | `authStyle` | `"bearer" \| "api-key" \| "none"` | Auth header style. Default `bearer`. |
35
+ | `buildBodyExtra` | `(request) => JsonObject \| undefined` | Optional provider-specific body fields (thinking/reasoning/cache); merged over the base body. |
36
+ | `mapMessages` | `(request) => readonly Message[]` | Optional message transform before serialization (e.g. cache-control markers). Defaults to `request.messages`. |
37
+ | `mapUsage` | `(usage: unknown) => Usage \| undefined` | Optional usage mapping override (e.g. OpenRouter cost fields). Defaults to `mapOpenAIChatUsage`. |
38
+ | `serializeMessage` | `(message, request) => JsonObject` | Optional custom message serializer (e.g. Z.AI `reasoning_content` replay). Defaults to assert + `serializeOpenAIChatMessage`. |
39
+ | `doneUsage` | `boolean` | Emit the final stream usage on the `done` event (without strict completion checks). |
40
+ | `mapHttpError` | `(response, bodyText, secrets) => Error` | Custom HTTP error mapping (e.g. NeuralWatt retry classification). Receives the response and redacted body text. |
41
+ | `onComment` | `(text) => ProviderEvent \| undefined` | Handle SSE comment lines (text after `:`), e.g. NeuralWatt `: energy` / `: cost` telemetry. Returned events are yielded in stream order. |
42
+ | `extraHeaders` | `(request) => Record<string, string>` | Optional extra request headers; provider auth and `content-type` still win. |
43
+ | `transformBody` | `(body, request) => JsonObject` | Optional final body transform, applied last (token limits, compat stripping); wins over everything. |
44
+ | `strictCompletion` | `boolean` | Require `[DONE]` and a `finish_reason`; truncated streams yield an `error` and `done` carries the final usage. |
45
+ | `requestFailedPrefix` | `string` | Prefix for HTTP error messages. Default `OpenAI-compatible request failed`. |
46
+
47
+ The subpath also exports the building blocks for provider packages that keep public body/stream helpers:
48
+
49
+ - `openAIChatEvents(body, { signal, strictCompletion, doneUsage, mapUsage, onComment })`: the shared SSE stream loop as an `AsyncIterable<ProviderEvent>`.
50
+ - `buildOpenAIChatBody(request, { mapMessages, serializeMessage, buildBodyExtra, transformBody })`: the base Chat Completions request body builder.
35
51
 
36
52
  Provider requests use the standard `ProviderRequest` shape: `model`, `messages`, optional `tools`, `metadata`, and `signal`.
37
53
 
@@ -49,7 +65,7 @@ The returned provider emits normalized `ProviderEvent` values:
49
65
  | `[DONE]` or stream end | `done` event. |
50
66
  | HTTP/stream/parsing error | `error` event with redacted `ErrorInfo`. |
51
67
 
52
- The adapter passes `request.signal` to `fetch` for abort propagation.
68
+ The adapter passes `request.signal` to `fetch` for abort propagation; an already-aborted signal throws before fetch.
53
69
 
54
70
  ## Request/response example
55
71
 
@@ -110,6 +126,17 @@ const provider = createOpenAICompatibleProvider({
110
126
  - The adapter resolves `apiKey` per request through `resolveCredentialValue()`.
111
127
  - This adapter currently targets Chat Completions streaming only.
112
128
  - The serializer preserves text, thinking (downgraded to text), assistant `tool_call` blocks as `tool_calls`, `tool_result` blocks as role `tool` messages, and image blocks when the model declares `capabilities.input` includes `"image"`. Unsupported block placements or unclaimed images fail before fetch.
129
+ - Vendor-specific OpenAI-compatible endpoints (cache markers, thinking bodies, reasoning fields, custom usage) plug in through `buildBodyExtra`/`mapMessages`/`mapUsage`/`extraHeaders` instead of duplicating the stream loop:
130
+
131
+ ```ts
132
+ const provider = createOpenAICompatibleProvider({
133
+ baseUrl: "https://vendor.example/v1",
134
+ apiKey: () => process.env.VENDOR_API_KEY,
135
+ buildBodyExtra: (request) => ({ thinking: { type: "enabled" } }),
136
+ extraHeaders: () => ({ "x-vendor-app": "my-app" }),
137
+ });
138
+ ```
139
+
113
140
  - Cache behavior is intentionally minimal: this Chat Completions adapter sends no `prompt_cache_key`, `prompt_cache_retention`, or `cache_control` fields. Endpoints that cache implicitly do so automatically; hosts needing OpenAI `prompt_cache_key`/`prompt_cache_retention` should use the [`@arnilo/prism-provider-openai`](openai.md) Responses package. The adapter still normalizes cache usage from `prompt_tokens_details.cached_tokens` (and `prompt_cache_hit_tokens`) into `Usage.cacheReadTokens`.
114
141
 
115
142
  ## Security and performance notes
@@ -145,7 +145,7 @@ Important request shapes:
145
145
  | `ConfigLayer` | Named JSON config layer consumed by `mergeConfigLayers()`. |
146
146
  | `PrismManifest` | Data-only package manifest with config defaults, contribution declarations, and resource declarations. |
147
147
  | `ProductionPersistenceStore` | Adapter-facing interface for durable, paginated, multi-tenant storage plus optional `checkpoints?: CheckpointStore`, `leases?: LeaseStore`, and `feedback?: RunFeedbackStore`. No SQL/ORM/host file storage/network dependency. |
148
- | `CheckpointStore` | Generic versioned checkpoint capability: save/load/bounded-list/delete by namespace and key, with ownership, exact-version CAS, and lease fencing. `createMemoryCheckpointStore()` is the reference implementation. |
148
+ | `CheckpointStore` | Generic versioned checkpoint capability: save/load/bounded-list/delete by namespace and key, with ownership, exact-version CAS, and lease fencing. `createMemoryCheckpointStore()` is the reference implementation; it is bounded — `maxRecords` (default 10,000, evicts least-recently-saved) and `maxValueBytes` (default 1 MiB per JSON value). |
149
149
  | `LeaseStore` | Atomic acquire/renew/release/get by namespace and key, with opaque claim tokens, expiry, ownership scope, and monotonically increasing takeover fences. `createMemoryLeaseStore()` is the reference implementation. |
150
150
  | `RunFeedbackStore` | Immutable append, bounded owned query, and owned deletion for ratings/comments/tags linked to existing run/trace/evaluation IDs. `createMemoryRunFeedbackStore()` is the reference implementation. |
151
151
  | `EventMultiplexer<T>` | Generic bounded fan-in from async sources. `createEventMultiplexer()` owns queue limits, overflow policy, abort, source teardown, and close behavior. |
@@ -463,4 +463,4 @@ void credentials;
463
463
  - `@arnilo/prism/providers/transport`: bounded SSE/event parsing, bounded HTTP error-body reads, and JSON-object tool-argument parsing for provider packages.
464
464
  - `@arnilo/prism/providers/openai`: OpenAI Chat Completions message/tool serialization, usage mapping, and indexed message validation helpers.
465
465
 
466
- Phase 10 public helpers include `createStaticSettingsProvider`, `createChainedSettingsProvider`, `createMemoryCredentialStore`, `createChainedCredentialResolver`, `createStaticTrustPolicy`, `assertTrusted`, `createStaticPermissionPolicy`, `assertPermission`, and `createSecretRedactor`. Phase 11 auth/request/prompt helpers include `createExplicitCredentialResolver`, `createEnvCredentialResolver`, `refreshOAuthCredential`, `createProviderRequestPolicyChain`, `createSessionCachePolicy`, `mergeProviderRequestOptions`, `composeSystemPrompt`, and `mergeSystemPromptConfig`; they do not read env vars, persist OAuth tokens, create cache stores, discover prompt files, or load packages unless the host supplies that behavior. `@arnilo/prism/testing/provider-conformance` exports network-free provider assertion helpers. Node subpaths `@arnilo/prism/node/settings` and `@arnilo/prism/node/trust` are explicit filesystem/path helpers.
466
+ Phase 10 public helpers include `createStaticSettingsProvider`, `createChainedSettingsProvider`, `createMemoryCredentialStore`, `createChainedCredentialResolver`, `createStaticTrustPolicy`, `assertTrusted`, `createStaticPermissionPolicy`, `assertPermission`, and `createSecretRedactor`. Phase 11 auth/request/prompt helpers include `createExplicitCredentialResolver`, `createEnvCredentialResolver`, `refreshOAuthCredential`, `createProviderRequestPolicyChain`, `createSessionCachePolicy`, `mergeProviderRequestOptions`, `composeSystemPrompt`, and `mergeSystemPromptConfig`; 0.0.16 adds `resolveRedactor(redactor?, secrets?)`, which resolves the active redactor from an explicit redactor plus known secret values (the single survivor of the former per-package copies). They do not read env vars, persist OAuth tokens, create cache stores, discover prompt files, or load packages unless the host supplies that behavior. `@arnilo/prism/testing/provider-conformance` exports network-free provider assertion helpers. Node subpaths `@arnilo/prism/node/settings` and `@arnilo/prism/node/trust` are explicit filesystem/path helpers.