@arnilo/prism 0.0.14 → 0.0.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/CHANGELOG.md +22 -2
  2. package/README.md +5 -4
  3. package/dist/agent-definitions.js +2 -3
  4. package/dist/agent-loops.d.ts +4 -0
  5. package/dist/agent-loops.js +28 -10
  6. package/dist/agent-run-lifecycle.d.ts +1 -2
  7. package/dist/agent-run-lifecycle.js +1 -1
  8. package/dist/agent-run-state.js +16 -3
  9. package/dist/agents.d.ts +1 -1
  10. package/dist/agents.js +121 -51
  11. package/dist/cache-helpers.js +18 -9
  12. package/dist/checkpoints.js +5 -9
  13. package/dist/cli-init.js +3 -7
  14. package/dist/cli-runner.d.ts +1 -1
  15. package/dist/cli-runner.js +74 -13
  16. package/dist/compaction.js +5 -4
  17. package/dist/config.js +7 -4
  18. package/dist/content.js +26 -24
  19. package/dist/context-budget.js +12 -8
  20. package/dist/contracts.d.ts +79 -3
  21. package/dist/contracts.js +4 -1
  22. package/dist/contribution-parsing.js +6 -2
  23. package/dist/conversations.js +2 -1
  24. package/dist/credentials.d.ts +1 -1
  25. package/dist/credentials.js +3 -1
  26. package/dist/event-multiplexer.js +1 -3
  27. package/dist/feedback.js +11 -9
  28. package/dist/guardrails.d.ts +1 -1
  29. package/dist/guardrails.js +17 -14
  30. package/dist/identity.js +10 -2
  31. package/dist/index.d.ts +83 -84
  32. package/dist/index.js +42 -42
  33. package/dist/input.d.ts +2 -2
  34. package/dist/input.js +40 -24
  35. package/dist/instruction-injection.d.ts +1 -1
  36. package/dist/node/agent-definitions.js +16 -8
  37. package/dist/node/contribution-discovery.d.ts +1 -2
  38. package/dist/node/contribution-discovery.js +3 -3
  39. package/dist/node/session-store-jsonl.js +10 -7
  40. package/dist/node/settings.d.ts +1 -1
  41. package/dist/node/settings.js +1 -1
  42. package/dist/node/system-project-prompts.js +2 -4
  43. package/dist/node/trust.js +1 -1
  44. package/dist/persistence-lifecycle.js +1 -3
  45. package/dist/provider-events.d.ts +1 -0
  46. package/dist/provider-events.js +6 -1
  47. package/dist/provider-request-policy.js +3 -4
  48. package/dist/providers/media.d.ts +1 -1
  49. package/dist/providers/openai-compatible.js +3 -4
  50. package/dist/providers/openai-primitives.js +7 -7
  51. package/dist/redaction.d.ts +1 -0
  52. package/dist/redaction.js +5 -2
  53. package/dist/resources.d.ts +2 -2
  54. package/dist/resources.js +2 -2
  55. package/dist/rpc.js +42 -9
  56. package/dist/run-ledger.js +13 -4
  57. package/dist/run-limits.js +49 -10
  58. package/dist/secure-agent.js +1 -1
  59. package/dist/security.js +7 -2
  60. package/dist/session-stores.d.ts +1 -1
  61. package/dist/session-stores.js +13 -13
  62. package/dist/structured-output.js +2 -2
  63. package/dist/system-prompts.js +7 -2
  64. package/dist/testing/compaction-conformance.js +5 -1
  65. package/dist/testing/extension-conformance.js +15 -3
  66. package/dist/testing/feedback.d.ts +1 -3
  67. package/dist/testing/feedback.js +1 -1
  68. package/dist/testing/persistence-schema.js +206 -37
  69. package/dist/testing/provider-conformance.js +3 -3
  70. package/dist/testing/run-ledger-conformance.js +1 -1
  71. package/dist/testing/session-store-conformance.js +1 -1
  72. package/dist/testing/tool-conformance.js +30 -5
  73. package/dist/thinking.js +4 -1
  74. package/dist/tools.d.ts +2 -2
  75. package/dist/tools.js +24 -5
  76. package/docs/0.1.0-readiness.md +139 -0
  77. package/docs/host-security.md +4 -1
  78. package/docs/index.md +14 -11
  79. package/docs/migration.md +58 -1
  80. package/docs/multimodal-content.md +8 -5
  81. package/docs/performance.md +67 -0
  82. package/docs/provider-caching.md +8 -0
  83. package/docs/provider-conformance.md +29 -5
  84. package/docs/provider-packages.md +22 -1
  85. package/docs/providers/ai-sdk.md +23 -7
  86. package/docs/providers/openai.md +22 -3
  87. package/docs/public-contracts.md +1 -1
  88. package/docs/rag.md +41 -12
  89. package/docs/release-and-install.md +153 -16
  90. package/docs/resource-loading.md +3 -0
  91. package/docs/working-and-semantic-memory.md +22 -4
  92. package/package.json +14 -6
  93. package/docs/review-coverage-2026-07-14.md +0 -260
  94. package/docs/review-coverage-2026-07-15.md +0 -193
  95. package/docs/review-coverage-2026-07-17-provider-validation.md +0 -192
  96. package/docs/review-coverage-2026-07-19-phase-3.md +0 -174
  97. package/docs/review-coverage-2026-07-20-phase-4.md +0 -175
  98. package/docs/review-coverage-2026-07-21-phase-5.md +0 -172
  99. package/docs/review-coverage-2026-07-22-phase-6.md +0 -209
  100. package/docs/review-coverage-2026-07-22-phase-7.md +0 -173
  101. package/docs/review-coverage-2026-07-23-phase-8.md +0 -245
  102. package/docs/review-coverage-2026-07-25-phase-9.md +0 -256
package/dist/tools.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import { isJsonObject } from "./config.js";
2
- import { createId } from "./ids.js";
3
2
  import { GuardrailError, runGuardrails } from "./guardrails.js";
4
3
  import { assertIdentityActive, assertIdentityMatchesOwnership } from "./identity.js";
4
+ import { createId } from "./ids.js";
5
5
  import { errorToErrorInfo, redactRunLedgerRecord, redactSecrets } from "./redaction.js";
6
6
  import { assertCanRegister } from "./registry-options.js";
7
7
  import { assertPermission, assertTrusted } from "./security.js";
@@ -52,7 +52,9 @@ export function createToolRegistry(tools = [], options = {}) {
52
52
  export function filterTools(tools, filter) {
53
53
  const filters = Array.isArray(filter) ? filter : filter ? [filter] : [];
54
54
  const denied = new Set(filters.flatMap((item) => item.deny ?? []));
55
- const allows = filters.map((item) => item.allow?.length ? new Set(item.allow) : undefined).filter((item) => Boolean(item));
55
+ const allows = filters
56
+ .map((item) => (item.allow?.length ? new Set(item.allow) : undefined))
57
+ .filter((item) => Boolean(item));
56
58
  return tools.filter((tool) => !denied.has(tool.name) && allows.every((allow) => allow.has(tool.name)));
57
59
  }
58
60
  function toolExecutionMetadata(startedAt, status) {
@@ -114,8 +116,18 @@ export async function dispatchToolCall(options) {
114
116
  assertIdentityActive(context.identity);
115
117
  assertIdentityMatchesOwnership(context.identity, options.ownership);
116
118
  }
117
- await assertTrusted(options.trust, { kind: "tool", target: mediatedCall.name, capability: "execute", metadata: options.context.metadata });
118
- await assertPermission(options.permission, { kind: "tool", action: "execute", target: mediatedCall.name, metadata: options.context.metadata });
119
+ await assertTrusted(options.trust, {
120
+ kind: "tool",
121
+ target: mediatedCall.name,
122
+ capability: "execute",
123
+ metadata: options.context.metadata,
124
+ });
125
+ await assertPermission(options.permission, {
126
+ kind: "tool",
127
+ action: "execute",
128
+ target: mediatedCall.name,
129
+ metadata: options.context.metadata,
130
+ });
119
131
  }
120
132
  catch (error) {
121
133
  return blocked(mediatedCall, context, "permission_denied", errorToErrorInfo(error, secrets), options, startedAt);
@@ -170,7 +182,14 @@ export async function dispatchToolCall(options) {
170
182
  const result = { toolCallId: mediatedCall.id, name: mediatedCall.name, error: info };
171
183
  const finishedAt = new Date().toISOString();
172
184
  const metadata = toolExecutionMetadata(startedAt, "error");
173
- await options.emit?.({ type: "tool_execution_error", sessionId: context.sessionId, runId: context.runId, call: mediatedCall, error: info, metadata });
185
+ await options.emit?.({
186
+ type: "tool_execution_error",
187
+ sessionId: context.sessionId,
188
+ runId: context.runId,
189
+ call: mediatedCall,
190
+ error: info,
191
+ metadata,
192
+ });
174
193
  await appendToolCallRecord(options, "error", mediatedCall, startedAt, { finishedAt, result });
175
194
  return result;
176
195
  }
@@ -0,0 +1,139 @@
1
+ # 0.1.0 / 1.0 Readiness Gates
2
+
3
+ Status: **0.0.16 is a 1.0 readiness review, not an automatic 1.0 release.**
4
+ This page distills the Phase 11 (0.0.16) gates into one command-per-gate table
5
+ so readiness is checkable, not prose. Every gate below is a runnable command
6
+ with last evidence captured on 2026-07-26 (release 0.0.16, Node v24.18.0,
7
+ Linux x86_64). The decision to cut 1.0 stays with the operator after the
8
+ operator-gated legs run in a protected environment and the Phase 12 demand
9
+ evidence exists.
10
+
11
+ Evidence trail: [`docs/review-coverage-2026-07-26-phase-11.md`](./review-coverage-2026-07-26-phase-11.md)
12
+ (addenda 0–9), [`docs/release-and-install.md`](./release-and-install.md),
13
+ [`docs/migration.md`](./migration.md), [`docs/performance.md`](./performance.md).
14
+
15
+ ## Gate table
16
+
17
+ | Gate | Command | Last evidence (2026-07-26) | Owner |
18
+ |---|---|---|---|
19
+ | Full quality gate | `npm run sdk:ready` | RC=0: typecheck (+examples), lint 0, format clean, full test, coverage, pack, release:gate | CI |
20
+ | Exact version graph | `node scripts/release.mjs check --version <v>` | 0.0.16 pass: exact versions/ranges/lockfile/access + registry-collision check, 44 manifests | CI + operator |
21
+ | Frozen public API surface + compat gate | `node scripts/release.mjs gate` | 0 breaks / 0 errors vs 44 checked-in baselines (`scripts/compat-baseline/`); only additive delta is `resolveRedactor` | CI |
22
+ | Migration coverage 0.0.5→0.0.16 | `node --test dist/__tests__/docs.test.js` | 112/112; `docs/migration.md` has one section per release 0.0.5→0.0.16, each tripwired | Maintainer |
23
+ | Deterministic artifact budget | `node --test dist/__tests__/budget-gate.test.mjs` | root 579.2 kB / 2.1 MB / 270 files within +5% of baseline; startup 38 ms < 250 ms ceiling | CI (in `npm test`) |
24
+ | Performance benchmark medians | `node scripts/benchmark-0.0.16.mjs` | 6 network-free scenarios within ±25%; 0 backpressure / 0 resource-limit signals | On-demand release evidence |
25
+ | Secret scan | `node scripts/scan-secrets.mjs` | 3095 files / 0 findings | CI |
26
+ | License / SBOM | `node scripts/verify-sbom.mjs` | 188 packages / 8 licenses, all allow-listed | CI |
27
+ | Dependency audit | `npm audit --audit-level=high` | rc=0 (2 moderate, 0 high) | CI |
28
+ | Whitespace hygiene | `git diff --check` | clean | CI |
29
+ | Publish order + tarball validation | `node scripts/release.mjs publish --version <v> --dry-run --allow-dirty --allow-untagged` | 44/44 packages `dry-run`, deterministic dependency order, no failures | Operator (dry-run), CI |
30
+ | Node 20 compatibility | CI `node20-compat` (build + public-import smoke) | all 21 root exports import cleanly on Node 20.20.2 | CI |
31
+ | PostgreSQL suite | `npm run test:postgres` | **operator-gated** (requires live PostgreSQL) | Operator |
32
+ | Keychain / live-provider suites | `npm run test:live` (protected) | **operator-gated** (requires credentials) | Operator |
33
+ | SAST | GitHub CodeQL | **operator-gated** (runs in CI workflow) | CI |
34
+ | Signed, provenance publication | `npm run release:publish` (clean tagged tree, OIDC) | **operator-gated** (see "Remaining for 1.0") | Operator |
35
+
36
+ ## Frozen public API surface
37
+
38
+ The compat gate diffs every package's generated `.d.ts` export surface against
39
+ checked-in baselines in `scripts/compat-baseline/` (one file per package,
40
+ regenerated at 0.0.16). It fails on any **removed** export or changed
41
+ declaration; additive exports are allowed. `scripts/release-gates.mjs` also
42
+ enforces a tarball deny list (no `docs/review-coverage-*` in published
43
+ artifacts) and exact version-range drift. A genuine break requires
44
+ `--allow-break` **and** a `docs/migration.md` entry mentioning the version.
45
+
46
+ **Baseline maintenance:** `scripts/compat-baseline/` must stay committed.
47
+ Regenerate only after review with `node scripts/release.mjs gate --update-baseline`,
48
+ having first confirmed zero removed exports (the gate's order-sensitive
49
+ signatures can drift on a TypeScript bump without any real API change).
50
+
51
+ ## Migration coverage 0.0.5 → 0.0.16
52
+
53
+ `docs/migration.md` carries one section per release from 0.0.5 through 0.0.16;
54
+ `docs.test.ts` tripwires each section heading and key phrase, so a missing or
55
+ gutted migration section fails the suite. 0.0.16's only user-facing change is
56
+ the additive `resolveRedactor` export (no breaking changes).
57
+
58
+ ## Budget table
59
+
60
+ Deterministic budgets (CI gate, `scripts/budget-gate.test.mjs`):
61
+
62
+ | Metric | Baseline | Tolerance | 0.0.16 measured |
63
+ |---|---|---|---|
64
+ | Root packed bytes | 575,680 | +5% | 579.2 kB (within) |
65
+ | Root unpacked bytes | 2,043,402 | +5% | 2.1 MB (within) |
66
+ | Root file count | 270 | +5% | 270 |
67
+ | Cold-startup import | 38 ms | ceiling 250 ms | ~38 ms |
68
+ | Aggregate packed (44 manifests, reference only) | 1,217,694 | +10% | not gated in fast test |
69
+
70
+ Benchmark medians (on-demand evidence, `scripts/benchmark-0.0.16.mjs`, ±25%):
71
+
72
+ | Scenario | throughput/s | p50 ms | p95 ms |
73
+ |---|---|---|---|
74
+ | openai-hosted-continuation | 5,386.0 | 0.1317 | 0.2907 |
75
+ | openai-realtime-envelope | 880.3 | 1.1293 | 1.2399 |
76
+ | ai-sdk-v4-stream-mapping | 22,403.0 | 0.0230 | 0.0734 |
77
+ | provider-package-metadata | 55,829.2 | 0.0065 | 0.0394 |
78
+ | rag-parse-replace-rerank-retrieve | 4,800.3 | 0.1432 | 0.4064 |
79
+ | memory-retention-export-rebuild | 8,763.2 | 0.0686 | 0.1892 |
80
+
81
+ Baselines are a 2026-07-26 snapshot (`scripts/budgets.json`); raise them only
82
+ after a deliberate reviewed performance change.
83
+
84
+ ## Live-suite matrix (operator-gated)
85
+
86
+ | Suite | Command | Environment |
87
+ |---|---|---|
88
+ | PostgreSQL persistence | `npm run test:postgres` | live PostgreSQL |
89
+ | Keychain credentials | protected live suite | OS keychain |
90
+ | Provider live-canary (OpenAI/Anthropic/Google/Kimi/Ollama/…) | protected live-canary matrix | vendor credentials |
91
+ | RAG / memory / workflows live journeys | protected live-canary matrix | vendor + DB credentials |
92
+
93
+ These do not run on a contributor machine; their evidence is recorded in the
94
+ protected environment, never faked.
95
+
96
+ ## Security matrix
97
+
98
+ | Control | Command / source | 0.0.16 status |
99
+ |---|---|---|
100
+ | Secret scan | `node scripts/scan-secrets.mjs` | 3095 files / 0 findings |
101
+ | License / SBOM | `node scripts/verify-sbom.mjs` | 188 packages / 8 licenses, allow-listed |
102
+ | Dependency audit | `npm audit --audit-level=high` | 0 high (2 moderate) |
103
+ | SAST | GitHub CodeQL workflow | CI-gated |
104
+ | Sandbox / protocol / tenant threat suites | `npm test` (coding-security, MCP, policy, guardrail suites) | green |
105
+ | Signed deterministic publication | `release.mjs publish` on clean tagged tree | operator-gated |
106
+
107
+ ## Remaining for 1.0 (operator / protected environment)
108
+
109
+ Exact prerequisites that must be satisfied before cutting 1.0:
110
+
111
+ 1. **Signed tag + commits:** create and sign `v0.1.0` (or `v1.0.0`) on a clean
112
+ tree; `release.mjs publish` refuses real publication with `--allow-dirty`
113
+ or `--allow-untagged`.
114
+ 2. **npm authentication + OIDC provenance/attestation:** publish with
115
+ `--provenance` and `--access public` from the protected registry identity.
116
+ 3. **Protected live-canary matrix green:** provider/RAG/memory/workflows live
117
+ journeys pass with real credentials.
118
+ 4. **PostgreSQL + keychain protected suites green.**
119
+ 5. **CodeQL SAST green** on the release commit.
120
+ 6. **`scripts/compat-baseline/` committed** so CI's compat gate has a
121
+ checked-in baseline.
122
+ 7. **Phase 12 demand evidence** (below) recorded for any capability that 1.0
123
+ is expected to anchor.
124
+
125
+ ## Phase 12 demand-evidence entry criteria
126
+
127
+ Phase 12 (demand-gated 0.1.x) promotes no capability on comparison-table parity
128
+ alone. Each candidate must present, before it becomes a numbered plan:
129
+
130
+ - **Named user** (a concrete person/team who will use it),
131
+ - **Concrete integration** (the real system it connects to),
132
+ - **Operational owner** (who runs and pages for it),
133
+ - **Measurable acceptance criteria** (scale/cost/latency/storage budgets that
134
+ do not expand default core/install/runtime cost),
135
+ - then the pipeline: demand evidence → primitive review → threat model →
136
+ optional package/service → conformance → release gate.
137
+
138
+ The 0.0.16 readiness gates above are the stable API/compat/budget/security
139
+ floor that Phase 12 capabilities must consume and must not regress.
@@ -134,8 +134,11 @@ Wire those values where they matter: provider adapters receive the resolved cred
134
134
  - Prism does not sandbox host tools, extensions, provider adapters, credential resolvers, or custom middleware. Use OS/container/process isolation when code is untrusted.
135
135
  - Redaction is exact known-secret replacement only. It is not arbitrary secret detection, entropy scanning, or DLP.
136
136
  - Known secrets must be passed into redactors before data is emitted or persisted. Redact again in host adapters if they transform records after Prism redaction.
137
+ - OpenAI Realtime sessions require a stable host owner identifier, use header-only credentials, and bind to the server `session.created` id. Treat returned audio/transcripts as untrusted; use a `SecretRedactor`, retain finite event/byte/wall caps, and close on disconnect or an identity/budget breach.
137
138
  - Tool `parameters` metadata is not validated by default. Add a `ToolValidator`, use `createToolParameterValidator()` with a schema adapter, or install `@arnilo/prism-tool-validator-json-schema` before side effects. Its untrusted-schema adapter rejects non-local refs, forbidden keys/cycles/non-finite values and bounds bytes/depth/properties/keywords/refs plus its LRU cache before Ajv compilation; do not raise caps above documented hard limits.
138
- - Treat embeddings as untrusted numeric input. `@arnilo/prism-memory` rejects empty, non-number, NaN, and infinite vectors before in-memory similarity or pgvector parameters; custom `Embedder`/`VectorStore` implementations must retain the same boundary. Memory entries carry consent/source/visibility; revoked, invisible, or (in strict mode) consent-less entries never enter prompts, events, exports, or telemetry, and `forget`/`applyRetention` are real bounded deletes.
139
+ - RAG `replaceSource()` only accepts a store with scoped `getBySource()` plus a real transaction; it stages bounded embeddings before mutation and otherwise fails closed. `deleteSource()` rechecks returned tenant/resource/corpus/source metadata. `createResourceDocumentLoader()` receives only a host-authorized `ResourceLoader`; `createWebFetchDocumentLoader()` never opens I/O and rejects local/private/IP-literal URLs before delegating to host-configured web-tools. HTML scripts/styles are stripped, PDF parsing has byte/page/time caps and rejects compressed PDFs.
140
+ - RAG retrieval always emits `trust: { untrusted: true, inert: true, injectionCapable: true }` plus attributable citation provenance. Context blocks repeat this metadata and never gain tool authority. Host `Reranker`s see redacted finite candidates, are hard-capped by bytes/time/concurrency, must return only a permutation of candidate IDs, and cannot overwrite provenance/trust. Ingestion status errors are redacted; status storage/listing stays exact-scope and capped.
141
+ - Treat embeddings as untrusted numeric input. `@arnilo/prism-memory` rejects empty, non-number, NaN, and infinite vectors before in-memory similarity, pgvector parameters, export, or rebuild; custom `Embedder`/`VectorStore` implementations must retain the same boundary. Memory entries carry consent/source/visibility; revoked/invisible entries never enter prompts, events, exports, or telemetry. `exportMemory()` additionally excludes consent-less legacy records regardless of recall mode and requires exact host identity equal to its tenant/resource/thread scope. Save rebuild cursors only in host-authorized storage; `rebuildIndex()` is one abortable capped page, never an implicit corpus job. `forget`/`applyRetention` are real bounded deletes.
139
142
  - Evaluation trace readers require exact supplied ownership plus session/run identity, reject cursor/identity drift, and redact before bounded scorer/judge input. Model-judge callbacks receive no credential resolver, tools, or workspace; keep live judges outside default CI and redact report artifacts.
140
143
  - Prism-generated session/run/tool/workflow/evaluation IDs use Node cryptographic UUIDs. Keep host-provided IDs authorization-scoped and validate them as untrusted identifiers; do not substitute timestamps or `Math.random()` for durable/security-relevant IDs.
141
144
  - MCP client tools from `@arnilo/prism-mcp` are untrusted remote servers. Stdio remains an explicit host executable. Streamable HTTP requires exact HTTPS origins, rejects credentials/fragments/redirects/private or mixed DNS, pins a validated address on every SDK request/reconnect, and bounds each response; plaintext is explicit loopback-only development mode. Discovery has finite page/tool/cursor/metadata/schema totals and commits atomically. Every result branch shares byte/depth/property bounds before core dispatch; supply a known-secret `SecretRedactor`, `PermissionPolicy`, and `ToolValidator` there. MCP server direction exposes only passed tools/commands/resources/prompts, requires per-operation `authorize`, and retains core gates. Sampling, roots, model/credential selection, and elicitation consent stay host-owned; URL elicitation is never opened automatically. Stateful web mode requires host `resolveAuthInfo` plus `resolveIdentity`, exact origin policy, and binds every POST/GET/DELETE/SSE request to one non-secret principal; mismatches return 404. Handler still needs TLS and edge rate limiting. See [MCP client/server exposure](mcp-tools.md).
package/docs/index.md CHANGED
@@ -19,14 +19,14 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
19
19
  - [Observability](observability.md): OTel GenAI agent/provider/tool hierarchy, host context parenting, bounded trace linkage, safe evaluation events, controlled metrics, and exporter isolation.
20
20
  - [Evaluations](evaluations.md): deterministic and bounded trace/model-judge/pairwise scoring, CI thresholds, OTel trace-reference linkage, coding/browser adversarial fixtures, and ID-only linkage to immutable owned run feedback.
21
21
  - [Runs and usage ledger](runs-and-usage.md): durable run/event/tool/usage persistence, optional bounded FIFO durability policies, session snapshot caching, and immutable run/trace feedback.
22
- - [Performance limits](performance.md): bounded evaluation traces/judges/reports, 0.0.12 frontend interoperability benchmark evidence/caps, 0.0.11 search/budget, 0.0.10 workspace-mode, and 0.0.9 coding/browser benchmark evidence, security scan/live-canary backstops, live subscriber queues, branch-read pagination expectations, JSONL/dev-store limits, and production sizing assumptions.
22
+ - [Performance limits](performance.md): 0.0.15 network-free provider/RAG/memory benchmark evidence and frozen caps, bounded evaluation traces/judges/reports, 0.0.12 frontend interoperability, 0.0.11 search/budget, 0.0.10 workspace-mode, 0.0.9 coding/browser, security scan/live-canary backstops, live subscriber queues, branch-read pagination expectations, JSONL/dev-store limits, and production sizing assumptions.
23
23
  - [Structured output](structured-output.md): the `Artifact*` seam plus provider-native `StructuredOutputOptions` / `structuredOutputMode` for capable models.
24
24
 
25
25
  ## Compaction/session memory
26
26
  - [Compaction and retry policies](compaction-and-retry.md): summarize branch history and retry transient provider failures with host-replaceable policies.
27
27
  - [LLM compaction package](compaction-llm.md): optional provider-backed strategy with finite summary/reserve/error caps, bounded redacted streaming retention, mandatory finite post-policy `model.parameters.maxTokens`, and `createCodingCompactionStrategy()` for coding handoff focus.
28
28
  - [Observational memory compaction package](compaction-observational-memory.md): optional source-backed memory with owned append callback, finite turn/call/argument/result/transcript/error worker limits, redacted provider-valid transcripts, fast compaction, recall, and status/view commands; worker model falls back to host-supplied `sessionModel`.
29
- - [Working and semantic memory](working-and-semantic-memory.md): optional `@arnilo/prism-memory` working-memory store, semantic recall, finite Embedder/VectorStore contracts, in-memory adapters, PostgreSQL/pgvector path, and consent/source/visibility lifecycle (grant/correct/forget/retention) enforced at injection.
29
+ - [Working and semantic memory](working-and-semantic-memory.md): optional `@arnilo/prism-memory` working-memory store, semantic recall, finite Embedder/VectorStore contracts, PostgreSQL/pgvector path, consent lifecycle, identity-bound redacted export, and resumable bounded rebuild.
30
30
  - [Session stores](session-stores.md): `SessionStore` contract, `SessionAppendOptions`, `SessionAppendConflictError`, branch handles, `readBranchPath`, optional bounded `searchSessions` / `SessionIndex` (memory linear|unsupported), and dev-vs-production branch reads — start here for session persistence.
31
31
  - [Conversations](conversations.md): durable user-scoped conversation threads (create/list/continue/branch/archive/export/delete) on session + event-ledger seams, thread-bound reconnectable replay, frozen caps, and legal-hold-aware deletion.
32
32
  - [Work artifacts and review](work-artifacts-and-review.md): durable artifact co-work review — authorized attach (MIME/hash/version, producer run, citations, preview metadata), revision compare, approve/reject with last-validated recovery, and authorized expiring delivery links; records persist as versioned checkpoints, never file bodies.
@@ -34,7 +34,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
34
34
  - [Database persistence](database-persistence.md): production persistence contracts, shared checksummed migration/full-shape catalog primitives (`@arnilo/prism/testing/persistence-schema`), conditional append, indexes, `readBranchPath`, reference relational schema, retention/legal-hold/quota lifecycle (`lifecycle`), and NoSQL mapping.
35
35
  - [SQLite persistence](sqlite-persistence.md): optional `better-sqlite3` adapter with session/run storage, checkpoints/leases, feedback, FTS `searchSessions` (migration-v4), and transactionally verified/backfilled migration metadata.
36
36
  - [PostgreSQL persistence](postgres-persistence.md): optional pooled `pg` adapter with session/run/checkpoint/lease/feedback storage, FTS `searchSessions` (migration-v4), advisory-locked checksummed/full-shape migrations, and opt-in live conformance.
37
- - [Migration guide](migration.md): **0.0.14** personal/work-agent conversations, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth connectors, browser checkpoints, device contracts, and Alibaba/Ollama providers; **0.0.13** enterprise identity/policy/router/work connectors, cloud providers, server deployment seams, and persistence schema v5; plus 0.0.12 AG-UI/ACP, 0.0.11 coding-harness fundamentals, 0.0.10 workspace modes, and 0.0.9 coding/browser surfaces.
37
+ - [Migration guide](migration.md): **0.0.15** OpenAI hosted tools/continuation/Realtime, exact AI SDK v4 matrix, RAG lifecycle/reranking/trust/status, and memory export/rebuild; **0.0.14** conversations, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth connectors, browser checkpoints, device contracts, and Alibaba/Ollama providers; plus prior release migrations.
38
38
  - [Node JSONL session store](node-jsonl-session-store.md): development-only JSONL file adapter for single-process Node hosts; no cross-process safety; `searchSessions` throws `SessionSearchUnsupportedError`.
39
39
  - [Persistence, credentials, and multimodality primitives](persistence-credentials-multimodality-primitives.md): Plan 056 inventory — session/run-ledger/persistence contracts, credential/OAuth seams, content/resource/model capabilities, package dependency matrix, conformance matrix, and threat model for production adapters.
40
40
 
@@ -42,24 +42,24 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
42
42
  - [Provider primitives](provider-primitives.md): shared bounded transport and OpenAI serialization helpers — migrated across first-party providers; native structured-output and observability contracts.
43
43
  - [Provider layer](provider-layer.md): register and resolve host-owned providers/models, choose replace-or-error duplicate policy, create provider events, stream/reconstruct tool-call deltas, use generic provider request options, and test with the mock provider; deprecated provider-level timeout/retry hints point to runtime abort/retry.
44
44
  - [Model registry](model-registry.md): register and resolve `ModelConfig` records with capabilities, limits, cost, cache support metadata, compat data, and duplicate policy.
45
- - [Provider caching](provider-caching.md): use `PromptCacheHints`, `PromptCacheBreakpoint`, `ModelCacheCapabilities`, cache-aware stable-prefix guidance, and shared cache diagnostics helpers; includes a per-provider explicit/implicit cache matrix for OpenAI, OpenRouter, OpenCode Go, Z.AI, Kimi, NeuralWatt, and the host-owned AI SDK adapter; cache hints are best-effort and cache keys are never secrets.
45
+ - [Provider caching](provider-caching.md): use `PromptCacheHints`, `PromptCacheBreakpoint`, `ModelCacheCapabilities`, cache-aware stable-prefix guidance, and shared cache diagnostics helpers; includes the complete per-provider explicit/implicit cache matrix plus no-Prism-cache entries (including Anthropic, Google, Alibaba, Ollama, cloud adapters, and host-owned AI SDK); cache hints are best-effort and cache keys are never secrets.
46
46
  - [Thinking and reasoning](thinking-and-reasoning.md): portable `ThinkingLevel` helpers (`applyThinkingLevel` / `thinkingCompatFor`) map per-turn effort into provider `compat` fields; model defaults stay on `ModelConfig.compat`; no second options tree.
47
47
  - [Use-case model selection](use-case-model-selection.md): bind `{ model?, provider?, thinkingLevel? }` for observational memory, LLM compaction, and other non-session LLM jobs with explicit session-model fallback via `resolveUseCaseModel`.
48
48
  - [Provider request policies](provider-request-policies.md): chain `ProviderRequestPolicy` hooks, use `createSessionCachePolicy`, and merge legacy/structured cache options safely.
49
- - [Provider packages](provider-packages.md): define explicit provider packages, model metadata, auth descriptors, request/cache policies, provider-owned header precedence, and the 0.0.12 provider-authorized OAuth matrix without package discovery or provider-specific core behavior; includes a first-party cache behavior summary and the **caller-gated on-demand model discovery** contract (`list*Models`, setup zero-fetch).
50
- - Phase 12 package workspaces: [`@arnilo/prism-provider-openai`](providers/openai.md), [`@arnilo/prism-provider-anthropic`](providers/anthropic.md) (native Messages, `cache_control`, thinking, caller-gated `listAnthropicModels`), [`@arnilo/prism-provider-google`](providers/google.md) (native Gemini `generateContent` SSE, caller-gated `listGoogleModels`), [`@arnilo/prism-provider-opencode-go`](providers/opencode-go.md) (official Go open models, dual-route Anthropic/OpenAI, caller-gated `listOpenCodeGoModels`, `reasoning_content`/thinking preserve), [`@arnilo/prism-provider-openrouter`](providers/openrouter.md) (app-controlled catalog, caller-gated `listOpenRouterModels`, `reasoning` merge/preserve, `cache_control` + sticky `session_id`), [`@arnilo/prism-provider-zai`](providers/zai.md) (official `thinking`/`reasoning_effort`/`tool_stream`, implicit cache, caller-gated `listZaiModels`), [`@arnilo/prism-provider-kimi`](providers/kimi.md), [`@arnilo/prism-provider-alibaba`](providers/alibaba.md) (Alibaba Cloud Model Studio / DashScope + Coding Plan, OpenAI-compatible, caller-gated `listAlibabaModels`, implicit + explicit `cache_control` caching, Qwen `enable_thinking`), [`@arnilo/prism-provider-ollama`](providers/ollama.md) (Ollama Cloud + local, OpenAI-compatible, caller-gated `listOllamaModels`, implicit-only caching, `reasoning_effort`), and [`@arnilo/prism-provider-neuralwatt`](providers/neuralwatt.md) with implicit vLLM prefix caching, reasoning controls (`reasoning_effort`/`thinking_token_budget`/`enable_thinking`/`preserve_thinking`/`clear_thinking`), reasoning preservation, OpenAI-style tool-call loop, quota, telemetry, and retry classification helpers.
49
+ - [Provider packages](provider-packages.md): define explicit provider packages, model metadata, auth descriptors, request/cache policies, provider-owned header precedence, the provider-authorized OAuth matrix, and the Phase 10 first-party compatibility matrix without package discovery or provider-specific core behavior; includes a cache behavior summary and **caller-gated on-demand model discovery** (`list*Models`, setup zero-fetch).
50
+ - Phase 12 package workspaces: [`@arnilo/prism-provider-openai`](providers/openai.md) (Responses hosted-tool attribution, bounded continuation, Realtime session seam), [`@arnilo/prism-provider-anthropic`](providers/anthropic.md) (native Messages, `cache_control`, thinking, caller-gated `listAnthropicModels`), [`@arnilo/prism-provider-google`](providers/google.md) (native Gemini `generateContent` SSE, caller-gated `listGoogleModels`), [`@arnilo/prism-provider-opencode-go`](providers/opencode-go.md) (official Go open models, dual-route Anthropic/OpenAI, caller-gated `listOpenCodeGoModels`, `reasoning_content`/thinking preserve), [`@arnilo/prism-provider-openrouter`](providers/openrouter.md) (app-controlled catalog, caller-gated `listOpenRouterModels`, `reasoning` merge/preserve, `cache_control` + sticky `session_id`), [`@arnilo/prism-provider-zai`](providers/zai.md) (official `thinking`/`reasoning_effort`/`tool_stream`, implicit cache, caller-gated `listZaiModels`), [`@arnilo/prism-provider-kimi`](providers/kimi.md), [`@arnilo/prism-provider-alibaba`](providers/alibaba.md) (Alibaba Cloud Model Studio / DashScope + Coding Plan, OpenAI-compatible, caller-gated `listAlibabaModels`, implicit + explicit `cache_control` caching, Qwen `enable_thinking`), [`@arnilo/prism-provider-ollama`](providers/ollama.md) (Ollama Cloud + local, OpenAI-compatible, caller-gated `listOllamaModels`, implicit-only caching, `reasoning_effort`), and [`@arnilo/prism-provider-neuralwatt`](providers/neuralwatt.md) with implicit vLLM prefix caching, reasoning controls (`reasoning_effort`/`thinking_token_budget`/`enable_thinking`/`preserve_thinking`/`clear_thinking`), reasoning preservation, OpenAI-style tool-call loop, quota, telemetry, and retry classification helpers.
51
51
  - Phase 8 enterprise cloud (workload identity; separate from consumer Anthropic/Google): [`@arnilo/prism-provider-azure`](providers/azure.md) (Entra / Foundry), [`@arnilo/prism-provider-bedrock`](providers/bedrock.md) (IAM/IRSA + region/PrivateLink), [`@arnilo/prism-provider-vertex`](providers/vertex.md) (ADC / Vertex OpenAPI).
52
- - Optional AI SDK adapter: [`@arnilo/prism-provider-ai-sdk`](providers/ai-sdk.md) maps host-owned `LanguageModelV4` models onto Prism `AIProvider` streams (specification v4; no Prism catalog; maps `finish.usage` cache read/write tokens; reasoning is host-model-owned).
52
+ - Optional AI SDK adapter: [`@arnilo/prism-provider-ai-sdk`](providers/ai-sdk.md) maps host-owned pinned `LanguageModelV4` models onto Prism `AIProvider` streams (offline-tested `@ai-sdk/provider` version matrix; no Prism catalog; maps metadata/tool authority/`finish.usage` cache tokens; reasoning is host-model-owned).
53
53
  - [OpenAI-compatible provider](providers/openai-compatible.md): optional provider subpath using native or injected `fetch` for Chat Completions streaming (`chatCompletionsUrl` / `authStyle` overrides for enterprise adapters).
54
54
 
55
55
  ## Input, prompt, and context assembly
56
56
  - [SDK customization guide](customization.md): map provider resolution, middleware, context, builders, injectors, loops, compaction, retry, stores, and skills to explicit host-wired APIs.
57
57
  - [Input and prompt assembly](input-and-prompt-assembly.md): render tiny prompt templates and turn common host input, history, attachments, explicit resources, summaries, and tool results into messages with replaceable builders, provider-input assembly, legacy default order, opt-in cache-aware ordering, and optional `contextBudget` eviction + omission reports. Audio/file/document `ContentBlock` types and capability checks are documented there.
58
- - [Multimodal content](multimodal-content.md): complete-request media resolution and aggregate bounds, DNS-classified/address-pinned URLs, SSRF/MIME policy, and `ModelCapabilities.input` tags.
58
+ - [Multimodal content](multimodal-content.md): complete-request media resolution and aggregate bounds, DNS-classified/address-pinned URLs, SSRF/MIME policy, `ModelCapabilities.input` tags, and first-party content-type mapping.
59
59
  - [System prompts](system-prompts.md): compose explicit user/package/app/run system prompt layers, auto-load the standard `AGENTS.md` (workspace) / `SYSTEM.md` prompt files via the Node `loadSystemPromptFiles` loader (trust-gated for `AGENTS.md`), and append `SYSTEM.md` → per-agent `AGENT.md` body → repo `AGENTS.md` layers from a discovered agent bundle via `resolveAgentBundle`.
60
60
  - [Instruction injection](instruction-injection.md): register package injectors that layer redacted instructions/context blocks without granting tools, permissions, or resource escapes.
61
61
  - [Context and skills](context-and-skills.md): resolve ordered context providers and keep context/skill selection host-owned; omitted declarative skills stay inactive by default, `toolNames` fail closed before provider turns, and strict skill registries prevent silent shadowing.
62
- - [Retrieval-augmented generation](rag.md): optional bounded text/Markdown chunking, Phase 7 vector indexing/retrieval, stable citations, and explicit inert context injection.
62
+ - [Retrieval-augmented generation](rag.md): optional bounded source lifecycle, document adapters, host reranking, ingestion status, attributable citations, and inert context injection.
63
63
 
64
64
  ## Tools
65
65
  - [Tools](tools.md): register host-owned active tools with replace-or-error duplicate policy, apply exact allow/deny filtering, dispatch normal or opt-in bounded artifact-loop calls, and optionally bound untrusted JSON Schema compilation.
@@ -84,7 +84,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
84
84
  ## Configuration/manifests
85
85
  - [Configuration and manifests](configuration-and-manifests.md): merge in-memory JSON config layers and validate data-only package manifests with prototype-pollution key rejection.
86
86
  - [Node filesystem config loader](node-filesystem-config.md): explicitly read caller-named JSON config files in Node hosts.
87
- - [Resource loading](resource-loading.md): decode text, JSON, binary, and manifest resources through caller-provided loaders with bounded byte limits.
87
+ - [Resource loading](resource-loading.md): decode text, JSON, binary, and manifests through caller-provided loaders; bridge host-authorized artifacts to bounded RAG document loading.
88
88
 
89
89
  ## Server/API
90
90
  - [Web-standard server handler](server.md): optional framework-free authorized direct/SSE agent, durable agent lifecycle, durable workflow routes, plus optional health/drain/rate-limit/replay/deployment-lease seams; explicit bounds and zero default exposure.
@@ -117,7 +117,10 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
117
117
  - `examples/`: compile-checked typed examples and runnable mock demos (SDK basics, provider registration, auth, tools, [`examples/ag-ui-server.ts`](../examples/ag-ui-server.ts), [`examples/enterprise-identity.ts`](../examples/enterprise-identity.ts), [`examples/enterprise-policy-audit.ts`](../examples/enterprise-policy-audit.ts), [`examples/enterprise-work-connectors.ts`](../examples/enterprise-work-connectors.ts), [`examples/conversation-durable-replay.ts`](../examples/conversation-durable-replay.ts), [`examples/artifact-review-delivery.ts`](../examples/artifact-review-delivery.ts), [`examples/server-deployment-seams.ts`](../examples/server-deployment-seams.ts), cache-aware prompt assembly, NeuralWatt agent run ([`examples/neuralwatt-agent-run.ts`](../examples/neuralwatt-agent-run.ts)), [`examples/coding-compaction.ts`](../examples/coding-compaction.ts), stores/branching, structured-output/artifact-loop, CLI, RPC, workflow orchestration).
118
118
 
119
119
  ## Release and install
120
- - [Release and install](release-and-install.md): current **43**-package graph (Phase 9 conversations/artifacts/co-work events, scoped OAuth connectors, browser checkpoints, device contracts, and `@arnilo/prism-provider-alibaba`/`@arnilo/prism-provider-ollama` ship at 0.0.14), install/tarball rules, pinned CodeQL/dependency/SBOM/license/secret/attestation gates, deterministic resumable publication, offline tests, protected live canaries, and sandbox-browser Docker/Playwright gates.
120
+ - [Release and install](release-and-install.md): current **0.0.16** 44-package graph (Phase 11 simplification/readiness; one internal codec package, no retirement), exact-peer/install/tarball rules, deterministic resumable publication and publish dry-run, pinned supply-chain gates, offline tests, the 0.0.15 provider/AI-SDK/RAG/memory protected live-canary matrix (still standing for 0.0.16), and sandbox-browser Docker/Playwright gates.
121
+ - [0.1.0 / 1.0 readiness gates](0.1.0-readiness.md): command-per-gate 1.0 readiness table — frozen API surface + compat gate, migration coverage 0.0.5→0.0.16, budget table, live-suite matrix, security matrix, the signed-publication/live-canary prerequisites remaining for 1.0, and the Phase 12 demand-evidence entry criteria.
122
+ - [Review coverage (2026-07-26 Phase 11)](review-coverage-2026-07-26-phase-11.md): Plan 079 evidence freeze — baseline size/startup/benchmark budgets, hotspot domain extraction table, confirmed duplication survivors (redactor/cleanJson/row-codecs/checkpoints/exec-runner/approval/ownership), profile adoption recommendations, and tarball artifact-diet findings for 0.0.16.
123
+ - [Review coverage (2026-07-26 Phase 10)](review-coverage-2026-07-26-phase-10.md): Plan 078 evidence freeze — OpenAI hosted tools/continuation/realtime, AI SDK version matrix, remaining provider metadata parity, RAG replaceSource/loaders/parsers/reranker/provenance/ingestion-status, memory export/rebuild/conformance, and 0.0.15 (43 → 43 manifests; no new package) release gates.
121
124
  - [Review coverage (2026-07-25 Phase 9)](review-coverage-2026-07-25-phase-9.md): Plan 077 evidence freeze — conversation service, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth, browser checkpoint composition, and deny-by-default device contracts for 0.0.14 (41 → 43 manifests; only the two provider packages are new).
122
125
  - [Review coverage (2026-07-23 Phase 8)](review-coverage-2026-07-23-phase-8.md): Plan 076 evidence freeze — enterprise identity/policy/router packages, Azure/Bedrock/Vertex adapters, server deployment seams, persistence lifecycle hooks, and M365/GWS work-connector bounds for 0.0.13.
123
126
  - [Review coverage (2026-07-22 Phase 7)](review-coverage-2026-07-22-phase-7.md): Plan 075 evidence freeze — AG-UI/ACP package boundary, streamed durable resume, bounded replay/projection, coding compaction preset, and provider-authorized OAuth policy for 0.0.12.
package/docs/migration.md CHANGED
@@ -7,6 +7,63 @@ Prism 0.0.6 preserves documented 0.0.3 agent construction except for two intenti
7
7
  1. **`session.run()` / `session.prompt()` return `AgentRunResult`** and `session.stream()` starts one owned run after subscribing. Callers that ignored the previous `Promise<void>` keep working; failed/aborted runs reject with `AgentRunError` (`.result` attached).
8
8
  2. **`AgentConfig.extensions` / `settings` / `credentials` are removed.** Wire extensions through `createExtensionKernel()`, read settings in the host, and pass credential resolvers to the provider edge.
9
9
 
10
+ ## 0.0.15 → 0.0.16 simplification, shared survivors, and release gates (additive, pre-release)
11
+
12
+ Release **0.0.16** is a simplification/readiness release: no runtime behavior changes, no package retired, and the only public-surface change is one additive export plus one internal package. The published root tarball is smaller and the release now runs offline pre-publish gates. See [Phase 11 evidence](review-coverage-2026-07-26-phase-11.md).
13
+
14
+ ### New shared export: `resolveRedactor` (additive)
15
+
16
+ `@arnilo/prism` now exports `resolveRedactor(redactor?, secrets?)` from `src/redaction.ts` — the single survivor of four private copies previously duplicated across `evals`, `memory`, `rag`, and `workflows`. Those packages now source it from core; no package previously exported it, so this is purely additive (added to the frozen value-export surface deliberately). Hosts that resolved a redactor by hand can use it directly:
17
+
18
+ ```ts
19
+ import { resolveRedactor } from "@arnilo/prism";
20
+ const redactor = resolveRedactor(undefined, [apiKey, process.env.SECRET]);
21
+ ```
22
+
23
+ Provider JSON cleanup (`cleanJson`) was deliberately **not** consolidated: the nine provider copies are private one-liners with real wire-shape variants (neuralwatt/openrouter also strip `null`), so they remain per-package. Checkpoint codecs were already consolidated in `workflows/src/checkpoint-core.ts`, and the executable `spawn` sites stay per-domain because each encodes distinct security invariants.
24
+
25
+ ### New internal package: `@arnilo/prism-session-store-codecs`
26
+
27
+ The two 409-line SQLite/Postgres row-mapper files (which differed only in the `redacted` boolean representation) were replaced by a shared `createSessionRowMappers<R>(codec)` factory in the new `@arnilo/prism-session-store-codecs` package (44th manifest). It is an internal implementation detail of the two session stores — not enrolled in `prism-all` or any profile family — so no install recipe or import changes for consumers.
28
+
29
+ Surface note: `@arnilo/prism-session-store-sqlite` and `@arnilo/prism-session-store-postgres` no longer re-export the individual row-mapper functions (`rowToSessionRecord`, `sessionEntryToRow`, `encodeEntryCursor`, `decodeEntryCursor`, `parentKey`, and the other `*ToRow`/`rowTo*` helpers). These were persistence internals; the supported entry points remain `createSqlitePersistence` / `createPostgresPersistence` and friends. If you imported a mapper directly, build the equivalent with `createSessionRowMappers(codec)` from `@arnilo/prism-session-store-codecs` (pass the SQLite INTEGER or Postgres BOOLEAN `redacted` codec).
30
+
31
+ ### Profiles: all six retained (no migration)
32
+
33
+ Adoption evidence (manifest dependents + docs/examples) froze all six profiles — `prism-all`, `prism-base`, `prism-code`, `prism-compaction`, `prism-providers`, `prism-sdk` — as **retain**; zero retirements. Task 0's "compaction/base zero dependents" was a measurement error (profiles are manifest-only and never imported in `src`). The profiles form a layered DAG (`all → {code, sdk, providers}`, `code/sdk → base → compaction`). Install recipes are unchanged except a new standalone `prism-compaction` recipe in [release-and-install.md](release-and-install.md). No profile migration is needed.
34
+
35
+ ### Smaller root tarball + offline release gates (no runtime impact)
36
+
37
+ The root package no longer ships the historical `docs/review-coverage-*.md` evidence (11 files, ~283 KB): the packed tarball dropped from 659,478 to ≈575,680 bytes (281 → 270 files). `npm run release:gate` now runs offline pre-publish gates (API-surface `.d.ts` diff vs `scripts/compat-baseline/`, tarball deny-list, exact version ranges) and is part of `npm run sdk:ready`. Performance budgets are recorded in `scripts/budgets.json` and enforced by `scripts/budget-gate.test.mjs` (in `npm test`) and `scripts/benchmark-0.0.16.mjs`; see [performance.md](performance.md). None of this changes SDK runtime behavior.
38
+
39
+ ## 0.0.14 → 0.0.15 OpenAI hosted tools, continuation, and realtime (additive, pre-release)
40
+
41
+ `@arnilo/prism-provider-openai` now distinguishes server-executed calls with `authority: "provider-hosted"`; host dispatchers must not execute or reply to them. Incomplete Responses streams self-resume with an opaque `previous_response_id` cursor (at most 4 KiB, at most eight hops) and surface `continuation_required`; cap or duplicate-cursor failure now ends with a provider error instead of a silent partial response.
42
+
43
+ Realtime is opt-in through `createOpenAIRealtimeSession({ model, ownerId, apiKey, ... })`. Supply a stable host-owned `ownerId`; the session uses documented WebSocket headers, waits for `session.created`, exposes audio/transcript/interrupt/close events, and fails closed on disconnect, identity, audio/byte, or wall-time limits. It does not add a vendor package or automatic voice capture/playback.
44
+
45
+ ## 0.0.14 → 0.0.15 AI SDK adapter matrix (additive, pre-release)
46
+
47
+ `@arnilo/prism-provider-ai-sdk` now pins and verifies `@ai-sdk/provider@4.0.3` at setup rather than accepting any v4 minor. Upgrade the peer package to the documented matrix entry. An unlisted installed version fails with typed `AiSdkProviderError` code `unsupported_version`; add a tested matrix row before changing it.
48
+
49
+ Stream output now maps `response-metadata.id` to `message_start`, preserves `providerExecuted` tool authority as `"provider-hosted"`, and rejects unsupported output parts or `structuredOutput.strict` with `unsupported_mapping` rather than dropping them. Pass `redactor` when using the adapter directly; agents retain their existing active-redactor behavior.
50
+
51
+ ## 0.0.14 → 0.0.15 RAG source lifecycle and document adapters (additive, pre-release)
52
+
53
+ `@arnilo/prism-rag` now adds `replaceSource()`, `deleteSource()`, and `replaceDocument()` plus `DocumentLoader` / `Parser` seams. Existing `indexChunks()` behavior is unchanged; use `replaceSource()` when a source can shrink or must retain its old index if re-embedding fails.
54
+
55
+ Atomic replacement deliberately requires a scoped source-aware transaction (`getBySource()` + `transaction()`). The in-memory reference vector store supplies both; durable custom stores must add equivalent exact tenant/resource/corpus behavior before using replacement. Prism rejects a generic upsert-only store rather than offering a non-atomic fallback.
56
+
57
+ Reference parsers (`textParser`, `markdownParser`, `htmlParser`, `pdfParser`) are available from root and `@arnilo/prism-rag/parsers`; loaders are available from `@arnilo/prism-rag/loaders`. HTML removes script/style text. The PDF parser only accepts bounded uncompressed text PDFs (8 MiB / 256 pages / 30 s); install no new parser dependency—supply a host `Parser` for compressed or scanned files. `createWebFetchDocumentLoader()` accepts an existing `@arnilo/prism-web-tools` adapter and preserves its citation/untrusted metadata; it does not add a crawler.
58
+
59
+ RAG retrieval now optionally accepts host-owned `Reranker`; it receives redacted bounded hits and must return their exact IDs once each. Results add `trust`, `provenance`, and `retrievalRank`; context blocks now repeat untrusted/inert/injection-capable metadata. Add `statusStore` to indexing/replacement when hosts need per-source pending/indexed/failed/partial progress, use `listIngestionStatus()` for capped exact-scope pages, and supply durable storage if process restart durability matters. `createMemoryIngestionStatusStore()` is only a reference adapter.
60
+
61
+ ## 0.0.14 → 0.0.15 memory export and rebuild (additive, pre-release)
62
+
63
+ `@arnilo/prism-memory` adds `exportMemory({ identity, cursor?, ... })` and `rebuildIndex({ cursor?, ... })`. Export is not a generic admin dump: provide the exact host-verified tenant/resource/thread identity used to construct `createMemory()`. It excludes revoked, invisible, and consent-less legacy entries, redacts each returned record, and caps one page at 100 entries / 4 MiB / 10 seconds by default (200 / 32 MiB / 60 seconds hard).
64
+
65
+ `rebuildIndex()` re-embeds one 32-record page by default (128 hard), validates existing and new finite vectors, and returns `nextCursor`; persist that cursor in host-owned authorized state and call again to resume after an abort/restart. Neither API scans a corpus or starts a background worker. They require a semantic `VectorStore.listByThread()` implementation; `applyRetention()` now also requires `countByThread()` for bounded oldest-first deletion. The shipped in-memory adapter and PostgreSQL/pgvector adapter conform. `@arnilo/prism-session-store-sqlite` remains a session/run persistence package, not a semantic-vector adapter.
66
+
10
67
  ## 0.0.13 → 0.0.14 personal/work-agent conversations, co-work review, and channel/device gates (additive, pre-release)
11
68
 
12
69
  Release **0.0.14** is strictly additive: every surface extends a shipped package and reuses the AG-UI adapter shipped in 0.0.12. The only new packages are two optional provider adapters (41 → 43 manifests): `@arnilo/prism-provider-alibaba` and `@arnilo/prism-provider-ollama`, both enrolled via the `@arnilo/prism-providers` family. No permission broadening — channel/device/co-work features cannot widen consent, memory, network, file, browser, connector, or tool permissions (roadmap gate 8). See [Phase 9 evidence](review-coverage-2026-07-25-phase-9.md).
@@ -213,7 +270,7 @@ Phase 4 adds optional `@arnilo/prism-evals` for deterministic scorers/datasets/e
213
270
 
214
271
  Phase 5 adds `prism init <dir>` to the existing CLI. It scaffolds a tiny TypeScript project with one selected provider and an offline mock test. Optional `--with-workflows` / `--with-evals` flags add only those packages; storage and telemetry stay opt-in elsewhere.
215
272
 
216
- Phase 6 adds optional `@arnilo/prism-provider-ai-sdk` for AI SDK `LanguageModelV4` interoperability. Install it with `@ai-sdk/provider@^4`, through `@arnilo/prism-providers`, or through `@arnilo/prism-all`; it is not a core dependency.
273
+ Phase 6 adds optional `@arnilo/prism-provider-ai-sdk` for AI SDK `LanguageModelV4` interoperability. For 0.0.15 install its exact supported peer `@ai-sdk/provider@4.0.3` (not `^4`); an unlisted version fails at setup. Install the adapter directly, through `@arnilo/prism-providers`, or through `@arnilo/prism-all`; it is not a core dependency.
217
274
 
218
275
  Phase 7 adds optional `@arnilo/prism-memory` for schema/template-backed working memory and embedding-based semantic recall. Install it directly or through `@arnilo/prism-all`; in-memory adapters are default, and PostgreSQL/pgvector is opt-in. It is not a core dependency.
219
276
 
@@ -49,11 +49,13 @@ Known `ModelCapabilities.input` tags are exported as `MODEL_INPUT_CAPABILITIES`:
49
49
 
50
50
  | Tag | Block type | First-party mapping (declared capability required) |
51
51
  | --- | --- | --- |
52
- | `text` | `text` (default) | All providers |
53
- | `image` | `image` | OpenAI Responses, OpenRouter, OpenCode Go Anthropic route, Kimi, NeuralWatt |
54
- | `audio` | `audio` | OpenAI Responses (`input_audio`) |
55
- | `file` | `file` | OpenAI Responses (`input_file`); Anthropic routes map PDF only |
56
- | `document` | `document` | OpenAI Responses (`input_file`); OpenCode Go Anthropic route; Kimi |
52
+ | `text` | `text` (default) | All first-party providers; Azure/Bedrock/Vertex use their host-selected OpenAI-compatible endpoint/model. |
53
+ | `image` | `image` | OpenAI Responses; Anthropic; Google; Kimi; Z.AI; OpenRouter; OpenCode Go OpenAI route; Alibaba; Ollama; NeuralWatt. Enterprise OpenAI-compatible packages require the host model/endpoint to declare and accept image input. |
54
+ | `audio` | `audio` | OpenAI Responses (`input_audio`) and Google `generateContent` inline data. OpenAI Realtime instead receives `RealtimeSession.sendAudio()` chunks, not an `audio` `ContentBlock`. |
55
+ | `file` | `file` | OpenAI Responses (`input_file`); Anthropic/Kimi/OpenCode Go Anthropic route accept PDF file/document forms; Google maps inline file data. |
56
+ | `document` | `document` | OpenAI Responses (`input_file`); Anthropic/Kimi/OpenCode Go Anthropic route map PDF; Google maps inline document data. |
57
+
58
+ The AI SDK adapter maps declared user text/image/audio/file/document blocks (and assistant text/image/file/document) to AI SDK file parts; `resourceUri` remains host-resolved before `doStream`. Its output `file`, `reasoning-file`, and `source` parts are deliberately rejected as `unsupported_mapping`, not converted to trusted Prism content. Provider capability metadata is the gate—this matrix never upgrades a model that does not declare the matching input tag.
57
59
 
58
60
  ## Outputs / response / events
59
61
 
@@ -136,6 +138,7 @@ try {
136
138
  - Local filesystem paths should use trust policies such as `createPathTrustPolicy()` before exposing URIs to loaders.
137
139
  - Provider upload/create/delete lifecycles are provider-package-local. `@arnilo/prism-provider-openai` inlines files under 4 MiB as `data:<mediaType>;base64,...` `file_data`, otherwise uses a bounded per-run upload cache and best-effort `DELETE /v1/files` cleanup after each stream.
138
140
  - Shared wire helpers live in `@arnilo/prism/providers/media` (`resolveProviderMediaMessages`, `serializeOpenAIResponsesInputFile`, `serializePdfDocumentWireBlock`, `createBoundedUploadCache`). OpenAI Responses, Kimi, and OpenCode Go Anthropic routes resolve their complete media collection once before serialization or upload.
141
+ - OpenAI Realtime audio is a bidirectional `RealtimeSession` stream, not a `ContentBlock`: provide host-captured `Uint8Array` chunks with `sendAudio()` and consume untrusted `audio_delta` / transcript events. It has a fixed 256 events/s, 1 MiB/s, and 600 s default ceiling.
139
142
 
140
143
  ## Security and performance notes
141
144
 
@@ -6,6 +6,73 @@ Evaluation defaults are finite: 100 trace rows × 20 pages and 4 MiB aggregate t
6
6
 
7
7
  This page states Prism runtime limits that keep slow consumers and long sessions from becoming unbounded memory or latency problems.
8
8
 
9
+ ## Release 0.0.16 performance budgets and artifact diet
10
+
11
+ Release 0.0.16 is a simplification/readiness release: it added no performance-affecting code, so the six network-free scenario medians are held at the 0.0.15 baseline and the win is a smaller published artifact. Budgets live in `scripts/budgets.json` (measured baselines + tolerance) and are enforced two ways:
12
+
13
+ - **Fast gate (every `npm test`)** — `scripts/budget-gate.test.mjs` re-packs the root tarball (`npm pack --dry-run --json`) and fails if packed bytes, unpacked bytes, or file count exceed baseline + 5%, and fails if cold-process `import('./dist/index.js')` exceeds the 250 ms sanity ceiling. Negative fixtures prove an inflated/regressed value fails.
14
+ - **Release evidence runner** — `node scripts/benchmark-0.0.16.mjs` re-measures root pack + startup, spawns `benchmark-0.0.15.mjs` for the six scenario medians (reused unchanged), compares every value to `budgets.json` (throughput floor / latency ceiling at ±25%), prints the evidence report below, and exits non-zero on any regression.
15
+
16
+ **Artifact diet (the 0.0.16 finding).** The Task 1 tarball deny list dropped the historical `docs/review-coverage-*.md` (11 files, 283,022 bytes) from the root package: the root tarball went from **659,478 packed / 2,310,686 unpacked / 281 files** (0.0.15) to a budgeted **≈575,680 packed / 2,043,402 unpacked / 270 files**. The per-release `scripts/benchmark-0.0.*.mjs` history never shipped in artifacts (root `files` is `dist`/`docs`/`templates`/`CHANGELOG.md` only — zero `scripts/` entries packed), so no archive move was needed; `benchmark-0.0.16.mjs` consolidates the current evidence behind one budget-gating runner.
17
+
18
+ **Recorded budgets (`scripts/budgets.json`, measured 2026-07-26, Node v24.18.0, Linux x86_64):**
19
+
20
+ | Budget | Baseline | Tolerance |
21
+ | --- | --- | --- |
22
+ | Root packed bytes | 575,680 | +5% |
23
+ | Root unpacked bytes | 2,043,402 | +5% |
24
+ | Root file count | 270 | +5% |
25
+ | Aggregate packed bytes (44 manifests, reference only) | 1,217,694 | +10% |
26
+ | Startup `import('./dist/index.js')` | ~38 ms | ceiling 250 ms |
27
+ | Six scenario medians (below) | 0.0.15 baseline | ±25% |
28
+
29
+ **0.0.16 measured evidence** (`node scripts/benchmark-0.0.16.mjs`, 100 iterations each, network-free, 0 backpressure / 0 resource-limit signals; all 22 budget checks passed):
30
+
31
+ | Scenario | throughput/s | p50 ms | p95 ms |
32
+ | --- | --- | --- | --- |
33
+ | openai-hosted-continuation | 5,514.9 | 0.1305 | 0.2735 |
34
+ | openai-realtime-envelope | 900.5 | 1.1277 | 1.2002 |
35
+ | ai-sdk-v4-stream-mapping | 23,850.4 | 0.0225 | 0.0795 |
36
+ | provider-package-metadata | 54,097.0 | 0.0066 | 0.0386 |
37
+ | rag-parse-replace-rerank-retrieve | 5,176.6 | 0.1428 | 0.3671 |
38
+ | memory-retention-export-rebuild | 13,952.1 | 0.0470 | 0.1339 |
39
+
40
+ Root startup measured ≈37.7 ms (ceiling 250 ms). Timing is machine-dependent, so medians carry a wide ±25% band and are release evidence rather than tight cross-machine guarantees; the deterministic artifact-size gate is the hard CI tripwire. Raise the baselines in `scripts/budgets.json` after a deliberate, reviewed performance change.
41
+
42
+ ## Release 0.0.15 provider, RAG, and memory evidence
43
+
44
+ Run `node scripts/benchmark-0.0.15.mjs`; `PRISM_BENCH_ITERATIONS` accepts 10–100,000 (default 100). Schema/bounds test: `node --test scripts/benchmark-0.0.15.test.mjs`. Default mode is network-free: fake Responses SSE/WebSocket transports, a fake AI SDK v4 model, zero-fetch provider-package registration, hash embeddings, in-memory RAG replacement/reranking/retrieval/status, and in-memory memory retention/export/rebuild.
45
+
46
+ Scenarios: `openai-hosted-continuation`, `openai-realtime-envelope`, `ai-sdk-v4-stream-mapping`, `provider-package-metadata`, `rag-parse-replace-rerank-retrieve`, and `memory-retention-export-rebuild`.
47
+
48
+ Every row reports throughput, p50/p95 latency, heap, disk, queue/backpressure, and safety signals. `resourceLimitSignals` must be zero: hosted calls remain provider-owned, continuation stops after its finite path, Realtime credentials are absent from events, provider setup does not resolve credentials, retrieved RAG content stays inert, and memory export redacts the fixture secret. These are behavior/bound gates; host-local timings are comparison evidence, not portable release thresholds.
49
+
50
+ | Resource | Default / hard |
51
+ | --- | --- |
52
+ | OpenAI continuation hops | 8 |
53
+ | Realtime audio events / bytes per second | 64 / 256 · 1 MiB / 8 MiB |
54
+ | RAG document bytes | 1 MiB / 8 MiB |
55
+ | RAG rerank input / time / active calls | 64 KiB / 256 KiB · 2 s / 10 s · 2 / 8 |
56
+ | RAG ingestion-status page | 50 / 200 |
57
+ | Memory retention batch | 500 / 5,000 |
58
+ | Memory export | 100 / 200 entries · 4 MiB / 32 MiB · 10 s / 60 s |
59
+ | Memory rebuild | 32 / 128 entries · 10 s / 60 s |
60
+
61
+ This task adds no package or runtime dependency: package/install delta is zero and the frozen graph remains 43 publishable manifests. Credentialed protocol checks are documented in the [0.0.15 protected live-canary matrix](release-and-install.md#015-protected-live-canary-matrix); they never run in this benchmark, `npm test`, or `sdk:ready`.
62
+
63
+ 2026-07-26 baseline: Node v24.18.0, Linux x64, 100 iterations/scenario, network=false, credentials=false.
64
+
65
+ | Scenario | ops/s | p95 ms | heap bytes | backpressure | resource limits |
66
+ | --- | ---: | ---: | ---: | ---: | ---: |
67
+ | OpenAI hosted continuation | 4,885 | 0.2924 | 15,475,920 | 0 | 0 |
68
+ | OpenAI Realtime envelope | 858 | 1.3130 | 12,092,232 | 0 | 0 |
69
+ | AI SDK v4 mapping | 27,580 | 0.0573 | 14,848,288 | 0 | 0 |
70
+ | Provider package metadata | 79,823 | 0.0288 | 16,077,080 | 0 | 0 |
71
+ | RAG lifecycle/reranking | 5,324 | 0.3586 | 13,894,576 | 0 | 0 |
72
+ | Memory lifecycle | 12,892 | 0.1608 | 13,923,936 | 0 | 0 |
73
+
74
+ These values are dated local comparison evidence, not portable thresholds.
75
+
9
76
  ## Release 0.0.12 frontend interoperability caps and evidence
10
77
 
11
78
  `@arnilo/prism-ag-ui` uses finite handler/projection limits, all defaults / hard: request 64 KiB / 1 MiB; input 128 / 1024 messages and 64 KiB / 1 MiB text; event 64 KiB / 1 MiB; error 8 KiB / 64 KiB; replay cursor 4 / 16 KiB; replay page 100 / 500; subscriber queue 128 / 4096; stream 10,000 / 100,000 events and 10 / 64 MiB; request wall time 120 seconds / 30 minutes. Tool arguments/results/progress, frontend tools, and mutable frontend state default to zero exposure; hosts may only add bounded safe projection.
@@ -146,6 +146,8 @@ Provider request policies can set `ProviderRequestOptions.cache` or the legacy `
146
146
  | Provider package | Cache kind | Explicit cache hints | Multi-turn reuse notes | Caveats |
147
147
  | --- | --- | --- | --- | --- |
148
148
  | `@arnilo/prism-provider-openai` | `openai_key` | Sends sanitized `prompt_cache_key`; `prompt_cache_retention: "24h"` only when the model declares `longRetention`. | Stable cache key + stable prefix can improve reuse. | Best-effort only; `"short"`/`"none"` omit retention. |
149
+ | `@arnilo/prism-provider-anthropic` | `cache_control` | Marks only selected Anthropic message anchors; `"long"` maps to documented `ttl: "1h"`. | Keep selected anchors stable. | Best-effort; never stamp every block. |
150
+ | `@arnilo/prism-provider-google` | none | Sends no Prism cache marker. | Host/model may have upstream behavior. | Gemini cache controls are not mapped in this package. |
149
151
  | `@arnilo/prism-provider-openrouter` | `cache_control` | Top-level automatic `cache_control` when enabled without breakpoints; otherwise markers only on caller-selected `cache.breakpoints`; `"long"` may add `ttl: "1h"`. Sticky `session_id` routing. | Breakpoint-stable / automatic prefixes can be reused by upstream providers. | Best-effort only; top-level automatic may exclude some backends from routing. |
150
152
  | `@arnilo/prism-provider-opencode-go` | route-specific | Sends sanitized `x-opencode-session`; Anthropic route applies selected `cache_control` breakpoints; OpenAI route sends none. | Session id + unchanged selected anchors can help route-native caches. | Best-effort and route-dependent. |
151
153
  | `@arnilo/prism-provider-zai` | `implicit` | No explicit cache payload; GLM context caching is automatic. | Resend unchanged prior history for implicit context-cache reuse. | Best-effort only; cache options do not force hits. |
@@ -154,11 +156,16 @@ Provider request policies can set `ProviderRequestOptions.cache` or the legacy `
154
156
  | `@arnilo/prism-provider-ai-sdk` | host-owned | No Prism cache payload; host `LanguageModelV4` owns upstream caching. | Host model/provider decides cache keys, breakpoints, and sticky routing. | Adapter maps `inputTokens.cacheRead`/`cacheWrite` from `finish.usage` only; does not invent cache fields. |
155
157
  | `@arnilo/prism-provider-alibaba` | implicit by default, optional `cache_control` | DashScope implicit prefix caching is automatic; opt-in `cache_control: {"type":"ephemeral"}` markers only on caller-selected `cache.breakpoints`, capped at 4. | Keep selected anchors and prior history stable; each cached prefix needs ≥1024 tokens and lives ~5 minutes upstream. | Best-effort and model-dependent; `cached_tokens`→read, `cache_creation_input_tokens`→write. |
156
158
  | `@arnilo/prism-provider-ollama` | `implicit` | No `cache_control`, `cacheKey`, `prompt_cache`, or `cacheRetention` payload; Ollama KV/prefix caching is automatic with no request knob. | Resend unchanged prior history for implicit KV reuse. | Best-effort only; Ollama reports no cached-token count, so `Usage.cacheReadTokens` stays `undefined`. |
159
+ | `@arnilo/prism-provider-azure` | none | No Prism cache mapping. | Endpoint/model-specific. | Host owns Azure cache policy. |
160
+ | `@arnilo/prism-provider-bedrock` | none | No Prism cache mapping. | Endpoint/model-specific. | Host owns Bedrock cache policy. |
161
+ | `@arnilo/prism-provider-vertex` | none | No Prism cache mapping. | Endpoint/model-specific. | Host owns Vertex cache policy. |
157
162
 
158
163
  Detailed first-party provider notes:
159
164
 
160
165
  - OpenAI Responses (`@arnilo/prism-provider-openai`): `kind: "openai_key"`. Sanitizes/clamps `prompt_cache_key` to 64 chars; `"long"` retention maps to `prompt_cache_retention: "24h"` only when the model declares `cache.longRetention`; `"short"`/`"none"` omit the field. GPT-5.6+ official docs use `prompt_cache_options` / breakpoints instead of retention — `listOpenAIModels` sets `longRetention: false` for those ids. `input_tokens_details.cached_tokens` maps to `Usage.cacheReadTokens`.
161
166
  - OpenAI-compatible Chat Completions adapter: minimal scope, sends no cache payload; see [OpenAI-compatible provider](providers/openai-compatible.md).
167
+ - Anthropic (`@arnilo/prism-provider-anthropic`): `kind: "cache_control"`; selected Anthropic message anchors receive `cache_control` and eligible long retention maps to `ttl: "1h"`. Cache read/create usage maps to normalized cache read/write tokens.
168
+ - Google (`@arnilo/prism-provider-google`): sends no Prism cache-control payload. Do not infer cache hits or cache token counts from absent Gemini fields.
162
169
  - OpenRouter (`@arnilo/prism-provider-openrouter`): `kind: "cache_control"`. Sanitizes/clamps `session_id`/`X-Session-Id` to 256 chars for sticky routing; with no breakpoints emits top-level automatic `cache_control: { type: "ephemeral" }`; with breakpoints applies Anthropic-style markers only to caller-selected locations (last content block of each selected message); `"long"` retention adds `ttl: "1h"` when the model allows it. `prompt_tokens_details.cached_tokens`/`cache_write_tokens` map to `Usage.cacheReadTokens`/`cacheWriteTokens`. Optional `listOpenRouterModels()` may populate `ModelConfig.cache`/`cost` from live pricing.
163
170
  - OpenCode Go (`@arnilo/prism-provider-opencode-go`): default base `https://opencode.ai/zen/go/v1`; `x-opencode-session` from `cacheKey ?? sessionId`, sanitized to 128 chars; the Anthropic route (MiniMax/Qwen) applies `cache_control` markers only to selected breakpoints (`"long"` → `ttl: "1h"`), the OpenAI route (Grok/GLM/Kimi/MiMo/DeepSeek) sends none and preserves `reasoning_content`. OpenAI route maps `prompt_tokens_details.cached_tokens`/`cache_write_tokens`; Anthropic route maps `cache_read_input_tokens`/`cache_creation_input_tokens`. Caller-gated `listOpenCodeGoModels` against official `GET /zen/go/v1/models`.
164
171
  - Z.AI (`@arnilo/prism-provider-zai`): `kind: "implicit"`. GLM context caching is automatic; sends no explicit cache payload regardless of cache options. `prompt_tokens_details.cached_tokens`/`cache_write_tokens` map to `Usage.cacheReadTokens`/`cacheWriteTokens`.
@@ -167,6 +174,7 @@ Detailed first-party provider notes:
167
174
  - AI SDK adapter (`@arnilo/prism-provider-ai-sdk`): **host-owned**. Sends no Prism cache payload; the supplied `LanguageModelV4` and its upstream provider own request caching. Maps AI SDK v4 `finish.usage.inputTokens.cacheRead`/`cacheWrite` to `Usage.cacheReadTokens`/`cacheWriteTokens`. No `list*Models()` export.
168
175
  - Alibaba Cloud (`@arnilo/prism-provider-alibaba`): implicit by default, optional `cache_control`. DashScope implicit prefix caching is automatic (no marker); explicit opt-in `cache_control: {"type":"ephemeral"}` markers apply only to selected breakpoints when `ModelConfig.cache.kind: "cache_control"` and the caller supplies breakpoints, capped at 4 (each prefix ≥1024 tokens, ~5 minute TTL). `prompt_tokens_details.cached_tokens`/`cache_creation_input_tokens` map to `Usage.cacheReadTokens`/`cacheWriteTokens`. Caller-gated `listAlibabaModels` against OpenAI-compatible `GET {base}/models`.
169
176
  - Ollama (`@arnilo/prism-provider-ollama`): `kind: "implicit"`. Ollama reuses its KV/prompt cache automatically; there is no request knob and no wire marker, so Prism never emits `cache_control`. Ollama reports no cached-token count, so `Usage.cacheReadTokens` is intentionally left `undefined` (not `0`). Caller-gated `listOllamaModels` against OpenAI-compatible `GET {base}/models`.
177
+ - Azure, Bedrock, and Vertex: their OpenAI-compatible packages intentionally emit no Prism cache fields. Endpoint/model-specific cache controls remain host-owned rather than guessed from another provider family.
170
178
 
171
179
  ### NeuralWatt cache-aware limiter
172
180