@arnilo/prism 0.0.14 → 0.0.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,16 @@
1
1
  # Changelog
2
2
 
3
+ ## [0.0.15] - 2026-07-26
4
+
5
+ ### Added
6
+
7
+ - Phase 10 provider, memory, and RAG parity: OpenAI hosted-tool attribution, bounded Responses continuation and Realtime seam; exact AI SDK V4 mapping; bounded RAG source lifecycle, document adapters, reranking, citation provenance, content trust, and ingestion status; memory export/rebuild with production-store conformance.
8
+
9
+ ### Changed
10
+
11
+ - Versioned all **43** publishable manifests, exact internal ranges, and lockfile entries to `0.0.15`; no package was added.
12
+ - Added network-free Phase 10 evidence: `scripts/benchmark-0.0.15.mjs`.
13
+
3
14
  ## [0.0.14] - 2026-07-26
4
15
 
5
16
  ### Added
@@ -28,8 +39,6 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
28
39
 
29
40
  All notable changes to this project will be documented in this file.
30
41
 
31
- ## [Unreleased]
32
-
33
42
  ## [0.0.12] - 2026-07-22
34
43
 
35
44
  ### Added
package/README.md CHANGED
@@ -34,10 +34,11 @@ packages. Prism defines contracts, not apps.
34
34
  - **Config, settings, security**: layered config merge, settings providers,
35
35
  credential resolvers, trust/permission policies, and secret redaction.
36
36
  - **CLI/RPC/server**: `prism --mode print|json|rpc`, `prism init`, optional framework-free authorized Web agent/workflow routes, and explicit MCP server exposure.
37
- - **Co-work contracts (0.0.14)**: conversation thread and artifact/review types, a
38
- deny-by-default device-adapter contract (`resolveDevicePolicy`/`assertDeviceAdmit`),
39
- and OAuth refresh/revoke helpers; services ship in `@arnilo/prism-server` and
40
- `@arnilo/prism-credentials-node`.
37
+ - **Ecosystem parity (0.0.15)**: OpenAI hosted-tool attribution, bounded Responses
38
+ continuation/Realtime, exact AI SDK V4 mapping, bounded RAG lifecycle/reranking/trust,
39
+ and consent-bound memory export/rebuild; provider, RAG, and memory packages remain optional.
40
+ - **Co-work contracts (0.0.14)**: conversation/artifact review types, deny-by-default device
41
+ contracts, and OAuth refresh/revoke helpers; services stay in optional packages.
41
42
 
42
43
  ## Install
43
44
 
@@ -14,6 +14,10 @@ export declare function resolveToolConcurrency(options: {
14
14
  }, config: {
15
15
  loop?: AgentLoopStrategy | AgentLoopOptions;
16
16
  }): number;
17
+ /** Calls the host must dispatch. Provider-hosted calls (`authority: "provider-hosted"`)
18
+ * were already executed server-side; the assistant response text carries their effect, so
19
+ * the host neither dispatches them nor appends a `tool_result`. */
20
+ export declare function dispatchableToolCalls(calls: readonly ToolCallContent[]): readonly ToolCallContent[];
17
21
  /** Dispatch tool calls with bounded concurrency; append transcript rows in call order. */
18
22
  export declare function dispatchToolCallsInOrder(calls: readonly ToolCallContent[], ctx: LoopContext): Promise<void>;
19
23
  export declare function resolveLoop(options: {
@@ -39,7 +39,8 @@ export const singleShotLoop = {
39
39
  ctx.emit({ type: "message_finished", sessionId: ctx.sessionId, runId: ctx.runId, message });
40
40
  }
41
41
  ctx.emit({ type: "turn_finished", sessionId: ctx.sessionId, runId: ctx.runId, turn });
42
- if (calls.length === 0 || toolRounds >= ctx.maxToolRounds) {
42
+ const dispatchable = dispatchableToolCalls(calls);
43
+ if (dispatchable.length === 0 || toolRounds >= ctx.maxToolRounds) {
43
44
  // Soft-interrupt / late steer: keep same run going when queue still has text.
44
45
  if (await ctx.applyPendingSteers?.()) {
45
46
  nextInput = [];
@@ -48,7 +49,7 @@ export const singleShotLoop = {
48
49
  break;
49
50
  }
50
51
  toolRounds += 1;
51
- await dispatchToolCallsInOrder(calls, ctx);
52
+ await dispatchToolCallsInOrder(dispatchable, ctx);
52
53
  nextInput = [];
53
54
  }
54
55
  return usage;
@@ -113,13 +114,19 @@ export function generateValidateReviseLoop(opts) {
113
114
  }
114
115
  ctx.emit({ type: "turn_finished", sessionId: ctx.sessionId, runId: ctx.runId, turn });
115
116
  if (opts.toolCalls === "bounded" && calls.length > 0) {
117
+ const dispatchable = dispatchableToolCalls(calls);
118
+ if (dispatchable.length === 0) {
119
+ // Only provider-hosted calls; no host tool to run. Continue without charging a round.
120
+ nextInput = [];
121
+ continue;
122
+ }
116
123
  if (toolRounds >= ctx.maxToolRounds) {
117
124
  const result = { ok: false, errors: [{ message: "maximum tool rounds exceeded" }], metadata: { reason: "tool_round_limit" } };
118
125
  ctx.emit({ type: "artifact_failed", sessionId: ctx.sessionId, runId: ctx.runId, turn, attempt: attempts + 1, result });
119
126
  return usage;
120
127
  }
121
128
  toolRounds += 1;
122
- await dispatchToolCallsInOrder(calls, { ...ctx, toolConcurrency: 1 });
129
+ await dispatchToolCallsInOrder(dispatchable, { ...ctx, toolConcurrency: 1 });
123
130
  nextInput = [];
124
131
  continue;
125
132
  }
@@ -195,6 +202,12 @@ export function resolveToolConcurrency(options, config) {
195
202
  return 1;
196
203
  return Math.floor(value);
197
204
  }
205
+ /** Calls the host must dispatch. Provider-hosted calls (`authority: "provider-hosted"`)
206
+ * were already executed server-side; the assistant response text carries their effect, so
207
+ * the host neither dispatches them nor appends a `tool_result`. */
208
+ export function dispatchableToolCalls(calls) {
209
+ return calls.filter((call) => call.authority !== "provider-hosted");
210
+ }
198
211
  /** Dispatch tool calls with bounded concurrency; append transcript rows in call order. */
199
212
  export async function dispatchToolCallsInOrder(calls, ctx) {
200
213
  if (calls.length === 0)
@@ -43,7 +43,11 @@ export interface ToolCallDeltaContent {
43
43
  readonly id?: string;
44
44
  readonly name?: string;
45
45
  readonly argumentsText?: string;
46
+ /** Who executes the call. `"provider-hosted"` = the provider runs it server-side;
47
+ * the host must NOT dispatch it or send a `tool_result`. Defaults to `"host"`. */
48
+ readonly authority?: ToolCallAuthority;
46
49
  }
50
+ export type ToolCallAuthority = "host" | "provider-hosted";
47
51
  export interface ToolCallContent {
48
52
  readonly type: "tool_call";
49
53
  readonly id: string;
@@ -51,6 +55,10 @@ export interface ToolCallContent {
51
55
  readonly arguments: JsonObject;
52
56
  /** Set when streamed arguments failed JSON parse; dispatch blocks without execute(). */
53
57
  readonly argumentsError?: ErrorInfo;
58
+ /** Who executes the call. `"provider-hosted"` = the provider already ran it
59
+ * server-side; the host must NOT dispatch it or append a `tool_result`. The
60
+ * assistant response text already incorporates the call's effect. */
61
+ readonly authority?: ToolCallAuthority;
54
62
  }
55
63
  export interface ToolResultContent {
56
64
  readonly type: "tool_result";
@@ -229,6 +237,11 @@ export interface ProviderRequestOptions {
229
237
  readonly extra?: JsonObject;
230
238
  /** Provider-neutral JSON-schema structured output request. Requires model `capabilities.structuredOutput`. */
231
239
  readonly structuredOutput?: StructuredOutputOptions;
240
+ /** Opaque provider continuation cursor (e.g. OpenAI `previous_response_id`). When set,
241
+ * the provider resumes from this cursor instead of re-sending full history. */
242
+ readonly continuation?: {
243
+ readonly cursor: string;
244
+ };
232
245
  }
233
246
  export interface ProviderRequest {
234
247
  readonly model: ModelConfig;
@@ -251,12 +264,17 @@ export type ProviderEvent = {
251
264
  readonly id?: string;
252
265
  readonly name?: string;
253
266
  readonly argumentsText?: string;
267
+ readonly authority?: ToolCallAuthority;
254
268
  } | {
255
269
  readonly type: "tool_call";
256
270
  readonly call: ToolCallContent;
257
271
  } | {
258
272
  readonly type: "usage";
259
273
  readonly usage: Usage;
274
+ } | {
275
+ readonly type: "continuation_required";
276
+ readonly cursor: string;
277
+ readonly reason?: string;
260
278
  } | {
261
279
  readonly type: "done";
262
280
  readonly usage?: Usage;
@@ -269,6 +287,64 @@ export interface AIProvider {
269
287
  generate(request: ProviderRequest): AsyncIterable<ProviderEvent>;
270
288
  }
271
289
  export type ProviderResolver = (model: ModelConfig) => AIProvider | undefined;
290
+ /** Realtime audio/session event. Realtime is a bidirectional session, not a request/response
291
+ * stream, so it is a separate neutral seam from `AIProvider.generate()`. Credentials are
292
+ * bound to the session handshake only and never appear in events. */
293
+ export type RealtimeEvent = {
294
+ readonly type: "session_started";
295
+ readonly sessionId?: string;
296
+ } | {
297
+ readonly type: "audio_delta";
298
+ readonly audio: Uint8Array;
299
+ } | {
300
+ readonly type: "transcript_delta";
301
+ readonly text: string;
302
+ readonly role: "user" | "assistant";
303
+ } | {
304
+ readonly type: "tool_call";
305
+ readonly call: ToolCallContent;
306
+ } | {
307
+ readonly type: "interrupted";
308
+ } | {
309
+ readonly type: "session_closed";
310
+ readonly reason?: string;
311
+ } | {
312
+ readonly type: "error";
313
+ readonly error: ErrorInfo;
314
+ };
315
+ /** Neutral bidirectional realtime session seam. The provider owns the transport
316
+ * (e.g. WebSocket); the host owns audio capture/playback and session lifecycle. */
317
+ export interface RealtimeSession {
318
+ readonly id: string;
319
+ readonly provider: string;
320
+ /** Send an audio chunk (PCM/Opus; provider-specific format set at creation). */
321
+ sendAudio(chunk: Uint8Array, options?: {
322
+ readonly signal?: AbortSignal;
323
+ }): Promise<void>;
324
+ /** Inbound events (audio out, transcripts, hosted tool calls, interruption, close, error). */
325
+ events(): AsyncIterable<RealtimeEvent>;
326
+ /** Request the provider stop the current response mid-stream. */
327
+ interrupt(options?: {
328
+ readonly signal?: AbortSignal;
329
+ }): Promise<void>;
330
+ /** Close the session and release the transport. Idempotent. */
331
+ close(reason?: string, options?: {
332
+ readonly signal?: AbortSignal;
333
+ }): Promise<void>;
334
+ }
335
+ /** Factory a provider exposes for realtime sessions; not part of `AIProvider`. */
336
+ export type RealtimeSessionFactory = (options: RealtimeSessionOptions) => RealtimeSession;
337
+ export interface RealtimeSessionOptions {
338
+ readonly model: ModelConfig;
339
+ readonly signal?: AbortSignal;
340
+ /** Provider-specific caps override; providers enforce finite defaults. */
341
+ readonly caps?: RealtimeCaps;
342
+ }
343
+ export interface RealtimeCaps {
344
+ readonly maxAudioEventsPerSecond?: number;
345
+ readonly maxBytesPerSecond?: number;
346
+ readonly maxWallMs?: number;
347
+ }
272
348
  export type InputAssemblyLayout = "legacy" | "cache_aware";
273
349
  export interface RunOptions {
274
350
  readonly signal?: AbortSignal;
package/dist/index.d.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  export type * from "./contracts.js";
2
- export type { RunLimitCounters, RunLimitName, SecureAgentOptions } from "./contracts.js";
2
+ export type { RealtimeCaps, RealtimeEvent, RealtimeSession, RealtimeSessionFactory, RealtimeSessionOptions, RunLimitCounters, RunLimitName, SecureAgentOptions, ToolCallAuthority, } from "./contracts.js";
3
3
  export { isSessionEntryKind, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SessionAppendConflictError, isSessionAppendConflict, AgentRunError, AgentRunStateError, assertSessionMetadataKey, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SESSION_SEARCH_UNSUPPORTED_CODE, SessionSearchUnsupportedError, isSessionSearchUnsupported, DEFAULT_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_LIMIT, DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES, DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES, HARD_MAX_SESSION_SEARCH_CURSOR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, HARD_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES, HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES, resolveSessionSearchQuery, DEFAULT_MAX_PENDING_STEERS, HARD_MAX_PENDING_STEERS, DEFAULT_MAX_PENDING_STEER_BYTES, HARD_MAX_PENDING_STEER_BYTES } from "./contracts.js";
4
4
  export { createAgent, createAgentSession, resumeAgentRun, resumeAgentRunStream } from "./agents.js";
5
5
  export { createBatchedRunLedger, isFlushableRunLedger, DEFAULT_LEDGER_BATCH_ENTRIES, HARD_LEDGER_BATCH_ENTRIES, DEFAULT_LEDGER_BATCH_BYTES, HARD_LEDGER_BATCH_BYTES, DEFAULT_LEDGER_BATCH_DELAY_MS, HARD_LEDGER_BATCH_DELAY_MS, } from "./run-ledger.js";
@@ -70,7 +70,7 @@ export { DEFAULT_DEVICE_MAX_CHUNK_BYTES, DEFAULT_DEVICE_MAX_CONCURRENT_SESSIONS,
70
70
  export type { DeviceAdapter, DeviceAdmitRequest, DeviceChunkResult, DeviceConformanceResult, DeviceKind, DevicePolicyErrorCode, DevicePolicyOptions, DeviceStreamLimits, ResolvedDevicePolicy, } from "./devices.js";
71
71
  export type { CreateMemorySessionStoreOptions, CreateSessionEntryOptions, MemorySessionSearchMode, SessionBranch, SessionBranchOptions, SessionContextSnapshot } from "./session-stores.js";
72
72
  export type { MockProviderOptions } from "./mock-provider.js";
73
- export { providerContentDelta, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, toolCallFromArgumentsText, } from "./provider-events.js";
73
+ export { providerContentDelta, providerContinuationRequired, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, toolCallFromArgumentsText, } from "./provider-events.js";
74
74
  export type { ProviderResolver } from "./contracts.js";
75
75
  export { createProviderRegistry, createProviderResolver } from "./providers.js";
76
76
  export type { ProviderRegistry, ProviderRegistryOptions } from "./providers.js";
@@ -95,5 +95,5 @@ export type { DispatchToolCallOptions, ToolArgumentValidationError, ToolArgument
95
95
  export type { DuplicateRegistrationOptions, DuplicateRegistrationPolicy } from "./registry-options.js";
96
96
  export { dispatchToolCallsInOrder, generateValidateReviseLoop, isAgentLoopOptions, resolveLoop, resolveToolConcurrency, singleShotLoop } from "./agent-loops.js";
97
97
  export declare const name = "prism";
98
- export declare const version = "0.0.14";
98
+ export declare const version = "0.0.15";
99
99
  export declare const description = "Agent harness for AI providers, agents, sessions, and tools.";
package/dist/index.js CHANGED
@@ -38,7 +38,7 @@ export { createMemorySessionStore, createSessionEntry, getSessionBranchEntries,
38
38
  export { CONVERSATION_METADATA_KEY, DEFAULT_MAX_CONVERSATION_CURSOR_BYTES, HARD_MAX_CONVERSATION_CURSOR_BYTES, ConversationError, conversationMarkerMetadata, conversationThreadFromRecord, decodeConversationReplayCursor, encodeConversationReplayCursor, } from "./conversations.js";
39
39
  export { ARTIFACT_CHECKPOINT_NAMESPACE, ArtifactError, artifactApprovalState, artifactCheckpointKey, } from "./artifacts.js";
40
40
  export { DEFAULT_DEVICE_MAX_CHUNK_BYTES, DEFAULT_DEVICE_MAX_CONCURRENT_SESSIONS, HARD_DEVICE_MAX_CHUNK_BYTES, HARD_DEVICE_MAX_CONCURRENT_SESSIONS, DevicePolicyError, acceptDeviceChunk, assertDeviceAdmit, redactDeviceTelemetry, resolveDevicePolicy, runDevicePolicyConformance, } from "./devices.js";
41
- export { providerContentDelta, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, toolCallFromArgumentsText, } from "./provider-events.js";
41
+ export { providerContentDelta, providerContinuationRequired, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, toolCallFromArgumentsText, } from "./provider-events.js";
42
42
  export { createProviderRegistry, createProviderResolver } from "./providers.js";
43
43
  export { createSecretRedactor, errorToErrorInfo, redactAgentEvent, redactMessage, redactProviderRequest, redactRunLedgerRecord, redactSecrets, redactSessionEntry } from "./redaction.js";
44
44
  export { assertPermission, assertTrusted, checkPermission, createStaticPermissionPolicy, createStaticTrustPolicy, denialToErrorInfo, isTrusted, PermissionDeniedError, TrustDeniedError } from "./security.js";
@@ -51,6 +51,6 @@ export { assertGuardrailsAllowed, GuardrailError, MAX_GUARDRAIL_CONCURRENCY, run
51
51
  export { createRunLimitTracker, DEFAULT_RUN_LIMITS, HARD_MAX_RUN_COST, HARD_RUN_LIMITS, RunLimitError, RunLimitTracker, resolveRunLimits } from "./run-limits.js";
52
52
  export { dispatchToolCallsInOrder, generateValidateReviseLoop, isAgentLoopOptions, resolveLoop, resolveToolConcurrency, singleShotLoop } from "./agent-loops.js";
53
53
  export const name = "prism";
54
- export const version = "0.0.14";
54
+ export const version = "0.0.15";
55
55
  export const description = "Agent harness for AI providers, agents, sessions, and tools.";
56
56
  //# sourceMappingURL=index.js.map
@@ -10,6 +10,7 @@ export declare function providerToolCallDelta(delta: {
10
10
  readonly argumentsText?: string;
11
11
  }): ProviderEvent;
12
12
  export declare function providerToolCallDeltaContent(delta: Omit<ToolCallDeltaContent, "type">): ToolCallDeltaContent;
13
+ export declare function providerContinuationRequired(cursor: string, reason?: string): ProviderEvent;
13
14
  export declare function reconstructToolCallDeltas(events: readonly ProviderEvent[]): readonly ToolCallContent[];
14
15
  export declare function providerUsage(usage: Usage): ProviderEvent;
15
16
  export declare function providerDone(usage?: Usage): ProviderEvent;
@@ -18,6 +18,9 @@ export function providerToolCallDelta(delta) {
18
18
  export function providerToolCallDeltaContent(delta) {
19
19
  return { type: "tool_call_delta", ...delta };
20
20
  }
21
+ export function providerContinuationRequired(cursor, reason) {
22
+ return { type: "continuation_required", cursor, reason };
23
+ }
21
24
  export function reconstructToolCallDeltas(events) {
22
25
  const partials = new Map();
23
26
  for (const event of events) {
@@ -134,8 +134,11 @@ Wire those values where they matter: provider adapters receive the resolved cred
134
134
  - Prism does not sandbox host tools, extensions, provider adapters, credential resolvers, or custom middleware. Use OS/container/process isolation when code is untrusted.
135
135
  - Redaction is exact known-secret replacement only. It is not arbitrary secret detection, entropy scanning, or DLP.
136
136
  - Known secrets must be passed into redactors before data is emitted or persisted. Redact again in host adapters if they transform records after Prism redaction.
137
+ - OpenAI Realtime sessions require a stable host owner identifier, use header-only credentials, and bind to the server `session.created` id. Treat returned audio/transcripts as untrusted; use a `SecretRedactor`, retain finite event/byte/wall caps, and close on disconnect or an identity/budget breach.
137
138
  - Tool `parameters` metadata is not validated by default. Add a `ToolValidator`, use `createToolParameterValidator()` with a schema adapter, or install `@arnilo/prism-tool-validator-json-schema` before side effects. Its untrusted-schema adapter rejects non-local refs, forbidden keys/cycles/non-finite values and bounds bytes/depth/properties/keywords/refs plus its LRU cache before Ajv compilation; do not raise caps above documented hard limits.
138
- - Treat embeddings as untrusted numeric input. `@arnilo/prism-memory` rejects empty, non-number, NaN, and infinite vectors before in-memory similarity or pgvector parameters; custom `Embedder`/`VectorStore` implementations must retain the same boundary. Memory entries carry consent/source/visibility; revoked, invisible, or (in strict mode) consent-less entries never enter prompts, events, exports, or telemetry, and `forget`/`applyRetention` are real bounded deletes.
139
+ - RAG `replaceSource()` only accepts a store with scoped `getBySource()` plus a real transaction; it stages bounded embeddings before mutation and otherwise fails closed. `deleteSource()` rechecks returned tenant/resource/corpus/source metadata. `createResourceDocumentLoader()` receives only a host-authorized `ResourceLoader`; `createWebFetchDocumentLoader()` never opens I/O and rejects local/private/IP-literal URLs before delegating to host-configured web-tools. HTML scripts/styles are stripped, PDF parsing has byte/page/time caps and rejects compressed PDFs.
140
+ - RAG retrieval always emits `trust: { untrusted: true, inert: true, injectionCapable: true }` plus attributable citation provenance. Context blocks repeat this metadata and never gain tool authority. Host `Reranker`s see redacted finite candidates, are hard-capped by bytes/time/concurrency, must return only a permutation of candidate IDs, and cannot overwrite provenance/trust. Ingestion status errors are redacted; status storage/listing stays exact-scope and capped.
141
+ - Treat embeddings as untrusted numeric input. `@arnilo/prism-memory` rejects empty, non-number, NaN, and infinite vectors before in-memory similarity, pgvector parameters, export, or rebuild; custom `Embedder`/`VectorStore` implementations must retain the same boundary. Memory entries carry consent/source/visibility; revoked/invisible entries never enter prompts, events, exports, or telemetry. `exportMemory()` additionally excludes consent-less legacy records regardless of recall mode and requires exact host identity equal to its tenant/resource/thread scope. Save rebuild cursors only in host-authorized storage; `rebuildIndex()` is one abortable capped page, never an implicit corpus job. `forget`/`applyRetention` are real bounded deletes.
139
142
  - Evaluation trace readers require exact supplied ownership plus session/run identity, reject cursor/identity drift, and redact before bounded scorer/judge input. Model-judge callbacks receive no credential resolver, tools, or workspace; keep live judges outside default CI and redact report artifacts.
140
143
  - Prism-generated session/run/tool/workflow/evaluation IDs use Node cryptographic UUIDs. Keep host-provided IDs authorization-scoped and validate them as untrusted identifiers; do not substitute timestamps or `Math.random()` for durable/security-relevant IDs.
141
144
  - MCP client tools from `@arnilo/prism-mcp` are untrusted remote servers. Stdio remains an explicit host executable. Streamable HTTP requires exact HTTPS origins, rejects credentials/fragments/redirects/private or mixed DNS, pins a validated address on every SDK request/reconnect, and bounds each response; plaintext is explicit loopback-only development mode. Discovery has finite page/tool/cursor/metadata/schema totals and commits atomically. Every result branch shares byte/depth/property bounds before core dispatch; supply a known-secret `SecretRedactor`, `PermissionPolicy`, and `ToolValidator` there. MCP server direction exposes only passed tools/commands/resources/prompts, requires per-operation `authorize`, and retains core gates. Sampling, roots, model/credential selection, and elicitation consent stay host-owned; URL elicitation is never opened automatically. Stateful web mode requires host `resolveAuthInfo` plus `resolveIdentity`, exact origin policy, and binds every POST/GET/DELETE/SSE request to one non-secret principal; mismatches return 404. Handler still needs TLS and edge rate limiting. See [MCP client/server exposure](mcp-tools.md).
package/docs/index.md CHANGED
@@ -19,14 +19,14 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
19
19
  - [Observability](observability.md): OTel GenAI agent/provider/tool hierarchy, host context parenting, bounded trace linkage, safe evaluation events, controlled metrics, and exporter isolation.
20
20
  - [Evaluations](evaluations.md): deterministic and bounded trace/model-judge/pairwise scoring, CI thresholds, OTel trace-reference linkage, coding/browser adversarial fixtures, and ID-only linkage to immutable owned run feedback.
21
21
  - [Runs and usage ledger](runs-and-usage.md): durable run/event/tool/usage persistence, optional bounded FIFO durability policies, session snapshot caching, and immutable run/trace feedback.
22
- - [Performance limits](performance.md): bounded evaluation traces/judges/reports, 0.0.12 frontend interoperability benchmark evidence/caps, 0.0.11 search/budget, 0.0.10 workspace-mode, and 0.0.9 coding/browser benchmark evidence, security scan/live-canary backstops, live subscriber queues, branch-read pagination expectations, JSONL/dev-store limits, and production sizing assumptions.
22
+ - [Performance limits](performance.md): 0.0.15 network-free provider/RAG/memory benchmark evidence and frozen caps, bounded evaluation traces/judges/reports, 0.0.12 frontend interoperability, 0.0.11 search/budget, 0.0.10 workspace-mode, 0.0.9 coding/browser, security scan/live-canary backstops, live subscriber queues, branch-read pagination expectations, JSONL/dev-store limits, and production sizing assumptions.
23
23
  - [Structured output](structured-output.md): the `Artifact*` seam plus provider-native `StructuredOutputOptions` / `structuredOutputMode` for capable models.
24
24
 
25
25
  ## Compaction/session memory
26
26
  - [Compaction and retry policies](compaction-and-retry.md): summarize branch history and retry transient provider failures with host-replaceable policies.
27
27
  - [LLM compaction package](compaction-llm.md): optional provider-backed strategy with finite summary/reserve/error caps, bounded redacted streaming retention, mandatory finite post-policy `model.parameters.maxTokens`, and `createCodingCompactionStrategy()` for coding handoff focus.
28
28
  - [Observational memory compaction package](compaction-observational-memory.md): optional source-backed memory with owned append callback, finite turn/call/argument/result/transcript/error worker limits, redacted provider-valid transcripts, fast compaction, recall, and status/view commands; worker model falls back to host-supplied `sessionModel`.
29
- - [Working and semantic memory](working-and-semantic-memory.md): optional `@arnilo/prism-memory` working-memory store, semantic recall, finite Embedder/VectorStore contracts, in-memory adapters, PostgreSQL/pgvector path, and consent/source/visibility lifecycle (grant/correct/forget/retention) enforced at injection.
29
+ - [Working and semantic memory](working-and-semantic-memory.md): optional `@arnilo/prism-memory` working-memory store, semantic recall, finite Embedder/VectorStore contracts, PostgreSQL/pgvector path, consent lifecycle, identity-bound redacted export, and resumable bounded rebuild.
30
30
  - [Session stores](session-stores.md): `SessionStore` contract, `SessionAppendOptions`, `SessionAppendConflictError`, branch handles, `readBranchPath`, optional bounded `searchSessions` / `SessionIndex` (memory linear|unsupported), and dev-vs-production branch reads — start here for session persistence.
31
31
  - [Conversations](conversations.md): durable user-scoped conversation threads (create/list/continue/branch/archive/export/delete) on session + event-ledger seams, thread-bound reconnectable replay, frozen caps, and legal-hold-aware deletion.
32
32
  - [Work artifacts and review](work-artifacts-and-review.md): durable artifact co-work review — authorized attach (MIME/hash/version, producer run, citations, preview metadata), revision compare, approve/reject with last-validated recovery, and authorized expiring delivery links; records persist as versioned checkpoints, never file bodies.
@@ -34,7 +34,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
34
34
  - [Database persistence](database-persistence.md): production persistence contracts, shared checksummed migration/full-shape catalog primitives (`@arnilo/prism/testing/persistence-schema`), conditional append, indexes, `readBranchPath`, reference relational schema, retention/legal-hold/quota lifecycle (`lifecycle`), and NoSQL mapping.
35
35
  - [SQLite persistence](sqlite-persistence.md): optional `better-sqlite3` adapter with session/run storage, checkpoints/leases, feedback, FTS `searchSessions` (migration-v4), and transactionally verified/backfilled migration metadata.
36
36
  - [PostgreSQL persistence](postgres-persistence.md): optional pooled `pg` adapter with session/run/checkpoint/lease/feedback storage, FTS `searchSessions` (migration-v4), advisory-locked checksummed/full-shape migrations, and opt-in live conformance.
37
- - [Migration guide](migration.md): **0.0.14** personal/work-agent conversations, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth connectors, browser checkpoints, device contracts, and Alibaba/Ollama providers; **0.0.13** enterprise identity/policy/router/work connectors, cloud providers, server deployment seams, and persistence schema v5; plus 0.0.12 AG-UI/ACP, 0.0.11 coding-harness fundamentals, 0.0.10 workspace modes, and 0.0.9 coding/browser surfaces.
37
+ - [Migration guide](migration.md): **0.0.15** OpenAI hosted tools/continuation/Realtime, exact AI SDK v4 matrix, RAG lifecycle/reranking/trust/status, and memory export/rebuild; **0.0.14** conversations, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth connectors, browser checkpoints, device contracts, and Alibaba/Ollama providers; plus prior release migrations.
38
38
  - [Node JSONL session store](node-jsonl-session-store.md): development-only JSONL file adapter for single-process Node hosts; no cross-process safety; `searchSessions` throws `SessionSearchUnsupportedError`.
39
39
  - [Persistence, credentials, and multimodality primitives](persistence-credentials-multimodality-primitives.md): Plan 056 inventory — session/run-ledger/persistence contracts, credential/OAuth seams, content/resource/model capabilities, package dependency matrix, conformance matrix, and threat model for production adapters.
40
40
 
@@ -42,24 +42,24 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
42
42
  - [Provider primitives](provider-primitives.md): shared bounded transport and OpenAI serialization helpers — migrated across first-party providers; native structured-output and observability contracts.
43
43
  - [Provider layer](provider-layer.md): register and resolve host-owned providers/models, choose replace-or-error duplicate policy, create provider events, stream/reconstruct tool-call deltas, use generic provider request options, and test with the mock provider; deprecated provider-level timeout/retry hints point to runtime abort/retry.
44
44
  - [Model registry](model-registry.md): register and resolve `ModelConfig` records with capabilities, limits, cost, cache support metadata, compat data, and duplicate policy.
45
- - [Provider caching](provider-caching.md): use `PromptCacheHints`, `PromptCacheBreakpoint`, `ModelCacheCapabilities`, cache-aware stable-prefix guidance, and shared cache diagnostics helpers; includes a per-provider explicit/implicit cache matrix for OpenAI, OpenRouter, OpenCode Go, Z.AI, Kimi, NeuralWatt, and the host-owned AI SDK adapter; cache hints are best-effort and cache keys are never secrets.
45
+ - [Provider caching](provider-caching.md): use `PromptCacheHints`, `PromptCacheBreakpoint`, `ModelCacheCapabilities`, cache-aware stable-prefix guidance, and shared cache diagnostics helpers; includes the complete per-provider explicit/implicit cache matrix plus no-Prism-cache entries (including Anthropic, Google, Alibaba, Ollama, cloud adapters, and host-owned AI SDK); cache hints are best-effort and cache keys are never secrets.
46
46
  - [Thinking and reasoning](thinking-and-reasoning.md): portable `ThinkingLevel` helpers (`applyThinkingLevel` / `thinkingCompatFor`) map per-turn effort into provider `compat` fields; model defaults stay on `ModelConfig.compat`; no second options tree.
47
47
  - [Use-case model selection](use-case-model-selection.md): bind `{ model?, provider?, thinkingLevel? }` for observational memory, LLM compaction, and other non-session LLM jobs with explicit session-model fallback via `resolveUseCaseModel`.
48
48
  - [Provider request policies](provider-request-policies.md): chain `ProviderRequestPolicy` hooks, use `createSessionCachePolicy`, and merge legacy/structured cache options safely.
49
- - [Provider packages](provider-packages.md): define explicit provider packages, model metadata, auth descriptors, request/cache policies, provider-owned header precedence, and the 0.0.12 provider-authorized OAuth matrix without package discovery or provider-specific core behavior; includes a first-party cache behavior summary and the **caller-gated on-demand model discovery** contract (`list*Models`, setup zero-fetch).
50
- - Phase 12 package workspaces: [`@arnilo/prism-provider-openai`](providers/openai.md), [`@arnilo/prism-provider-anthropic`](providers/anthropic.md) (native Messages, `cache_control`, thinking, caller-gated `listAnthropicModels`), [`@arnilo/prism-provider-google`](providers/google.md) (native Gemini `generateContent` SSE, caller-gated `listGoogleModels`), [`@arnilo/prism-provider-opencode-go`](providers/opencode-go.md) (official Go open models, dual-route Anthropic/OpenAI, caller-gated `listOpenCodeGoModels`, `reasoning_content`/thinking preserve), [`@arnilo/prism-provider-openrouter`](providers/openrouter.md) (app-controlled catalog, caller-gated `listOpenRouterModels`, `reasoning` merge/preserve, `cache_control` + sticky `session_id`), [`@arnilo/prism-provider-zai`](providers/zai.md) (official `thinking`/`reasoning_effort`/`tool_stream`, implicit cache, caller-gated `listZaiModels`), [`@arnilo/prism-provider-kimi`](providers/kimi.md), [`@arnilo/prism-provider-alibaba`](providers/alibaba.md) (Alibaba Cloud Model Studio / DashScope + Coding Plan, OpenAI-compatible, caller-gated `listAlibabaModels`, implicit + explicit `cache_control` caching, Qwen `enable_thinking`), [`@arnilo/prism-provider-ollama`](providers/ollama.md) (Ollama Cloud + local, OpenAI-compatible, caller-gated `listOllamaModels`, implicit-only caching, `reasoning_effort`), and [`@arnilo/prism-provider-neuralwatt`](providers/neuralwatt.md) with implicit vLLM prefix caching, reasoning controls (`reasoning_effort`/`thinking_token_budget`/`enable_thinking`/`preserve_thinking`/`clear_thinking`), reasoning preservation, OpenAI-style tool-call loop, quota, telemetry, and retry classification helpers.
49
+ - [Provider packages](provider-packages.md): define explicit provider packages, model metadata, auth descriptors, request/cache policies, provider-owned header precedence, the provider-authorized OAuth matrix, and the Phase 10 first-party compatibility matrix without package discovery or provider-specific core behavior; includes a cache behavior summary and **caller-gated on-demand model discovery** (`list*Models`, setup zero-fetch).
50
+ - Phase 12 package workspaces: [`@arnilo/prism-provider-openai`](providers/openai.md) (Responses hosted-tool attribution, bounded continuation, Realtime session seam), [`@arnilo/prism-provider-anthropic`](providers/anthropic.md) (native Messages, `cache_control`, thinking, caller-gated `listAnthropicModels`), [`@arnilo/prism-provider-google`](providers/google.md) (native Gemini `generateContent` SSE, caller-gated `listGoogleModels`), [`@arnilo/prism-provider-opencode-go`](providers/opencode-go.md) (official Go open models, dual-route Anthropic/OpenAI, caller-gated `listOpenCodeGoModels`, `reasoning_content`/thinking preserve), [`@arnilo/prism-provider-openrouter`](providers/openrouter.md) (app-controlled catalog, caller-gated `listOpenRouterModels`, `reasoning` merge/preserve, `cache_control` + sticky `session_id`), [`@arnilo/prism-provider-zai`](providers/zai.md) (official `thinking`/`reasoning_effort`/`tool_stream`, implicit cache, caller-gated `listZaiModels`), [`@arnilo/prism-provider-kimi`](providers/kimi.md), [`@arnilo/prism-provider-alibaba`](providers/alibaba.md) (Alibaba Cloud Model Studio / DashScope + Coding Plan, OpenAI-compatible, caller-gated `listAlibabaModels`, implicit + explicit `cache_control` caching, Qwen `enable_thinking`), [`@arnilo/prism-provider-ollama`](providers/ollama.md) (Ollama Cloud + local, OpenAI-compatible, caller-gated `listOllamaModels`, implicit-only caching, `reasoning_effort`), and [`@arnilo/prism-provider-neuralwatt`](providers/neuralwatt.md) with implicit vLLM prefix caching, reasoning controls (`reasoning_effort`/`thinking_token_budget`/`enable_thinking`/`preserve_thinking`/`clear_thinking`), reasoning preservation, OpenAI-style tool-call loop, quota, telemetry, and retry classification helpers.
51
51
  - Phase 8 enterprise cloud (workload identity; separate from consumer Anthropic/Google): [`@arnilo/prism-provider-azure`](providers/azure.md) (Entra / Foundry), [`@arnilo/prism-provider-bedrock`](providers/bedrock.md) (IAM/IRSA + region/PrivateLink), [`@arnilo/prism-provider-vertex`](providers/vertex.md) (ADC / Vertex OpenAPI).
52
- - Optional AI SDK adapter: [`@arnilo/prism-provider-ai-sdk`](providers/ai-sdk.md) maps host-owned `LanguageModelV4` models onto Prism `AIProvider` streams (specification v4; no Prism catalog; maps `finish.usage` cache read/write tokens; reasoning is host-model-owned).
52
+ - Optional AI SDK adapter: [`@arnilo/prism-provider-ai-sdk`](providers/ai-sdk.md) maps host-owned pinned `LanguageModelV4` models onto Prism `AIProvider` streams (offline-tested `@ai-sdk/provider` version matrix; no Prism catalog; maps metadata/tool authority/`finish.usage` cache tokens; reasoning is host-model-owned).
53
53
  - [OpenAI-compatible provider](providers/openai-compatible.md): optional provider subpath using native or injected `fetch` for Chat Completions streaming (`chatCompletionsUrl` / `authStyle` overrides for enterprise adapters).
54
54
 
55
55
  ## Input, prompt, and context assembly
56
56
  - [SDK customization guide](customization.md): map provider resolution, middleware, context, builders, injectors, loops, compaction, retry, stores, and skills to explicit host-wired APIs.
57
57
  - [Input and prompt assembly](input-and-prompt-assembly.md): render tiny prompt templates and turn common host input, history, attachments, explicit resources, summaries, and tool results into messages with replaceable builders, provider-input assembly, legacy default order, opt-in cache-aware ordering, and optional `contextBudget` eviction + omission reports. Audio/file/document `ContentBlock` types and capability checks are documented there.
58
- - [Multimodal content](multimodal-content.md): complete-request media resolution and aggregate bounds, DNS-classified/address-pinned URLs, SSRF/MIME policy, and `ModelCapabilities.input` tags.
58
+ - [Multimodal content](multimodal-content.md): complete-request media resolution and aggregate bounds, DNS-classified/address-pinned URLs, SSRF/MIME policy, `ModelCapabilities.input` tags, and first-party content-type mapping.
59
59
  - [System prompts](system-prompts.md): compose explicit user/package/app/run system prompt layers, auto-load the standard `AGENTS.md` (workspace) / `SYSTEM.md` prompt files via the Node `loadSystemPromptFiles` loader (trust-gated for `AGENTS.md`), and append `SYSTEM.md` → per-agent `AGENT.md` body → repo `AGENTS.md` layers from a discovered agent bundle via `resolveAgentBundle`.
60
60
  - [Instruction injection](instruction-injection.md): register package injectors that layer redacted instructions/context blocks without granting tools, permissions, or resource escapes.
61
61
  - [Context and skills](context-and-skills.md): resolve ordered context providers and keep context/skill selection host-owned; omitted declarative skills stay inactive by default, `toolNames` fail closed before provider turns, and strict skill registries prevent silent shadowing.
62
- - [Retrieval-augmented generation](rag.md): optional bounded text/Markdown chunking, Phase 7 vector indexing/retrieval, stable citations, and explicit inert context injection.
62
+ - [Retrieval-augmented generation](rag.md): optional bounded source lifecycle, document adapters, host reranking, ingestion status, attributable citations, and inert context injection.
63
63
 
64
64
  ## Tools
65
65
  - [Tools](tools.md): register host-owned active tools with replace-or-error duplicate policy, apply exact allow/deny filtering, dispatch normal or opt-in bounded artifact-loop calls, and optionally bound untrusted JSON Schema compilation.
@@ -84,7 +84,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
84
84
  ## Configuration/manifests
85
85
  - [Configuration and manifests](configuration-and-manifests.md): merge in-memory JSON config layers and validate data-only package manifests with prototype-pollution key rejection.
86
86
  - [Node filesystem config loader](node-filesystem-config.md): explicitly read caller-named JSON config files in Node hosts.
87
- - [Resource loading](resource-loading.md): decode text, JSON, binary, and manifest resources through caller-provided loaders with bounded byte limits.
87
+ - [Resource loading](resource-loading.md): decode text, JSON, binary, and manifests through caller-provided loaders; bridge host-authorized artifacts to bounded RAG document loading.
88
88
 
89
89
  ## Server/API
90
90
  - [Web-standard server handler](server.md): optional framework-free authorized direct/SSE agent, durable agent lifecycle, durable workflow routes, plus optional health/drain/rate-limit/replay/deployment-lease seams; explicit bounds and zero default exposure.
@@ -117,7 +117,8 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
117
117
  - `examples/`: compile-checked typed examples and runnable mock demos (SDK basics, provider registration, auth, tools, [`examples/ag-ui-server.ts`](../examples/ag-ui-server.ts), [`examples/enterprise-identity.ts`](../examples/enterprise-identity.ts), [`examples/enterprise-policy-audit.ts`](../examples/enterprise-policy-audit.ts), [`examples/enterprise-work-connectors.ts`](../examples/enterprise-work-connectors.ts), [`examples/conversation-durable-replay.ts`](../examples/conversation-durable-replay.ts), [`examples/artifact-review-delivery.ts`](../examples/artifact-review-delivery.ts), [`examples/server-deployment-seams.ts`](../examples/server-deployment-seams.ts), cache-aware prompt assembly, NeuralWatt agent run ([`examples/neuralwatt-agent-run.ts`](../examples/neuralwatt-agent-run.ts)), [`examples/coding-compaction.ts`](../examples/coding-compaction.ts), stores/branching, structured-output/artifact-loop, CLI, RPC, workflow orchestration).
118
118
 
119
119
  ## Release and install
120
- - [Release and install](release-and-install.md): current **43**-package graph (Phase 9 conversations/artifacts/co-work events, scoped OAuth connectors, browser checkpoints, device contracts, and `@arnilo/prism-provider-alibaba`/`@arnilo/prism-provider-ollama` ship at 0.0.14), install/tarball rules, pinned CodeQL/dependency/SBOM/license/secret/attestation gates, deterministic resumable publication, offline tests, protected live canaries, and sandbox-browser Docker/Playwright gates.
120
+ - [Release and install](release-and-install.md): current **0.0.15** 43-package graph (Phase 10 provider/AI-SDK/RAG/memory parity; no new package), exact-peer/install/tarball rules, deterministic resumable publication and publish dry-run, pinned supply-chain gates, offline tests, the 0.0.15 provider/AI-SDK/RAG/memory protected live-canary matrix, and sandbox-browser Docker/Playwright gates.
121
+ - [Review coverage (2026-07-26 Phase 10)](review-coverage-2026-07-26-phase-10.md): Plan 078 evidence freeze — OpenAI hosted tools/continuation/realtime, AI SDK version matrix, remaining provider metadata parity, RAG replaceSource/loaders/parsers/reranker/provenance/ingestion-status, memory export/rebuild/conformance, and 0.0.15 (43 → 43 manifests; no new package) release gates.
121
122
  - [Review coverage (2026-07-25 Phase 9)](review-coverage-2026-07-25-phase-9.md): Plan 077 evidence freeze — conversation service, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth, browser checkpoint composition, and deny-by-default device contracts for 0.0.14 (41 → 43 manifests; only the two provider packages are new).
122
123
  - [Review coverage (2026-07-23 Phase 8)](review-coverage-2026-07-23-phase-8.md): Plan 076 evidence freeze — enterprise identity/policy/router packages, Azure/Bedrock/Vertex adapters, server deployment seams, persistence lifecycle hooks, and M365/GWS work-connector bounds for 0.0.13.
123
124
  - [Review coverage (2026-07-22 Phase 7)](review-coverage-2026-07-22-phase-7.md): Plan 075 evidence freeze — AG-UI/ACP package boundary, streamed durable resume, bounded replay/projection, coding compaction preset, and provider-authorized OAuth policy for 0.0.12.
package/docs/migration.md CHANGED
@@ -7,6 +7,34 @@ Prism 0.0.6 preserves documented 0.0.3 agent construction except for two intenti
7
7
  1. **`session.run()` / `session.prompt()` return `AgentRunResult`** and `session.stream()` starts one owned run after subscribing. Callers that ignored the previous `Promise<void>` keep working; failed/aborted runs reject with `AgentRunError` (`.result` attached).
8
8
  2. **`AgentConfig.extensions` / `settings` / `credentials` are removed.** Wire extensions through `createExtensionKernel()`, read settings in the host, and pass credential resolvers to the provider edge.
9
9
 
10
+ ## 0.0.14 → 0.0.15 OpenAI hosted tools, continuation, and realtime (additive, pre-release)
11
+
12
+ `@arnilo/prism-provider-openai` now distinguishes server-executed calls with `authority: "provider-hosted"`; host dispatchers must not execute or reply to them. Incomplete Responses streams self-resume with an opaque `previous_response_id` cursor (at most 4 KiB, at most eight hops) and surface `continuation_required`; cap or duplicate-cursor failure now ends with a provider error instead of a silent partial response.
13
+
14
+ Realtime is opt-in through `createOpenAIRealtimeSession({ model, ownerId, apiKey, ... })`. Supply a stable host-owned `ownerId`; the session uses documented WebSocket headers, waits for `session.created`, exposes audio/transcript/interrupt/close events, and fails closed on disconnect, identity, audio/byte, or wall-time limits. It does not add a vendor package or automatic voice capture/playback.
15
+
16
+ ## 0.0.14 → 0.0.15 AI SDK adapter matrix (additive, pre-release)
17
+
18
+ `@arnilo/prism-provider-ai-sdk` now pins and verifies `@ai-sdk/provider@4.0.3` at setup rather than accepting any v4 minor. Upgrade the peer package to the documented matrix entry. An unlisted installed version fails with typed `AiSdkProviderError` code `unsupported_version`; add a tested matrix row before changing it.
19
+
20
+ Stream output now maps `response-metadata.id` to `message_start`, preserves `providerExecuted` tool authority as `"provider-hosted"`, and rejects unsupported output parts or `structuredOutput.strict` with `unsupported_mapping` rather than dropping them. Pass `redactor` when using the adapter directly; agents retain their existing active-redactor behavior.
21
+
22
+ ## 0.0.14 → 0.0.15 RAG source lifecycle and document adapters (additive, pre-release)
23
+
24
+ `@arnilo/prism-rag` now adds `replaceSource()`, `deleteSource()`, and `replaceDocument()` plus `DocumentLoader` / `Parser` seams. Existing `indexChunks()` behavior is unchanged; use `replaceSource()` when a source can shrink or must retain its old index if re-embedding fails.
25
+
26
+ Atomic replacement deliberately requires a scoped source-aware transaction (`getBySource()` + `transaction()`). The in-memory reference vector store supplies both; durable custom stores must add equivalent exact tenant/resource/corpus behavior before using replacement. Prism rejects a generic upsert-only store rather than offering a non-atomic fallback.
27
+
28
+ Reference parsers (`textParser`, `markdownParser`, `htmlParser`, `pdfParser`) are available from root and `@arnilo/prism-rag/parsers`; loaders are available from `@arnilo/prism-rag/loaders`. HTML removes script/style text. The PDF parser only accepts bounded uncompressed text PDFs (8 MiB / 256 pages / 30 s); install no new parser dependency—supply a host `Parser` for compressed or scanned files. `createWebFetchDocumentLoader()` accepts an existing `@arnilo/prism-web-tools` adapter and preserves its citation/untrusted metadata; it does not add a crawler.
29
+
30
+ RAG retrieval now optionally accepts host-owned `Reranker`; it receives redacted bounded hits and must return their exact IDs once each. Results add `trust`, `provenance`, and `retrievalRank`; context blocks now repeat untrusted/inert/injection-capable metadata. Add `statusStore` to indexing/replacement when hosts need per-source pending/indexed/failed/partial progress, use `listIngestionStatus()` for capped exact-scope pages, and supply durable storage if process restart durability matters. `createMemoryIngestionStatusStore()` is only a reference adapter.
31
+
32
+ ## 0.0.14 → 0.0.15 memory export and rebuild (additive, pre-release)
33
+
34
+ `@arnilo/prism-memory` adds `exportMemory({ identity, cursor?, ... })` and `rebuildIndex({ cursor?, ... })`. Export is not a generic admin dump: provide the exact host-verified tenant/resource/thread identity used to construct `createMemory()`. It excludes revoked, invisible, and consent-less legacy entries, redacts each returned record, and caps one page at 100 entries / 4 MiB / 10 seconds by default (200 / 32 MiB / 60 seconds hard).
35
+
36
+ `rebuildIndex()` re-embeds one 32-record page by default (128 hard), validates existing and new finite vectors, and returns `nextCursor`; persist that cursor in host-owned authorized state and call again to resume after an abort/restart. Neither API scans a corpus or starts a background worker. They require a semantic `VectorStore.listByThread()` implementation; `applyRetention()` now also requires `countByThread()` for bounded oldest-first deletion. The shipped in-memory adapter and PostgreSQL/pgvector adapter conform. `@arnilo/prism-session-store-sqlite` remains a session/run persistence package, not a semantic-vector adapter.
37
+
10
38
  ## 0.0.13 → 0.0.14 personal/work-agent conversations, co-work review, and channel/device gates (additive, pre-release)
11
39
 
12
40
  Release **0.0.14** is strictly additive: every surface extends a shipped package and reuses the AG-UI adapter shipped in 0.0.12. The only new packages are two optional provider adapters (41 → 43 manifests): `@arnilo/prism-provider-alibaba` and `@arnilo/prism-provider-ollama`, both enrolled via the `@arnilo/prism-providers` family. No permission broadening — channel/device/co-work features cannot widen consent, memory, network, file, browser, connector, or tool permissions (roadmap gate 8). See [Phase 9 evidence](review-coverage-2026-07-25-phase-9.md).
@@ -213,7 +241,7 @@ Phase 4 adds optional `@arnilo/prism-evals` for deterministic scorers/datasets/e
213
241
 
214
242
  Phase 5 adds `prism init <dir>` to the existing CLI. It scaffolds a tiny TypeScript project with one selected provider and an offline mock test. Optional `--with-workflows` / `--with-evals` flags add only those packages; storage and telemetry stay opt-in elsewhere.
215
243
 
216
- Phase 6 adds optional `@arnilo/prism-provider-ai-sdk` for AI SDK `LanguageModelV4` interoperability. Install it with `@ai-sdk/provider@^4`, through `@arnilo/prism-providers`, or through `@arnilo/prism-all`; it is not a core dependency.
244
+ Phase 6 adds optional `@arnilo/prism-provider-ai-sdk` for AI SDK `LanguageModelV4` interoperability. For 0.0.15 install its exact supported peer `@ai-sdk/provider@4.0.3` (not `^4`); an unlisted version fails at setup. Install the adapter directly, through `@arnilo/prism-providers`, or through `@arnilo/prism-all`; it is not a core dependency.
217
245
 
218
246
  Phase 7 adds optional `@arnilo/prism-memory` for schema/template-backed working memory and embedding-based semantic recall. Install it directly or through `@arnilo/prism-all`; in-memory adapters are default, and PostgreSQL/pgvector is opt-in. It is not a core dependency.
219
247
 
@@ -49,11 +49,13 @@ Known `ModelCapabilities.input` tags are exported as `MODEL_INPUT_CAPABILITIES`:
49
49
 
50
50
  | Tag | Block type | First-party mapping (declared capability required) |
51
51
  | --- | --- | --- |
52
- | `text` | `text` (default) | All providers |
53
- | `image` | `image` | OpenAI Responses, OpenRouter, OpenCode Go Anthropic route, Kimi, NeuralWatt |
54
- | `audio` | `audio` | OpenAI Responses (`input_audio`) |
55
- | `file` | `file` | OpenAI Responses (`input_file`); Anthropic routes map PDF only |
56
- | `document` | `document` | OpenAI Responses (`input_file`); OpenCode Go Anthropic route; Kimi |
52
+ | `text` | `text` (default) | All first-party providers; Azure/Bedrock/Vertex use their host-selected OpenAI-compatible endpoint/model. |
53
+ | `image` | `image` | OpenAI Responses; Anthropic; Google; Kimi; Z.AI; OpenRouter; OpenCode Go OpenAI route; Alibaba; Ollama; NeuralWatt. Enterprise OpenAI-compatible packages require the host model/endpoint to declare and accept image input. |
54
+ | `audio` | `audio` | OpenAI Responses (`input_audio`) and Google `generateContent` inline data. OpenAI Realtime instead receives `RealtimeSession.sendAudio()` chunks, not an `audio` `ContentBlock`. |
55
+ | `file` | `file` | OpenAI Responses (`input_file`); Anthropic/Kimi/OpenCode Go Anthropic route accept PDF file/document forms; Google maps inline file data. |
56
+ | `document` | `document` | OpenAI Responses (`input_file`); Anthropic/Kimi/OpenCode Go Anthropic route map PDF; Google maps inline document data. |
57
+
58
+ The AI SDK adapter maps declared user text/image/audio/file/document blocks (and assistant text/image/file/document) to AI SDK file parts; `resourceUri` remains host-resolved before `doStream`. Its output `file`, `reasoning-file`, and `source` parts are deliberately rejected as `unsupported_mapping`, not converted to trusted Prism content. Provider capability metadata is the gate—this matrix never upgrades a model that does not declare the matching input tag.
57
59
 
58
60
  ## Outputs / response / events
59
61
 
@@ -136,6 +138,7 @@ try {
136
138
  - Local filesystem paths should use trust policies such as `createPathTrustPolicy()` before exposing URIs to loaders.
137
139
  - Provider upload/create/delete lifecycles are provider-package-local. `@arnilo/prism-provider-openai` inlines files under 4 MiB as `data:<mediaType>;base64,...` `file_data`, otherwise uses a bounded per-run upload cache and best-effort `DELETE /v1/files` cleanup after each stream.
138
140
  - Shared wire helpers live in `@arnilo/prism/providers/media` (`resolveProviderMediaMessages`, `serializeOpenAIResponsesInputFile`, `serializePdfDocumentWireBlock`, `createBoundedUploadCache`). OpenAI Responses, Kimi, and OpenCode Go Anthropic routes resolve their complete media collection once before serialization or upload.
141
+ - OpenAI Realtime audio is a bidirectional `RealtimeSession` stream, not a `ContentBlock`: provide host-captured `Uint8Array` chunks with `sendAudio()` and consume untrusted `audio_delta` / transcript events. It has a fixed 256 events/s, 1 MiB/s, and 600 s default ceiling.
139
142
 
140
143
  ## Security and performance notes
141
144
 
@@ -6,6 +6,40 @@ Evaluation defaults are finite: 100 trace rows × 20 pages and 4 MiB aggregate t
6
6
 
7
7
  This page states Prism runtime limits that keep slow consumers and long sessions from becoming unbounded memory or latency problems.
8
8
 
9
+ ## Release 0.0.15 provider, RAG, and memory evidence
10
+
11
+ Run `node scripts/benchmark-0.0.15.mjs`; `PRISM_BENCH_ITERATIONS` accepts 10–100,000 (default 100). Schema/bounds test: `node --test scripts/benchmark-0.0.15.test.mjs`. Default mode is network-free: fake Responses SSE/WebSocket transports, a fake AI SDK v4 model, zero-fetch provider-package registration, hash embeddings, in-memory RAG replacement/reranking/retrieval/status, and in-memory memory retention/export/rebuild.
12
+
13
+ Scenarios: `openai-hosted-continuation`, `openai-realtime-envelope`, `ai-sdk-v4-stream-mapping`, `provider-package-metadata`, `rag-parse-replace-rerank-retrieve`, and `memory-retention-export-rebuild`.
14
+
15
+ Every row reports throughput, p50/p95 latency, heap, disk, queue/backpressure, and safety signals. `resourceLimitSignals` must be zero: hosted calls remain provider-owned, continuation stops after its finite path, Realtime credentials are absent from events, provider setup does not resolve credentials, retrieved RAG content stays inert, and memory export redacts the fixture secret. These are behavior/bound gates; host-local timings are comparison evidence, not portable release thresholds.
16
+
17
+ | Resource | Default / hard |
18
+ | --- | --- |
19
+ | OpenAI continuation hops | 8 |
20
+ | Realtime audio events / bytes per second | 64 / 256 · 1 MiB / 8 MiB |
21
+ | RAG document bytes | 1 MiB / 8 MiB |
22
+ | RAG rerank input / time / active calls | 64 KiB / 256 KiB · 2 s / 10 s · 2 / 8 |
23
+ | RAG ingestion-status page | 50 / 200 |
24
+ | Memory retention batch | 500 / 5,000 |
25
+ | Memory export | 100 / 200 entries · 4 MiB / 32 MiB · 10 s / 60 s |
26
+ | Memory rebuild | 32 / 128 entries · 10 s / 60 s |
27
+
28
+ This task adds no package or runtime dependency: package/install delta is zero and the frozen graph remains 43 publishable manifests. Credentialed protocol checks are documented in the [0.0.15 protected live-canary matrix](release-and-install.md#015-protected-live-canary-matrix); they never run in this benchmark, `npm test`, or `sdk:ready`.
29
+
30
+ 2026-07-26 baseline: Node v24.18.0, Linux x64, 100 iterations/scenario, network=false, credentials=false.
31
+
32
+ | Scenario | ops/s | p95 ms | heap bytes | backpressure | resource limits |
33
+ | --- | ---: | ---: | ---: | ---: | ---: |
34
+ | OpenAI hosted continuation | 4,885 | 0.2924 | 15,475,920 | 0 | 0 |
35
+ | OpenAI Realtime envelope | 858 | 1.3130 | 12,092,232 | 0 | 0 |
36
+ | AI SDK v4 mapping | 27,580 | 0.0573 | 14,848,288 | 0 | 0 |
37
+ | Provider package metadata | 79,823 | 0.0288 | 16,077,080 | 0 | 0 |
38
+ | RAG lifecycle/reranking | 5,324 | 0.3586 | 13,894,576 | 0 | 0 |
39
+ | Memory lifecycle | 12,892 | 0.1608 | 13,923,936 | 0 | 0 |
40
+
41
+ These values are dated local comparison evidence, not portable thresholds.
42
+
9
43
  ## Release 0.0.12 frontend interoperability caps and evidence
10
44
 
11
45
  `@arnilo/prism-ag-ui` uses finite handler/projection limits, all defaults / hard: request 64 KiB / 1 MiB; input 128 / 1024 messages and 64 KiB / 1 MiB text; event 64 KiB / 1 MiB; error 8 KiB / 64 KiB; replay cursor 4 / 16 KiB; replay page 100 / 500; subscriber queue 128 / 4096; stream 10,000 / 100,000 events and 10 / 64 MiB; request wall time 120 seconds / 30 minutes. Tool arguments/results/progress, frontend tools, and mutable frontend state default to zero exposure; hosts may only add bounded safe projection.
@@ -146,6 +146,8 @@ Provider request policies can set `ProviderRequestOptions.cache` or the legacy `
146
146
  | Provider package | Cache kind | Explicit cache hints | Multi-turn reuse notes | Caveats |
147
147
  | --- | --- | --- | --- | --- |
148
148
  | `@arnilo/prism-provider-openai` | `openai_key` | Sends sanitized `prompt_cache_key`; `prompt_cache_retention: "24h"` only when the model declares `longRetention`. | Stable cache key + stable prefix can improve reuse. | Best-effort only; `"short"`/`"none"` omit retention. |
149
+ | `@arnilo/prism-provider-anthropic` | `cache_control` | Marks only selected Anthropic message anchors; `"long"` maps to documented `ttl: "1h"`. | Keep selected anchors stable. | Best-effort; never stamp every block. |
150
+ | `@arnilo/prism-provider-google` | none | Sends no Prism cache marker. | Host/model may have upstream behavior. | Gemini cache controls are not mapped in this package. |
149
151
  | `@arnilo/prism-provider-openrouter` | `cache_control` | Top-level automatic `cache_control` when enabled without breakpoints; otherwise markers only on caller-selected `cache.breakpoints`; `"long"` may add `ttl: "1h"`. Sticky `session_id` routing. | Breakpoint-stable / automatic prefixes can be reused by upstream providers. | Best-effort only; top-level automatic may exclude some backends from routing. |
150
152
  | `@arnilo/prism-provider-opencode-go` | route-specific | Sends sanitized `x-opencode-session`; Anthropic route applies selected `cache_control` breakpoints; OpenAI route sends none. | Session id + unchanged selected anchors can help route-native caches. | Best-effort and route-dependent. |
151
153
  | `@arnilo/prism-provider-zai` | `implicit` | No explicit cache payload; GLM context caching is automatic. | Resend unchanged prior history for implicit context-cache reuse. | Best-effort only; cache options do not force hits. |
@@ -154,11 +156,16 @@ Provider request policies can set `ProviderRequestOptions.cache` or the legacy `
154
156
  | `@arnilo/prism-provider-ai-sdk` | host-owned | No Prism cache payload; host `LanguageModelV4` owns upstream caching. | Host model/provider decides cache keys, breakpoints, and sticky routing. | Adapter maps `inputTokens.cacheRead`/`cacheWrite` from `finish.usage` only; does not invent cache fields. |
155
157
  | `@arnilo/prism-provider-alibaba` | implicit by default, optional `cache_control` | DashScope implicit prefix caching is automatic; opt-in `cache_control: {"type":"ephemeral"}` markers only on caller-selected `cache.breakpoints`, capped at 4. | Keep selected anchors and prior history stable; each cached prefix needs ≥1024 tokens and lives ~5 minutes upstream. | Best-effort and model-dependent; `cached_tokens`→read, `cache_creation_input_tokens`→write. |
156
158
  | `@arnilo/prism-provider-ollama` | `implicit` | No `cache_control`, `cacheKey`, `prompt_cache`, or `cacheRetention` payload; Ollama KV/prefix caching is automatic with no request knob. | Resend unchanged prior history for implicit KV reuse. | Best-effort only; Ollama reports no cached-token count, so `Usage.cacheReadTokens` stays `undefined`. |
159
+ | `@arnilo/prism-provider-azure` | none | No Prism cache mapping. | Endpoint/model-specific. | Host owns Azure cache policy. |
160
+ | `@arnilo/prism-provider-bedrock` | none | No Prism cache mapping. | Endpoint/model-specific. | Host owns Bedrock cache policy. |
161
+ | `@arnilo/prism-provider-vertex` | none | No Prism cache mapping. | Endpoint/model-specific. | Host owns Vertex cache policy. |
157
162
 
158
163
  Detailed first-party provider notes:
159
164
 
160
165
  - OpenAI Responses (`@arnilo/prism-provider-openai`): `kind: "openai_key"`. Sanitizes/clamps `prompt_cache_key` to 64 chars; `"long"` retention maps to `prompt_cache_retention: "24h"` only when the model declares `cache.longRetention`; `"short"`/`"none"` omit the field. GPT-5.6+ official docs use `prompt_cache_options` / breakpoints instead of retention — `listOpenAIModels` sets `longRetention: false` for those ids. `input_tokens_details.cached_tokens` maps to `Usage.cacheReadTokens`.
161
166
  - OpenAI-compatible Chat Completions adapter: minimal scope, sends no cache payload; see [OpenAI-compatible provider](providers/openai-compatible.md).
167
+ - Anthropic (`@arnilo/prism-provider-anthropic`): `kind: "cache_control"`; selected Anthropic message anchors receive `cache_control` and eligible long retention maps to `ttl: "1h"`. Cache read/create usage maps to normalized cache read/write tokens.
168
+ - Google (`@arnilo/prism-provider-google`): sends no Prism cache-control payload. Do not infer cache hits or cache token counts from absent Gemini fields.
162
169
  - OpenRouter (`@arnilo/prism-provider-openrouter`): `kind: "cache_control"`. Sanitizes/clamps `session_id`/`X-Session-Id` to 256 chars for sticky routing; with no breakpoints emits top-level automatic `cache_control: { type: "ephemeral" }`; with breakpoints applies Anthropic-style markers only to caller-selected locations (last content block of each selected message); `"long"` retention adds `ttl: "1h"` when the model allows it. `prompt_tokens_details.cached_tokens`/`cache_write_tokens` map to `Usage.cacheReadTokens`/`cacheWriteTokens`. Optional `listOpenRouterModels()` may populate `ModelConfig.cache`/`cost` from live pricing.
163
170
  - OpenCode Go (`@arnilo/prism-provider-opencode-go`): default base `https://opencode.ai/zen/go/v1`; `x-opencode-session` from `cacheKey ?? sessionId`, sanitized to 128 chars; the Anthropic route (MiniMax/Qwen) applies `cache_control` markers only to selected breakpoints (`"long"` → `ttl: "1h"`), the OpenAI route (Grok/GLM/Kimi/MiMo/DeepSeek) sends none and preserves `reasoning_content`. OpenAI route maps `prompt_tokens_details.cached_tokens`/`cache_write_tokens`; Anthropic route maps `cache_read_input_tokens`/`cache_creation_input_tokens`. Caller-gated `listOpenCodeGoModels` against official `GET /zen/go/v1/models`.
164
171
  - Z.AI (`@arnilo/prism-provider-zai`): `kind: "implicit"`. GLM context caching is automatic; sends no explicit cache payload regardless of cache options. `prompt_tokens_details.cached_tokens`/`cache_write_tokens` map to `Usage.cacheReadTokens`/`cacheWriteTokens`.
@@ -167,6 +174,7 @@ Detailed first-party provider notes:
167
174
  - AI SDK adapter (`@arnilo/prism-provider-ai-sdk`): **host-owned**. Sends no Prism cache payload; the supplied `LanguageModelV4` and its upstream provider own request caching. Maps AI SDK v4 `finish.usage.inputTokens.cacheRead`/`cacheWrite` to `Usage.cacheReadTokens`/`cacheWriteTokens`. No `list*Models()` export.
168
175
  - Alibaba Cloud (`@arnilo/prism-provider-alibaba`): implicit by default, optional `cache_control`. DashScope implicit prefix caching is automatic (no marker); explicit opt-in `cache_control: {"type":"ephemeral"}` markers apply only to selected breakpoints when `ModelConfig.cache.kind: "cache_control"` and the caller supplies breakpoints, capped at 4 (each prefix ≥1024 tokens, ~5 minute TTL). `prompt_tokens_details.cached_tokens`/`cache_creation_input_tokens` map to `Usage.cacheReadTokens`/`cacheWriteTokens`. Caller-gated `listAlibabaModels` against OpenAI-compatible `GET {base}/models`.
169
176
  - Ollama (`@arnilo/prism-provider-ollama`): `kind: "implicit"`. Ollama reuses its KV/prompt cache automatically; there is no request knob and no wire marker, so Prism never emits `cache_control`. Ollama reports no cached-token count, so `Usage.cacheReadTokens` is intentionally left `undefined` (not `0`). Caller-gated `listOllamaModels` against OpenAI-compatible `GET {base}/models`.
177
+ - Azure, Bedrock, and Vertex: their OpenAI-compatible packages intentionally emit no Prism cache fields. Endpoint/model-specific cache controls remain host-owned rather than guessed from another provider family.
170
178
 
171
179
  ### NeuralWatt cache-aware limiter
172
180