@arnilo/prism 0.0.14 → 0.0.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -2
- package/README.md +5 -4
- package/dist/agent-loops.d.ts +4 -0
- package/dist/agent-loops.js +16 -3
- package/dist/contracts.d.ts +76 -0
- package/dist/index.d.ts +3 -3
- package/dist/index.js +2 -2
- package/dist/provider-events.d.ts +1 -0
- package/dist/provider-events.js +3 -0
- package/docs/host-security.md +4 -1
- package/docs/index.md +12 -11
- package/docs/migration.md +29 -1
- package/docs/multimodal-content.md +8 -5
- package/docs/performance.md +34 -0
- package/docs/provider-caching.md +8 -0
- package/docs/provider-conformance.md +29 -5
- package/docs/provider-packages.md +22 -1
- package/docs/providers/ai-sdk.md +23 -7
- package/docs/providers/openai.md +22 -3
- package/docs/rag.md +41 -12
- package/docs/release-and-install.md +62 -14
- package/docs/resource-loading.md +3 -0
- package/docs/review-coverage-2026-07-26-phase-10.md +132 -0
- package/docs/working-and-semantic-memory.md +22 -4
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,16 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [0.0.15] - 2026-07-26
|
|
4
|
+
|
|
5
|
+
### Added
|
|
6
|
+
|
|
7
|
+
- Phase 10 provider, memory, and RAG parity: OpenAI hosted-tool attribution, bounded Responses continuation and Realtime seam; exact AI SDK V4 mapping; bounded RAG source lifecycle, document adapters, reranking, citation provenance, content trust, and ingestion status; memory export/rebuild with production-store conformance.
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- Versioned all **43** publishable manifests, exact internal ranges, and lockfile entries to `0.0.15`; no package was added.
|
|
12
|
+
- Added network-free Phase 10 evidence: `scripts/benchmark-0.0.15.mjs`.
|
|
13
|
+
|
|
3
14
|
## [0.0.14] - 2026-07-26
|
|
4
15
|
|
|
5
16
|
### Added
|
|
@@ -28,8 +39,6 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
28
39
|
|
|
29
40
|
All notable changes to this project will be documented in this file.
|
|
30
41
|
|
|
31
|
-
## [Unreleased]
|
|
32
|
-
|
|
33
42
|
## [0.0.12] - 2026-07-22
|
|
34
43
|
|
|
35
44
|
### Added
|
package/README.md
CHANGED
|
@@ -34,10 +34,11 @@ packages. Prism defines contracts, not apps.
|
|
|
34
34
|
- **Config, settings, security**: layered config merge, settings providers,
|
|
35
35
|
credential resolvers, trust/permission policies, and secret redaction.
|
|
36
36
|
- **CLI/RPC/server**: `prism --mode print|json|rpc`, `prism init`, optional framework-free authorized Web agent/workflow routes, and explicit MCP server exposure.
|
|
37
|
-
- **
|
|
38
|
-
|
|
39
|
-
and
|
|
40
|
-
|
|
37
|
+
- **Ecosystem parity (0.0.15)**: OpenAI hosted-tool attribution, bounded Responses
|
|
38
|
+
continuation/Realtime, exact AI SDK V4 mapping, bounded RAG lifecycle/reranking/trust,
|
|
39
|
+
and consent-bound memory export/rebuild; provider, RAG, and memory packages remain optional.
|
|
40
|
+
- **Co-work contracts (0.0.14)**: conversation/artifact review types, deny-by-default device
|
|
41
|
+
contracts, and OAuth refresh/revoke helpers; services stay in optional packages.
|
|
41
42
|
|
|
42
43
|
## Install
|
|
43
44
|
|
package/dist/agent-loops.d.ts
CHANGED
|
@@ -14,6 +14,10 @@ export declare function resolveToolConcurrency(options: {
|
|
|
14
14
|
}, config: {
|
|
15
15
|
loop?: AgentLoopStrategy | AgentLoopOptions;
|
|
16
16
|
}): number;
|
|
17
|
+
/** Calls the host must dispatch. Provider-hosted calls (`authority: "provider-hosted"`)
|
|
18
|
+
* were already executed server-side; the assistant response text carries their effect, so
|
|
19
|
+
* the host neither dispatches them nor appends a `tool_result`. */
|
|
20
|
+
export declare function dispatchableToolCalls(calls: readonly ToolCallContent[]): readonly ToolCallContent[];
|
|
17
21
|
/** Dispatch tool calls with bounded concurrency; append transcript rows in call order. */
|
|
18
22
|
export declare function dispatchToolCallsInOrder(calls: readonly ToolCallContent[], ctx: LoopContext): Promise<void>;
|
|
19
23
|
export declare function resolveLoop(options: {
|
package/dist/agent-loops.js
CHANGED
|
@@ -39,7 +39,8 @@ export const singleShotLoop = {
|
|
|
39
39
|
ctx.emit({ type: "message_finished", sessionId: ctx.sessionId, runId: ctx.runId, message });
|
|
40
40
|
}
|
|
41
41
|
ctx.emit({ type: "turn_finished", sessionId: ctx.sessionId, runId: ctx.runId, turn });
|
|
42
|
-
|
|
42
|
+
const dispatchable = dispatchableToolCalls(calls);
|
|
43
|
+
if (dispatchable.length === 0 || toolRounds >= ctx.maxToolRounds) {
|
|
43
44
|
// Soft-interrupt / late steer: keep same run going when queue still has text.
|
|
44
45
|
if (await ctx.applyPendingSteers?.()) {
|
|
45
46
|
nextInput = [];
|
|
@@ -48,7 +49,7 @@ export const singleShotLoop = {
|
|
|
48
49
|
break;
|
|
49
50
|
}
|
|
50
51
|
toolRounds += 1;
|
|
51
|
-
await dispatchToolCallsInOrder(
|
|
52
|
+
await dispatchToolCallsInOrder(dispatchable, ctx);
|
|
52
53
|
nextInput = [];
|
|
53
54
|
}
|
|
54
55
|
return usage;
|
|
@@ -113,13 +114,19 @@ export function generateValidateReviseLoop(opts) {
|
|
|
113
114
|
}
|
|
114
115
|
ctx.emit({ type: "turn_finished", sessionId: ctx.sessionId, runId: ctx.runId, turn });
|
|
115
116
|
if (opts.toolCalls === "bounded" && calls.length > 0) {
|
|
117
|
+
const dispatchable = dispatchableToolCalls(calls);
|
|
118
|
+
if (dispatchable.length === 0) {
|
|
119
|
+
// Only provider-hosted calls; no host tool to run. Continue without charging a round.
|
|
120
|
+
nextInput = [];
|
|
121
|
+
continue;
|
|
122
|
+
}
|
|
116
123
|
if (toolRounds >= ctx.maxToolRounds) {
|
|
117
124
|
const result = { ok: false, errors: [{ message: "maximum tool rounds exceeded" }], metadata: { reason: "tool_round_limit" } };
|
|
118
125
|
ctx.emit({ type: "artifact_failed", sessionId: ctx.sessionId, runId: ctx.runId, turn, attempt: attempts + 1, result });
|
|
119
126
|
return usage;
|
|
120
127
|
}
|
|
121
128
|
toolRounds += 1;
|
|
122
|
-
await dispatchToolCallsInOrder(
|
|
129
|
+
await dispatchToolCallsInOrder(dispatchable, { ...ctx, toolConcurrency: 1 });
|
|
123
130
|
nextInput = [];
|
|
124
131
|
continue;
|
|
125
132
|
}
|
|
@@ -195,6 +202,12 @@ export function resolveToolConcurrency(options, config) {
|
|
|
195
202
|
return 1;
|
|
196
203
|
return Math.floor(value);
|
|
197
204
|
}
|
|
205
|
+
/** Calls the host must dispatch. Provider-hosted calls (`authority: "provider-hosted"`)
|
|
206
|
+
* were already executed server-side; the assistant response text carries their effect, so
|
|
207
|
+
* the host neither dispatches them nor appends a `tool_result`. */
|
|
208
|
+
export function dispatchableToolCalls(calls) {
|
|
209
|
+
return calls.filter((call) => call.authority !== "provider-hosted");
|
|
210
|
+
}
|
|
198
211
|
/** Dispatch tool calls with bounded concurrency; append transcript rows in call order. */
|
|
199
212
|
export async function dispatchToolCallsInOrder(calls, ctx) {
|
|
200
213
|
if (calls.length === 0)
|
package/dist/contracts.d.ts
CHANGED
|
@@ -43,7 +43,11 @@ export interface ToolCallDeltaContent {
|
|
|
43
43
|
readonly id?: string;
|
|
44
44
|
readonly name?: string;
|
|
45
45
|
readonly argumentsText?: string;
|
|
46
|
+
/** Who executes the call. `"provider-hosted"` = the provider runs it server-side;
|
|
47
|
+
* the host must NOT dispatch it or send a `tool_result`. Defaults to `"host"`. */
|
|
48
|
+
readonly authority?: ToolCallAuthority;
|
|
46
49
|
}
|
|
50
|
+
export type ToolCallAuthority = "host" | "provider-hosted";
|
|
47
51
|
export interface ToolCallContent {
|
|
48
52
|
readonly type: "tool_call";
|
|
49
53
|
readonly id: string;
|
|
@@ -51,6 +55,10 @@ export interface ToolCallContent {
|
|
|
51
55
|
readonly arguments: JsonObject;
|
|
52
56
|
/** Set when streamed arguments failed JSON parse; dispatch blocks without execute(). */
|
|
53
57
|
readonly argumentsError?: ErrorInfo;
|
|
58
|
+
/** Who executes the call. `"provider-hosted"` = the provider already ran it
|
|
59
|
+
* server-side; the host must NOT dispatch it or append a `tool_result`. The
|
|
60
|
+
* assistant response text already incorporates the call's effect. */
|
|
61
|
+
readonly authority?: ToolCallAuthority;
|
|
54
62
|
}
|
|
55
63
|
export interface ToolResultContent {
|
|
56
64
|
readonly type: "tool_result";
|
|
@@ -229,6 +237,11 @@ export interface ProviderRequestOptions {
|
|
|
229
237
|
readonly extra?: JsonObject;
|
|
230
238
|
/** Provider-neutral JSON-schema structured output request. Requires model `capabilities.structuredOutput`. */
|
|
231
239
|
readonly structuredOutput?: StructuredOutputOptions;
|
|
240
|
+
/** Opaque provider continuation cursor (e.g. OpenAI `previous_response_id`). When set,
|
|
241
|
+
* the provider resumes from this cursor instead of re-sending full history. */
|
|
242
|
+
readonly continuation?: {
|
|
243
|
+
readonly cursor: string;
|
|
244
|
+
};
|
|
232
245
|
}
|
|
233
246
|
export interface ProviderRequest {
|
|
234
247
|
readonly model: ModelConfig;
|
|
@@ -251,12 +264,17 @@ export type ProviderEvent = {
|
|
|
251
264
|
readonly id?: string;
|
|
252
265
|
readonly name?: string;
|
|
253
266
|
readonly argumentsText?: string;
|
|
267
|
+
readonly authority?: ToolCallAuthority;
|
|
254
268
|
} | {
|
|
255
269
|
readonly type: "tool_call";
|
|
256
270
|
readonly call: ToolCallContent;
|
|
257
271
|
} | {
|
|
258
272
|
readonly type: "usage";
|
|
259
273
|
readonly usage: Usage;
|
|
274
|
+
} | {
|
|
275
|
+
readonly type: "continuation_required";
|
|
276
|
+
readonly cursor: string;
|
|
277
|
+
readonly reason?: string;
|
|
260
278
|
} | {
|
|
261
279
|
readonly type: "done";
|
|
262
280
|
readonly usage?: Usage;
|
|
@@ -269,6 +287,64 @@ export interface AIProvider {
|
|
|
269
287
|
generate(request: ProviderRequest): AsyncIterable<ProviderEvent>;
|
|
270
288
|
}
|
|
271
289
|
export type ProviderResolver = (model: ModelConfig) => AIProvider | undefined;
|
|
290
|
+
/** Realtime audio/session event. Realtime is a bidirectional session, not a request/response
|
|
291
|
+
* stream, so it is a separate neutral seam from `AIProvider.generate()`. Credentials are
|
|
292
|
+
* bound to the session handshake only and never appear in events. */
|
|
293
|
+
export type RealtimeEvent = {
|
|
294
|
+
readonly type: "session_started";
|
|
295
|
+
readonly sessionId?: string;
|
|
296
|
+
} | {
|
|
297
|
+
readonly type: "audio_delta";
|
|
298
|
+
readonly audio: Uint8Array;
|
|
299
|
+
} | {
|
|
300
|
+
readonly type: "transcript_delta";
|
|
301
|
+
readonly text: string;
|
|
302
|
+
readonly role: "user" | "assistant";
|
|
303
|
+
} | {
|
|
304
|
+
readonly type: "tool_call";
|
|
305
|
+
readonly call: ToolCallContent;
|
|
306
|
+
} | {
|
|
307
|
+
readonly type: "interrupted";
|
|
308
|
+
} | {
|
|
309
|
+
readonly type: "session_closed";
|
|
310
|
+
readonly reason?: string;
|
|
311
|
+
} | {
|
|
312
|
+
readonly type: "error";
|
|
313
|
+
readonly error: ErrorInfo;
|
|
314
|
+
};
|
|
315
|
+
/** Neutral bidirectional realtime session seam. The provider owns the transport
|
|
316
|
+
* (e.g. WebSocket); the host owns audio capture/playback and session lifecycle. */
|
|
317
|
+
export interface RealtimeSession {
|
|
318
|
+
readonly id: string;
|
|
319
|
+
readonly provider: string;
|
|
320
|
+
/** Send an audio chunk (PCM/Opus; provider-specific format set at creation). */
|
|
321
|
+
sendAudio(chunk: Uint8Array, options?: {
|
|
322
|
+
readonly signal?: AbortSignal;
|
|
323
|
+
}): Promise<void>;
|
|
324
|
+
/** Inbound events (audio out, transcripts, hosted tool calls, interruption, close, error). */
|
|
325
|
+
events(): AsyncIterable<RealtimeEvent>;
|
|
326
|
+
/** Request the provider stop the current response mid-stream. */
|
|
327
|
+
interrupt(options?: {
|
|
328
|
+
readonly signal?: AbortSignal;
|
|
329
|
+
}): Promise<void>;
|
|
330
|
+
/** Close the session and release the transport. Idempotent. */
|
|
331
|
+
close(reason?: string, options?: {
|
|
332
|
+
readonly signal?: AbortSignal;
|
|
333
|
+
}): Promise<void>;
|
|
334
|
+
}
|
|
335
|
+
/** Factory a provider exposes for realtime sessions; not part of `AIProvider`. */
|
|
336
|
+
export type RealtimeSessionFactory = (options: RealtimeSessionOptions) => RealtimeSession;
|
|
337
|
+
export interface RealtimeSessionOptions {
|
|
338
|
+
readonly model: ModelConfig;
|
|
339
|
+
readonly signal?: AbortSignal;
|
|
340
|
+
/** Provider-specific caps override; providers enforce finite defaults. */
|
|
341
|
+
readonly caps?: RealtimeCaps;
|
|
342
|
+
}
|
|
343
|
+
export interface RealtimeCaps {
|
|
344
|
+
readonly maxAudioEventsPerSecond?: number;
|
|
345
|
+
readonly maxBytesPerSecond?: number;
|
|
346
|
+
readonly maxWallMs?: number;
|
|
347
|
+
}
|
|
272
348
|
export type InputAssemblyLayout = "legacy" | "cache_aware";
|
|
273
349
|
export interface RunOptions {
|
|
274
350
|
readonly signal?: AbortSignal;
|
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
export type * from "./contracts.js";
|
|
2
|
-
export type { RunLimitCounters, RunLimitName, SecureAgentOptions } from "./contracts.js";
|
|
2
|
+
export type { RealtimeCaps, RealtimeEvent, RealtimeSession, RealtimeSessionFactory, RealtimeSessionOptions, RunLimitCounters, RunLimitName, SecureAgentOptions, ToolCallAuthority, } from "./contracts.js";
|
|
3
3
|
export { isSessionEntryKind, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SessionAppendConflictError, isSessionAppendConflict, AgentRunError, AgentRunStateError, assertSessionMetadataKey, SESSION_SEARCH_WORKSPACE_METADATA_KEY, SESSION_SEARCH_UNSUPPORTED_CODE, SessionSearchUnsupportedError, isSessionSearchUnsupported, DEFAULT_SESSION_SEARCH_LIMIT, HARD_MAX_SESSION_SEARCH_LIMIT, DEFAULT_MAX_SESSION_SEARCH_QUERY_BYTES, HARD_MAX_SESSION_SEARCH_QUERY_BYTES, DEFAULT_MAX_SESSION_SEARCH_SNIPPET_BYTES, HARD_MAX_SESSION_SEARCH_SNIPPET_BYTES, DEFAULT_MAX_SESSION_SEARCH_CURSOR_BYTES, HARD_MAX_SESSION_SEARCH_CURSOR_BYTES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_SESSIONS, HARD_MAX_SESSION_SEARCH_LINEAR_SESSIONS, DEFAULT_MAX_SESSION_SEARCH_LINEAR_ENTRIES, HARD_MAX_SESSION_SEARCH_LINEAR_ENTRIES, DEFAULT_MAX_SESSION_SEARCH_LINEAR_BYTES, HARD_MAX_SESSION_SEARCH_LINEAR_BYTES, DEFAULT_MAX_SESSION_SEARCH_FTS_CANDIDATES, HARD_MAX_SESSION_SEARCH_FTS_CANDIDATES, resolveSessionSearchQuery, DEFAULT_MAX_PENDING_STEERS, HARD_MAX_PENDING_STEERS, DEFAULT_MAX_PENDING_STEER_BYTES, HARD_MAX_PENDING_STEER_BYTES } from "./contracts.js";
|
|
4
4
|
export { createAgent, createAgentSession, resumeAgentRun, resumeAgentRunStream } from "./agents.js";
|
|
5
5
|
export { createBatchedRunLedger, isFlushableRunLedger, DEFAULT_LEDGER_BATCH_ENTRIES, HARD_LEDGER_BATCH_ENTRIES, DEFAULT_LEDGER_BATCH_BYTES, HARD_LEDGER_BATCH_BYTES, DEFAULT_LEDGER_BATCH_DELAY_MS, HARD_LEDGER_BATCH_DELAY_MS, } from "./run-ledger.js";
|
|
@@ -70,7 +70,7 @@ export { DEFAULT_DEVICE_MAX_CHUNK_BYTES, DEFAULT_DEVICE_MAX_CONCURRENT_SESSIONS,
|
|
|
70
70
|
export type { DeviceAdapter, DeviceAdmitRequest, DeviceChunkResult, DeviceConformanceResult, DeviceKind, DevicePolicyErrorCode, DevicePolicyOptions, DeviceStreamLimits, ResolvedDevicePolicy, } from "./devices.js";
|
|
71
71
|
export type { CreateMemorySessionStoreOptions, CreateSessionEntryOptions, MemorySessionSearchMode, SessionBranch, SessionBranchOptions, SessionContextSnapshot } from "./session-stores.js";
|
|
72
72
|
export type { MockProviderOptions } from "./mock-provider.js";
|
|
73
|
-
export { providerContentDelta, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, toolCallFromArgumentsText, } from "./provider-events.js";
|
|
73
|
+
export { providerContentDelta, providerContinuationRequired, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, toolCallFromArgumentsText, } from "./provider-events.js";
|
|
74
74
|
export type { ProviderResolver } from "./contracts.js";
|
|
75
75
|
export { createProviderRegistry, createProviderResolver } from "./providers.js";
|
|
76
76
|
export type { ProviderRegistry, ProviderRegistryOptions } from "./providers.js";
|
|
@@ -95,5 +95,5 @@ export type { DispatchToolCallOptions, ToolArgumentValidationError, ToolArgument
|
|
|
95
95
|
export type { DuplicateRegistrationOptions, DuplicateRegistrationPolicy } from "./registry-options.js";
|
|
96
96
|
export { dispatchToolCallsInOrder, generateValidateReviseLoop, isAgentLoopOptions, resolveLoop, resolveToolConcurrency, singleShotLoop } from "./agent-loops.js";
|
|
97
97
|
export declare const name = "prism";
|
|
98
|
-
export declare const version = "0.0.
|
|
98
|
+
export declare const version = "0.0.15";
|
|
99
99
|
export declare const description = "Agent harness for AI providers, agents, sessions, and tools.";
|
package/dist/index.js
CHANGED
|
@@ -38,7 +38,7 @@ export { createMemorySessionStore, createSessionEntry, getSessionBranchEntries,
|
|
|
38
38
|
export { CONVERSATION_METADATA_KEY, DEFAULT_MAX_CONVERSATION_CURSOR_BYTES, HARD_MAX_CONVERSATION_CURSOR_BYTES, ConversationError, conversationMarkerMetadata, conversationThreadFromRecord, decodeConversationReplayCursor, encodeConversationReplayCursor, } from "./conversations.js";
|
|
39
39
|
export { ARTIFACT_CHECKPOINT_NAMESPACE, ArtifactError, artifactApprovalState, artifactCheckpointKey, } from "./artifacts.js";
|
|
40
40
|
export { DEFAULT_DEVICE_MAX_CHUNK_BYTES, DEFAULT_DEVICE_MAX_CONCURRENT_SESSIONS, HARD_DEVICE_MAX_CHUNK_BYTES, HARD_DEVICE_MAX_CONCURRENT_SESSIONS, DevicePolicyError, acceptDeviceChunk, assertDeviceAdmit, redactDeviceTelemetry, resolveDevicePolicy, runDevicePolicyConformance, } from "./devices.js";
|
|
41
|
-
export { providerContentDelta, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, toolCallFromArgumentsText, } from "./provider-events.js";
|
|
41
|
+
export { providerContentDelta, providerContinuationRequired, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, toolCallFromArgumentsText, } from "./provider-events.js";
|
|
42
42
|
export { createProviderRegistry, createProviderResolver } from "./providers.js";
|
|
43
43
|
export { createSecretRedactor, errorToErrorInfo, redactAgentEvent, redactMessage, redactProviderRequest, redactRunLedgerRecord, redactSecrets, redactSessionEntry } from "./redaction.js";
|
|
44
44
|
export { assertPermission, assertTrusted, checkPermission, createStaticPermissionPolicy, createStaticTrustPolicy, denialToErrorInfo, isTrusted, PermissionDeniedError, TrustDeniedError } from "./security.js";
|
|
@@ -51,6 +51,6 @@ export { assertGuardrailsAllowed, GuardrailError, MAX_GUARDRAIL_CONCURRENCY, run
|
|
|
51
51
|
export { createRunLimitTracker, DEFAULT_RUN_LIMITS, HARD_MAX_RUN_COST, HARD_RUN_LIMITS, RunLimitError, RunLimitTracker, resolveRunLimits } from "./run-limits.js";
|
|
52
52
|
export { dispatchToolCallsInOrder, generateValidateReviseLoop, isAgentLoopOptions, resolveLoop, resolveToolConcurrency, singleShotLoop } from "./agent-loops.js";
|
|
53
53
|
export const name = "prism";
|
|
54
|
-
export const version = "0.0.
|
|
54
|
+
export const version = "0.0.15";
|
|
55
55
|
export const description = "Agent harness for AI providers, agents, sessions, and tools.";
|
|
56
56
|
//# sourceMappingURL=index.js.map
|
|
@@ -10,6 +10,7 @@ export declare function providerToolCallDelta(delta: {
|
|
|
10
10
|
readonly argumentsText?: string;
|
|
11
11
|
}): ProviderEvent;
|
|
12
12
|
export declare function providerToolCallDeltaContent(delta: Omit<ToolCallDeltaContent, "type">): ToolCallDeltaContent;
|
|
13
|
+
export declare function providerContinuationRequired(cursor: string, reason?: string): ProviderEvent;
|
|
13
14
|
export declare function reconstructToolCallDeltas(events: readonly ProviderEvent[]): readonly ToolCallContent[];
|
|
14
15
|
export declare function providerUsage(usage: Usage): ProviderEvent;
|
|
15
16
|
export declare function providerDone(usage?: Usage): ProviderEvent;
|
package/dist/provider-events.js
CHANGED
|
@@ -18,6 +18,9 @@ export function providerToolCallDelta(delta) {
|
|
|
18
18
|
export function providerToolCallDeltaContent(delta) {
|
|
19
19
|
return { type: "tool_call_delta", ...delta };
|
|
20
20
|
}
|
|
21
|
+
export function providerContinuationRequired(cursor, reason) {
|
|
22
|
+
return { type: "continuation_required", cursor, reason };
|
|
23
|
+
}
|
|
21
24
|
export function reconstructToolCallDeltas(events) {
|
|
22
25
|
const partials = new Map();
|
|
23
26
|
for (const event of events) {
|
package/docs/host-security.md
CHANGED
|
@@ -134,8 +134,11 @@ Wire those values where they matter: provider adapters receive the resolved cred
|
|
|
134
134
|
- Prism does not sandbox host tools, extensions, provider adapters, credential resolvers, or custom middleware. Use OS/container/process isolation when code is untrusted.
|
|
135
135
|
- Redaction is exact known-secret replacement only. It is not arbitrary secret detection, entropy scanning, or DLP.
|
|
136
136
|
- Known secrets must be passed into redactors before data is emitted or persisted. Redact again in host adapters if they transform records after Prism redaction.
|
|
137
|
+
- OpenAI Realtime sessions require a stable host owner identifier, use header-only credentials, and bind to the server `session.created` id. Treat returned audio/transcripts as untrusted; use a `SecretRedactor`, retain finite event/byte/wall caps, and close on disconnect or an identity/budget breach.
|
|
137
138
|
- Tool `parameters` metadata is not validated by default. Add a `ToolValidator`, use `createToolParameterValidator()` with a schema adapter, or install `@arnilo/prism-tool-validator-json-schema` before side effects. Its untrusted-schema adapter rejects non-local refs, forbidden keys/cycles/non-finite values and bounds bytes/depth/properties/keywords/refs plus its LRU cache before Ajv compilation; do not raise caps above documented hard limits.
|
|
138
|
-
-
|
|
139
|
+
- RAG `replaceSource()` only accepts a store with scoped `getBySource()` plus a real transaction; it stages bounded embeddings before mutation and otherwise fails closed. `deleteSource()` rechecks returned tenant/resource/corpus/source metadata. `createResourceDocumentLoader()` receives only a host-authorized `ResourceLoader`; `createWebFetchDocumentLoader()` never opens I/O and rejects local/private/IP-literal URLs before delegating to host-configured web-tools. HTML scripts/styles are stripped, PDF parsing has byte/page/time caps and rejects compressed PDFs.
|
|
140
|
+
- RAG retrieval always emits `trust: { untrusted: true, inert: true, injectionCapable: true }` plus attributable citation provenance. Context blocks repeat this metadata and never gain tool authority. Host `Reranker`s see redacted finite candidates, are hard-capped by bytes/time/concurrency, must return only a permutation of candidate IDs, and cannot overwrite provenance/trust. Ingestion status errors are redacted; status storage/listing stays exact-scope and capped.
|
|
141
|
+
- Treat embeddings as untrusted numeric input. `@arnilo/prism-memory` rejects empty, non-number, NaN, and infinite vectors before in-memory similarity, pgvector parameters, export, or rebuild; custom `Embedder`/`VectorStore` implementations must retain the same boundary. Memory entries carry consent/source/visibility; revoked/invisible entries never enter prompts, events, exports, or telemetry. `exportMemory()` additionally excludes consent-less legacy records regardless of recall mode and requires exact host identity equal to its tenant/resource/thread scope. Save rebuild cursors only in host-authorized storage; `rebuildIndex()` is one abortable capped page, never an implicit corpus job. `forget`/`applyRetention` are real bounded deletes.
|
|
139
142
|
- Evaluation trace readers require exact supplied ownership plus session/run identity, reject cursor/identity drift, and redact before bounded scorer/judge input. Model-judge callbacks receive no credential resolver, tools, or workspace; keep live judges outside default CI and redact report artifacts.
|
|
140
143
|
- Prism-generated session/run/tool/workflow/evaluation IDs use Node cryptographic UUIDs. Keep host-provided IDs authorization-scoped and validate them as untrusted identifiers; do not substitute timestamps or `Math.random()` for durable/security-relevant IDs.
|
|
141
144
|
- MCP client tools from `@arnilo/prism-mcp` are untrusted remote servers. Stdio remains an explicit host executable. Streamable HTTP requires exact HTTPS origins, rejects credentials/fragments/redirects/private or mixed DNS, pins a validated address on every SDK request/reconnect, and bounds each response; plaintext is explicit loopback-only development mode. Discovery has finite page/tool/cursor/metadata/schema totals and commits atomically. Every result branch shares byte/depth/property bounds before core dispatch; supply a known-secret `SecretRedactor`, `PermissionPolicy`, and `ToolValidator` there. MCP server direction exposes only passed tools/commands/resources/prompts, requires per-operation `authorize`, and retains core gates. Sampling, roots, model/credential selection, and elicitation consent stay host-owned; URL elicitation is never opened automatically. Stateful web mode requires host `resolveAuthInfo` plus `resolveIdentity`, exact origin policy, and binds every POST/GET/DELETE/SSE request to one non-secret principal; mismatches return 404. Handler still needs TLS and edge rate limiting. See [MCP client/server exposure](mcp-tools.md).
|
package/docs/index.md
CHANGED
|
@@ -19,14 +19,14 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
19
19
|
- [Observability](observability.md): OTel GenAI agent/provider/tool hierarchy, host context parenting, bounded trace linkage, safe evaluation events, controlled metrics, and exporter isolation.
|
|
20
20
|
- [Evaluations](evaluations.md): deterministic and bounded trace/model-judge/pairwise scoring, CI thresholds, OTel trace-reference linkage, coding/browser adversarial fixtures, and ID-only linkage to immutable owned run feedback.
|
|
21
21
|
- [Runs and usage ledger](runs-and-usage.md): durable run/event/tool/usage persistence, optional bounded FIFO durability policies, session snapshot caching, and immutable run/trace feedback.
|
|
22
|
-
- [Performance limits](performance.md): bounded evaluation traces/judges/reports, 0.0.12 frontend interoperability
|
|
22
|
+
- [Performance limits](performance.md): 0.0.15 network-free provider/RAG/memory benchmark evidence and frozen caps, bounded evaluation traces/judges/reports, 0.0.12 frontend interoperability, 0.0.11 search/budget, 0.0.10 workspace-mode, 0.0.9 coding/browser, security scan/live-canary backstops, live subscriber queues, branch-read pagination expectations, JSONL/dev-store limits, and production sizing assumptions.
|
|
23
23
|
- [Structured output](structured-output.md): the `Artifact*` seam plus provider-native `StructuredOutputOptions` / `structuredOutputMode` for capable models.
|
|
24
24
|
|
|
25
25
|
## Compaction/session memory
|
|
26
26
|
- [Compaction and retry policies](compaction-and-retry.md): summarize branch history and retry transient provider failures with host-replaceable policies.
|
|
27
27
|
- [LLM compaction package](compaction-llm.md): optional provider-backed strategy with finite summary/reserve/error caps, bounded redacted streaming retention, mandatory finite post-policy `model.parameters.maxTokens`, and `createCodingCompactionStrategy()` for coding handoff focus.
|
|
28
28
|
- [Observational memory compaction package](compaction-observational-memory.md): optional source-backed memory with owned append callback, finite turn/call/argument/result/transcript/error worker limits, redacted provider-valid transcripts, fast compaction, recall, and status/view commands; worker model falls back to host-supplied `sessionModel`.
|
|
29
|
-
- [Working and semantic memory](working-and-semantic-memory.md): optional `@arnilo/prism-memory` working-memory store, semantic recall, finite Embedder/VectorStore contracts,
|
|
29
|
+
- [Working and semantic memory](working-and-semantic-memory.md): optional `@arnilo/prism-memory` working-memory store, semantic recall, finite Embedder/VectorStore contracts, PostgreSQL/pgvector path, consent lifecycle, identity-bound redacted export, and resumable bounded rebuild.
|
|
30
30
|
- [Session stores](session-stores.md): `SessionStore` contract, `SessionAppendOptions`, `SessionAppendConflictError`, branch handles, `readBranchPath`, optional bounded `searchSessions` / `SessionIndex` (memory linear|unsupported), and dev-vs-production branch reads — start here for session persistence.
|
|
31
31
|
- [Conversations](conversations.md): durable user-scoped conversation threads (create/list/continue/branch/archive/export/delete) on session + event-ledger seams, thread-bound reconnectable replay, frozen caps, and legal-hold-aware deletion.
|
|
32
32
|
- [Work artifacts and review](work-artifacts-and-review.md): durable artifact co-work review — authorized attach (MIME/hash/version, producer run, citations, preview metadata), revision compare, approve/reject with last-validated recovery, and authorized expiring delivery links; records persist as versioned checkpoints, never file bodies.
|
|
@@ -34,7 +34,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
34
34
|
- [Database persistence](database-persistence.md): production persistence contracts, shared checksummed migration/full-shape catalog primitives (`@arnilo/prism/testing/persistence-schema`), conditional append, indexes, `readBranchPath`, reference relational schema, retention/legal-hold/quota lifecycle (`lifecycle`), and NoSQL mapping.
|
|
35
35
|
- [SQLite persistence](sqlite-persistence.md): optional `better-sqlite3` adapter with session/run storage, checkpoints/leases, feedback, FTS `searchSessions` (migration-v4), and transactionally verified/backfilled migration metadata.
|
|
36
36
|
- [PostgreSQL persistence](postgres-persistence.md): optional pooled `pg` adapter with session/run/checkpoint/lease/feedback storage, FTS `searchSessions` (migration-v4), advisory-locked checksummed/full-shape migrations, and opt-in live conformance.
|
|
37
|
-
- [Migration guide](migration.md): **0.0.
|
|
37
|
+
- [Migration guide](migration.md): **0.0.15** OpenAI hosted tools/continuation/Realtime, exact AI SDK v4 matrix, RAG lifecycle/reranking/trust/status, and memory export/rebuild; **0.0.14** conversations, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth connectors, browser checkpoints, device contracts, and Alibaba/Ollama providers; plus prior release migrations.
|
|
38
38
|
- [Node JSONL session store](node-jsonl-session-store.md): development-only JSONL file adapter for single-process Node hosts; no cross-process safety; `searchSessions` throws `SessionSearchUnsupportedError`.
|
|
39
39
|
- [Persistence, credentials, and multimodality primitives](persistence-credentials-multimodality-primitives.md): Plan 056 inventory — session/run-ledger/persistence contracts, credential/OAuth seams, content/resource/model capabilities, package dependency matrix, conformance matrix, and threat model for production adapters.
|
|
40
40
|
|
|
@@ -42,24 +42,24 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
42
42
|
- [Provider primitives](provider-primitives.md): shared bounded transport and OpenAI serialization helpers — migrated across first-party providers; native structured-output and observability contracts.
|
|
43
43
|
- [Provider layer](provider-layer.md): register and resolve host-owned providers/models, choose replace-or-error duplicate policy, create provider events, stream/reconstruct tool-call deltas, use generic provider request options, and test with the mock provider; deprecated provider-level timeout/retry hints point to runtime abort/retry.
|
|
44
44
|
- [Model registry](model-registry.md): register and resolve `ModelConfig` records with capabilities, limits, cost, cache support metadata, compat data, and duplicate policy.
|
|
45
|
-
- [Provider caching](provider-caching.md): use `PromptCacheHints`, `PromptCacheBreakpoint`, `ModelCacheCapabilities`, cache-aware stable-prefix guidance, and shared cache diagnostics helpers; includes
|
|
45
|
+
- [Provider caching](provider-caching.md): use `PromptCacheHints`, `PromptCacheBreakpoint`, `ModelCacheCapabilities`, cache-aware stable-prefix guidance, and shared cache diagnostics helpers; includes the complete per-provider explicit/implicit cache matrix plus no-Prism-cache entries (including Anthropic, Google, Alibaba, Ollama, cloud adapters, and host-owned AI SDK); cache hints are best-effort and cache keys are never secrets.
|
|
46
46
|
- [Thinking and reasoning](thinking-and-reasoning.md): portable `ThinkingLevel` helpers (`applyThinkingLevel` / `thinkingCompatFor`) map per-turn effort into provider `compat` fields; model defaults stay on `ModelConfig.compat`; no second options tree.
|
|
47
47
|
- [Use-case model selection](use-case-model-selection.md): bind `{ model?, provider?, thinkingLevel? }` for observational memory, LLM compaction, and other non-session LLM jobs with explicit session-model fallback via `resolveUseCaseModel`.
|
|
48
48
|
- [Provider request policies](provider-request-policies.md): chain `ProviderRequestPolicy` hooks, use `createSessionCachePolicy`, and merge legacy/structured cache options safely.
|
|
49
|
-
- [Provider packages](provider-packages.md): define explicit provider packages, model metadata, auth descriptors, request/cache policies, provider-owned header precedence,
|
|
50
|
-
- Phase 12 package workspaces: [`@arnilo/prism-provider-openai`](providers/openai.md), [`@arnilo/prism-provider-anthropic`](providers/anthropic.md) (native Messages, `cache_control`, thinking, caller-gated `listAnthropicModels`), [`@arnilo/prism-provider-google`](providers/google.md) (native Gemini `generateContent` SSE, caller-gated `listGoogleModels`), [`@arnilo/prism-provider-opencode-go`](providers/opencode-go.md) (official Go open models, dual-route Anthropic/OpenAI, caller-gated `listOpenCodeGoModels`, `reasoning_content`/thinking preserve), [`@arnilo/prism-provider-openrouter`](providers/openrouter.md) (app-controlled catalog, caller-gated `listOpenRouterModels`, `reasoning` merge/preserve, `cache_control` + sticky `session_id`), [`@arnilo/prism-provider-zai`](providers/zai.md) (official `thinking`/`reasoning_effort`/`tool_stream`, implicit cache, caller-gated `listZaiModels`), [`@arnilo/prism-provider-kimi`](providers/kimi.md), [`@arnilo/prism-provider-alibaba`](providers/alibaba.md) (Alibaba Cloud Model Studio / DashScope + Coding Plan, OpenAI-compatible, caller-gated `listAlibabaModels`, implicit + explicit `cache_control` caching, Qwen `enable_thinking`), [`@arnilo/prism-provider-ollama`](providers/ollama.md) (Ollama Cloud + local, OpenAI-compatible, caller-gated `listOllamaModels`, implicit-only caching, `reasoning_effort`), and [`@arnilo/prism-provider-neuralwatt`](providers/neuralwatt.md) with implicit vLLM prefix caching, reasoning controls (`reasoning_effort`/`thinking_token_budget`/`enable_thinking`/`preserve_thinking`/`clear_thinking`), reasoning preservation, OpenAI-style tool-call loop, quota, telemetry, and retry classification helpers.
|
|
49
|
+
- [Provider packages](provider-packages.md): define explicit provider packages, model metadata, auth descriptors, request/cache policies, provider-owned header precedence, the provider-authorized OAuth matrix, and the Phase 10 first-party compatibility matrix without package discovery or provider-specific core behavior; includes a cache behavior summary and **caller-gated on-demand model discovery** (`list*Models`, setup zero-fetch).
|
|
50
|
+
- Phase 12 package workspaces: [`@arnilo/prism-provider-openai`](providers/openai.md) (Responses hosted-tool attribution, bounded continuation, Realtime session seam), [`@arnilo/prism-provider-anthropic`](providers/anthropic.md) (native Messages, `cache_control`, thinking, caller-gated `listAnthropicModels`), [`@arnilo/prism-provider-google`](providers/google.md) (native Gemini `generateContent` SSE, caller-gated `listGoogleModels`), [`@arnilo/prism-provider-opencode-go`](providers/opencode-go.md) (official Go open models, dual-route Anthropic/OpenAI, caller-gated `listOpenCodeGoModels`, `reasoning_content`/thinking preserve), [`@arnilo/prism-provider-openrouter`](providers/openrouter.md) (app-controlled catalog, caller-gated `listOpenRouterModels`, `reasoning` merge/preserve, `cache_control` + sticky `session_id`), [`@arnilo/prism-provider-zai`](providers/zai.md) (official `thinking`/`reasoning_effort`/`tool_stream`, implicit cache, caller-gated `listZaiModels`), [`@arnilo/prism-provider-kimi`](providers/kimi.md), [`@arnilo/prism-provider-alibaba`](providers/alibaba.md) (Alibaba Cloud Model Studio / DashScope + Coding Plan, OpenAI-compatible, caller-gated `listAlibabaModels`, implicit + explicit `cache_control` caching, Qwen `enable_thinking`), [`@arnilo/prism-provider-ollama`](providers/ollama.md) (Ollama Cloud + local, OpenAI-compatible, caller-gated `listOllamaModels`, implicit-only caching, `reasoning_effort`), and [`@arnilo/prism-provider-neuralwatt`](providers/neuralwatt.md) with implicit vLLM prefix caching, reasoning controls (`reasoning_effort`/`thinking_token_budget`/`enable_thinking`/`preserve_thinking`/`clear_thinking`), reasoning preservation, OpenAI-style tool-call loop, quota, telemetry, and retry classification helpers.
|
|
51
51
|
- Phase 8 enterprise cloud (workload identity; separate from consumer Anthropic/Google): [`@arnilo/prism-provider-azure`](providers/azure.md) (Entra / Foundry), [`@arnilo/prism-provider-bedrock`](providers/bedrock.md) (IAM/IRSA + region/PrivateLink), [`@arnilo/prism-provider-vertex`](providers/vertex.md) (ADC / Vertex OpenAPI).
|
|
52
|
-
- Optional AI SDK adapter: [`@arnilo/prism-provider-ai-sdk`](providers/ai-sdk.md) maps host-owned `LanguageModelV4` models onto Prism `AIProvider` streams (
|
|
52
|
+
- Optional AI SDK adapter: [`@arnilo/prism-provider-ai-sdk`](providers/ai-sdk.md) maps host-owned pinned `LanguageModelV4` models onto Prism `AIProvider` streams (offline-tested `@ai-sdk/provider` version matrix; no Prism catalog; maps metadata/tool authority/`finish.usage` cache tokens; reasoning is host-model-owned).
|
|
53
53
|
- [OpenAI-compatible provider](providers/openai-compatible.md): optional provider subpath using native or injected `fetch` for Chat Completions streaming (`chatCompletionsUrl` / `authStyle` overrides for enterprise adapters).
|
|
54
54
|
|
|
55
55
|
## Input, prompt, and context assembly
|
|
56
56
|
- [SDK customization guide](customization.md): map provider resolution, middleware, context, builders, injectors, loops, compaction, retry, stores, and skills to explicit host-wired APIs.
|
|
57
57
|
- [Input and prompt assembly](input-and-prompt-assembly.md): render tiny prompt templates and turn common host input, history, attachments, explicit resources, summaries, and tool results into messages with replaceable builders, provider-input assembly, legacy default order, opt-in cache-aware ordering, and optional `contextBudget` eviction + omission reports. Audio/file/document `ContentBlock` types and capability checks are documented there.
|
|
58
|
-
- [Multimodal content](multimodal-content.md): complete-request media resolution and aggregate bounds, DNS-classified/address-pinned URLs, SSRF/MIME policy,
|
|
58
|
+
- [Multimodal content](multimodal-content.md): complete-request media resolution and aggregate bounds, DNS-classified/address-pinned URLs, SSRF/MIME policy, `ModelCapabilities.input` tags, and first-party content-type mapping.
|
|
59
59
|
- [System prompts](system-prompts.md): compose explicit user/package/app/run system prompt layers, auto-load the standard `AGENTS.md` (workspace) / `SYSTEM.md` prompt files via the Node `loadSystemPromptFiles` loader (trust-gated for `AGENTS.md`), and append `SYSTEM.md` → per-agent `AGENT.md` body → repo `AGENTS.md` layers from a discovered agent bundle via `resolveAgentBundle`.
|
|
60
60
|
- [Instruction injection](instruction-injection.md): register package injectors that layer redacted instructions/context blocks without granting tools, permissions, or resource escapes.
|
|
61
61
|
- [Context and skills](context-and-skills.md): resolve ordered context providers and keep context/skill selection host-owned; omitted declarative skills stay inactive by default, `toolNames` fail closed before provider turns, and strict skill registries prevent silent shadowing.
|
|
62
|
-
- [Retrieval-augmented generation](rag.md): optional bounded
|
|
62
|
+
- [Retrieval-augmented generation](rag.md): optional bounded source lifecycle, document adapters, host reranking, ingestion status, attributable citations, and inert context injection.
|
|
63
63
|
|
|
64
64
|
## Tools
|
|
65
65
|
- [Tools](tools.md): register host-owned active tools with replace-or-error duplicate policy, apply exact allow/deny filtering, dispatch normal or opt-in bounded artifact-loop calls, and optionally bound untrusted JSON Schema compilation.
|
|
@@ -84,7 +84,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
84
84
|
## Configuration/manifests
|
|
85
85
|
- [Configuration and manifests](configuration-and-manifests.md): merge in-memory JSON config layers and validate data-only package manifests with prototype-pollution key rejection.
|
|
86
86
|
- [Node filesystem config loader](node-filesystem-config.md): explicitly read caller-named JSON config files in Node hosts.
|
|
87
|
-
- [Resource loading](resource-loading.md): decode text, JSON, binary, and
|
|
87
|
+
- [Resource loading](resource-loading.md): decode text, JSON, binary, and manifests through caller-provided loaders; bridge host-authorized artifacts to bounded RAG document loading.
|
|
88
88
|
|
|
89
89
|
## Server/API
|
|
90
90
|
- [Web-standard server handler](server.md): optional framework-free authorized direct/SSE agent, durable agent lifecycle, durable workflow routes, plus optional health/drain/rate-limit/replay/deployment-lease seams; explicit bounds and zero default exposure.
|
|
@@ -117,7 +117,8 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
|
|
|
117
117
|
- `examples/`: compile-checked typed examples and runnable mock demos (SDK basics, provider registration, auth, tools, [`examples/ag-ui-server.ts`](../examples/ag-ui-server.ts), [`examples/enterprise-identity.ts`](../examples/enterprise-identity.ts), [`examples/enterprise-policy-audit.ts`](../examples/enterprise-policy-audit.ts), [`examples/enterprise-work-connectors.ts`](../examples/enterprise-work-connectors.ts), [`examples/conversation-durable-replay.ts`](../examples/conversation-durable-replay.ts), [`examples/artifact-review-delivery.ts`](../examples/artifact-review-delivery.ts), [`examples/server-deployment-seams.ts`](../examples/server-deployment-seams.ts), cache-aware prompt assembly, NeuralWatt agent run ([`examples/neuralwatt-agent-run.ts`](../examples/neuralwatt-agent-run.ts)), [`examples/coding-compaction.ts`](../examples/coding-compaction.ts), stores/branching, structured-output/artifact-loop, CLI, RPC, workflow orchestration).
|
|
118
118
|
|
|
119
119
|
## Release and install
|
|
120
|
-
- [Release and install](release-and-install.md): current **43
|
|
120
|
+
- [Release and install](release-and-install.md): current **0.0.15** 43-package graph (Phase 10 provider/AI-SDK/RAG/memory parity; no new package), exact-peer/install/tarball rules, deterministic resumable publication and publish dry-run, pinned supply-chain gates, offline tests, the 0.0.15 provider/AI-SDK/RAG/memory protected live-canary matrix, and sandbox-browser Docker/Playwright gates.
|
|
121
|
+
- [Review coverage (2026-07-26 Phase 10)](review-coverage-2026-07-26-phase-10.md): Plan 078 evidence freeze — OpenAI hosted tools/continuation/realtime, AI SDK version matrix, remaining provider metadata parity, RAG replaceSource/loaders/parsers/reranker/provenance/ingestion-status, memory export/rebuild/conformance, and 0.0.15 (43 → 43 manifests; no new package) release gates.
|
|
121
122
|
- [Review coverage (2026-07-25 Phase 9)](review-coverage-2026-07-25-phase-9.md): Plan 077 evidence freeze — conversation service, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth, browser checkpoint composition, and deny-by-default device contracts for 0.0.14 (41 → 43 manifests; only the two provider packages are new).
|
|
122
123
|
- [Review coverage (2026-07-23 Phase 8)](review-coverage-2026-07-23-phase-8.md): Plan 076 evidence freeze — enterprise identity/policy/router packages, Azure/Bedrock/Vertex adapters, server deployment seams, persistence lifecycle hooks, and M365/GWS work-connector bounds for 0.0.13.
|
|
123
124
|
- [Review coverage (2026-07-22 Phase 7)](review-coverage-2026-07-22-phase-7.md): Plan 075 evidence freeze — AG-UI/ACP package boundary, streamed durable resume, bounded replay/projection, coding compaction preset, and provider-authorized OAuth policy for 0.0.12.
|
package/docs/migration.md
CHANGED
|
@@ -7,6 +7,34 @@ Prism 0.0.6 preserves documented 0.0.3 agent construction except for two intenti
|
|
|
7
7
|
1. **`session.run()` / `session.prompt()` return `AgentRunResult`** and `session.stream()` starts one owned run after subscribing. Callers that ignored the previous `Promise<void>` keep working; failed/aborted runs reject with `AgentRunError` (`.result` attached).
|
|
8
8
|
2. **`AgentConfig.extensions` / `settings` / `credentials` are removed.** Wire extensions through `createExtensionKernel()`, read settings in the host, and pass credential resolvers to the provider edge.
|
|
9
9
|
|
|
10
|
+
## 0.0.14 → 0.0.15 OpenAI hosted tools, continuation, and realtime (additive, pre-release)
|
|
11
|
+
|
|
12
|
+
`@arnilo/prism-provider-openai` now distinguishes server-executed calls with `authority: "provider-hosted"`; host dispatchers must not execute or reply to them. Incomplete Responses streams self-resume with an opaque `previous_response_id` cursor (at most 4 KiB, at most eight hops) and surface `continuation_required`; cap or duplicate-cursor failure now ends with a provider error instead of a silent partial response.
|
|
13
|
+
|
|
14
|
+
Realtime is opt-in through `createOpenAIRealtimeSession({ model, ownerId, apiKey, ... })`. Supply a stable host-owned `ownerId`; the session uses documented WebSocket headers, waits for `session.created`, exposes audio/transcript/interrupt/close events, and fails closed on disconnect, identity, audio/byte, or wall-time limits. It does not add a vendor package or automatic voice capture/playback.
|
|
15
|
+
|
|
16
|
+
## 0.0.14 → 0.0.15 AI SDK adapter matrix (additive, pre-release)
|
|
17
|
+
|
|
18
|
+
`@arnilo/prism-provider-ai-sdk` now pins and verifies `@ai-sdk/provider@4.0.3` at setup rather than accepting any v4 minor. Upgrade the peer package to the documented matrix entry. An unlisted installed version fails with typed `AiSdkProviderError` code `unsupported_version`; add a tested matrix row before changing it.
|
|
19
|
+
|
|
20
|
+
Stream output now maps `response-metadata.id` to `message_start`, preserves `providerExecuted` tool authority as `"provider-hosted"`, and rejects unsupported output parts or `structuredOutput.strict` with `unsupported_mapping` rather than dropping them. Pass `redactor` when using the adapter directly; agents retain their existing active-redactor behavior.
|
|
21
|
+
|
|
22
|
+
## 0.0.14 → 0.0.15 RAG source lifecycle and document adapters (additive, pre-release)
|
|
23
|
+
|
|
24
|
+
`@arnilo/prism-rag` now adds `replaceSource()`, `deleteSource()`, and `replaceDocument()` plus `DocumentLoader` / `Parser` seams. Existing `indexChunks()` behavior is unchanged; use `replaceSource()` when a source can shrink or must retain its old index if re-embedding fails.
|
|
25
|
+
|
|
26
|
+
Atomic replacement deliberately requires a scoped source-aware transaction (`getBySource()` + `transaction()`). The in-memory reference vector store supplies both; durable custom stores must add equivalent exact tenant/resource/corpus behavior before using replacement. Prism rejects a generic upsert-only store rather than offering a non-atomic fallback.
|
|
27
|
+
|
|
28
|
+
Reference parsers (`textParser`, `markdownParser`, `htmlParser`, `pdfParser`) are available from root and `@arnilo/prism-rag/parsers`; loaders are available from `@arnilo/prism-rag/loaders`. HTML removes script/style text. The PDF parser only accepts bounded uncompressed text PDFs (8 MiB / 256 pages / 30 s); install no new parser dependency—supply a host `Parser` for compressed or scanned files. `createWebFetchDocumentLoader()` accepts an existing `@arnilo/prism-web-tools` adapter and preserves its citation/untrusted metadata; it does not add a crawler.
|
|
29
|
+
|
|
30
|
+
RAG retrieval now optionally accepts host-owned `Reranker`; it receives redacted bounded hits and must return their exact IDs once each. Results add `trust`, `provenance`, and `retrievalRank`; context blocks now repeat untrusted/inert/injection-capable metadata. Add `statusStore` to indexing/replacement when hosts need per-source pending/indexed/failed/partial progress, use `listIngestionStatus()` for capped exact-scope pages, and supply durable storage if process restart durability matters. `createMemoryIngestionStatusStore()` is only a reference adapter.
|
|
31
|
+
|
|
32
|
+
## 0.0.14 → 0.0.15 memory export and rebuild (additive, pre-release)
|
|
33
|
+
|
|
34
|
+
`@arnilo/prism-memory` adds `exportMemory({ identity, cursor?, ... })` and `rebuildIndex({ cursor?, ... })`. Export is not a generic admin dump: provide the exact host-verified tenant/resource/thread identity used to construct `createMemory()`. It excludes revoked, invisible, and consent-less legacy entries, redacts each returned record, and caps one page at 100 entries / 4 MiB / 10 seconds by default (200 / 32 MiB / 60 seconds hard).
|
|
35
|
+
|
|
36
|
+
`rebuildIndex()` re-embeds one 32-record page by default (128 hard), validates existing and new finite vectors, and returns `nextCursor`; persist that cursor in host-owned authorized state and call again to resume after an abort/restart. Neither API scans a corpus or starts a background worker. They require a semantic `VectorStore.listByThread()` implementation; `applyRetention()` now also requires `countByThread()` for bounded oldest-first deletion. The shipped in-memory adapter and PostgreSQL/pgvector adapter conform. `@arnilo/prism-session-store-sqlite` remains a session/run persistence package, not a semantic-vector adapter.
|
|
37
|
+
|
|
10
38
|
## 0.0.13 → 0.0.14 personal/work-agent conversations, co-work review, and channel/device gates (additive, pre-release)
|
|
11
39
|
|
|
12
40
|
Release **0.0.14** is strictly additive: every surface extends a shipped package and reuses the AG-UI adapter shipped in 0.0.12. The only new packages are two optional provider adapters (41 → 43 manifests): `@arnilo/prism-provider-alibaba` and `@arnilo/prism-provider-ollama`, both enrolled via the `@arnilo/prism-providers` family. No permission broadening — channel/device/co-work features cannot widen consent, memory, network, file, browser, connector, or tool permissions (roadmap gate 8). See [Phase 9 evidence](review-coverage-2026-07-25-phase-9.md).
|
|
@@ -213,7 +241,7 @@ Phase 4 adds optional `@arnilo/prism-evals` for deterministic scorers/datasets/e
|
|
|
213
241
|
|
|
214
242
|
Phase 5 adds `prism init <dir>` to the existing CLI. It scaffolds a tiny TypeScript project with one selected provider and an offline mock test. Optional `--with-workflows` / `--with-evals` flags add only those packages; storage and telemetry stay opt-in elsewhere.
|
|
215
243
|
|
|
216
|
-
Phase 6 adds optional `@arnilo/prism-provider-ai-sdk` for AI SDK `LanguageModelV4` interoperability.
|
|
244
|
+
Phase 6 adds optional `@arnilo/prism-provider-ai-sdk` for AI SDK `LanguageModelV4` interoperability. For 0.0.15 install its exact supported peer `@ai-sdk/provider@4.0.3` (not `^4`); an unlisted version fails at setup. Install the adapter directly, through `@arnilo/prism-providers`, or through `@arnilo/prism-all`; it is not a core dependency.
|
|
217
245
|
|
|
218
246
|
Phase 7 adds optional `@arnilo/prism-memory` for schema/template-backed working memory and embedding-based semantic recall. Install it directly or through `@arnilo/prism-all`; in-memory adapters are default, and PostgreSQL/pgvector is opt-in. It is not a core dependency.
|
|
219
247
|
|
|
@@ -49,11 +49,13 @@ Known `ModelCapabilities.input` tags are exported as `MODEL_INPUT_CAPABILITIES`:
|
|
|
49
49
|
|
|
50
50
|
| Tag | Block type | First-party mapping (declared capability required) |
|
|
51
51
|
| --- | --- | --- |
|
|
52
|
-
| `text` | `text` (default) | All providers |
|
|
53
|
-
| `image` | `image` | OpenAI Responses
|
|
54
|
-
| `audio` | `audio` | OpenAI Responses (`input_audio`) |
|
|
55
|
-
| `file` | `file` | OpenAI Responses (`input_file`); Anthropic
|
|
56
|
-
| `document` | `document` | OpenAI Responses (`input_file`); OpenCode Go Anthropic route;
|
|
52
|
+
| `text` | `text` (default) | All first-party providers; Azure/Bedrock/Vertex use their host-selected OpenAI-compatible endpoint/model. |
|
|
53
|
+
| `image` | `image` | OpenAI Responses; Anthropic; Google; Kimi; Z.AI; OpenRouter; OpenCode Go OpenAI route; Alibaba; Ollama; NeuralWatt. Enterprise OpenAI-compatible packages require the host model/endpoint to declare and accept image input. |
|
|
54
|
+
| `audio` | `audio` | OpenAI Responses (`input_audio`) and Google `generateContent` inline data. OpenAI Realtime instead receives `RealtimeSession.sendAudio()` chunks, not an `audio` `ContentBlock`. |
|
|
55
|
+
| `file` | `file` | OpenAI Responses (`input_file`); Anthropic/Kimi/OpenCode Go Anthropic route accept PDF file/document forms; Google maps inline file data. |
|
|
56
|
+
| `document` | `document` | OpenAI Responses (`input_file`); Anthropic/Kimi/OpenCode Go Anthropic route map PDF; Google maps inline document data. |
|
|
57
|
+
|
|
58
|
+
The AI SDK adapter maps declared user text/image/audio/file/document blocks (and assistant text/image/file/document) to AI SDK file parts; `resourceUri` remains host-resolved before `doStream`. Its output `file`, `reasoning-file`, and `source` parts are deliberately rejected as `unsupported_mapping`, not converted to trusted Prism content. Provider capability metadata is the gate—this matrix never upgrades a model that does not declare the matching input tag.
|
|
57
59
|
|
|
58
60
|
## Outputs / response / events
|
|
59
61
|
|
|
@@ -136,6 +138,7 @@ try {
|
|
|
136
138
|
- Local filesystem paths should use trust policies such as `createPathTrustPolicy()` before exposing URIs to loaders.
|
|
137
139
|
- Provider upload/create/delete lifecycles are provider-package-local. `@arnilo/prism-provider-openai` inlines files under 4 MiB as `data:<mediaType>;base64,...` `file_data`, otherwise uses a bounded per-run upload cache and best-effort `DELETE /v1/files` cleanup after each stream.
|
|
138
140
|
- Shared wire helpers live in `@arnilo/prism/providers/media` (`resolveProviderMediaMessages`, `serializeOpenAIResponsesInputFile`, `serializePdfDocumentWireBlock`, `createBoundedUploadCache`). OpenAI Responses, Kimi, and OpenCode Go Anthropic routes resolve their complete media collection once before serialization or upload.
|
|
141
|
+
- OpenAI Realtime audio is a bidirectional `RealtimeSession` stream, not a `ContentBlock`: provide host-captured `Uint8Array` chunks with `sendAudio()` and consume untrusted `audio_delta` / transcript events. It has a fixed 256 events/s, 1 MiB/s, and 600 s default ceiling.
|
|
139
142
|
|
|
140
143
|
## Security and performance notes
|
|
141
144
|
|
package/docs/performance.md
CHANGED
|
@@ -6,6 +6,40 @@ Evaluation defaults are finite: 100 trace rows × 20 pages and 4 MiB aggregate t
|
|
|
6
6
|
|
|
7
7
|
This page states Prism runtime limits that keep slow consumers and long sessions from becoming unbounded memory or latency problems.
|
|
8
8
|
|
|
9
|
+
## Release 0.0.15 provider, RAG, and memory evidence
|
|
10
|
+
|
|
11
|
+
Run `node scripts/benchmark-0.0.15.mjs`; `PRISM_BENCH_ITERATIONS` accepts 10–100,000 (default 100). Schema/bounds test: `node --test scripts/benchmark-0.0.15.test.mjs`. Default mode is network-free: fake Responses SSE/WebSocket transports, a fake AI SDK v4 model, zero-fetch provider-package registration, hash embeddings, in-memory RAG replacement/reranking/retrieval/status, and in-memory memory retention/export/rebuild.
|
|
12
|
+
|
|
13
|
+
Scenarios: `openai-hosted-continuation`, `openai-realtime-envelope`, `ai-sdk-v4-stream-mapping`, `provider-package-metadata`, `rag-parse-replace-rerank-retrieve`, and `memory-retention-export-rebuild`.
|
|
14
|
+
|
|
15
|
+
Every row reports throughput, p50/p95 latency, heap, disk, queue/backpressure, and safety signals. `resourceLimitSignals` must be zero: hosted calls remain provider-owned, continuation stops after its finite path, Realtime credentials are absent from events, provider setup does not resolve credentials, retrieved RAG content stays inert, and memory export redacts the fixture secret. These are behavior/bound gates; host-local timings are comparison evidence, not portable release thresholds.
|
|
16
|
+
|
|
17
|
+
| Resource | Default / hard |
|
|
18
|
+
| --- | --- |
|
|
19
|
+
| OpenAI continuation hops | 8 |
|
|
20
|
+
| Realtime audio events / bytes per second | 64 / 256 · 1 MiB / 8 MiB |
|
|
21
|
+
| RAG document bytes | 1 MiB / 8 MiB |
|
|
22
|
+
| RAG rerank input / time / active calls | 64 KiB / 256 KiB · 2 s / 10 s · 2 / 8 |
|
|
23
|
+
| RAG ingestion-status page | 50 / 200 |
|
|
24
|
+
| Memory retention batch | 500 / 5,000 |
|
|
25
|
+
| Memory export | 100 / 200 entries · 4 MiB / 32 MiB · 10 s / 60 s |
|
|
26
|
+
| Memory rebuild | 32 / 128 entries · 10 s / 60 s |
|
|
27
|
+
|
|
28
|
+
This task adds no package or runtime dependency: package/install delta is zero and the frozen graph remains 43 publishable manifests. Credentialed protocol checks are documented in the [0.0.15 protected live-canary matrix](release-and-install.md#015-protected-live-canary-matrix); they never run in this benchmark, `npm test`, or `sdk:ready`.
|
|
29
|
+
|
|
30
|
+
2026-07-26 baseline: Node v24.18.0, Linux x64, 100 iterations/scenario, network=false, credentials=false.
|
|
31
|
+
|
|
32
|
+
| Scenario | ops/s | p95 ms | heap bytes | backpressure | resource limits |
|
|
33
|
+
| --- | ---: | ---: | ---: | ---: | ---: |
|
|
34
|
+
| OpenAI hosted continuation | 4,885 | 0.2924 | 15,475,920 | 0 | 0 |
|
|
35
|
+
| OpenAI Realtime envelope | 858 | 1.3130 | 12,092,232 | 0 | 0 |
|
|
36
|
+
| AI SDK v4 mapping | 27,580 | 0.0573 | 14,848,288 | 0 | 0 |
|
|
37
|
+
| Provider package metadata | 79,823 | 0.0288 | 16,077,080 | 0 | 0 |
|
|
38
|
+
| RAG lifecycle/reranking | 5,324 | 0.3586 | 13,894,576 | 0 | 0 |
|
|
39
|
+
| Memory lifecycle | 12,892 | 0.1608 | 13,923,936 | 0 | 0 |
|
|
40
|
+
|
|
41
|
+
These values are dated local comparison evidence, not portable thresholds.
|
|
42
|
+
|
|
9
43
|
## Release 0.0.12 frontend interoperability caps and evidence
|
|
10
44
|
|
|
11
45
|
`@arnilo/prism-ag-ui` uses finite handler/projection limits, all defaults / hard: request 64 KiB / 1 MiB; input 128 / 1024 messages and 64 KiB / 1 MiB text; event 64 KiB / 1 MiB; error 8 KiB / 64 KiB; replay cursor 4 / 16 KiB; replay page 100 / 500; subscriber queue 128 / 4096; stream 10,000 / 100,000 events and 10 / 64 MiB; request wall time 120 seconds / 30 minutes. Tool arguments/results/progress, frontend tools, and mutable frontend state default to zero exposure; hosts may only add bounded safe projection.
|
package/docs/provider-caching.md
CHANGED
|
@@ -146,6 +146,8 @@ Provider request policies can set `ProviderRequestOptions.cache` or the legacy `
|
|
|
146
146
|
| Provider package | Cache kind | Explicit cache hints | Multi-turn reuse notes | Caveats |
|
|
147
147
|
| --- | --- | --- | --- | --- |
|
|
148
148
|
| `@arnilo/prism-provider-openai` | `openai_key` | Sends sanitized `prompt_cache_key`; `prompt_cache_retention: "24h"` only when the model declares `longRetention`. | Stable cache key + stable prefix can improve reuse. | Best-effort only; `"short"`/`"none"` omit retention. |
|
|
149
|
+
| `@arnilo/prism-provider-anthropic` | `cache_control` | Marks only selected Anthropic message anchors; `"long"` maps to documented `ttl: "1h"`. | Keep selected anchors stable. | Best-effort; never stamp every block. |
|
|
150
|
+
| `@arnilo/prism-provider-google` | none | Sends no Prism cache marker. | Host/model may have upstream behavior. | Gemini cache controls are not mapped in this package. |
|
|
149
151
|
| `@arnilo/prism-provider-openrouter` | `cache_control` | Top-level automatic `cache_control` when enabled without breakpoints; otherwise markers only on caller-selected `cache.breakpoints`; `"long"` may add `ttl: "1h"`. Sticky `session_id` routing. | Breakpoint-stable / automatic prefixes can be reused by upstream providers. | Best-effort only; top-level automatic may exclude some backends from routing. |
|
|
150
152
|
| `@arnilo/prism-provider-opencode-go` | route-specific | Sends sanitized `x-opencode-session`; Anthropic route applies selected `cache_control` breakpoints; OpenAI route sends none. | Session id + unchanged selected anchors can help route-native caches. | Best-effort and route-dependent. |
|
|
151
153
|
| `@arnilo/prism-provider-zai` | `implicit` | No explicit cache payload; GLM context caching is automatic. | Resend unchanged prior history for implicit context-cache reuse. | Best-effort only; cache options do not force hits. |
|
|
@@ -154,11 +156,16 @@ Provider request policies can set `ProviderRequestOptions.cache` or the legacy `
|
|
|
154
156
|
| `@arnilo/prism-provider-ai-sdk` | host-owned | No Prism cache payload; host `LanguageModelV4` owns upstream caching. | Host model/provider decides cache keys, breakpoints, and sticky routing. | Adapter maps `inputTokens.cacheRead`/`cacheWrite` from `finish.usage` only; does not invent cache fields. |
|
|
155
157
|
| `@arnilo/prism-provider-alibaba` | implicit by default, optional `cache_control` | DashScope implicit prefix caching is automatic; opt-in `cache_control: {"type":"ephemeral"}` markers only on caller-selected `cache.breakpoints`, capped at 4. | Keep selected anchors and prior history stable; each cached prefix needs ≥1024 tokens and lives ~5 minutes upstream. | Best-effort and model-dependent; `cached_tokens`→read, `cache_creation_input_tokens`→write. |
|
|
156
158
|
| `@arnilo/prism-provider-ollama` | `implicit` | No `cache_control`, `cacheKey`, `prompt_cache`, or `cacheRetention` payload; Ollama KV/prefix caching is automatic with no request knob. | Resend unchanged prior history for implicit KV reuse. | Best-effort only; Ollama reports no cached-token count, so `Usage.cacheReadTokens` stays `undefined`. |
|
|
159
|
+
| `@arnilo/prism-provider-azure` | none | No Prism cache mapping. | Endpoint/model-specific. | Host owns Azure cache policy. |
|
|
160
|
+
| `@arnilo/prism-provider-bedrock` | none | No Prism cache mapping. | Endpoint/model-specific. | Host owns Bedrock cache policy. |
|
|
161
|
+
| `@arnilo/prism-provider-vertex` | none | No Prism cache mapping. | Endpoint/model-specific. | Host owns Vertex cache policy. |
|
|
157
162
|
|
|
158
163
|
Detailed first-party provider notes:
|
|
159
164
|
|
|
160
165
|
- OpenAI Responses (`@arnilo/prism-provider-openai`): `kind: "openai_key"`. Sanitizes/clamps `prompt_cache_key` to 64 chars; `"long"` retention maps to `prompt_cache_retention: "24h"` only when the model declares `cache.longRetention`; `"short"`/`"none"` omit the field. GPT-5.6+ official docs use `prompt_cache_options` / breakpoints instead of retention — `listOpenAIModels` sets `longRetention: false` for those ids. `input_tokens_details.cached_tokens` maps to `Usage.cacheReadTokens`.
|
|
161
166
|
- OpenAI-compatible Chat Completions adapter: minimal scope, sends no cache payload; see [OpenAI-compatible provider](providers/openai-compatible.md).
|
|
167
|
+
- Anthropic (`@arnilo/prism-provider-anthropic`): `kind: "cache_control"`; selected Anthropic message anchors receive `cache_control` and eligible long retention maps to `ttl: "1h"`. Cache read/create usage maps to normalized cache read/write tokens.
|
|
168
|
+
- Google (`@arnilo/prism-provider-google`): sends no Prism cache-control payload. Do not infer cache hits or cache token counts from absent Gemini fields.
|
|
162
169
|
- OpenRouter (`@arnilo/prism-provider-openrouter`): `kind: "cache_control"`. Sanitizes/clamps `session_id`/`X-Session-Id` to 256 chars for sticky routing; with no breakpoints emits top-level automatic `cache_control: { type: "ephemeral" }`; with breakpoints applies Anthropic-style markers only to caller-selected locations (last content block of each selected message); `"long"` retention adds `ttl: "1h"` when the model allows it. `prompt_tokens_details.cached_tokens`/`cache_write_tokens` map to `Usage.cacheReadTokens`/`cacheWriteTokens`. Optional `listOpenRouterModels()` may populate `ModelConfig.cache`/`cost` from live pricing.
|
|
163
170
|
- OpenCode Go (`@arnilo/prism-provider-opencode-go`): default base `https://opencode.ai/zen/go/v1`; `x-opencode-session` from `cacheKey ?? sessionId`, sanitized to 128 chars; the Anthropic route (MiniMax/Qwen) applies `cache_control` markers only to selected breakpoints (`"long"` → `ttl: "1h"`), the OpenAI route (Grok/GLM/Kimi/MiMo/DeepSeek) sends none and preserves `reasoning_content`. OpenAI route maps `prompt_tokens_details.cached_tokens`/`cache_write_tokens`; Anthropic route maps `cache_read_input_tokens`/`cache_creation_input_tokens`. Caller-gated `listOpenCodeGoModels` against official `GET /zen/go/v1/models`.
|
|
164
171
|
- Z.AI (`@arnilo/prism-provider-zai`): `kind: "implicit"`. GLM context caching is automatic; sends no explicit cache payload regardless of cache options. `prompt_tokens_details.cached_tokens`/`cache_write_tokens` map to `Usage.cacheReadTokens`/`cacheWriteTokens`.
|
|
@@ -167,6 +174,7 @@ Detailed first-party provider notes:
|
|
|
167
174
|
- AI SDK adapter (`@arnilo/prism-provider-ai-sdk`): **host-owned**. Sends no Prism cache payload; the supplied `LanguageModelV4` and its upstream provider own request caching. Maps AI SDK v4 `finish.usage.inputTokens.cacheRead`/`cacheWrite` to `Usage.cacheReadTokens`/`cacheWriteTokens`. No `list*Models()` export.
|
|
168
175
|
- Alibaba Cloud (`@arnilo/prism-provider-alibaba`): implicit by default, optional `cache_control`. DashScope implicit prefix caching is automatic (no marker); explicit opt-in `cache_control: {"type":"ephemeral"}` markers apply only to selected breakpoints when `ModelConfig.cache.kind: "cache_control"` and the caller supplies breakpoints, capped at 4 (each prefix ≥1024 tokens, ~5 minute TTL). `prompt_tokens_details.cached_tokens`/`cache_creation_input_tokens` map to `Usage.cacheReadTokens`/`cacheWriteTokens`. Caller-gated `listAlibabaModels` against OpenAI-compatible `GET {base}/models`.
|
|
169
176
|
- Ollama (`@arnilo/prism-provider-ollama`): `kind: "implicit"`. Ollama reuses its KV/prompt cache automatically; there is no request knob and no wire marker, so Prism never emits `cache_control`. Ollama reports no cached-token count, so `Usage.cacheReadTokens` is intentionally left `undefined` (not `0`). Caller-gated `listOllamaModels` against OpenAI-compatible `GET {base}/models`.
|
|
177
|
+
- Azure, Bedrock, and Vertex: their OpenAI-compatible packages intentionally emit no Prism cache fields. Endpoint/model-specific cache controls remain host-owned rather than guessed from another provider family.
|
|
170
178
|
|
|
171
179
|
### NeuralWatt cache-aware limiter
|
|
172
180
|
|