@arnilo/prism 0.1.5 → 0.1.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/CHANGELOG.md +10 -0
  2. package/dist/agent-run-lifecycle.d.ts +2 -0
  3. package/dist/agent-run-lifecycle.js +7 -0
  4. package/dist/agent-run-state.d.ts +2 -0
  5. package/dist/agent-run-state.js +10 -0
  6. package/dist/agent-session.d.ts +7 -0
  7. package/dist/agent-session.js +25 -2
  8. package/dist/cache-telemetry.d.ts +58 -0
  9. package/dist/cache-telemetry.js +102 -0
  10. package/dist/cli-provider-add.d.ts +37 -0
  11. package/dist/cli-provider-add.js +293 -0
  12. package/dist/cli-runner.d.ts +5 -1
  13. package/dist/cli-runner.js +13 -1
  14. package/dist/contracts-run-state.d.ts +13 -0
  15. package/dist/index.d.ts +5 -3
  16. package/dist/index.js +3 -2
  17. package/dist/skill-load.d.ts +23 -0
  18. package/dist/skill-load.js +74 -0
  19. package/docs/acp.md +2 -2
  20. package/docs/agent-session-runtime.md +1 -1
  21. package/docs/cli-rpc.md +33 -0
  22. package/docs/coding-agent-tools.md +12 -7
  23. package/docs/coding-security.md +3 -0
  24. package/docs/context-and-skills.md +2 -2
  25. package/docs/document-reader.md +85 -0
  26. package/docs/index.md +8 -8
  27. package/docs/model-routing.md +45 -0
  28. package/docs/provider-caching.md +63 -0
  29. package/docs/provider-packages.md +2 -0
  30. package/docs/release-and-install.md +54 -4
  31. package/package.json +3 -2
  32. package/templates/provider/CHANGELOG.md.tmpl +5 -0
  33. package/templates/provider/README.md.tmpl +41 -0
  34. package/templates/provider/docs/providers/NAME.md.tmpl +61 -0
  35. package/templates/provider/package.json.tmpl +49 -0
  36. package/templates/provider/src/cache.ts.tmpl +20 -0
  37. package/templates/provider/src/index.ts.tmpl +33 -0
  38. package/templates/provider/src/models.ts.tmpl +16 -0
  39. package/templates/provider/src/provider.ts.tmpl +23 -0
  40. package/templates/provider/src/tests/provider.test.ts.tmpl +104 -0
  41. package/templates/provider/tsconfig.json.tmpl +16 -0
package/CHANGELOG.md CHANGED
@@ -1,5 +1,15 @@
1
1
  # Changelog
2
2
 
3
+ ## [0.1.7] - 2026-08-12
4
+
5
+ ### Changed
6
+ - **Release 0.1.7 (plan 019)** is the performance-and-DX patch on the frozen 0.1.x line — additive-only vs 0.1.6 (freeze manifest `scripts/phase19-freeze-manifest.json`; every task's diff stayed inside its allowed files, enforced by the phase19 freeze machine; the async `AgUiProjection` item is a verification closeout, not new code). (1) **Prompt-cache telemetry surface** (`cache-telemetry`): dependency-free `createCacheTelemetry()` aggregator in core — host-activated (nothing subscribes by import), consumes `Usage` + `ModelConfig` pairs from the usage `ProviderEvent` or run-ledger records, and reports per-provider/model request counts, aggregate hit rate via the existing `cacheHitRate` math, cache-read/write token totals, and estimated savings via `cacheSavings` when cost metadata exists; bounded cardinality (default cap 256 distinct provider/model keys, overflow collapses into the `CACHE_TELEMETRY_OVERFLOW_KEY` `__overflow__` bucket with an `overflowed` flag, ponytail ceiling named — host-configurable caps or LRU eviction only if a real deployment exceeds it); reports carry token counters/rates only — never prompt content, cache keys, or identity; `record()` is O(1) with validated finite non-negative inputs; no OTel metric emission (demand-gated follow-up). (2) **Model-router selection policies** (`router-selection`): additive `selection` hook on `CreateModelRouterOptions` — `ModelRouterSelectionPolicy` (`name`/`rank`/`observe`); default ordered behavior byte-identical (regression test); reference `createCostLatencySelection` ranks candidates by `ModelCost` (input/output/cacheRead with per-million normalization) then in-memory EMA latency fed from `recordOutcome`'s new optional `latencyMs` (`latencyWeight` 0–1, default 0.5; pure-cost order on cold start); the policy is a permutation-only reorder of already-allowed candidates so it cannot widen allow-list/residency/budget decisions, and any drop/add/duplicate misbehavior fails closed with `ERR_PRISM_MODEL_ROUTER_POLICY`; policy name rides the still-redacted diagnostics; durable latency stats are a demand-gated follow-up requiring a `ModelRouterStateStore` contract change. (3) **Async AgUiProjection closeout** (`async-hooks`): plan 009 Task 15 surface verified with evidence — hooks are typed `Awaitable<T>` (17 hooks), `getMessages` accepts `() => readonly AgUiMessage[] | Promise<...>` so `messagesFromSession` can call async host APIs like `session.entries()`, snapshots are awaited strictly in event order (never `Promise.all`), rejection fails closed per event with sibling hooks still projected, caps apply to awaited values; evidence in `scripts/phase19-baseline.json` `asyncHooks` (`verified: true`, `gapFound: false`). (4) **`prism providers add <name>` scaffold** (`provider-scaffold`): new CLI subcommand (stdlib-only, mirrors `prism init`) scaffolds an OpenAI-compatible provider package into `./<name>` — `package.json` (peer dep `@arnilo/prism`, `sideEffects: false`, publish metadata), `tsconfig.json`, `README.md`, `CHANGELOG.md`, `src/index.ts` (`defineProviderPackage` + `api_key` auth-method registration), `src/provider.ts` (built on `createOpenAICompatibleProvider`), `src/models.ts` (starter `ModelConfig` list), `src/cache.ts` (cache-hint mapping via the shared core helpers), `src/__tests__/provider.test.ts` (offline conformance: stream shape + usage, header ownership, tool-call delta reconstruction, serialized-content coverage, secret-leak redaction), and `docs/providers/<name>.md` stub; flags `--base-url` (http(s) validated), `--env-key` (shell-safe identifier), `--model`, `--force`; npm package-name validation, path-traversal and symlink-escape refusal (nothing can land outside the destination), usage errors exit 2 with nothing written, generated code contains placeholders only — never secrets; scaffold output is host-chosen and never auto-registered into repo workspaces or resolvers; a fixture test proves the generated package typechecks and passes its conformance test offline against the repo build. Release graph stays **50** publishable manifests (root + 49 workspace packages — 14 provider adapters, 9 `prism-*` family/profile, 26 capability incl. `@arnilo/prism-document-reader`) at exact **0.1.7**. Exit gate green (core tests + script gates incl. phase19-freeze done-phase, `sdk:ready`, audit 0 moderate, pack dry-run 50/50 twice byte-identical, plain compat gate at 0.1.7 with 0 breaking deltas then version-literal baseline refresh — no `--allow-break` anywhere, evidence in `scripts/phase19-baseline.json`). Store compatibility with 0.1.6: **compatible, no migration** (additive-only; no persisted-shape change; `docs/migration.md` gains no entries). **Publication remains the operator handoff** (`docs/release-and-install.md` `0.1.7 publish handoff` — signed `v0.1.7` tag + npm OIDC). **CI hardening after the exit gate** (same day): the release `verify` job now runs the sdk:ready legs phase-by-phase (explicit rc per leg so a silent failure still names the failing phase) and uploads `sdk-ready.log` as an artifact on failure; the examples demo test compiles the demos in place with the repo tsc before spawning them, so the spawned children are plain JS instead of loading the amaro type-stripping WASM module (whose large per-process virtual reservation fails with `WebAssembly.Instance(): Out of memory` on memory-constrained CI runners); emitted .js files are removed in a finally block.
7
+
8
+ ## [0.1.6] - 2026-08-11
9
+
10
+ ### Changed
11
+ - **Release 0.1.6 (plan 018)** is the coding-agent capability-closeouts patch on the frozen 0.1.x line — five demand-gated closeouts, all shipped, additive-only vs 0.1.5 (freeze manifest `scripts/phase18-freeze-manifest.json`; every closeout flipped to `demanded` by named demand evidence before its task landed, then the demand-gate registry validated demanded ⇒ implemented, deferred ⇒ untouched). (1) **Durable ACP session store** (`acp-session-store`): `@arnilo/prism-ag-ui` gains the host-owned `AcpSessionStore` seam on `CreatePrismAcpAgentOptions` — `save` (upsert on session/new, set_mode, set_config_option, never on cancel/prompt-end), `loadAll` (lazy, once per agent instance, after authorization, cross-tenant entries refused `ERR_PRISM_ACP_INPUT`), `evict` (on close/delete); the persisted entry shape `{sessionId, ownership, modeId, configValues, cwd, additionalDirectories, updatedAt}` deliberately excludes client/controller/budget/pending state; fail-closed restore drops corrupt/oversized entries, re-validates modes/config options, keeps the in-memory registry caps (32 default / 128 hard), and re-resolves the live session binding; absent seam = byte-identical 0.1.5 in-memory behavior; the whole persisted entry rides the optional `SecretRedactor` at the save boundary. (2) **Network-free native sandbox backend** (`native-sandbox`): `@arnilo/prism-coding-security` gains `createNativeSandbox` — spawn + POSIX rlimits + existing path containment, zero new dependencies; every command runs in a fresh network namespace via the OS `unshare` binary (plain or `--map-root-user` preflighted once at creation, fail-closed on macOS/Windows and where netns cannot be created), ulimit chains (`-v`/`-t`/`-n`) with `|| exit 126`, argv-only `exec` (never shell-interpolated), cwd containment via `assertPathInsideRoots`, process-group kill on timeout/abort, env allow-list (host env never inherited), output cap, `close({export})` tar parity, and a documented honest boundary (runs as the invoking OS user; egress denial + rlimits + cwd containment only). (3) **Bounded PDF/Office document reader** (`doc-reader`): new optional package `@arnilo/prism-document-reader` (the 50th publishable manifest) — `createDocumentReader({ maxBytes, maxPages, maxTextBytes, parsers })` behind optional peer parsers `pdf-parse`/`mammoth` (dynamic-import, fail-closed at creation with an install hint when absent), magic-byte format gating (never extension sniffing), null fall-through to the 0.1.5 text path, refuse-over-truncate for over-page PDFs, byte-safe text truncation, optional `SecretRedactor` at the adapter boundary, no embedded-content execution, no external resource fetch (egress tripwire test), extraction envelope recorded in `scripts/budgets.json`; `createReadTool` gains the additive `documentReader` slot with input/page/text caps re-checked in the read flow. (4) **Recursive delete + brace-expanding glob** (`delete-glob`): `delete` gains the per-call opt-in `recursive: true` (symlink children unlinked never followed, iterative post-order walk, per-call fan-out cap 10,000 default / 100,000 hard, partial deletion reported never silent, `maxEntries` bound); `glob` gains host-selected + per-call `braceExpansion` (`{a,b}` textual expansion, max 128 alternatives / 4096 expanded bytes, unbalanced/nested/empty braces and overflow fail closed, default matcher semantics unchanged). (5) **Checkpoint persistence for loaded-skill bodies** (`checkpoint-bodies`): durable runs may set `includeSkillBodies: true` on BOTH run and resume options (alongside `persistSessionState`) — the exact loaded-skill instructions ride the checkpoint (`{name, instructions}` pairs, ≤64 bodies / ≤256-char names / ≤262144-byte bodies / ≤1 MiB total, validated fail-closed on save and load, redacted at rest) so resume re-renders them registry-independently with no `load_skill` round-trip; names-only stays the default and 0.1.3/0.1.2 checkpoint shapes are byte-identical; `maxStateBytes` refuses oversize bodies with a recorded error, never truncates. Release graph **50** publishable manifests (root + 49 workspace packages — 14 provider adapters, 9 `prism-*` family/profile, 26 capability incl. `@arnilo/prism-document-reader`) at exact **0.1.6**. Exit gate green (core 1,433/1,433 + 190 script gates incl. phase18-freeze done-phase, `sdk:ready`, audit 0 moderate, pack dry-run 50/50 twice byte-identical, plain compat gate at 0.1.6 with 0 breaking deltas then version-literal baseline refresh, evidence in `scripts/phase18-baseline.json`). Store compatibility with 0.1.5: **compatible, no migration** (additive-only; no persisted-shape change). **Publication remains the operator handoff** (`docs/release-and-install.md` `0.1.6 publish handoff` — signed `v0.1.6` tag + npm OIDC).
12
+
3
13
  ## [0.1.5] - 2026-08-11
4
14
 
5
15
  ### Changed
@@ -20,6 +20,8 @@ export interface AgentRunLifecycleRequest {
20
20
  readonly agentId?: string;
21
21
  /** Opt-in (plan 015 Task 4): restore persisted loaded-skill names on resume. */
22
22
  readonly persistSessionState?: boolean;
23
+ /** Opt-in (plan 018 Task 6): restore persisted loaded-skill bodies on resume (requires `persistSessionState` too). */
24
+ readonly includeSkillBodies?: boolean;
23
25
  }
24
26
  /** Bounded live-event options for a durable lifecycle resume. */
25
27
  export interface AgentRunLifecycleStreamRequest extends AgentRunLifecycleRequest, SubscribeOptions {
@@ -28,6 +28,7 @@ export function createAgentRunLifecycle(options) {
28
28
  fencingToken: options.fencingToken,
29
29
  definitionRevision: resolved.definitionRevision,
30
30
  persistSessionState: request.persistSessionState,
31
+ includeSkillBodies: request.includeSkillBodies,
31
32
  });
32
33
  },
33
34
  async *resumeStream(ref, resume, request = {}) {
@@ -45,6 +46,7 @@ export function createAgentRunLifecycle(options) {
45
46
  maxQueuedEvents: request.maxQueuedEvents,
46
47
  overflow: request.overflow,
47
48
  persistSessionState: request.persistSessionState,
49
+ includeSkillBodies: request.includeSkillBodies,
48
50
  });
49
51
  },
50
52
  };
@@ -95,6 +97,11 @@ async function prepareAgentRunResume(agent, ref, resume, options, signal) {
95
97
  if (options.persistSessionState && state.sessionState?.loadedSkillNames) {
96
98
  session.restoreLoadedSkills(state.sessionState.loadedSkillNames);
97
99
  }
100
+ // Plan 018 Task 6 (closeout `checkpoint-bodies`): restore exact instructions so the
101
+ // resumed session renders them registry-independently (no load_skill round-trip).
102
+ if (options.persistSessionState && options.includeSkillBodies && state.sessionState?.loadedSkillBodies) {
103
+ session.restoreLoadedSkillBodies(state.sessionState.loadedSkillBodies);
104
+ }
98
105
  if (resume.decision !== undefined && resume.decisions !== undefined) {
99
106
  throw new AgentDecisionError("ERR_PRISM_DECISION_INVALID", "Resume accepts exactly one of decision or decisions");
100
107
  }
@@ -1,5 +1,6 @@
1
1
  import type { Agent, AgentRunInterruption, AgentRunRef, AgentRunState, AgentRunStateOptions, AgentRunStatusResult, CheckpointRecord, CheckpointStore, JsonValue, Message, ModelConfig, NestedRunRef, OwnershipScope, RunDecision, RunLimitCounters, StickyDecision, ToolCallContent } from "./contracts.js";
2
2
  import type { SecretRedactor } from "./redaction.js";
3
+ import { type LoadedSkillBodiesEntry } from "./skill-load.js";
3
4
  export declare const AGENT_RUN_STATE_NAMESPACE = "prism.agent-run";
4
5
  export declare const AGENT_RUN_STATE_SCHEMA_VERSION: 1;
5
6
  export declare const DEFAULT_MAX_AGENT_RUN_STATE_BYTES: number;
@@ -41,6 +42,7 @@ export interface StoredAgentRunState extends AgentRunState {
41
42
  */
42
43
  readonly sessionState?: {
43
44
  readonly loadedSkillNames?: readonly string[];
45
+ readonly loadedSkillBodies?: readonly LoadedSkillBodiesEntry[];
44
46
  };
45
47
  }
46
48
  /** Session-state caps (plan 015 Task 4): bounded names charged against the run-state byte budget. */
@@ -1,5 +1,6 @@
1
1
  import { createHash } from "node:crypto";
2
2
  import { AgentLoopStateError, AgentRunStateError } from "./contracts.js";
3
+ import { validateLoadedSkillBodies } from "./skill-load.js";
3
4
  export const AGENT_RUN_STATE_NAMESPACE = "prism.agent-run";
4
5
  export const AGENT_RUN_STATE_SCHEMA_VERSION = 1;
5
6
  export const DEFAULT_MAX_AGENT_RUN_STATE_BYTES = 256 * 1024;
@@ -241,6 +242,15 @@ function validateSessionState(sessionState) {
241
242
  if (!sessionState || typeof sessionState !== "object") {
242
243
  throw new AgentRunStateError("Malformed agent run session state");
243
244
  }
245
+ const bodies = sessionState.loadedSkillBodies;
246
+ if (bodies !== undefined) {
247
+ try {
248
+ validateLoadedSkillBodies(bodies);
249
+ }
250
+ catch (error) {
251
+ throw new AgentRunStateError(error instanceof Error ? error.message : String(error));
252
+ }
253
+ }
244
254
  const names = sessionState.loadedSkillNames;
245
255
  if (names === undefined)
246
256
  return;
@@ -1,6 +1,7 @@
1
1
  import { type StoredAgentRunState } from "./agent-run-state.js";
2
2
  import type { Agent, AgentConfig, AgentEvent, AgentRunResult, AgentRunStateOptions, AgentSession, AgentSessionConfig, CompactionOptions, CompactionResult, OwnershipScope, RunDecision, RunOptions, SessionEntry, SteerOptions, SubscribeOptions } from "./contracts.js";
3
3
  import { type AgentInput } from "./input.js";
4
+ import { type LoadedSkillBodiesEntry } from "./skill-load.js";
4
5
  export declare function createAgent(config: AgentConfig): Agent;
5
6
  export declare function createAgentSession(config: AgentSessionConfig & {
6
7
  readonly agent: Agent;
@@ -36,8 +37,14 @@ export declare class RuntimeAgentSession implements AgentSession {
36
37
  private activeGatedRound?;
37
38
  private activeLoopTurn;
38
39
  private readonly loadedSkills;
40
+ /** Plan 018 Task 6 (closeout `checkpoint-bodies`): persisted exact instructions, registry-independent. */
41
+ private restoredSkillBodies;
42
+ /** Skills of the current run (for the bodies snapshot); replaced at each run start. */
43
+ private activeRunSkills;
39
44
  /** Plan 015 Task 4: re-add persisted loaded-skill names (names only; bodies re-resolve on demand). */
40
45
  restoreLoadedSkills(names: readonly string[]): void;
46
+ /** Plan 018 Task 6: restore persisted loaded-skill bodies (already validated fail-closed at load). */
47
+ restoreLoadedSkillBodies(bodies: readonly LoadedSkillBodiesEntry[]): void;
41
48
  private ledgerChain;
42
49
  private ledgerFailure;
43
50
  private snapshotGeneration;
@@ -16,6 +16,7 @@ import { RunLimitError, RunLimitTracker, resolveRunLimits } from "./run-limits.j
16
16
  import { createMemorySessionStore, createSessionEntry, getSessionBranchEntries, rebuildSessionContext, } from "./session-stores.js";
17
17
  import { resolveActiveSkills } from "./skills.js";
18
18
  import { createLoadedSkillSet, resolveSkillsDisclosure } from "./skill-disclosure.js";
19
+ import { applyRestoredSkillBodies, snapshotLoadedSkillBodies, validateLoadedSkillBodies, } from "./skill-load.js";
19
20
  import { resolveToolResultFold } from "./tool-result-fold.js";
20
21
  import { assertStructuredOutputRequestSupported, resolveRunProviderOptions } from "./structured-output.js";
21
22
  import { composeSystemPrompt, mergeSystemPromptConfig } from "./system-prompts.js";
@@ -65,11 +66,22 @@ export class RuntimeAgentSession {
65
66
  activeGatedRound;
66
67
  activeLoopTurn = 1;
67
68
  loadedSkills = createLoadedSkillSet();
69
+ /** Plan 018 Task 6 (closeout `checkpoint-bodies`): persisted exact instructions, registry-independent. */
70
+ restoredSkillBodies = [];
71
+ /** Skills of the current run (for the bodies snapshot); replaced at each run start. */
72
+ activeRunSkills = [];
68
73
  /** Plan 015 Task 4: re-add persisted loaded-skill names (names only; bodies re-resolve on demand). */
69
74
  restoreLoadedSkills(names) {
70
75
  for (const name of names)
71
76
  this.loadedSkills.add(name);
72
77
  }
78
+ /** Plan 018 Task 6: restore persisted loaded-skill bodies (already validated fail-closed at load). */
79
+ restoreLoadedSkillBodies(bodies) {
80
+ validateLoadedSkillBodies(bodies);
81
+ this.restoredSkillBodies = bodies;
82
+ for (const entry of bodies)
83
+ this.loadedSkills.add(entry.name);
84
+ }
73
85
  ledgerChain = Promise.resolve();
74
86
  ledgerFailure;
75
87
  snapshotGeneration = 0;
@@ -255,6 +267,7 @@ export class RuntimeAgentSession {
255
267
  await this.rebuildHistory();
256
268
  const { registry, tools } = activeTools(this.agent.config.tools);
257
269
  const activeSkills = this.resolveRunSkills(options, tools);
270
+ this.activeRunSkills = activeSkills; // for the durable bodies snapshot (plan 018 Task 6)
258
271
  if (options.model && JSON.stringify(options.model) !== JSON.stringify(this.agent.config.model)) {
259
272
  await this.appendEntry(createSessionEntry({
260
273
  sessionId: this.id,
@@ -464,7 +477,7 @@ export class RuntimeAgentSession {
464
477
  inputBuilder: this.agent.config.inputBuilder,
465
478
  promptBuilder: this.agent.config.promptBuilder,
466
479
  contextProviders,
467
- skills: activeSkills,
480
+ skills: this.restoredSkillBodies.length ? applyRestoredSkillBodies(activeSkills, this.restoredSkillBodies) : activeSkills,
468
481
  skillsDisclosure: resolveSkillsDisclosure(options.skillsDisclosure, this.agent.config.skillsDisclosure),
469
482
  toolResultFold: resolveToolResultFold(options.toolResultFold, this.agent.config.toolResultFold),
470
483
  loadedSkills: this.loadedSkills,
@@ -1108,7 +1121,17 @@ export class RuntimeAgentSession {
1108
1121
  if (!durable)
1109
1122
  throw new AgentRunStateError("Durable run state is not configured");
1110
1123
  const persisted = durable.options.persistSessionState
1111
- ? { ...state, sessionState: { loadedSkillNames: this.loadedSkills.list() } }
1124
+ ? {
1125
+ ...state,
1126
+ sessionState: {
1127
+ loadedSkillNames: this.loadedSkills.list(),
1128
+ ...(durable.options.includeSkillBodies
1129
+ ? {
1130
+ loadedSkillBodies: snapshotLoadedSkillBodies(this.activeRunSkills, this.loadedSkills, this.restoredSkillBodies.length ? new Map(this.restoredSkillBodies.map((e) => [e.name, e.instructions])) : undefined),
1131
+ }
1132
+ : {}),
1133
+ },
1134
+ }
1112
1135
  : state;
1113
1136
  const saved = await saveAgentRunState({
1114
1137
  checkpoints: durable.options.checkpoints,
@@ -0,0 +1,58 @@
1
+ import type { ModelConfig, Usage } from "./contracts-core.js";
2
+ /**
3
+ * Prompt-cache telemetry surface (0.1.7, plan 019 Task 2).
4
+ *
5
+ * Dependency-free aggregator hosts attach to their `usage` `ProviderEvent`
6
+ * stream (or run-ledger usage records) to get per-provider/model cache
7
+ * statistics for tuning the `cache_aware` input layout. Explicit activation:
8
+ * nothing subscribes by import — the host calls `record()`.
9
+ */
10
+ /** Single provider/model statistics sample. */
11
+ export interface CacheTelemetrySample {
12
+ readonly provider: string;
13
+ readonly model: string;
14
+ readonly requests: number;
15
+ readonly cacheReadTokens: number;
16
+ readonly cacheWriteTokens: number;
17
+ readonly inputTokens: number;
18
+ /** Cached-input ratio across the sample (`cacheHitRate` math, aggregated). */
19
+ readonly hitRate?: number;
20
+ /** Estimated read-token savings via `cacheSavings` math; present only when
21
+ * the sample's model carries cost metadata (`ModelCost.input`/`cacheRead`). */
22
+ readonly estimatedSavings?: number;
23
+ readonly currency?: string;
24
+ }
25
+ /** Aggregated report. Samples are sorted by provider then model. */
26
+ export interface CacheTelemetryReport {
27
+ readonly samples: readonly CacheTelemetrySample[];
28
+ readonly overflowed: boolean;
29
+ readonly totalRequests: number;
30
+ readonly totalCacheReadTokens: number;
31
+ readonly totalCacheWriteTokens: number;
32
+ }
33
+ export interface CacheTelemetryOptions {
34
+ /** Distinct provider/model keys before excess keys collapse into the
35
+ * `__overflow__` bucket. Default {@link DEFAULT_CACHE_TELEMETRY_CAP}. */
36
+ readonly maxKeys?: number;
37
+ }
38
+ export interface CacheTelemetry {
39
+ /** Aggregate one usage record attributed to `model` (or an unknown bucket
40
+ * when no model is supplied). Rejects non-finite/negative token counts
41
+ * with {@link CacheTelemetryError}; validates before mutating. */
42
+ record(usage: Usage, model?: ModelConfig): void;
43
+ /** Snapshot of all samples (O(keys)); never throws. */
44
+ report(): CacheTelemetryReport;
45
+ /** Clear all samples (host rotation / long-run reset). */
46
+ reset(): void;
47
+ /** Number of distinct provider/model keys held (excluding the overflow bucket). */
48
+ readonly size: number;
49
+ }
50
+ /** Cardinality ceiling: keys beyond this collapse into `__overflow__`. */
51
+ export declare const DEFAULT_CACHE_TELEMETRY_CAP = 256;
52
+ /** Sample bucket key for provider/model keys beyond the cap. */
53
+ export declare const CACHE_TELEMETRY_OVERFLOW_KEY = "__overflow__";
54
+ export declare class CacheTelemetryError extends Error {
55
+ readonly code = "ERR_PRISM_CACHE_TELEMETRY";
56
+ constructor(message: string);
57
+ }
58
+ export declare function createCacheTelemetry(options?: CacheTelemetryOptions): CacheTelemetry;
@@ -0,0 +1,102 @@
1
+ import { cacheHitRate, cacheSavings } from "./cache-helpers.js";
2
+ /** Cardinality ceiling: keys beyond this collapse into `__overflow__`. */
3
+ export const DEFAULT_CACHE_TELEMETRY_CAP = 256;
4
+ /** Sample bucket key for provider/model keys beyond the cap. */
5
+ export const CACHE_TELEMETRY_OVERFLOW_KEY = "__overflow__";
6
+ export class CacheTelemetryError extends Error {
7
+ code = "ERR_PRISM_CACHE_TELEMETRY";
8
+ constructor(message) {
9
+ super(message);
10
+ this.name = "CacheTelemetryError";
11
+ }
12
+ }
13
+ function validateTokens(name, value) {
14
+ if (value === undefined)
15
+ return;
16
+ if (!Number.isSafeInteger(value) || value < 0) {
17
+ throw new CacheTelemetryError(`${name} must be a non-negative safe integer, got ${value}`);
18
+ }
19
+ }
20
+ function sampleFor(usage, model) {
21
+ if (model)
22
+ return { provider: model.provider, model: model.model };
23
+ // Provider-only aggregation: no model supplied, attribute to the unknown bucket.
24
+ return { provider: "unknown", model: "unknown" };
25
+ }
26
+ export function createCacheTelemetry(options = {}) {
27
+ const maxKeys = options.maxKeys ?? DEFAULT_CACHE_TELEMETRY_CAP;
28
+ if (!Number.isSafeInteger(maxKeys) || maxKeys < 1) {
29
+ throw new CacheTelemetryError(`maxKeys must be a positive safe integer, got ${options.maxKeys}`);
30
+ }
31
+ // ponytail: fixed per-provider/model key cap with a single __overflow__ bucket;
32
+ // if real deployments exceed it, upgrade to host-configurable caps or LRU eviction.
33
+ const samples = new Map();
34
+ let overflow = false;
35
+ let overflowSample;
36
+ function bucket(provider, model) {
37
+ const key = `${provider}\u0000${model}`;
38
+ let sample = samples.get(key);
39
+ if (sample)
40
+ return sample;
41
+ if (samples.size >= maxKeys) {
42
+ overflow = true;
43
+ if (!overflowSample) {
44
+ overflowSample = {
45
+ provider: CACHE_TELEMETRY_OVERFLOW_KEY,
46
+ model: CACHE_TELEMETRY_OVERFLOW_KEY,
47
+ requests: 0,
48
+ cacheReadTokens: 0,
49
+ cacheWriteTokens: 0,
50
+ inputTokens: 0,
51
+ };
52
+ }
53
+ return overflowSample;
54
+ }
55
+ sample = { provider, model, requests: 0, cacheReadTokens: 0, cacheWriteTokens: 0, inputTokens: 0 };
56
+ samples.set(key, sample);
57
+ return sample;
58
+ }
59
+ return {
60
+ record(usage, model) {
61
+ // Validate everything before mutating: a bad record mutates nothing.
62
+ validateTokens("usage.cacheReadTokens", usage.cacheReadTokens);
63
+ validateTokens("usage.cacheWriteTokens", usage.cacheWriteTokens);
64
+ validateTokens("usage.inputTokens", usage.inputTokens);
65
+ const { provider, model: modelName } = sampleFor(usage, model);
66
+ const sample = bucket(provider, modelName);
67
+ sample.requests += 1;
68
+ sample.cacheReadTokens += usage.cacheReadTokens ?? 0;
69
+ sample.cacheWriteTokens += usage.cacheWriteTokens ?? 0;
70
+ sample.inputTokens += usage.inputTokens ?? 0;
71
+ sample.hitRate = cacheHitRate({
72
+ cacheReadTokens: sample.cacheReadTokens,
73
+ inputTokens: sample.inputTokens,
74
+ });
75
+ if (model?.cost) {
76
+ // cacheSavings depends only on read tokens + cost metadata, so the
77
+ // aggregate equals the sum of per-call savings; feed it the totals
78
+ // to reuse the exact cache-helpers math rather than reimplementing it.
79
+ const savings = cacheSavings({ cacheReadTokens: sample.cacheReadTokens }, model);
80
+ sample.estimatedSavings = savings;
81
+ sample.currency = model.cost.currency;
82
+ }
83
+ },
84
+ report() {
85
+ const samplesAll = [...samples.values()].sort((a, b) => a.provider === b.provider ? a.model.localeCompare(b.model) : a.provider.localeCompare(b.provider));
86
+ const listed = overflowSample ? [...samplesAll, overflowSample] : samplesAll;
87
+ const totalRequests = listed.reduce((sum, s) => sum + s.requests, 0);
88
+ const totalCacheReadTokens = listed.reduce((sum, s) => sum + s.cacheReadTokens, 0);
89
+ const totalCacheWriteTokens = listed.reduce((sum, s) => sum + s.cacheWriteTokens, 0);
90
+ return { samples: listed, overflowed: overflow, totalRequests, totalCacheReadTokens, totalCacheWriteTokens };
91
+ },
92
+ reset() {
93
+ samples.clear();
94
+ overflow = false;
95
+ overflowSample = undefined;
96
+ },
97
+ get size() {
98
+ return samples.size;
99
+ },
100
+ };
101
+ }
102
+ //# sourceMappingURL=cache-telemetry.js.map
@@ -0,0 +1,37 @@
1
+ import type { Writable } from "node:stream";
2
+ export declare class ProviderAddUsageError extends Error {
3
+ }
4
+ export interface ProviderAddOptions {
5
+ /** npm-validated provider/package name; also the target directory name. */
6
+ readonly name: string;
7
+ readonly baseUrl: string;
8
+ /** Shell-safe identifier, e.g. `ACME_API_KEY`. Never a secret value. */
9
+ readonly envKey: string;
10
+ readonly model: string;
11
+ readonly force: boolean;
12
+ readonly help: boolean;
13
+ }
14
+ export interface ProviderAddRuntime {
15
+ readonly stdout: Writable;
16
+ readonly stderr: Writable;
17
+ /** Override template root (tests). Defaults to package `templates/provider`. */
18
+ readonly templatesRoot?: string;
19
+ /** Override package version stamped into the generated package.json. */
20
+ readonly packageVersion?: string;
21
+ /** Working directory used to resolve relative destinations. Defaults to process.cwd(). */
22
+ readonly cwd?: string;
23
+ }
24
+ export interface ProviderAddResult {
25
+ readonly targetDir: string;
26
+ readonly writtenFiles: readonly string[];
27
+ readonly name: string;
28
+ readonly totalBytes: number;
29
+ }
30
+ export declare function getProviderAddUsage(): string;
31
+ export declare const providerAddUsage: string;
32
+ export declare function parseProviderAddArgs(argv: readonly string[]): ProviderAddOptions;
33
+ export declare function runProviderAddCommand(argv: readonly string[], runtime: ProviderAddRuntime): Promise<number>;
34
+ export declare function createProviderProject(options: ProviderAddOptions, runtime?: ProviderAddRuntime): Promise<ProviderAddResult>;
35
+ export declare function defaultProviderTemplatesRoot(): string;
36
+ /** npm package-name rules plus traversal refusal. Throws `ProviderAddUsageError` on violation. */
37
+ export declare function validateProviderName(name: string): void;