@arnilo/prism 0.1.1 → 0.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,15 @@
1
1
  # Changelog
2
2
 
3
+ ## [0.1.3] - 2026-08-10
4
+
5
+ ### Changed
6
+ - **Release 0.1.3 (plan 015)** is the dead-code and deprecation hygiene patch on the frozen 0.1.x line, additive-only vs 0.1.2 (freeze manifest `scripts/phase15-freeze-manifest.json`). (1) **Benchmark-runner consolidation** (Task 1): one parameterized runner `scripts/benchmark.mjs --scenario <name>` replaces the per-version runners; the six live legs moved to `scripts/benchmark-scenarios/` as named scenarios (`phase6-postgres`, `phase7-postgres`, `phase8-loops-hitl`, `phase9-coding`, `phase10-acp`, `phase11-auth`) and the 0.1.0 envelope orchestrator composes them through the runner; **removed files**: `scripts/benchmark-0.0.{8,9,10,11,12,13,14,15,16}.mjs` and `scripts/benchmark-0.0.{9,10,11,12,13,14,15}.test.mjs` (orphaned, unreferenced by `npm test`); all `benchmark-*.json` evidence files kept byte-identical; the CI benchmark-schema leg now runs `scripts/benchmark.test.mjs`. (2) **Review-coverage archive** (Task 2): the 12 `docs/review-coverage-2026-07-*.md` per-phase evidence files moved to `docs/_evidence/` (tarball-excluded via the `files` field; index/migration/performance links updated; archived evidence is not part of the shipped docs surface). (3) **Non-blocking unused-code sweep** (Task 3): `npm run sweep:unused` runs tsc `--noUnusedLocals`/`--noUnusedParameters` over core + every workspace tsconfig plus a zero-dep dead-export scan (`scripts/dead-exports.mjs`), writes the combined report to `scripts/unused-sweep-report.txt`, and always exits 0; CI runs it as a `continue-on-error` step with a retained artifact; 43 internal unused diagnostics (22 test files + 13 source files) removed in-tree, public-but-unused exports are report-only (removal is the 0.1.5 breaking cut). (4) **Opt-in checkpoint persistence** (Task 4): durable runs may set `persistSessionState: true` on the run and resume options — the loaded-skill **name catalog** (≤64 names, ≤256 chars each, validated fail-closed on every save and load) rides the run-state checkpoint and is restored into the resumed session's `LoadedSkillSet`; skill **bodies are never persisted** and re-resolve from the live registry; flag off keeps the checkpoint shape byte-identical to 0.1.2. `@arnilo/prism-coding-agent` adds `createReadPathSetPersistence({ checkpoints, key, ownership })` for the read-before-write path set (≤1024 paths / ≤1024 chars each, CAS read-modify-write, cross-ownership restore fails closed). Store compatibility with 0.1.2: **compatible, no migration**; declaration surface additive-only vs the frozen 0.1.x contract.
7
+
8
+ ## [0.1.2] - 2026-08-10
9
+
10
+ ### Changed
11
+ - **Release 0.1.2 (plan 014)** is the Alibaba Cloud provider enrichment patch on the frozen 0.1.x line, additive-only vs 0.1.1 (freeze manifest `scripts/phase14-freeze-manifest.json`): (1) **embeddings** — `createAlibabaEmbedder` in `@arnilo/prism-provider-alibaba` over the OpenAI-compatible `POST {base}/embeddings` (text-embedding-v3/v4), a structural `Embedder` assignable to `@arnilo/prism-memory`'s without a dependency; inputs chunked at the DashScope cap (10/request), vectors in input order, dimensions 64–2048 (default 1024) + `encoding_format` passthrough, key resolved per call and redacted from errors; (2) **video input** — `file` blocks with `video/*` media types serialize to compatible-mode `video_url` content parts on Qwen-VL models, gated on the `file` input capability (`mapAlibabaModel` advertises `["text", "image", "file"]` for the qwen-vl family); (3) **documented deferrals** — document input (compatible path is the OpenAI Files API `file-extract` + `fileid://` reference, an upload/status lifecycle) and rerank (only workspace-dedicated `compatible-api/v1/reranks` exists, not on the public presets) are recorded in the verified decision table in [docs/providers/alibaba.md](docs/providers/alibaba.md) as demand-gated follow-ups; (4) **opt-in live probe** — `PRISM_LIVE_DASHSCOPE_KEY`-gated `test:live` script (skips when absent, never in CI). Store compatibility with 0.1.1: **compatible, no migration**; declaration surface additive-only vs the frozen 0.1.x contract.
12
+
3
13
  ## [0.1.1] - 2026-08-10
4
14
 
5
15
  ### Changed
@@ -18,6 +18,8 @@ export interface AgentRunLifecycleRequest {
18
18
  readonly signal?: AbortSignal;
19
19
  /** Adapter-selected capability; stored runs for another agent are non-enumerable. */
20
20
  readonly agentId?: string;
21
+ /** Opt-in (plan 015 Task 4): restore persisted loaded-skill names on resume. */
22
+ readonly persistSessionState?: boolean;
21
23
  }
22
24
  /** Bounded live-event options for a durable lifecycle resume. */
23
25
  export interface AgentRunLifecycleStreamRequest extends AgentRunLifecycleRequest, SubscribeOptions {
@@ -26,6 +26,7 @@ export function createAgentRunLifecycle(options) {
26
26
  ownership: request.ownership,
27
27
  fencingToken: options.fencingToken,
28
28
  definitionRevision: resolved.definitionRevision,
29
+ persistSessionState: request.persistSessionState,
29
30
  });
30
31
  },
31
32
  async *resumeStream(ref, resume, request = {}) {
@@ -42,6 +43,7 @@ export function createAgentRunLifecycle(options) {
42
43
  signal: request.signal,
43
44
  maxQueuedEvents: request.maxQueuedEvents,
44
45
  overflow: request.overflow,
46
+ persistSessionState: request.persistSessionState,
45
47
  });
46
48
  },
47
49
  };
@@ -34,7 +34,18 @@ export interface StoredAgentRunState extends AgentRunState {
34
34
  readonly revision: string;
35
35
  readonly snapshot: JsonValue;
36
36
  };
37
+ /**
38
+ * Opt-in session-level state (plan 015 Task 4): loaded-skill names only; bodies are
39
+ * never persisted and reload on demand from the live registry via `load_skill`.
40
+ * Absent by default (0.1.x checkpoints parse unchanged).
41
+ */
42
+ readonly sessionState?: {
43
+ readonly loadedSkillNames?: readonly string[];
44
+ };
37
45
  }
46
+ /** Session-state caps (plan 015 Task 4): bounded names charged against the run-state byte budget. */
47
+ export declare const MAX_PERSISTED_SKILL_NAMES = 64;
48
+ export declare const MAX_PERSISTED_SKILL_NAME_CHARS = 256;
38
49
  /** Revision stamps of the built-in loops; custom strategies declare their own `revision`. */
39
50
  export declare const BUILT_IN_LOOP_REVISIONS: Readonly<Record<string, string>>;
40
51
  /** Validate a strategy snapshot as JSON-compatible and package it for the durable envelope. */
@@ -6,6 +6,9 @@ export const DEFAULT_MAX_AGENT_RUN_STATE_BYTES = 256 * 1024;
6
6
  export const HARD_MAX_AGENT_RUN_STATE_BYTES = 1024 * 1024;
7
7
  const MAX_DEPTH = 32;
8
8
  const MAX_PROPERTIES = 256;
9
+ /** Session-state caps (plan 015 Task 4): bounded names charged against the run-state byte budget. */
10
+ export const MAX_PERSISTED_SKILL_NAMES = 64;
11
+ export const MAX_PERSISTED_SKILL_NAME_CHARS = 256;
9
12
  /** Revision stamps of the built-in loops; custom strategies declare their own `revision`. */
10
13
  export const BUILT_IN_LOOP_REVISIONS = {
11
14
  "single-shot": "1",
@@ -207,6 +210,7 @@ export function parseAgentRunState(value, version) {
207
210
  return boundState({ ...state, version }, HARD_MAX_AGENT_RUN_STATE_BYTES);
208
211
  }
209
212
  function boundState(state, maxBytes) {
213
+ validateSessionState(state.sessionState);
210
214
  checkShape(state, 0);
211
215
  let text;
212
216
  try {
@@ -230,4 +234,23 @@ function checkShape(value, depth) {
230
234
  for (const item of entries)
231
235
  checkShape(item, depth + 1);
232
236
  }
237
+ /** Fail-closed validation of the opt-in session-state block (load and save sides). */
238
+ function validateSessionState(sessionState) {
239
+ if (sessionState === undefined)
240
+ return;
241
+ if (!sessionState || typeof sessionState !== "object") {
242
+ throw new AgentRunStateError("Malformed agent run session state");
243
+ }
244
+ const names = sessionState.loadedSkillNames;
245
+ if (names === undefined)
246
+ return;
247
+ if (!Array.isArray(names) || names.length > MAX_PERSISTED_SKILL_NAMES) {
248
+ throw new AgentRunStateError(`Loaded-skill names exceed ${MAX_PERSISTED_SKILL_NAMES} entries`);
249
+ }
250
+ for (const name of names) {
251
+ if (typeof name !== "string" || name.length > MAX_PERSISTED_SKILL_NAME_CHARS) {
252
+ throw new AgentRunStateError(`Loaded-skill name exceeds ${MAX_PERSISTED_SKILL_NAME_CHARS} chars`);
253
+ }
254
+ }
255
+ }
233
256
  //# sourceMappingURL=agent-run-state.js.map
package/dist/agents.js CHANGED
@@ -73,6 +73,11 @@ async function prepareAgentRunResume(agent, ref, resume, options, signal) {
73
73
  throw new AgentRunStateError("Stale or non-suspended agent run resume");
74
74
  }
75
75
  const session = new RuntimeAgentSession({ agent, id: state.sessionId, leafId: state.leafId });
76
+ // Opt-in session-state restore (plan 015 Task 4): names only; bodies re-resolve from
77
+ // the live registry the next time the model (re)loads them via load_skill.
78
+ if (options.persistSessionState && state.sessionState?.loadedSkillNames) {
79
+ session.restoreLoadedSkills(state.sessionState.loadedSkillNames);
80
+ }
76
81
  if (resume.decision !== undefined && resume.decisions !== undefined) {
77
82
  throw new AgentDecisionError("ERR_PRISM_DECISION_INVALID", "Resume accepts exactly one of decision or decisions");
78
83
  }
@@ -507,6 +512,11 @@ class RuntimeAgentSession {
507
512
  activeGatedRound;
508
513
  activeLoopTurn = 1;
509
514
  loadedSkills = createLoadedSkillSet();
515
+ /** Plan 015 Task 4: re-add persisted loaded-skill names (names only; bodies re-resolve on demand). */
516
+ restoreLoadedSkills(names) {
517
+ for (const name of names)
518
+ this.loadedSkills.add(name);
519
+ }
510
520
  ledgerChain = Promise.resolve();
511
521
  ledgerFailure;
512
522
  snapshotGeneration = 0;
@@ -1539,9 +1549,12 @@ class RuntimeAgentSession {
1539
1549
  const durable = this.activeDurable;
1540
1550
  if (!durable)
1541
1551
  throw new AgentRunStateError("Durable run state is not configured");
1552
+ const persisted = durable.options.persistSessionState
1553
+ ? { ...state, sessionState: { loadedSkillNames: this.loadedSkills.list() } }
1554
+ : state;
1542
1555
  const saved = await saveAgentRunState({
1543
1556
  checkpoints: durable.options.checkpoints,
1544
- state,
1557
+ state: persisted,
1545
1558
  expectedVersion: durable.version,
1546
1559
  ownership: this.activeOwnership,
1547
1560
  fencingToken: durable.options.fencingToken,
@@ -656,6 +656,12 @@ export interface AgentRunStateOptions {
656
656
  readonly fencingToken?: number;
657
657
  /** Enables sticky auto-apply when a nested suspension first surfaces during this run. */
658
658
  readonly resumeNestedRun?: ResumeNestedRun;
659
+ /**
660
+ * Opt-in (plan 015 Task 4): persist the session's loaded-skill names in the run-state
661
+ * checkpoint and restore them on resume. Names only — bodies reload via `load_skill`.
662
+ * Default off: checkpoint shape is identical to 0.1.2.
663
+ */
664
+ readonly persistSessionState?: boolean;
659
665
  }
660
666
  /** Versioned, redacted checkpoint payload. Treat as opaque except status/version/interruption. */
661
667
  export interface AgentRunState {
@@ -686,6 +692,8 @@ export interface AgentRunResumeOptions {
686
692
  readonly fencingToken?: number;
687
693
  /** Routes root decisions for nested-run approvals back to the child (e.g. supervisor). */
688
694
  readonly resumeNestedRun?: ResumeNestedRun;
695
+ /** Opt-in (plan 015 Task 4): restore persisted loaded-skill names into the resumed session catalog. */
696
+ readonly persistSessionState?: boolean;
689
697
  }
690
698
  /** Bounded, abortable options for `resumeAgentRunStream()`. */
691
699
  export interface AgentRunResumeStreamOptions extends AgentRunResumeOptions, SubscribeOptions {
package/dist/index.d.ts CHANGED
@@ -105,5 +105,5 @@ export { createToolParameterValidator, createToolRegistry, dispatchToolCall, fil
105
105
  export type { ResolvedUseCaseModel, ResolveUseCaseModelInput, UseCaseModelBinding, } from "./use-case-model.js";
106
106
  export { resolveUseCaseModel, resolveUseCaseModelBinding, useCaseCredentialProviderId, } from "./use-case-model.js";
107
107
  export declare const name = "prism";
108
- export declare const version = "0.1.1";
108
+ export declare const version = "0.1.3";
109
109
  export declare const description = "Agent harness for AI providers, agents, sessions, and tools.";
package/dist/index.js CHANGED
@@ -57,6 +57,6 @@ export { DEFAULT_TOOL_RESULT_FOLD_MAX_SUMMARY_BYTES, DEFAULT_TOOL_RESULT_FOLD_MI
57
57
  export { createToolParameterValidator, createToolRegistry, dispatchToolCall, filterTools } from "./tools.js";
58
58
  export { resolveUseCaseModel, resolveUseCaseModelBinding, useCaseCredentialProviderId, } from "./use-case-model.js";
59
59
  export const name = "prism";
60
- export const version = "0.1.1";
60
+ export const version = "0.1.3";
61
61
  export const description = "Agent harness for AI providers, agents, sessions, and tools.";
62
62
  //# sourceMappingURL=index.js.map
@@ -10,10 +10,12 @@ the release tree before cutting 1.0. The decision to cut 1.0 stays with the
10
10
  operator after operator-gated legs run in a protected environment and Phase
11
11
  12 demand evidence exists.
12
12
 
13
- Evidence trail: [`docs/review-coverage-2026-07-26-phase-11.md`](./review-coverage-2026-07-26-phase-11.md)
13
+ Evidence trail: [`docs/_evidence/review-coverage-2026-07-26-phase-11.md`](./_evidence/review-coverage-2026-07-26-phase-11.md)
14
14
  (addenda 0–9), [`docs/release-and-install.md`](./release-and-install.md),
15
15
  [`docs/migration.md`](./migration.md), [`docs/performance.md`](./performance.md),
16
16
  [`docs/public-contracts.md`](./public-contracts.md) (frozen 0.1.x contract).
17
+ The per-phase review-coverage evidence archive lives in [`docs/_evidence/`](./_evidence/)
18
+ (plans 067–079, releases 0.0.4–0.0.16; tarball-excluded, kept in-repo for audit).
17
19
  Historical release lines (0.0.16 floor → 0.0.27 Phase 10 ACP interop → 0.1.0)
18
20
  keep their per-phase evidence in the pages above; this page records the 0.1.1
19
21
  snapshot (plan 013) with the 0.1.0 table below as the previous line.
@@ -130,7 +130,7 @@ Identity is optional. Hosts that only set `ownership` keep prior behavior. When
130
130
  - Delegation only narrows scopes; tenant/account/user cannot widen on propagation.
131
131
  - Credential refs never expand to secrets in events, ledgers, or telemetry attributes.
132
132
  - Checks are O(fields) and network-free in core; remote auth stays in the host verifier.
133
- - Raising hard caps requires updating `docs/review-coverage-2026-07-23-phase-8.md`, tests, and docs.
133
+ - Raising hard caps requires updating `docs/_evidence/review-coverage-2026-07-23-phase-8.md`, tests, and docs.
134
134
 
135
135
  ## Related APIs
136
136
 
@@ -196,7 +196,7 @@ if (result.status === "suspended") {
196
196
  }
197
197
  ```
198
198
 
199
- Resume requires exact checkpoint ownership, version, agent fingerprint, and revision. The fingerprint hashes the agent id/name, `definitionRevision`, model, instructions, system-prompt contributions, skills (name/instructions/tool names), tool definitions (name/parameters/exclusive), guardrail definitions (name/stage/revision), and loop strategy — changing any of them without bumping `definitionRevision` fails resume closed instead of silently continuing with different agent semantics. Prism CAS-claims approval before work, rechecks normal guardrail/permission/validation/limit paths, and marks a pending tool dispatched before its side effect. `createAgentRunLifecycle()` wraps the same core path for server/MCP hosts: adapters pass only authorized ownership, status returns only `{ state, version }`, and `resolveAgent()` supplies current agent/revision. `resumeStream()` uses that same claim path and bounded subscriber, so adapters do not poll or duplicate resume logic. Remote restart requires both checkpoint and session stores to be durable. A crash after that mark is ambiguous and is never replayed automatically; use host tool idempotency keyed by `runId`/`toolCallId` or resolve it manually. Checkpoints contain bounded redacted state plus session/leaf references, never provider objects, callbacks, signals, credentials, or raw secrets. State is bounded at save by `runState.maxStateBytes` (default 256 KB, at most the 1 MB hard cap); load bounds against the 1 MB hard cap only, so state saved with a raised limit stays resumable while oversized records are still rejected. Built-in loop options are durable; custom `AgentLoopStrategy` instances are durable when they declare `snapshot`/`restore` hooks (see [Agent loops § Durable runs](agent-loops.md#durable-runs)) and reject before provider work otherwise.
199
+ Resume requires exact checkpoint ownership, version, agent fingerprint, and revision. The fingerprint hashes the agent id/name, `definitionRevision`, model, instructions, system-prompt contributions, skills (name/instructions/tool names), tool definitions (name/parameters/exclusive), guardrail definitions (name/stage/revision), and loop strategy — changing any of them without bumping `definitionRevision` fails resume closed instead of silently continuing with different agent semantics. Prism CAS-claims approval before work, rechecks normal guardrail/permission/validation/limit paths, and marks a pending tool dispatched before its side effect. `createAgentRunLifecycle()` wraps the same core path for server/MCP hosts: adapters pass only authorized ownership, status returns only `{ state, version }`, and `resolveAgent()` supplies current agent/revision. `resumeStream()` uses that same claim path and bounded subscriber, so adapters do not poll or duplicate resume logic. Remote restart requires both checkpoint and session stores to be durable. A crash after that mark is ambiguous and is never replayed automatically; use host tool idempotency keyed by `runId`/`toolCallId` or resolve it manually. Checkpoints contain bounded redacted state plus session/leaf references, never provider objects, callbacks, signals, credentials, or raw secrets. State is bounded at save by `runState.maxStateBytes` (default 256 KB, at most the 1 MB hard cap); load bounds against the 1 MB hard cap only, so state saved with a raised limit stays resumable while oversized records are still rejected. Since 0.1.3 (plan 015 Task 4), durable runs may opt in to session-state persistence with `persistSessionState: true` on both the run and resume options: the loaded-skill **name catalog** (≤64 names, ≤256 chars each) rides the checkpoint and is restored into the resumed session's `LoadedSkillSet`; skill **bodies are never persisted** and re-resolve from the live registry via `load_skill`. Default off keeps the checkpoint shape byte-identical to 0.1.2. Built-in loop options are durable; custom `AgentLoopStrategy` instances are durable when they declare `snapshot`/`restore` hooks (see [Agent loops § Durable runs](agent-loops.md#durable-runs)) and reject before provider work otherwise.
200
200
 
201
201
  ## Secure composition
202
202
 
@@ -206,6 +206,19 @@ const write = createWriteTool(cwd, { requireReadBeforeWrite: true, readPathSet:
206
206
  const edit = createEditTool(cwd, { requireReadBeforeWrite: true, readPathSet: readPaths });
207
207
  ```
208
208
 
209
+ Since 0.1.3 (plan 015 Task 4) hosts may opt in to persisting the set across restarts via the host-owned `CheckpointStore`:
210
+
211
+ ```ts
212
+ import { createReadPathSet, createReadPathSetPersistence } from "@arnilo/prism-coding-agent";
213
+
214
+ const readPaths = createReadPathSet();
215
+ const persistence = createReadPathSetPersistence({ checkpoints, key: sessionId, ownership });
216
+ await persistence.restore(readPaths); // on session attach (returns restored count)
217
+ await persistence.save(readPaths); // after reads, before session close
218
+ ```
219
+
220
+ Names only (paths are bounded at 1024 entries / 1024 chars each; larger sets fail closed with no partial write). Records live under the `prism.coding-agent.read-path-set` namespace keyed by session id, and `ownership` is part of the trust boundary: restoring under a different tenant/user throws instead of leaking paths. Default is **off** — the set stays in-memory unless the host wires the helper explicitly.
221
+
209
222
  ### `edit`
210
223
 
211
224
  Precise text replacement in an existing file via exact-then-fuzzy matching.
@@ -170,7 +170,7 @@ Catalog caps: **64** entries default / **256** hard; descriptions **512 B** defa
170
170
  ```ts
171
171
  import { assembleProviderInput, createLoadedSkillSet } from "@arnilo/prism";
172
172
 
173
- const loaded = createLoadedSkillSet(); // session-owned; not checkpoint-persisted in 0.0.20
173
+ const loaded = createLoadedSkillSet(); // session-owned; opt-in checkpoint-persisted via runState.persistSessionState (names only) since 0.1.3
174
174
  const request = await assembleProviderInput({
175
175
  model,
176
176
  input: "Hi",
@@ -256,7 +256,7 @@ Use `activateAllCapabilities: true` only as a temporary all-skills/all-tools com
256
256
  - Skill registry lookup is `Map`-backed, and selection is linear in requested skills plus active tools. Strict duplicate mode adds one O(1) `Map.has()` check during registration only.
257
257
  - Progressive catalog render is O(active skills) with byte/count caps; `load_skill` lookup is O(1). Budget eviction over context/skills is O(n log n) worst case.
258
258
  - `load_skill` cannot grant tools; loaded instructions are untrusted text bounded by hard caps. `toolResultFold` summarizer output is untrusted and capped; failures keep raw tool results.
259
- - Loaded-skill names are session-scoped in memory only in 0.0.20 — not checkpoint-persisted; new sessions start catalog-only until reload.
259
+ - Loaded-skill names are session-scoped. Since 0.1.3 (plan 015 Task 4) a durable run may opt in to persistence with `runState.persistSessionState: true` (and the same flag on resume options): the name catalog rides the run-state checkpoint (≤64 names, ≤256 chars each, charged against `maxStateBytes`) and is restored into the session `LoadedSkillSet` on resume, so progressive disclosure survives restart. **Bodies are never persisted** — they re-resolve from the live skill registry the next time the model loads the skill. Default off: checkpoint shape is identical to 0.1.2.
260
260
  - These helpers perform no provider calls, tool execution, resource loading, package discovery, filesystem/network access, retries, timers, or watchers by themselves.
261
261
  - Context and skill output is host/extension data. Do not include secrets unless the host explicitly accepts that prompt exposure.
262
262
  - Active tools remain host-supplied; skills and middleware do not activate tools or grant permissions. Use `duplicate: "error"` when loading third-party skills to prevent silent name shadowing.
package/docs/index.md CHANGED
@@ -129,18 +129,7 @@ Prism is a TypeScript/Node.js agent harness. Host apps and extension packages ow
129
129
  - [Ponytail behavior integration](ponytail.md): optional `@arnilo/prism-ponytail` — upstream Ponytail skills/commands, `ponytail-mode` injector, session `ponytail-mode` persistence; resolves peer `@dietrichgebert/ponytail` or `upstreamPath`; opt-in (not in code/sdk profiles).
130
130
 
131
131
  ## Release and install
132
- - [Release and install](release-and-install.md): current **0.1.1** 49-package graph (root + 48 workspace packages; plan 013 post-release hardening on the frozen 0.1.x line — build single-flight, MCP SSE relay test, combined coverage summary, canonical manifest-count narrative, ACP modes/config persistence guidance; Phase 12 release-candidate hardening; plan 012 — freeze manifest, compatibility matrix, upgrade matrix, packed-install e2e journeys, restart-recovery evidence, capacity envelopes, security policy), exact-peer/install/tarball rules, deterministic resumable publication and publish dry-run, frozen 0.1.x compatibility and support matrix (Node/PostgreSQL/platform/provider/protocol pins and unsupported combinations, machine-checked against `scripts/phase12-freeze-manifest.json`), protected PostgreSQL gate, pinned supply-chain gates, offline tests, the 0.0.15 provider/AI-SDK/RAG/memory protected live-canary matrix, and sandbox-browser Docker/Playwright gates.
132
+ - [Release and install](release-and-install.md): current **0.1.3** 49-package graph (root + 48 workspace packages; plan 015 dead-code and deprecation hygiene on the frozen 0.1.x line — parameterized benchmark runner `scripts/benchmark.mjs` absorbing the per-version runners, archived review-coverage evidence in `docs/_evidence/`, non-blocking unused-code sweep `npm run sweep:unused`, opt-in checkpoint persistence for loaded-skill names and read-path sets; plan 014 Alibaba provider enrichment — embeddings, video input, verified compatible-mode surface decision table; plan 013 post-release hardening — build single-flight, MCP SSE relay test, combined coverage summary, canonical manifest-count narrative, ACP modes/config persistence guidance; Phase 12 release-candidate hardening; plan 012 — freeze manifest, compatibility matrix, upgrade matrix, packed-install e2e journeys, restart-recovery evidence, capacity envelopes, security policy), exact-peer/install/tarball rules, deterministic resumable publication and publish dry-run, frozen 0.1.x compatibility and support matrix (Node/PostgreSQL/platform/provider/protocol pins and unsupported combinations, machine-checked against `scripts/phase12-freeze-manifest.json`), protected PostgreSQL gate, pinned supply-chain gates, offline tests, the 0.0.15 provider/AI-SDK/RAG/memory protected live-canary matrix, and sandbox-browser Docker/Playwright gates.
133
133
  - [0.1.0 / 1.0 readiness gates](0.1.0-readiness.md): command-per-gate 1.0 readiness table — frozen API surface + compat gate, migration/docs tripwires, budget table, live-suite matrix, security matrix, current-line status (**0.0.23** published target), signed-publication/live-canary prerequisites for 1.0, and Phase 12 demand-evidence entry criteria.
134
- - [Review coverage (2026-07-26 Phase 11)](review-coverage-2026-07-26-phase-11.md): Plan 079 evidence freeze — baseline size/startup/benchmark budgets, hotspot domain extraction table, confirmed duplication survivors (redactor/cleanJson/row-codecs/checkpoints/exec-runner/approval/ownership), profile adoption recommendations, and tarball artifact-diet findings for 0.0.16.
135
- - [Review coverage (2026-07-26 Phase 10)](review-coverage-2026-07-26-phase-10.md): Plan 078 evidence freeze — OpenAI hosted tools/continuation/realtime, AI SDK version matrix, remaining provider metadata parity, RAG replaceSource/loaders/parsers/reranker/provenance/ingestion-status, memory export/rebuild/conformance, and 0.0.15 (43 → 43 manifests; no new package) release gates.
136
- - [Review coverage (2026-07-25 Phase 9)](review-coverage-2026-07-25-phase-9.md): Plan 077 evidence freeze — conversation service, memory consent/lifecycle, artifact co-work review, AG-UI co-work events, scoped M365/GWS OAuth, browser checkpoint composition, and deny-by-default device contracts for 0.0.14 (41 → 43 manifests; only the two provider packages are new).
137
- - [Review coverage (2026-07-23 Phase 8)](review-coverage-2026-07-23-phase-8.md): Plan 076 evidence freeze — enterprise identity/policy/router packages, Azure/Bedrock/Vertex adapters, server deployment seams, persistence lifecycle hooks, and M365/GWS work-connector bounds for 0.0.13.
138
- - [Review coverage (2026-07-22 Phase 7)](review-coverage-2026-07-22-phase-7.md): Plan 075 evidence freeze — AG-UI/ACP package boundary, streamed durable resume, bounded replay/projection, coding compaction preset, and provider-authorized OAuth policy for 0.0.12.
139
- - [Review coverage (2026-07-22 Phase 6)](review-coverage-2026-07-22-phase-6.md): Plan 074 evidence freeze — SessionIndex/search, contextBudget, native Anthropic/Google packages, goal→verify, steer, ask_user_decision (multi/free-text/suspend), finite limits, threats, and 0.0.11 release gates.
140
- - [Review coverage (2026-07-21 Phase 5)](review-coverage-2026-07-21-phase-5.md): Plan 073 evidence freeze — unified workspace modes, primitive ownership, reused finite limits, threats, and 0.0.10 release gates.
141
- - [Review coverage (2026-07-20 Phase 4)](review-coverage-2026-07-20-phase-4.md): Plan 072 evidence freeze — revised coding/browser-only scope, external revisions, primitive ownership, finite limits, threats, and 0.0.9 release gates.
142
- - [Review coverage (2026-07-19 Phase 3)](review-coverage-2026-07-19-phase-3.md): Plan 070 evidence freeze — exact protocol/vendor references, capability/primitive/limit matrices, supported boundaries, and 0.0.8 release evidence.
143
- - [Review coverage (2026-07-17 provider validation)](review-coverage-2026-07-17-provider-validation.md): Plan 067 evidence freeze — P0–P2 re-verification owners, seven first-party provider packages mapped to official-doc URLs, Pi secondary refs, cache/thinking/discovery surfaces, credential canaries, and use-case model-binding inventory.
144
- - [Review coverage (2026-07-15)](review-coverage-2026-07-15.md): frozen 0.0.5 finding/feature ownership, existing-primitive inventory, package decisions, threat boundaries, exclusions, and measured Phase 0 baseline.
145
- - [Review coverage (2026-07-14)](review-coverage-2026-07-14.md): traceability matrix linking review findings and bug-report fixes to plan tasks, tests, and documentation for release 0.0.4.
134
+ - [Review coverage archive](_evidence/): per-phase evidence freezes (plans 067–079, releases 0.0.4–0.0.16) — traceability matrices, provider validation, capability/primitive/limit matrices, benchmark budgets, and artifact-diet findings; tarball-excluded, kept in-repo for audit.
146
135
 
package/docs/migration.md CHANGED
@@ -1,5 +1,9 @@
1
1
  # Migration guide
2
2
 
3
+ ## 0.1.2 → 0.1.3 dead-code and deprecation hygiene (additive, no migration)
4
+
5
+ Release **0.1.3** (plan 015) is the dead-code and deprecation hygiene patch on the frozen 0.1.x line: benchmark-runner consolidation (one parameterized `scripts/benchmark.mjs --scenario <name>` replaces the per-version runners; 16 orphaned `benchmark-0.0.{8..16}` runner/test files removed, all `benchmark-*.json` evidence kept), the 12 `docs/review-coverage-2026-07-*.md` evidence files archived to the tarball-excluded `docs/_evidence/`, a non-blocking unused-code sweep (`npm run sweep:unused`, always exits 0, report to `scripts/unused-sweep-report.txt`), and opt-in checkpoint persistence (`persistSessionState: true` on durable run/resume options persists the loaded-skill name catalog ≤64 names in the run-state checkpoint and restores it on resume — bodies re-resolve from the live registry; `createReadPathSetPersistence` in `@arnilo/prism-coding-agent` persists the read-before-write path set through the host `CheckpointStore`, ≤1024 paths, ownership-scoped). **Store compatibility: compatible** — the persisted run-state schema stays at version 1 (the optional `sessionState` field is absent by default, so 0.1.2 checkpoints parse unchanged and opt-out checkpoints are byte-identical); no upgrade or rollback step exists (rollback = restore the 0.1.2 manifests/tag; stores never change). Declaration surface is additive-only vs the frozen 0.1.x contract (`scripts/compat-baseline` regenerated at 0.1.3 with zero breaking deltas, enforced by `node scripts/release.mjs gate`). No breaking defaults.
6
+
3
7
  ## 0.1.0 → 0.1.1 post-release hardening (additive, no migration)
4
8
 
5
9
  Release **0.1.1** (plan 013) is a hardening patch on the frozen 0.1.x line: five scoped fixes — build single-flight (`npm run clean` removed from `npm run build`, standalone), deterministic MCP SSE relay test (`relayStatelessBody` internal export in `@arnilo/prism-mcp`, not in the package entry surface), combined core + workspace coverage summary (`scripts/coverage-summary.mjs`), canonical manifest-count narrative (49 publishable manifests = root + 48 workspace packages), and ACP modes/config ownership-scoped persistence guidance (the agent never persists `modeId`/`configValues`; host stores MUST key by `sessions.ownership`). **Store compatibility: compatible** — no persisted shape, event schema, or default behavior changed; the 0.0.28 → 0.1.0 → 0.1.1 lines all stay on the same checksum-protected contract, so no upgrade or rollback step exists (rollback = restore the 0.1.0 manifests/tag; stores never change). Declaration surface is additive-only vs the frozen 0.1.x contract (`scripts/compat-baseline` regenerated at 0.1.1 with zero breaking deltas, enforced by `node scripts/release.mjs gate`). No breaking defaults.
@@ -261,7 +265,7 @@ Prism 0.0.6 preserves documented 0.0.3 agent construction except for two intenti
261
265
 
262
266
  ## 0.0.15 → 0.0.16 simplification, shared survivors, and release gates (additive, pre-release)
263
267
 
264
- Release **0.0.16** is a simplification/readiness release: no runtime behavior changes, no package retired, and the only public-surface change is one additive export plus one internal package. The published root tarball is smaller and the release now runs offline pre-publish gates. See [Phase 11 evidence](review-coverage-2026-07-26-phase-11.md).
268
+ Release **0.0.16** is a simplification/readiness release: no runtime behavior changes, no package retired, and the only public-surface change is one additive export plus one internal package. The published root tarball is smaller and the release now runs offline pre-publish gates. See [Phase 11 evidence](_evidence/review-coverage-2026-07-26-phase-11.md).
265
269
 
266
270
  ### New shared export: `resolveRedactor` (additive)
267
271
 
@@ -318,7 +322,7 @@ RAG retrieval now optionally accepts host-owned `Reranker`; it receives redacted
318
322
 
319
323
  ## 0.0.13 → 0.0.14 personal/work-agent conversations, co-work review, and channel/device gates (additive, pre-release)
320
324
 
321
- Release **0.0.14** is strictly additive: every surface extends a shipped package and reuses the AG-UI adapter shipped in 0.0.12. The only new packages are two optional provider adapters (41 → 43 manifests): `@arnilo/prism-provider-alibaba` and `@arnilo/prism-provider-ollama`, both enrolled via the `@arnilo/prism-providers` family. No permission broadening — channel/device/co-work features cannot widen consent, memory, network, file, browser, connector, or tool permissions (roadmap gate 8). See [Phase 9 evidence](review-coverage-2026-07-25-phase-9.md).
325
+ Release **0.0.14** is strictly additive: every surface extends a shipped package and reuses the AG-UI adapter shipped in 0.0.12. The only new packages are two optional provider adapters (41 → 43 manifests): `@arnilo/prism-provider-alibaba` and `@arnilo/prism-provider-ollama`, both enrolled via the `@arnilo/prism-providers` family. No permission broadening — channel/device/co-work features cannot widen consent, memory, network, file, browser, connector, or tool permissions (roadmap gate 8). See [Phase 9 evidence](_evidence/review-coverage-2026-07-25-phase-9.md).
322
326
 
323
327
  | Surface | Before (0.0.13) | After (0.0.14) |
324
328
  | --- | --- | --- |
@@ -350,7 +354,7 @@ Optional `@arnilo/prism-policy` records allow/deny/modify/approval decisions wit
350
354
  | Model governance | Host wraps resolver ad hoc | Optional `@arnilo/prism-model-router` before provider I/O |
351
355
  | Work connectors | n/a | Optional `@arnilo/prism-work-tools` M365 + GWS; draft-then-approve; hard-coded CLI argv |
352
356
 
353
- **Deferred to 0.0.14+:** conversation storage/service, Studio/control plane, internal auth DB, Redis/SQS queue adapters, local Office binaries. See [Phase 8 evidence](review-coverage-2026-07-23-phase-8.md).
357
+ **Deferred to 0.0.14+:** conversation storage/service, Studio/control plane, internal auth DB, Redis/SQS queue adapters, local Office binaries. See [Phase 8 evidence](_evidence/review-coverage-2026-07-23-phase-8.md).
354
358
 
355
359
  Benchmark placeholder: `node scripts/benchmark-0.0.13.mjs` (release Task 10). Caps documented in [Performance limits](performance.md).
356
360
 
@@ -378,7 +382,7 @@ Release **0.0.12** adds optional `@arnilo/prism-ag-ui` (root AG-UI and stable `.
378
382
 
379
383
  **Host actions:** install the optional package only when a frontend protocol is needed; keep authorization, session/thread/run mapping, durable correlation, storage, redaction, and projection in the host. Reject frontend tools and state unless an explicit host policy accepts them. For a durable approval, persist protocol-run correlation before exposing the exact `${runId}:${version}` interrupt, then resume through the lifecycle with current ownership/version. Configure a redacted `ProductionPersistenceStore` before enabling replay. Use `createCodingCompactionStrategy()` only when the host already supplies a summary provider/model.
380
384
 
381
- AG-UI defaults/hard caps: request 64 KiB/1 MiB; projected event 64 KiB/1 MiB; replay page 100/500; subscriber queue 128/4096; stream 10k/100k events and 10/64 MiB; wall time 120 seconds/30 minutes. Benchmark results remain a release-gate placeholder: `node scripts/benchmark-0.0.12.mjs` lands in Task 8. See [Frontend interoperability](ag-ui.md), [LLM compaction package](compaction-llm.md), and [Phase 7 evidence](review-coverage-2026-07-22-phase-7.md).
385
+ AG-UI defaults/hard caps: request 64 KiB/1 MiB; projected event 64 KiB/1 MiB; replay page 100/500; subscriber queue 128/4096; stream 10k/100k events and 10/64 MiB; wall time 120 seconds/30 minutes. Benchmark results remain a release-gate placeholder: `node scripts/benchmark-0.0.12.mjs` lands in Task 8. See [Frontend interoperability](ag-ui.md), [LLM compaction package](compaction-llm.md), and [Phase 7 evidence](_evidence/review-coverage-2026-07-22-phase-7.md).
382
386
 
383
387
  ## 0.0.10 → 0.0.11 coding harness fundamentals (additive)
384
388
 
@@ -394,7 +398,7 @@ Release **0.0.11** adds SessionIndex/search, assembler `contextBudget`, native A
394
398
  | Ask user | n/a | Opt-in `createAskUserDecisionTool`; durable `suspendAskUserDecision` (no new agent interruption kinds) |
395
399
  | Structured output + tools | Native schema attached every GVR provider turn | Opt-in `structuredOutputTiming: "final-turn-only"` (default `"every-turn"`): tool-eligible turns omit schema; artifact/revision turns schema-on / tools-off |
396
400
 
397
- **Host actions:** reopen SQLite/Postgres stores so migration 004 applies; set `metadata.workspaceRoot` when filtering by workspace; wire Anthropic/Google packages explicitly; do not expect JSONL search. Benchmarks: `scripts/benchmark-0.0.11.mjs` (lands with release Task 13). See [Phase 6 evidence](review-coverage-2026-07-22-phase-6.md).
401
+ **Host actions:** reopen SQLite/Postgres stores so migration 004 applies; set `metadata.workspaceRoot` when filtering by workspace; wire Anthropic/Google packages explicitly; do not expect JSONL search. Benchmarks: `scripts/benchmark-0.0.11.mjs` (lands with release Task 13). See [Phase 6 evidence](_evidence/review-coverage-2026-07-22-phase-6.md).
398
402
 
399
403
  ## 0.0.9 / 0.0.96 → 0.0.10 coding workspace modes (breaking composition)
400
404
 
@@ -165,7 +165,7 @@ The same run accepted 1,000 rate claims, accumulated 16,000 budget tokens, grant
165
165
  Release 0.0.16 is a simplification/readiness release: it added no performance-affecting code, so the six network-free scenario medians are held at the 0.0.15 baseline and the win is a smaller published artifact. Budgets live in `scripts/budgets.json` (measured baselines + tolerance) and are enforced two ways:
166
166
 
167
167
  - **Fast gate (every `npm test`)** — `scripts/budget-gate.test.mjs` re-packs the root tarball (`npm pack --dry-run --json`) and fails if packed bytes, unpacked bytes, or file count exceed baseline + 5%, and fails if cold-process `import('./dist/index.js')` exceeds the 250 ms sanity ceiling. Negative fixtures prove an inflated/regressed value fails.
168
- - **Release evidence runner** — `node scripts/benchmark-0.0.16.mjs` re-measures root pack + startup, spawns `benchmark-0.0.15.mjs` for the six scenario medians (reused unchanged), compares every value to `budgets.json` (throughput floor / latency ceiling at ±25%), prints the evidence report below, and exits non-zero on any regression.
168
+ - **Release evidence runner** — `node scripts/benchmark-0.0.16.mjs` re-measures root pack + startup, spawns `benchmark-0.0.15.mjs` for the six scenario medians (reused unchanged), compares every value to `budgets.json` (throughput floor / latency ceiling at ±25%), prints the evidence report below, and exits non-zero on any regression. *(0.1.3, plan 015 Task 1: the per-version runners were consolidated into the parameterized runner `scripts/benchmark.mjs --scenario <name>`; the 0.0.16 evidence below is the historical record, budgets.json medians unchanged.)*
169
169
 
170
170
  **Artifact diet (the 0.0.16 finding).** The Task 1 tarball deny list dropped the historical `docs/review-coverage-*.md` (11 files, 283,022 bytes) from the root package: the root tarball went from **659,478 packed / 2,310,686 unpacked / 281 files** (0.0.15) to a budgeted **≈575,680 packed / 2,043,402 unpacked / 270 files**. The per-release `scripts/benchmark-0.0.*.mjs` history never shipped in artifacts (root `files` is `dist`/`docs`/`templates`/`CHANGELOG.md` only — zero `scripts/` entries packed), so no archive move was needed; `benchmark-0.0.16.mjs` consolidates the current evidence behind one budget-gating runner.
171
171
 
@@ -248,7 +248,7 @@ No network, credentials, provider summary call, durable database, or live subscr
248
248
 
249
249
  ## Release 0.0.11 session search / context budget / steer caps
250
250
 
251
- Finite caps (defaults / hard) — full matrix in [Phase 6 evidence](review-coverage-2026-07-22-phase-6.md):
251
+ Finite caps (defaults / hard) — full matrix in [Phase 6 evidence](_evidence/review-coverage-2026-07-22-phase-6.md):
252
252
 
253
253
  | Resource | Default / hard |
254
254
  | --- | --- |
@@ -315,7 +315,7 @@ Durable coding plan/checkpoint defaults/hard caps: plan Markdown 256 KiB/1 MiB;
315
315
 
316
316
  Browser automation defaults/hard caps from `@arnilo/prism-browser`: pages 4/16; actions 100/256; queued actions 16/64; snapshot refs 2,000/10,000; depth 30/100; snapshot bytes 256 KiB/2 MiB; navigation 30 s/120 s; action 10 s/60 s; wait 30 s/120 s; run wall 20 min/30 min; popups 4/16; dialogs 16/64; listeners 64/256; action input 64 KiB/256 KiB; close grace 5 s/30 s; network requests 1,000/10,000 with 10/32 redirects per request and 8/32 WebSockets; screenshots 16/64 with 16/64 megapixels and 10 MiB/32 MiB encoded; uploads 8/32 files, 16 MiB/64 MiB each, 64 MiB/256 MiB aggregate; downloads 8/32 files, 32 MiB/256 MiB each, 64 MiB/512 MiB aggregate. Caps charge before context/page/action/queue/snapshot/network/artifact retention. Host supplies Playwright and egress proxy attestation; package import launches nothing.
317
317
 
318
- 0.0.14 co-work defaults/hard caps (frozen in [Phase 9 evidence](review-coverage-2026-07-25-phase-9.md)): conversation thread list pages 50/200, active branches per thread 16/64, replay/export page 100/500 events; artifact revisions per artifact 32/128, artifacts per thread 64/256, metadata record 8/64 KiB, preview 16/64 KiB, citations 32/128 (2/8 KiB each), delivery-link TTL 5 min/24 h, delivery token 4/16 KiB, compare exactly 2 revisions; memory retention batch 500/5000; proactive capability TTL 24 h/31 d, capability token record 16 KiB; browser checkpoint URL 8 KiB/16 KiB, domain-state hash 256 B/1 KiB, host-data ref 2 KiB/8 KiB, 16/64 checkpoints per run; device stream chunk 1 MiB/8 MiB, concurrent device sessions per identity 1/4 (device wall/turns/tool calls consume shared `RunLimits`). All caps charge before persist/emit and fail closed on overflow. Benchmark placeholder: `node scripts/benchmark-0.0.14.mjs` (release Task 12) reports conversation replay, memory injection/consent, artifact revision/delivery, AG-UI co-work mapping, and connector refresh overhead against these budgets.
318
+ 0.0.14 co-work defaults/hard caps (frozen in [Phase 9 evidence](_evidence/review-coverage-2026-07-25-phase-9.md)): conversation thread list pages 50/200, active branches per thread 16/64, replay/export page 100/500 events; artifact revisions per artifact 32/128, artifacts per thread 64/256, metadata record 8/64 KiB, preview 16/64 KiB, citations 32/128 (2/8 KiB each), delivery-link TTL 5 min/24 h, delivery token 4/16 KiB, compare exactly 2 revisions; memory retention batch 500/5000; proactive capability TTL 24 h/31 d, capability token record 16 KiB; browser checkpoint URL 8 KiB/16 KiB, domain-state hash 256 B/1 KiB, host-data ref 2 KiB/8 KiB, 16/64 checkpoints per run; device stream chunk 1 MiB/8 MiB, concurrent device sessions per identity 1/4 (device wall/turns/tool calls consume shared `RunLimits`). All caps charge before persist/emit and fail closed on overflow. Benchmark placeholder: `node scripts/benchmark-0.0.14.mjs` (release Task 12) reports conversation replay, memory injection/consent, artifact revision/delivery, AG-UI co-work mapping, and connector refresh overhead against these budgets.
319
319
 
320
320
  Current surfaces:
321
321
 
@@ -499,7 +499,7 @@ Repository size at the same commit, counted from `src/` and `packages/` while ex
499
499
 
500
500
  Prism has no project generator before Phase 5, so a generated-Prism-project install/build size is **not applicable** at this baseline. The closest current install figure is the 72 MiB development workspace; it is not a scaffold target. The comparison Mastra default scaffold measured during the review used 439 MB `node_modules`, 300 MB build output, and 427 installed packages. Phase 5 must establish a real generated Prism project baseline and keep unselected storage, telemetry, eval, memory, server, and workflow dependencies absent.
501
501
 
502
- See [Review coverage — 2026-07-15](review-coverage-2026-07-15.md) for scope, primitive, package, and threat-boundary ownership.
502
+ See [Review coverage — 2026-07-15](_evidence/review-coverage-2026-07-15.md) for scope, primitive, package, and threat-boundary ownership.
503
503
 
504
504
  ### 0.0.5 Phase 2 verification (2026-07-15)
505
505
 
@@ -300,4 +300,4 @@ See **Threat model summary** and **Performance notes** above. Cross-cutting rule
300
300
  - [Resource loading](resource-loading.md): `ResourceLoader` decode helpers
301
301
  - [Model registry](model-registry.md): `ModelCapabilities` metadata
302
302
  - [Provider conformance](provider-conformance.md): content preservation and secret leak checks
303
- - [Review coverage (2026-07-14)](review-coverage-2026-07-14.md): traceability for C-005, C-010, C-011
303
+ - [Review coverage (2026-07-14)](_evidence/review-coverage-2026-07-14.md): traceability for C-005, C-010, C-011
@@ -273,7 +273,7 @@ Every migrated provider must pass this shared matrix (implemented in Task 1 test
273
273
  - [OpenAI-compatible provider](providers/openai-compatible.md): reference adapter subpath
274
274
  - [Structured output](structured-output.md): artifact loop fallback
275
275
  - [Provider request policies](provider-request-policies.md): cache and request hooks
276
- - [Review coverage (2026-07-14)](review-coverage-2026-07-14.md): finding → plan traceability
276
+ - [Review coverage (2026-07-14)](_evidence/review-coverage-2026-07-14.md): finding → plan traceability
277
277
 
278
278
  ## Task ownership map
279
279
 
@@ -29,12 +29,38 @@ serialization, dynamic model discovery, and explicit/implicit cache accounting.
29
29
  Do not use it for automatic credential discovery, setup-time catalog fetches, or
30
30
  real-network tests (live tests stay opt-in).
31
31
 
32
+ ## Compatible-mode surface (verified 2026-08-10)
33
+
34
+ Decision record for which DashScope / Model Studio surfaces are reachable through
35
+ OpenAI-compatible endpoints on the package's public presets. Sources retrieved
36
+ 2026-08-10; links in the table. This table is the authority for what the package
37
+ implements vs defers (plan 014 Task 1).
38
+
39
+ | Surface | OpenAI-compatible? | Verified route | Decision |
40
+ | --- | --- | --- | --- |
41
+ | Embeddings | Yes | `POST {base}/embeddings` on all public presets (intl/beijing/us); `text-embedding-v3`/`v4`; dimensions 64–2048 (default 1024); max 10 inputs per request, 8,192 tokens each | Implemented in 0.1.2 (`createAlibabaEmbedder`) |
42
+ | Video input | Yes | Chat content part `{"type":"video_url","video_url":{"url":…},"fps":2}` on Qwen-VL models; URL must be publicly reachable with correct `Content-Length`/`Content-Type`; `fps` 0.1–10 (default 2) | Implemented in 0.1.2 (video `file` blocks → `video_url`) |
43
+ | Document input | Partial | OpenAI Files API `POST {base}/files` (`purpose: "file-extract"`, ≤150 MB) then reference `fileid://<id>` as a system message (qwen-long, ≤100 files); no document content part exists in compatible mode; `doc_url` parts are native-only (qwen-doc-turbo) | Deferred — upload + status lifecycle, not a serialization mapping; demand-gated follow-up |
44
+ | Rerank | Partial | `POST {workspaceId}.{region}.maas.aliyuncs.com/compatible-api/v1/reranks` (`qwen3-rerank`, ≤500 documents, 4,000 tokens/item) — workspace-dedicated only, base path `compatible-api/v1` (not `compatible-mode/v1`); no rerank route on the public presets | Deferred — no route on public presets; workspace-dedicated route recorded for a future `baseUrl`-supplied reranker |
45
+ | Text-to-SQL | n/a | No dedicated endpoint; SQL generation is a chat prompt use case on `chat/completions` | Nothing to implement — covered by the existing chat provider |
46
+ | Async task polling | No | `X-DashScope-Async: enable` + `GET /api/v1/tasks/{id}` — native-only | Deferred (documented) |
47
+
48
+ Sources:
49
+
50
+ - OpenAI compatibility overview: <https://help.aliyun.com/en/model-studio/compatibility-of-openai-with-dashscope>
51
+ - Embeddings (models, dimensions): <https://www.alibabacloud.com/help/en/model-studio/models>; batch limits: <https://docs.qwencloud.com/resources/faq-embedding-reranking>
52
+ - Video input (`video_url` part): <https://help.aliyun.com/en/model-studio/qwen-api-via-openai-chat-completions>
53
+ - Document input (file-extract): <https://help.aliyun.com/en/model-studio/long-context-qwen-long> and <https://help.aliyun.com/en/model-studio/openai-file-interface>; native `doc_url`: <https://help.aliyun.com/en/model-studio/data-mining-qwen-doc>
54
+ - Rerank (`compatible-api/v1/reranks`): <https://www.alibabacloud.com/help/en/model-studio/rerank>
55
+ - Async task polling (native): <https://help.aliyun.com/en/model-studio/asynchronous-call-api-reference>
56
+
32
57
  ## Inputs / request
33
58
 
34
59
  ```ts
35
60
  import {
36
61
  createAlibabaProviderPackage,
37
62
  createAlibabaProvider,
63
+ createAlibabaEmbedder,
38
64
  listAlibabaModels,
39
65
  defineAlibabaModel,
40
66
  alibabaBaseUrl,
@@ -42,6 +68,7 @@ import {
42
68
 
43
69
  createAlibabaProviderPackage(options: AlibabaProviderPackageOptions): ProviderPackage
44
70
  createAlibabaProvider(options?: AlibabaProviderOptions): AIProvider
71
+ createAlibabaEmbedder(options: AlibabaEmbedderOptions): AlibabaEmbedder
45
72
  listAlibabaModels(options?: ListAlibabaModelsOptions): Promise<ModelConfig[]>
46
73
  defineAlibabaModel(config: AlibabaModelConfig): ModelConfig
47
74
  alibabaBaseUrl(options?: { baseUrl?: string; preset?: AlibabaBasePreset }): string
@@ -69,6 +96,70 @@ Workspace-dedicated endpoints
69
96
  (`https://{workspaceId}.{region}.maas.aliyuncs.com/compatible-mode/v1`) are supplied
70
97
  verbatim via `baseUrl`.
71
98
 
99
+ ## Embeddings
100
+
101
+ `createAlibabaEmbedder()` calls the OpenAI-compatible `POST {base}/embeddings`
102
+ (text-embedding-v3/v4) and returns a structural `Embedder` — assignable to
103
+ `@arnilo/prism-memory`'s `Embedder` without importing it (the package stays
104
+ dependency-free).
105
+
106
+ ```ts
107
+ import { createAlibabaEmbedder } from "@arnilo/prism-provider-alibaba";
108
+
109
+ const embedder = createAlibabaEmbedder({
110
+ apiKey: process.env.DASHSCOPE_API_KEY,
111
+ model: "text-embedding-v4",
112
+ dimensions: 1024, // 64–2048, default 1024
113
+ });
114
+
115
+ const vectors = await embedder.embed(["hello", "world"]); // number[2][1024]
116
+ ```
117
+
118
+ - Inputs are chunked at `ALIBABA_EMBEDDING_BATCH_SIZE` (10) per request — the
119
+ DashScope cap (8,192 tokens per text) — and vectors are returned in input order.
120
+ Empty input returns `[]` without a fetch.
121
+ - `dimensions` (64–2048, default 1024) and `encoding_format` (default `float`)
122
+ pass through on the wire; `baseUrl`/`preset`/`fetch`/`headers` mirror the
123
+ provider options.
124
+ - Caller-gated like discovery: construction never fetches; the key is resolved per
125
+ call and redacted from all thrown errors; provider-owned headers
126
+ (`authorization`, `content-type`) cannot be overridden by caller headers.
127
+
128
+ ## Multimodal input
129
+
130
+ Video input (0.1.2): a `file` content block with a `video/*` media type serializes
131
+ to the compatible-mode `video_url` content part on Qwen-VL models:
132
+
133
+ ```ts
134
+ // host side
135
+ { type: "file", mediaType: "video/mp4", url: "https://example.com/clip.mp4" }
136
+ // wire shape emitted by serializeAlibabaMessage
137
+ { "type": "video_url", "video_url": { "url": "https://example.com/clip.mp4" } }
138
+ ```
139
+
140
+ - Gated on the `file` input capability (no core `"video"` capability in 0.1.2);
141
+ `mapAlibabaModel()` advertises `["text", "image", "file"]` for the qwen-vl
142
+ family; `defineAlibabaModel` capability overrides still win.
143
+ - `url` (publicly reachable, correct `Content-Length`/`Content-Type`) or base64
144
+ `data:` URL pass through; `resourceUri`-only blocks throw before fetch (the
145
+ provider never fetches). `fps` defaults upstream to 2.0.
146
+ - Document input is **deferred**: compatible-mode chat has no document content
147
+ part — the compatible path is the OpenAI Files API (`purpose: file-extract`,
148
+ ≤150 MB) plus a `fileid://<id>` system-message reference (qwen-long, ≤100
149
+ files), an upload/status lifecycle outside serialization. `document` and
150
+ non-video `file` blocks keep failing before fetch.
151
+
152
+ ## Rerank (deferred)
153
+
154
+ No OpenAI-compatible rerank route exists on the public presets, so 0.1.2 ships no
155
+ reranker. The verified compatible route is workspace-dedicated only:
156
+ `POST {workspaceId}.{region}.maas.aliyuncs.com/compatible-api/v1/reranks`
157
+ (`qwen3-rerank`, ≤500 documents, 4,000 tokens/item; base path `compatible-api/v1`,
158
+ not `compatible-mode/v1`). A future `createAlibabaReranker` over that route is
159
+ demand-gated: implement when a caller supplies a workspace-dedicated `baseUrl` and
160
+ needs rerank (structural `Reranker` shape from `@arnilo/prism-rag`, no new
161
+ dependency). Multimodal rerank (`qwen3-vl-rerank`) is native-only and stays out.
162
+
72
163
  ## Outputs / response / events
73
164
 
74
165
  | Surface | Behavior |
@@ -163,6 +254,10 @@ await kernel.load([
163
254
  - The API key is resolved per request via `resolveCredentialValue` and sent only as
164
255
  `Authorization: Bearer`; keys are redacted from all thrown errors (including
165
256
  discovery failures). No local filesystem paths enter request payloads.
257
+ - Opt-in live probe (never part of `npm test`/CI):
258
+ `PRISM_LIVE_DASHSCOPE_KEY=… npm run test:live --workspace @arnilo/prism-provider-alibaba`
259
+ exercises an embeddings round-trip against the real endpoint (model override via
260
+ `PRISM_LIVE_DASHSCOPE_MODEL`); absent env = documented skip, never a failure.
166
261
  - Caller-supplied `ProviderRequest.options.headers` can add non-owned headers, but
167
262
  provider-owned headers (`content-type`, `authorization`) are applied last and
168
263
  cannot be overridden.
@@ -51,6 +51,7 @@ Consumers install the core package for the runtime and add first-party packages
51
51
  | Resume interrupted tagged publication | `npm run release:publish -- --version 0.1.0 --resume --report release-artifacts/publish-report.json` |
52
52
  | Protected PostgreSQL enterprise suite | `PRISM_TEST_POSTGRES_URL="$DATABASE_URL" npm run test:postgres` |
53
53
  | Full SDK readiness gate (typecheck + offline tests + pack) | `npm run sdk:ready` |
54
+ | Non-blocking unused-code sweep (report to `scripts/unused-sweep-report.txt`, always exits 0) | `npm run sweep:unused` |
54
55
 
55
56
  > **Build notes (0.1.1+).** `npm run build` no longer runs `npm run clean` first: concurrent builds/tests (`npm test`, `npm run typecheck`) are now race-free because the only destructive step was the `rm -rf` clean, and concurrent `tsc` is write-only and idempotent on identical input (two processes emitting the same files end byte-identical regardless of interleaving).
56
57
  >
@@ -241,6 +242,70 @@ git push origin v0.1.1 # tag push triggers release.yml publish job (prove
241
242
 
242
243
  **Rollback notes.** `release:publish --version 0.1.1 --resume --report release-artifacts/publish-report.json` resumes an interrupted publication and skips only registry versions whose internal dependency fingerprint matches the local manifest. A failed package aborts the run with its status written to the report; re-run after fixing the cause. npm cannot unpublish the `0.1.1` line after 72 hours — a post-publication defect ships as a `0.1.x` patch (additive-only compat promise, `release:gate` enforced), or as a documented break in the next line with a `docs/migration.md` entry. `0.1.1` is store-compatible with `0.1.0` in **both directions** (no migration ran — same checksum-protected contract), so an operator may defer or roll back the patch without a database rollback.
243
244
 
245
+ ### 0.1.2 publish handoff (plan 014 Task 6)
246
+
247
+ **Decision: GO when the operator prerequisites below are recorded.** Release **0.1.2** (plan 014) is the Alibaba Cloud provider enrichment patch on the frozen 0.1.x line: `createAlibabaEmbedder` over the OpenAI-compatible `POST {base}/embeddings` (structural `Embedder`, no new dependency), video input via `video_url` content parts on Qwen-VL models (gated on the `file` input capability), a verified compatible-mode surface decision table in [providers/alibaba.md](providers/alibaba.md) (document input and rerank deferred as demand-gated follow-ups), and an opt-in `PRISM_LIVE_DASHSCOPE_KEY` live probe. Publishable graph stays **49** manifests (root + 48 workspace) at exact **0.1.2**. Store compatibility with 0.1.1: **compatible, no migration**; declaration surface additive-only vs the frozen 0.1.x contract (`scripts/compat-baseline` regenerated at 0.1.2 with zero breaking deltas).
248
+
249
+ ```bash
250
+ # Operator prerequisites (each a named blocked gate — none may be skipped):
251
+ # 1. protected live-canary matrix green (live-canaries.yml, canary-report.json retained)
252
+ # 2. PostgreSQL + keychain protected suites green (test:postgres, keychain suite)
253
+ # 3. CodeQL SAST green on the release commit (security.yml / release.yml codeql-release)
254
+ # 4. npm OIDC trusted publishing identity authenticated (NPM_TOKEN with id-token, provenance)
255
+
256
+ git diff --check
257
+ npm ci
258
+ npm run sdk:ready # includes typecheck, lint, format, full test, coverage, pack, release:gate
259
+ npm run security:threat-suites
260
+ PRISM_TEST_POSTGRES_URL="$DATABASE_URL" npm run test:postgres # Phase 7 + Phase 12 restart-recovery
261
+ npm audit --audit-level=moderate
262
+ npm run release:check -- --version 0.1.2 --report /tmp/prism-0.1.2-preflight.json
263
+ npm run release:publish -- --version 0.1.2 --dry-run --allow-dirty --allow-untagged --report /tmp/prism-0.1.2-dry-run.json
264
+ # run the dry-run twice and diff the reports: deterministic, byte-identical
265
+
266
+ # Sign the release on the clean tagged tree (operator GPG key):
267
+ git tag -s v0.1.2 -m "Prism 0.1.2"
268
+ git verify-tag v0.1.2
269
+ git push origin v0.1.2 # tag push triggers release.yml publish job (provenance, attestations)
270
+
271
+ # Real publication never bypasses the gates: release.mjs refuses
272
+ # --allow-dirty/--allow-untagged without --dry-run.
273
+ ```
274
+
275
+ **Rollback notes.** `release:publish --version 0.1.2 --resume --report release-artifacts/publish-report.json` resumes an interrupted publication and skips only registry versions whose internal dependency fingerprint matches the local manifest. A failed package aborts the run with its status written to the report; re-run after fixing the cause. npm cannot unpublish the `0.1.2` line after 72 hours — a post-publication defect ships as a `0.1.x` patch (additive-only compat promise, `release:gate` enforced), or as a documented break in the next line with a `docs/migration.md` entry. `0.1.2` is store-compatible with `0.1.1` in **both directions** (no migration ran — same checksum-protected contract), so an operator may defer or roll back the patch without a database rollback.
276
+
277
+ ### 0.1.3 publish handoff (plan 015 Task 5)
278
+
279
+ **Decision: GO when the operator prerequisites below are recorded.** Release **0.1.3** (plan 015) is the dead-code and deprecation hygiene patch on the frozen 0.1.x line: one parameterized benchmark runner `scripts/benchmark.mjs --scenario <name>` replaces the per-version runners (16 orphaned `benchmark-0.0.{8..16}` runner/test files removed, evidence JSON kept; the CI schema leg runs `scripts/benchmark.test.mjs`), the 12 `docs/review-coverage-2026-07-*.md` evidence files moved to the tarball-excluded `docs/_evidence/` archive, a non-blocking unused-code sweep (`npm run sweep:unused` — tsc `noUnusedLocals`/`noUnusedParameters` over core + all workspace tsconfigs plus a zero-dep dead-export scan; always exits 0, report to `scripts/unused-sweep-report.txt`), and opt-in checkpoint persistence (`persistSessionState: true` on durable run/resume options — loaded-skill name catalog ≤64 names rides the run-state checkpoint and restores on resume, bodies re-resolve from the live registry; `createReadPathSetPersistence` in `@arnilo/prism-coding-agent` persists the read-before-write path set through the host `CheckpointStore`, ≤1024 paths, ownership-scoped). Publishable graph stays **49** manifests (root + 48 workspace) at exact **0.1.3**. Store compatibility with 0.1.2: **compatible, no migration**; declaration surface additive-only vs the frozen 0.1.x contract (`scripts/compat-baseline` regenerated at 0.1.3 with zero breaking deltas).
280
+
281
+ ```bash
282
+ # Operator prerequisites (each a named blocked gate — none may be skipped):
283
+ # 1. protected live-canary matrix green (live-canaries.yml, canary-report.json retained)
284
+ # 2. PostgreSQL + keychain protected suites green (test:postgres, keychain suite)
285
+ # 3. CodeQL SAST green on the release commit (security.yml / release.yml codeql-release)
286
+ # 4. npm OIDC trusted publishing identity authenticated (NPM_TOKEN with id-token, provenance)
287
+
288
+ git diff --check
289
+ npm ci
290
+ npm run sdk:ready # includes typecheck, lint, format, full test, coverage, pack, release:gate
291
+ npm run security:threat-suites
292
+ PRISM_TEST_POSTGRES_URL="$DATABASE_URL" npm run test:postgres # Phase 7 + Phase 12 restart-recovery
293
+ npm audit --audit-level=moderate
294
+ npm run release:check -- --version 0.1.3 --report /tmp/prism-0.1.3-preflight.json
295
+ npm run release:publish -- --version 0.1.3 --dry-run --allow-dirty --allow-untagged --report /tmp/prism-0.1.3-dry-run.json
296
+ # run the dry-run twice and diff the reports: deterministic, byte-identical
297
+
298
+ # Sign the release on the clean tagged tree (operator GPG key):
299
+ git tag -s v0.1.3 -m "Prism 0.1.3"
300
+ git verify-tag v0.1.3
301
+ git push origin v0.1.3 # tag push triggers release.yml publish job (provenance, attestations)
302
+
303
+ # Real publication never bypasses the gates: release.mjs refuses
304
+ # --allow-dirty/--allow-untagged without --dry-run.
305
+ ```
306
+
307
+ **Rollback notes.** `release:publish --version 0.1.3 --resume --report release-artifacts/publish-report.json` resumes an interrupted publication and skips only registry versions whose internal dependency fingerprint matches the local manifest. A failed package aborts the run with its status written to the report; re-run after fixing the cause. npm cannot unpublish the `0.1.3` line after 72 hours — a post-publication defect ships as a `0.1.x` patch (additive-only compat promise, `release:gate` enforced), or as a documented break in the next line with a `docs/migration.md` entry. `0.1.3` is store-compatible with `0.1.2` in **both directions** (no migration ran — same checksum-protected contract), so an operator may defer or roll back the patch without a database rollback.
308
+
244
309
  ### 0.0.28 publish handoff (historical)
245
310
 
246
311
  **Decision: GO after protected operator prerequisites below.** Release **0.0.28** (Phase 11, plan 011) ships the optional enterprise adapter seams: OIDC/JWKS identity verification (`@arnilo/prism-credentials-node/oidc`), OPA policy evaluation into the durable ledger (`@arnilo/prism-policy/opa`), MCP OAuth client/server support (`@arnilo/prism-mcp`), host-selected OpenAPI operations as effect-gated tools (`@arnilo/prism-openapi-tools`), and an S3-compatible artifact body store behind the new core body contract (`@arnilo/prism-server/artifact-bodies`). Every seam is opt-in and fail-closed; hosts that wire none keep exact prior behavior. Publishable graph stays **49** publishable manifests (root + 48 workspace packages; `prism-openapi-tools` joined the graph in this release). See [migration](migration.md) `0.0.27 → 0.0.28`.
@@ -126,7 +126,7 @@ const page = await store.searchSessions!({
126
126
  // Opt out: createMemorySessionStore([], { sessionSearchMode: "unsupported" })
127
127
  ```
128
128
 
129
- Finite caps (defaults / hard): page 20/100; query string 4 KiB/16 KiB; snippet 512 B/4 KiB; cursor 1 KiB/4 KiB; memory linear sessions 1000/5000, entries 10000/50000, bytes 8 MiB/64 MiB; DB FTS candidates 1000/5000. Overflow fails closed via `resolveSessionSearchQuery`. See [Phase 6 evidence](review-coverage-2026-07-22-phase-6.md).
129
+ Finite caps (defaults / hard): page 20/100; query string 4 KiB/16 KiB; snippet 512 B/4 KiB; cursor 1 KiB/4 KiB; memory linear sessions 1000/5000, entries 10000/50000, bytes 8 MiB/64 MiB; DB FTS candidates 1000/5000. Overflow fails closed via `resolveSessionSearchQuery`. See [Phase 6 evidence](_evidence/review-coverage-2026-07-22-phase-6.md).
130
130
 
131
131
  ## Session search (0.0.11)
132
132
 
@@ -144,7 +144,7 @@ const page = await store.searchSessions!({
144
144
  // Opt out: createMemorySessionStore([], { sessionSearchMode: "unsupported" })
145
145
  ```
146
146
 
147
- Finite caps (defaults / hard): page 20/100; query string 4 KiB/16 KiB; snippet 512 B/4 KiB; cursor 1 KiB/4 KiB; memory linear sessions 1000/5000, entries 10000/50000, bytes 8 MiB/64 MiB; DB FTS candidates 1000/5000. Overflow fails closed via `resolveSessionSearchQuery`. See [Phase 6 evidence](review-coverage-2026-07-22-phase-6.md).
147
+ Finite caps (defaults / hard): page 20/100; query string 4 KiB/16 KiB; snippet 512 B/4 KiB; cursor 1 KiB/4 KiB; memory linear sessions 1000/5000, entries 10000/50000, bytes 8 MiB/64 MiB; DB FTS candidates 1000/5000. Overflow fails closed via `resolveSessionSearchQuery`. See [Phase 6 evidence](_evidence/review-coverage-2026-07-22-phase-6.md).
148
148
 
149
149
  ## Security and performance notes
150
150
 
@@ -95,4 +95,4 @@ LLM compaction and observational memory accept `thinkingLevel?: string`. They ca
95
95
  - [Provider request policies](provider-request-policies.md) — `mergeProviderRequestOptions`
96
96
  - [Agent/session runtime](agent-session-runtime.md) — prior-reasoning preservation across turns
97
97
  - Per-provider pages under [docs/providers](providers/)
98
- - Evidence matrix: [Review coverage (2026-07-17 provider validation)](review-coverage-2026-07-17-provider-validation.md)
98
+ - Evidence matrix: [Review coverage (2026-07-17 provider validation)](_evidence/review-coverage-2026-07-17-provider-validation.md)
@@ -4,7 +4,7 @@
4
4
 
5
5
  This page freezes the reusable tool validation, parallel dispatch, MCP bridge, and coding execution-policy designs for Plan 055. It inventories existing `@arnilo/prism` tool harness seams, `@arnilo/prism-coding-agent` behavior, extension/contribution boundaries, and the MCP mapping surface Tasks 1–6 will implement against.
6
6
 
7
- Implementation is **shipped** for JSON Schema tool argument validation (Plan 055 Task 1), parallel single-shot tool dispatch (Task 2), the MCP client bridge (Task 3), coding execution policy (Task 4), and bounded image reads (Task 5). Task 6 verification evidence is recorded in [review coverage](review-coverage-2026-07-14.md).
7
+ Implementation is **shipped** for JSON Schema tool argument validation (Plan 055 Task 1), parallel single-shot tool dispatch (Task 2), the MCP client bridge (Task 3), coding execution policy (Task 4), and bounded image reads (Task 5). Task 6 verification evidence is recorded in [review coverage](_evidence/review-coverage-2026-07-14.md).
8
8
 
9
9
  ## When to use it
10
10
 
@@ -360,7 +360,7 @@ Core remains dependency-free: validators, MCP bridges, coding policy, sandboxes,
360
360
  - [Host security guide](host-security.md): permission, trust, validation checklist
361
361
  - [Coding agent tools](coding-agent-tools.md): first-party tool package behavior and limits
362
362
  - [Extensions](extensions.md): inert tool contributions
363
- - [Review coverage (2026-07-14)](review-coverage-2026-07-14.md): finding → plan traceability
363
+ - [Review coverage (2026-07-14)](_evidence/review-coverage-2026-07-14.md): finding → plan traceability
364
364
 
365
365
  ## Task ownership map
366
366
 
@@ -106,4 +106,4 @@ createLlmCompactionStrategy({
106
106
  - [LLM compaction package](compaction-llm.md)
107
107
  - [Agent/session runtime](agent-session-runtime.md) — `RunOptions.model` / `model_change`
108
108
  - [Working and semantic memory](working-and-semantic-memory.md) — `Embedder` (non-chat)
109
- - Evidence matrix: [Review coverage (2026-07-17 provider validation)](review-coverage-2026-07-17-provider-validation.md)
109
+ - Evidence matrix: [Review coverage (2026-07-17 provider validation)](_evidence/review-coverage-2026-07-17-provider-validation.md)
@@ -579,4 +579,4 @@ await session.run("Hi", { signal: AbortSignal.timeout(60_000) });
579
579
  - [Tool execution primitives](tool-execution-primitives.md): `ExecutionPolicy` and approval pattern.
580
580
  - [Host security guide](host-security.md): fail-closed checklist for workflow hosts.
581
581
  - [Performance limits](performance.md): subscriber queue defaults workflow tightens.
582
- - [Review coverage (2026-07-14)](review-coverage-2026-07-14.md): C-009/C-012 traceability.
582
+ - [Review coverage (2026-07-14)](_evidence/review-coverage-2026-07-14.md): C-009/C-012 traceability.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@arnilo/prism",
3
- "version": "0.1.1",
3
+ "version": "0.1.3",
4
4
  "description": "Agent harness for AI providers, agents, sessions, and tools.",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",
@@ -107,7 +107,7 @@
107
107
  "!dist/__tests__",
108
108
  "!dist/**/*.map",
109
109
  "docs",
110
- "!docs/review-coverage-*",
110
+ "!docs/_evidence",
111
111
  "templates",
112
112
  "CHANGELOG.md"
113
113
  ],
@@ -141,7 +141,8 @@
141
141
  "clean": "rm -rf dist packages/*/dist",
142
142
  "build": "npm run build:core && npm run build --workspaces --if-present",
143
143
  "typecheck": "npm run build && npm run typecheck --workspaces --if-present && tsc -p examples --noEmit",
144
- "test": "npm run build && node --test dist/__tests__/*.test.js && node --test scripts/release-gate.test.mjs scripts/tooling-gate.test.mjs scripts/budget-gate.test.mjs scripts/phase8-conformance.test.mjs scripts/phase9-conformance.test.mjs scripts/phase10-conformance.test.mjs scripts/phase11-conformance.test.mjs scripts/phase11-freeze.test.mjs scripts/phase12-freeze.test.mjs scripts/phase13-freeze.test.mjs scripts/benchmark-0.1.0.test.mjs scripts/e2e-enterprise-journey.test.mjs scripts/e2e-coding-journey.test.mjs && npm run test --workspaces --if-present",
144
+ "sweep:unused": "node scripts/sweep-unused.mjs",
145
+ "test": "npm run build && node --test dist/__tests__/*.test.js && node --test scripts/release-gate.test.mjs scripts/tooling-gate.test.mjs scripts/budget-gate.test.mjs scripts/phase8-conformance.test.mjs scripts/phase9-conformance.test.mjs scripts/phase10-conformance.test.mjs scripts/phase11-conformance.test.mjs scripts/phase11-freeze.test.mjs scripts/phase12-freeze.test.mjs scripts/phase13-freeze.test.mjs scripts/phase14-freeze.test.mjs scripts/phase15-freeze.test.mjs scripts/benchmark-0.1.0.test.mjs scripts/sweep-unused.test.mjs scripts/e2e-enterprise-journey.test.mjs scripts/e2e-coding-journey.test.mjs && npm run test --workspaces --if-present",
145
146
  "test:coverage": "node --test --experimental-test-coverage --test-coverage-lines=60 --test-coverage-functions=70 --test-coverage-branches=75 --test-coverage-exclude='**/__tests__/**' --test-coverage-exclude='**/node_modules/**' --test-coverage-exclude='**/scripts/**' --test-coverage-exclude='**/packages/**' --test-coverage-exclude='**/examples/**' dist/__tests__/*.test.js && node scripts/coverage-summary.mjs",
146
147
  "coverage:summary": "node scripts/coverage-summary.mjs",
147
148
  "lint": "biome lint .",