@arnilo/prism 0.2.5 → 0.2.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -1
- package/README.md +1 -0
- package/dist/field-policy.d.ts +119 -0
- package/dist/field-policy.js +418 -0
- package/dist/index.d.ts +3 -1
- package/dist/index.js +2 -1
- package/dist/redaction.d.ts +6 -5
- package/dist/redaction.js +18 -10
- package/docs/0.1.0-readiness.md +10 -10
- package/docs/acp.md +2 -0
- package/docs/audit-export.md +151 -0
- package/docs/browser-automation.md +1 -1
- package/docs/coding-agent-tools.md +5 -4
- package/docs/coding-review-and-diagnostics.md +76 -0
- package/docs/coding-security.md +2 -0
- package/docs/coding-workspaces.md +69 -0
- package/docs/data-classification.md +82 -0
- package/docs/disaster-recovery.md +71 -0
- package/docs/enterprise-postgres-state.md +60 -7
- package/docs/evaluations.md +45 -0
- package/docs/forge-integration.md +6 -0
- package/docs/host-security.md +4 -0
- package/docs/index.md +12 -7
- package/docs/indexed-code-search.md +82 -0
- package/docs/language-intelligence.md +15 -0
- package/docs/migration.md +29 -0
- package/docs/operations.md +104 -0
- package/docs/policy-and-audit.md +34 -0
- package/docs/process-sessions.md +58 -3
- package/docs/release-0.2.7-evidence.md +514 -0
- package/docs/release-and-install.md +74 -3
- package/docs/work-artifacts-and-review.md +4 -0
- package/docs/workflows.md +48 -3
- package/package.json +2 -2
package/dist/redaction.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { applyFieldPolicy } from "./field-policy.js";
|
|
1
2
|
const REDACTED = "[REDACTED]";
|
|
2
3
|
// Depth bound matching agent-run-state.ts; hostile deep structures yield a placeholder
|
|
3
4
|
// instead of a stack overflow.
|
|
@@ -5,20 +6,27 @@ const MAX_REDACT_DEPTH = 32;
|
|
|
5
6
|
export function createSecretRedactor(secrets) {
|
|
6
7
|
return { redact: (value) => redactSecrets(value, secrets) };
|
|
7
8
|
}
|
|
8
|
-
|
|
9
|
-
|
|
9
|
+
/** Applies a field policy after secret redaction; identity when absent (compat). */
|
|
10
|
+
function afterSecrets(value, redactor, policy, destination, labelFor) {
|
|
11
|
+
const secretRedacted = redactor?.redact(value) ?? value;
|
|
12
|
+
return policy
|
|
13
|
+
? applyFieldPolicy(secretRedacted, policy, { destination, direction: "outbound", ...(labelFor ? { labelFor } : {}) })
|
|
14
|
+
: secretRedacted;
|
|
10
15
|
}
|
|
11
|
-
export function
|
|
12
|
-
return
|
|
16
|
+
export function redactMessage(message, redactor, fieldPolicy, destination = "prompt", labelFor) {
|
|
17
|
+
return afterSecrets(message, redactor, fieldPolicy, destination, labelFor);
|
|
13
18
|
}
|
|
14
|
-
export function
|
|
15
|
-
return
|
|
19
|
+
export function redactAgentEvent(event, redactor, fieldPolicy, destination = "telemetry", labelFor) {
|
|
20
|
+
return afterSecrets(event, redactor, fieldPolicy, destination, labelFor);
|
|
16
21
|
}
|
|
17
|
-
export function
|
|
18
|
-
return
|
|
22
|
+
export function redactSessionEntry(entry, redactor, fieldPolicy, destination = "persistence", labelFor) {
|
|
23
|
+
return afterSecrets(entry, redactor, fieldPolicy, destination, labelFor);
|
|
19
24
|
}
|
|
20
|
-
export function
|
|
21
|
-
return
|
|
25
|
+
export function redactProviderRequest(request, redactor, fieldPolicy, destination = "prompt", labelFor) {
|
|
26
|
+
return afterSecrets(request, redactor, fieldPolicy, destination, labelFor);
|
|
27
|
+
}
|
|
28
|
+
export function redactRunLedgerRecord(record, redactor, fieldPolicy, destination = "run-ledger", labelFor) {
|
|
29
|
+
return afterSecrets(record, redactor, fieldPolicy, destination, labelFor);
|
|
22
30
|
}
|
|
23
31
|
export function resolveRedactor(redactor, secrets) {
|
|
24
32
|
return redactor ?? (secrets?.some((secret) => Boolean(secret)) ? createSecretRedactor(secrets) : undefined);
|
package/docs/0.1.0-readiness.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# 0.1.0 / 1.0 Readiness Gates
|
|
2
2
|
|
|
3
|
-
Status: **0.2.
|
|
3
|
+
Status: **0.2.7** is the current release line (the 0.2.x review-remediation line: fail-closed runtime/sandbox security, provider completion and outbound trust boundaries, concurrent-state/durability integrity, build/coverage/release-evidence integrity, package/documentation/compatibility truth, maintainability and bounded performance, fully featured coding-agent readiness); **0.1.7** was the terminal 0.1.x baseline; **1.0** readiness remains operator-gated, not automatic.
|
|
4
4
|
|
|
5
5
|
This page distills runnable readiness gates into one command-per-gate table.
|
|
6
6
|
The **Last evidence** column records the 0.1.0-tree snapshot (plan 012 Tasks
|
|
@@ -17,19 +17,19 @@ Evidence trail: [`docs/_evidence/review-coverage-2026-07-26-phase-11.md`](./_evi
|
|
|
17
17
|
The per-phase review-coverage evidence archive lives in [`docs/_evidence/`](./_evidence/)
|
|
18
18
|
(plans 067–079, releases 0.0.4–0.0.16; tarball-excluded, kept in-repo for audit).
|
|
19
19
|
Historical release lines (0.0.16 floor → 0.0.27 Phase 10 ACP interop → 0.1.0)
|
|
20
|
-
keep their per-phase evidence in the pages above; this page records the 0.2.
|
|
21
|
-
snapshot (plan
|
|
20
|
+
keep their per-phase evidence in the pages above; this page records the 0.2.6
|
|
21
|
+
snapshot (plan 026) with the 0.1.x tables below as the historical record.
|
|
22
22
|
|
|
23
|
-
## Current line (0.2.
|
|
23
|
+
## Current line (0.2.7)
|
|
24
24
|
|
|
25
25
|
| Item | Status |
|
|
26
26
|
|---|---|
|
|
27
|
-
| Published graph | **50** publishable manifests at exact **0.2.
|
|
28
|
-
| Current-line cut | The 0.2.x review-remediation line, additive-only vs the frozen 0.1.x contract: 0.2.0 fail-closed runtime/sandbox security (durable-resume decision validation, work-tool env isolation, explicit sandbox capabilities), 0.2.1 provider completion + outbound trust boundaries (strict stream completion, bounded success bodies, DNS-pinned OIDC/OPA fetches), 0.2.2 concurrent-state/durability integrity (model-budget reservation, conversation-metadata CAS, single-consumer EventMultiplexer, NATS durable identity), 0.2.3 build/coverage/release-evidence integrity (build single-flight, corrected coverage denominators, release skip manifest, stabilized quality gates), 0.2.4 package/documentation/compatibility truth (umbrella wording matches manifests, manifest-derived package truth, peer-version policy Decision A, current-line truth), 0.2.5 maintainability and bounded performance (god-module splits into cohesive family files behind preserved barrels, persistence-mechanics dedup into `session-store-codecs`, quadratic `Buffer.concat` removed from framing/tar, dead-code cleanup internal-only, 76 behavior-backed coverage regressions) |
|
|
29
|
-
| Upgrade path | `docs/migration.md` `0.2.
|
|
30
|
-
| Compat promise | Additive-only vs the frozen 0.1.x contract; `scripts/compat-baseline` regenerated at 0.2.
|
|
31
|
-
| Security policy | `npm audit --audit-level=moderate` 0 at 0.2.
|
|
32
|
-
| Docs freeze | tripwires green including the canonical manifest-count tripwire (50/49/14/9/26), the plan 024 package-truth tests (generator reproducibility + artifact equality + closure asserts + derived docs truth),
|
|
27
|
+
| Published graph | **50** publishable manifests at exact **0.2.7** (root + 49 workspace packages: 14 provider adapters + 9 `prism-*` family/profile + 26 capability; generated by `node scripts/package-truth.mjs` → `scripts/package-truth.json`) |
|
|
28
|
+
| Current-line cut | The 0.2.x review-remediation line, additive-only vs the frozen 0.1.x contract: 0.2.0 fail-closed runtime/sandbox security (durable-resume decision validation, work-tool env isolation, explicit sandbox capabilities), 0.2.1 provider completion + outbound trust boundaries (strict stream completion, bounded success bodies, DNS-pinned OIDC/OPA fetches), 0.2.2 concurrent-state/durability integrity (model-budget reservation, conversation-metadata CAS, single-consumer EventMultiplexer, NATS durable identity), 0.2.3 build/coverage/release-evidence integrity (build single-flight, corrected coverage denominators, release skip manifest, stabilized quality gates), 0.2.4 package/documentation/compatibility truth (umbrella wording matches manifests, manifest-derived package truth, peer-version policy Decision A, current-line truth), 0.2.5 maintainability and bounded performance (god-module splits into cohesive family files behind preserved barrels, persistence-mechanics dedup into `session-store-codecs`, quadratic `Buffer.concat` removed from framing/tar, dead-code cleanup internal-only, 76 behavior-backed coverage regressions), 0.2.6 fully featured coding-agent readiness (host-selected PTY backend, scalable indexed code-search seam, multi-worktree/repository lifecycle, forge breadth demand-gated, durable ACP/process recovery, patch review + incremental diagnostics, protected real coding journey) |
|
|
29
|
+
| Upgrade path | `docs/migration.md` `0.2.6 → 0.2.7` (additive; two new forward-only ERP migrations 004/005 — outbox/inbox + approvals tables; rollback = stop 0.2.7 workers, drop the ERP tables, restore 0.2.6 manifests/tag); store-compatible throughout 0.2.x |
|
|
30
|
+
| Compat promise | Additive-only vs the frozen 0.1.x contract; `scripts/compat-baseline` regenerated at 0.2.7 (version literal + additive PTY/index/workspace/recovery/review exports), zero breaking deltas |
|
|
31
|
+
| Security policy | `npm audit --audit-level=moderate` 0 at 0.2.7; threat-suites legs (phase8–11 + phase20–26) green; protected Postgres recovery/workspace conformance, PTY, real coding journey, NATS/live-canary legs operator-gated |
|
|
32
|
+
| Docs freeze | tripwires green including the canonical manifest-count tripwire (50/49/14/9/26), the plan 024 package-truth tests (generator reproducibility + artifact equality + closure asserts + derived docs truth), the plan 025 bounded-accumulation near-limit probe, and the plan 026 freeze tripwires (per-task markers, threat T1–T8 test mapping, exit gate green) |
|
|
33
33
|
| 0.1.x line | **0.1.7** (plan 019) is the terminal 0.1.x baseline; the 0.1.1 table below keeps the plan 013 snapshot; the 0.1.0 table keeps the plan 012 snapshot; the **0.0.16** values remain the historical network-free floor |
|
|
34
34
|
|
|
35
35
|
## Previous line (0.1.1)
|
package/docs/acp.md
CHANGED
|
@@ -111,6 +111,8 @@ const agent = createPrismAcpAgent({
|
|
|
111
111
|
|
|
112
112
|
### Persistence and ownership
|
|
113
113
|
|
|
114
|
+
- **Active-run recovery (0.2.6, plan 026 Task 5).** When `sessionStore` and the `recovery` seam (checkpoints + leases + ownerId, all three together) are wired, the agent records a bounded `activeRun` reference on `PersistedAcpSession` while a durable run is live (first run event → `running`, suspension → `suspended` + version, finish/deny/error → `terminal`; frozen 512-byte cap; advisory only — the authoritative status is re-queried from `AgentRunLifecycle.status`). After a restart, `restore` re-attaches the ref to the live session and hosts re-resolve it with `createAcpRunRecovery` (exported from `@arnilo/prism-ag-ui/acp`): suspended runs report their pending approval ids and durable version, terminal runs report terminal, and unprovable in-flight streams report `unknown` — the prompt is never restarted automatically. Durable cancellation (`recovery.cancel`) is ownership/version/fence checked, terminal/idempotent, aborts no unrelated run, and never replays a pending/dispatched tool: a cancelled run reports `cancelled` and must not be resumed. `session/cancel` on a live agent aborts the controller (0.2.5 parity) and, for restored runs, writes the durable marker under the session's ownership. Cancel markers live in `prism.coding-agent.cancel.v1` (schemaVersion 1, CAS + lease fenced). A host-side terminal whose managed process is unattestable after restart reports `unknown` (exitCode null); the agent never fabricates an exit or replays input (`terminal-client`).
|
|
115
|
+
|
|
114
116
|
- **Without the durability seam the agent never persists `modeId`/`configValues`.** Defaults are recomputed per session from the `modes`/`configOptions` seams — a fresh `session/new`, `load`, or `resume` always starts from `defaultModeId` / option `defaultValue`, and the agent's per-session registry is in-memory only. Persisting mode/config across sessions is a **host** decision, and host-side persistence MUST be ownership-scoped.
|
|
115
117
|
- **Host persistence MUST key by `sessions.ownership`.** `authorize` binds transport identity to ownership; a host store that persists `modeId`/`configValues` must refuse any restore whose stored ownership differs from the current session's ownership — a `sessionId` alone is never a sufficient key (session ids may collide across tenants). A cross-tenant restore rejects with `ERR_PRISM_ACP_INPUT` and never returns the other tenant's mode/config.
|
|
116
118
|
- **Ownership-scoped restore (host-owned store).** The store is keyed by `sessionId` and records the owning `userId`; restore refuses on mismatch (this exact pattern is asserted in `packages/ag-ui/src/__tests__/acp-modes-config.test.ts`):
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
# Signed, hash-chained audit export
|
|
2
|
+
|
|
3
|
+
## What it does
|
|
4
|
+
|
|
5
|
+
`@arnilo/prism-policy` exports tenant-scoped audit records as signed, hash-chained
|
|
6
|
+
batches: each record envelope is canonicalized (RFC 8785 semantics), hashed with
|
|
7
|
+
SHA-256 including the prior digest, so records form a tamper-evident chain. A
|
|
8
|
+
batch of chained records is wrapped in a manifest that a host-provided
|
|
9
|
+
`AuditSigner` signs; the signed artifact is written to an immutable WORM sink
|
|
10
|
+
which must acknowledge the exact artifact digest before the export cursor
|
|
11
|
+
advances, and is mirrored to an optional SIEM sink with replayable status. An
|
|
12
|
+
independent verifier (`verifyAuditBatch` or `scripts/verify-audit-export.mjs`)
|
|
13
|
+
re-derives everything from the artifact bytes and a public key — no ledger,
|
|
14
|
+
sink, or private key needed.
|
|
15
|
+
|
|
16
|
+
## When to use it
|
|
17
|
+
|
|
18
|
+
- You keep an append-only policy/audit ledger and must prove to auditors that
|
|
19
|
+
exported records were not reordered, edited, inserted, or truncated after the
|
|
20
|
+
fact.
|
|
21
|
+
- You need a durable WORM copy with an integrity receipt and a replayable SIEM
|
|
22
|
+
mirror, without embedding cloud SDKs or key storage in Prism.
|
|
23
|
+
- You must export only the redacted bytes an external verifier will actually
|
|
24
|
+
see, with redaction provenance and legal-hold provenance preserved.
|
|
25
|
+
|
|
26
|
+
Do not use it for exactly-once delivery claims: like the messaging outbox,
|
|
27
|
+
exports are at-least-once, and the exporter never proves that a record exists —
|
|
28
|
+
it proves that what was exported is exactly what the verifier can reproduce.
|
|
29
|
+
|
|
30
|
+
## Inputs / request
|
|
31
|
+
|
|
32
|
+
- `createAuditExporter({ source, cursorStore, signer, wormSink, siemSink?, redact? })`
|
|
33
|
+
- `source` — tenant-scoped, stable-order `AuditPageSource`; page cursors are
|
|
34
|
+
one-shot tokens (re-reading a cursor that already served its final page
|
|
35
|
+
yields an empty page).
|
|
36
|
+
- `cursorStore` — CAS-versioned `AuditCursorStore`; the exporter advances the
|
|
37
|
+
cursor only on a matched version after the WORM acknowledgement.
|
|
38
|
+
- `signer` — host `AuditSigner` (`sign(bytes) -> Uint8Array`, optional
|
|
39
|
+
`keyId`/`algorithm`); Prism never accepts raw private keys.
|
|
40
|
+
- `wormSink` — required immutable sink returning `{ batchId, digest }`; the
|
|
41
|
+
exporter refuses to advance unless both match.
|
|
42
|
+
- `siemSink` — optional replayable mirror.
|
|
43
|
+
- `redact` — optional `AuditRedactionPolicy` applied before hashing.
|
|
44
|
+
- `exportNext({ tenantId, maxRecords?, maxBytes?, signal? })` — processes one
|
|
45
|
+
page; returns `{ batchId, firstSequence, lastSequence, recordCount,
|
|
46
|
+
wormAcked, siemStatus: "disabled" | "sent" | "pending", nextDigest,
|
|
47
|
+
artifactBytes }`.
|
|
48
|
+
- `verifyAuditBatch({ artifactBytes, publicKey, expectedTenantId,
|
|
49
|
+
previousDigest?, expectedFirstSequence?, expectedLastSequence? })`.
|
|
50
|
+
|
|
51
|
+
## Outputs / response / events
|
|
52
|
+
|
|
53
|
+
A batch artifact written to the WORM sink contains:
|
|
54
|
+
|
|
55
|
+
```json
|
|
56
|
+
{
|
|
57
|
+
"schemaVersion": 1,
|
|
58
|
+
"document": "{ ...canonical signed manifest as a single JSON string... }",
|
|
59
|
+
"signature": { "algorithm": "sha256", "keyId": "k1", "value": "<base64>" }
|
|
60
|
+
}
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
The embedded document is the manifest: `tenantId`, `batchId`, `algorithm`,
|
|
64
|
+
`firstSequence`, `lastSequence`, `previousDigest`, `nextDigest`, and
|
|
65
|
+
`records` — each record carrying `sequence`, `priorDigest`, `digest`,
|
|
66
|
+
optional `legalHold`, optional `redactions` (`{ path, reason }[]`), and the
|
|
67
|
+
canonical `record` payload. `verifyAuditBatch` returns `{ ok, errors, batch }`.
|
|
68
|
+
|
|
69
|
+
## Request/response example
|
|
70
|
+
|
|
71
|
+
```ts
|
|
72
|
+
import { createAuditExporter, createMemoryAuditCursorStore } from "@arnilo/prism-policy";
|
|
73
|
+
|
|
74
|
+
const exporter = createAuditExporter({
|
|
75
|
+
source, // host: tenant-scoped record pages
|
|
76
|
+
cursorStore: createMemoryAuditCursorStore(), // host durable store in production
|
|
77
|
+
signer, // host: HSM/KMS-backed signer
|
|
78
|
+
wormSink, // host: S3 object-lock / WORM bucket
|
|
79
|
+
siemSink, // host: SIEM or event stream
|
|
80
|
+
});
|
|
81
|
+
const result = await exporter.exportNext({ tenantId: "acme", maxRecords: 1000 });
|
|
82
|
+
// result.artifactBytes -> store on WORM; result.nextDigest -> pass as
|
|
83
|
+
// previousDigest when verifying the next batch.
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## Implementation example
|
|
87
|
+
|
|
88
|
+
```ts
|
|
89
|
+
import { readFileSync } from "node:fs";
|
|
90
|
+
import { verifyAuditBatch } from "@arnilo/prism-policy";
|
|
91
|
+
|
|
92
|
+
const artifactBytes = new Uint8Array(readFileSync("./acme-000001.json"));
|
|
93
|
+
const verified = verifyAuditBatch({
|
|
94
|
+
artifactBytes,
|
|
95
|
+
publicKey: readFileSync("./audit-verification.pem", "utf8"),
|
|
96
|
+
expectedTenantId: "acme",
|
|
97
|
+
previousDigest: "0000000000000000000000000000000000000000000000000000000000000000",
|
|
98
|
+
});
|
|
99
|
+
if (!verified.ok) throw new Error(verified.errors.join("; "));
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
## Extension and configuration notes
|
|
103
|
+
|
|
104
|
+
- Key rotation: the artifact records the signer's `keyId`; verification fails
|
|
105
|
+
explicitly under a rotated key, so consumers select the key named by the
|
|
106
|
+
artifact's `signature.keyId`. Hosts own key lifecycle and the public-key
|
|
107
|
+
distribution path.
|
|
108
|
+
- Failed batches retain their one-page payload in exporter memory and replay
|
|
109
|
+
the same batch id on the next `exportNext`; WORM-failed or signer-failed
|
|
110
|
+
batches are never re-read from the source. A cursor CAS race after WORM
|
|
111
|
+
acknowledgment is surfaced as an explicit error and is not retried by this
|
|
112
|
+
exporter (the batch is already durable).
|
|
113
|
+
- SIEM failures do not fail the export: the batch reaches WORM, the cursor
|
|
114
|
+
advances, and a bounded `siemPending` list records what is not mirrored.
|
|
115
|
+
`retryPendingSiem` replays a pending batch when the host supplies its
|
|
116
|
+
artifact bytes (e.g. fetched from WORM) and verifies the digest matches.
|
|
117
|
+
- Legal holds: the exporter preserves a `legalHold` flag on envelope records
|
|
118
|
+
and never broadens tenant access; enforcing a hold is the host source's job.
|
|
119
|
+
Redaction (`redact`) strips values before hashing so the verifier sees
|
|
120
|
+
exactly the exported bytes; only `{ path, reason }` provenance survives.
|
|
121
|
+
- Caps are frozen: 1,000 records or 10 MiB per batch, at most 8 un-mirrored
|
|
122
|
+
SIEM batches retained.
|
|
123
|
+
|
|
124
|
+
## Security and performance notes
|
|
125
|
+
|
|
126
|
+
- Bytes are canonical JSON (RFC 8785 semantics: sorted keys, ECMAScript
|
|
127
|
+
shortest number round-trip, `-0` collapsed, lowercase control escapes);
|
|
128
|
+
non-finite numbers, BigInt, undefined, functions, symbols, and cyclic
|
|
129
|
+
values are rejected rather than coerced.
|
|
130
|
+
- Each record digest covers `schemaVersion`, `tenantId`, `sequence`,
|
|
131
|
+
`priorDigest`, `legalHold`, `redactions`, and the canonical record — a
|
|
132
|
+
cross-tenant record can never enter another tenant's chain, and the verifier
|
|
133
|
+
replays every envelope from the artifact's own bytes.
|
|
134
|
+
- The WORM acknowledgement must name the batch and match the artifact digest;
|
|
135
|
+
a lying or partial acknowledgement cannot falsely advance the cursor.
|
|
136
|
+
- Signer keys and raw values never enter logs, records, or artifacts.
|
|
137
|
+
- Performance: hashing and signing are linear in record bytes; batches are
|
|
138
|
+
bounded pages built without loading full history. Verification is also
|
|
139
|
+
linear and stateless. Prism does not certify compliance with NIST or any
|
|
140
|
+
SIEM/WORM vendor program; audit-export is the transport, hosts own the
|
|
141
|
+
custody chain and compliance posture.
|
|
142
|
+
|
|
143
|
+
## Related APIs
|
|
144
|
+
|
|
145
|
+
- `createPolicyDecisionStore` / `exportPolicyDecisions` — the ledger that
|
|
146
|
+
commonly backfills an `AuditPageSource`.
|
|
147
|
+
- `PersistenceLifecycleStore` — legal-hold and retention for lifecycle
|
|
148
|
+
records the source may consult.
|
|
149
|
+
- `createMemoryAuditCursorStore` — reference cursor store (replace with a
|
|
150
|
+
durable CAS store in production).
|
|
151
|
+
- `scripts/verify-audit-export.mjs` — standalone verifier CLI.
|
|
@@ -118,7 +118,7 @@ Observation tools declare `kind: none`; mutations are `external_mutation`/`unsup
|
|
|
118
118
|
|
|
119
119
|
Import is inert. Construction fails clearly when neither `browser` nor `manager` is supplied. Browser installation, launch, version, and control endpoint are host-owned. Prism never exposes init scripts, extensions, persistent profiles, or model-supplied Playwright launch options; CDP exposure is limited to the allowlisted Runtime/Network/Emulation surface above (evaluate is policy-gated arbitrary code execution — treat results as untrusted). Secrets and storage state must not appear in snapshots, tool results, logs, or checkpoints. Finite caps charge before context/page/action/queue/snapshot/network/artifact retention; snapshots retain no unbounded DOM, console, request, response, or trace history. Unreleased downloads are deleted on context close.
|
|
120
120
|
|
|
121
|
-
Default tests use fake Playwright APIs only. Protected live gate: `PRISM_LIVE_PLAYWRIGHT=1` (or `PRISM_TEST_PLAYWRIGHT=1`) `npm run test:live -w @arnilo/prism-browser` exercises a local loopback hostile HTML fixture for snapshot refs, stale-ref rejection, css/xpath targets, private/file deny, upload containment, screenshot bounds, download quarantine/release, and the CDP leg (real evaluate, observe, and emulate). Missing browser binaries fail closed when the gate is enabled. Adversarial network-free fixtures live in `eval-fixtures.test.ts`; see [Evaluations](evaluations.md) and `examples/coding-browser-evaluation.ts`.
|
|
121
|
+
Default tests use fake Playwright APIs only. Protected live gate: `PRISM_LIVE_PLAYWRIGHT=1` (or `PRISM_TEST_PLAYWRIGHT=1`) `npm run test:live -w @arnilo/prism-browser` exercises a local loopback hostile HTML fixture for snapshot refs, stale-ref rejection, css/xpath targets, private/file deny, upload containment, screenshot bounds, download quarantine/release, and the CDP leg (real evaluate, observe, and emulate). Missing browser binaries fail closed when the gate is enabled. The protected coding journey (0.2.6, plan 026 Task 7) additionally runs a real browser inspection leg (local loopback fixture page, snapshot text assertion, run-owned context closed before the host browser) inside the packed consumer as part of scripts/phase26-coding-journey.test.mjs, gated by PRISM_LIVE_PLAYWRIGHT with the pinned playwright-core installed into the consumer; browser storage never appears in the retained report. Adversarial network-free fixtures live in `eval-fixtures.test.ts`; see [Evaluations](evaluations.md) and `examples/coding-browser-evaluation.ts`.
|
|
122
122
|
|
|
123
123
|
## Related APIs
|
|
124
124
|
|
|
@@ -97,7 +97,7 @@ These are **out of scope** for the 0.0.21 package baseline (see roadmap Phase 9
|
|
|
97
97
|
|
|
98
98
|
- **No PDF / document reader** — text and supported images only via `read`.
|
|
99
99
|
- **No trash / recycle daemon** — `delete` / `move` are permanent; host undo is not automatic.
|
|
100
|
-
- **No PTY / interactive process control in `shell`** — `shell` stays one-shot; optional `createProcessSessions` covers long-running attach/input
|
|
100
|
+
- **No PTY / interactive process control in `shell`** — `shell` stays one-shot; optional `createProcessSessions` covers long-running attach/input with a host-selected PTY backend (`pty: true` requires the `ptyBackend` host option; without one it fails closed as unsupported — see [Process sessions](process-sessions.md)).
|
|
101
101
|
- **LSP language-server tools** — not in default aggregators; optional `createLanguageIntelligence` is Phase 9 (see [Language intelligence](language-intelligence.md)).
|
|
102
102
|
- **Managed process sessions** — not in default aggregators; optional `createProcessSessions` is Phase 9 (see [Process sessions](process-sessions.md)).
|
|
103
103
|
- **GitHub forge adapter** — not in default aggregators; optional `createGitHubForge` is Phase 9 (see [Forge integration](forge-integration.md)); no octokit dependency, no multi-forge abstraction.
|
|
@@ -285,7 +285,7 @@ Search text files under the workspace using literal substring match. Binary file
|
|
|
285
285
|
| --- | --- | --- |
|
|
286
286
|
| `query` | `string` | Literal substring (required). |
|
|
287
287
|
| `path` | `string` | Workspace-relative start path. |
|
|
288
|
-
| `mode` | `"literal"` | Literal only (
|
|
288
|
+
| `mode` | `"literal"` (default) \| `"indexed_literal"` \| `"semantic"` | Literal substring by default. Indexed modes exist only when the host enables them (`createRepoSearchTool({ modes })` with an indexed operations composite); missing capability, stale/failed index, or disabled mode returns a stable `ERR_PRISM_INDEX_*` error — never a silent fallback that changes query meaning. `regex` removed in 0.0.18. |
|
|
289
289
|
| `caseSensitive` | `boolean` | Default false. |
|
|
290
290
|
| `includeHidden` | `boolean` | Default false. |
|
|
291
291
|
| `context` | `number` | Context lines before/after each match (default 5, hard 20). Ignored for non-content `outputMode`. |
|
|
@@ -297,7 +297,7 @@ Search text files under the workspace using literal substring match. Binary file
|
|
|
297
297
|
- `files_with_matches`: unique matching paths only.
|
|
298
298
|
- `count`: totals (`N matches in M files`) without line bodies.
|
|
299
299
|
|
|
300
|
-
Metadata includes `matches`, `truncated`, scan/skip counts; non-content modes also expose `fileCount`.
|
|
300
|
+
Metadata includes `matches`, `truncated`, scan/skip counts; non-content modes also expose `fileCount`. Indexed modes add `untrusted_index`, `indexMode`, `indexState`, `indexRevision`, `indexUpdatedAt` and per-match `[score N.NNN]` suffixes — index text is untrusted and must be re-read before mutation. Full contract: see [Indexed code search](indexed-code-search.md).
|
|
301
301
|
|
|
302
302
|
### `glob`
|
|
303
303
|
|
|
@@ -354,10 +354,11 @@ Opt-in tools over a host-pinned Git executable (`gitPath`, default `/usr/bin/git
|
|
|
354
354
|
| `git_status` | `status --porcelain=v2 -z --branch` → structured branch + entries + `dirty`. |
|
|
355
355
|
| `git_diff` | Bounded `--no-ext-diff --no-textconv` diff; oversized output may spill via `artifactWriter`. |
|
|
356
356
|
| `git_branch` | `validate` / `list` / `create` / `switch` with `git check-ref-format --branch`. Switch refuses unrelated dirty trees unless `createCheckpoint=true`. |
|
|
357
|
-
| `git_worktree` | `list` / `add` / `remove` within finite worktree caps. |
|
|
357
|
+
| `git_worktree` | `list` / `add` / `lock` / `unlock` / `remove` within finite worktree caps; list exposes `locked`/`lockReason` from porcelain. One-shot tool: durable multi-repository worktree lifecycle (create/verify/cleanup with ownership, fencing, and cleanup policy) lives in `createCodingWorkspaceLifecycle` — see [Coding workspaces](coding-workspaces.md). |
|
|
358
358
|
| `git_apply` | `check` / `apply` / `reverse`; always `--check` before mutating apply. Apply requires clean/checkpoint; failures restore. |
|
|
359
359
|
| `git_commit` | Explicit-path `add` + `commit --no-verify -F <tempfile>`; requires host `commitIdentity`. Allows dirty entries that are exactly the requested paths; unrelated dirt requires checkpoint. Never pushes. |
|
|
360
360
|
| `git_pr_handoff` | Bounded `{ base, head, commits, changedPaths, diffstat, checks, artifact? }` for host PR creation. Never authenticates or opens a PR. |
|
|
361
|
+
| `git_pr_handoff` (0.2.6 review) | Handoff output feeds `createCodingPatchReviewManifest` — the review binds to base/head, the patch artifact digest, check summaries, and diagnostic summaries (`diagnosticDelta` output) with pending/accepted/rejected/superseded states; see [Coding review and diagnostics](coding-review-and-diagnostics.md). |
|
|
361
362
|
| `coding_check` | Included when `checks` are declared: model selects only a name; executable/args/env are host-fixed. |
|
|
362
363
|
|
|
363
364
|
```ts
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# Coding review and diagnostics
|
|
2
|
+
|
|
3
|
+
## What it does
|
|
4
|
+
|
|
5
|
+
Bounded patch-review manifests plus normalized LSP/check diagnostics for the coding agent runtime (plan 026 Task 6). The review side composes over the existing server `ArtifactService` — no second approval engine, no review database, no raw patch bodies persisted. The diagnostics side normalizes LSP push/pull results and host-parsed check output into one bounded shape with deterministic added/removed/unchanged deltas.
|
|
6
|
+
|
|
7
|
+
| Export | Purpose |
|
|
8
|
+
| --- | --- |
|
|
9
|
+
| `createCodingPatchReviewManifest(input)` | Build a bounded review manifest + structural artifact input (`preview.review` embeds the manifest). |
|
|
10
|
+
| `assertCodingPatchAccepted({ review, artifact })` | Derive `pending\|accepted\|rejected\|superseded` from the artifact record — digest/revision/identity checked. |
|
|
11
|
+
| `CodingPatchReviewError` | Typed fail-closed errors (`ERR_PRISM_REVIEW_*`). |
|
|
12
|
+
| `normalizeDiagnostics(raw, options)` | Validate/bound host-parsed check diagnostics into `NormalizedDiagnostic[]`. |
|
|
13
|
+
| `diagnosticDelta({ next, previous })` | Deterministic `added` / `removed` / `unchanged` across generations. |
|
|
14
|
+
| `diagnosticIdentity(diagnostic)` | Stable per-diagnostic key (`file:source:line:character:code`). |
|
|
15
|
+
| `LanguageIntelligence.syncDocument(file)` / `.diagnosticDelta({ files, previous })` | LSP document re-sync and bounded diagnostic refresh. |
|
|
16
|
+
|
|
17
|
+
## Review lifecycle
|
|
18
|
+
|
|
19
|
+
A review is created per patch handoff. The manifest binds: repository identity (credential-free remote fingerprint + default branch + optional worktree path), `base`/`head`, the patch artifact reference (`kind`, `uri`, `sha256`, `bytes`), changed paths, diffstat, named-check summaries, and diagnostic summaries. The `digest` is SHA-256 over the canonical manifest JSON; the structural artifact input carries the manifest in `preview.review` and the patch SHA-256 as the artifact hash.
|
|
20
|
+
|
|
21
|
+
```ts
|
|
22
|
+
import { createCodingPatchReviewManifest, assertCodingPatchAccepted } from "@arnilo/prism-coding-agent";
|
|
23
|
+
import { createArtifactService } from "@arnilo/prism-server";
|
|
24
|
+
|
|
25
|
+
const { review, artifactInput } = createCodingPatchReviewManifest({
|
|
26
|
+
threadId: "thread-1",
|
|
27
|
+
artifactId: "patch-1",
|
|
28
|
+
identity: { repositoryId: "app", remoteFingerprint: sha256Fingerprint, defaultBranch: "main" },
|
|
29
|
+
base: "main",
|
|
30
|
+
head: "feature-1",
|
|
31
|
+
patch: { kind: "patch", uri: "artifacts/patch-1.patch", sha256: patchSha, bytes: 4096 },
|
|
32
|
+
changedPaths: ["src/a.ts"],
|
|
33
|
+
diffstat: [{ file: "src/a.ts", additions: 10, deletions: 2 }],
|
|
34
|
+
checks: [{ name: "build", exitCode: 0, summary: "ok" }],
|
|
35
|
+
});
|
|
36
|
+
const record = await artifacts.attach({ ...artifactInput, ownership, identity });
|
|
37
|
+
|
|
38
|
+
// later, after a human approve/reject on the artifact:
|
|
39
|
+
const outcome = assertCodingPatchAccepted({ review, artifact: record });
|
|
40
|
+
// outcome.state: "accepted" | "rejected" | "pending" | "superseded"
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
State derivation rules:
|
|
44
|
+
|
|
45
|
+
- `pending` — no decision recorded for the bound revision.
|
|
46
|
+
- `accepted` — an `approved` decision exists for the exact artifact revision whose hash equals the patch digest, the revision is still the latest, and the embedded review digest and repository/worktree/base/head identity still match. Acceptance is a state assertion only — it never applies, commits, pushes, or merges automatically.
|
|
47
|
+
- `rejected` — a `rejected` decision exists for the bound revision (bounded reviewer reason).
|
|
48
|
+
- `superseded` — any change invalidates a prior acceptance: a new patch digest (no revision matches), a newer patch revision attached after the decision (stale acceptance refused), a changed review digest (patch/identity/base/head changed), or a tampered identity in the preview.
|
|
49
|
+
|
|
50
|
+
Caps (default / hard): review revisions 8/32, diagnostic summaries 500/5000, manifest bytes 64 KiB/256 KiB, delta entries 2000/10000; check summaries 8 KiB each; artifact URIs 2048 bytes. Every identity field is validated at manifest creation (`ERR_PRISM_REVIEW_INPUT`); caps charge before retention (`ERR_PRISM_REVIEW_LIMIT`); a record bound to another thread/artifact is refused (`ERR_PRISM_REVIEW_OWNERSHIP`). Raw patch bodies, commands, env, and secrets are never embedded in the manifest or artifact preview — the artifact hash is the patch digest, the body stays in the host artifact store.
|
|
51
|
+
|
|
52
|
+
## Diagnostics
|
|
53
|
+
|
|
54
|
+
`normalizeDiagnostics` accepts host-parsed raw diagnostics (hosts own the check parsers; there is no language/tool-specific parser catalog). Each entry is validated: workspace-relative path (absolute paths must stay inside the workspace root), non-negative finite positions, valid severity, non-empty message. Control characters are stripped, messages are UTF-8-truncated at the byte cap (4 KiB default / 16 KiB hard), and the per-file cap charges before retention (500 default / 5000 hard). Malformed entries are dropped fail-closed — never partially normalized.
|
|
55
|
+
|
|
56
|
+
`diagnosticDelta` computes a deterministic `added` / `removed` / `unchanged` view between generations using `diagnosticIdentity`. Duplicate identities in one side dedupe; previous views with a generation newer than the incoming view are treated as stale and ignored (a stale-version response never overwrites newer results). Same-generation views diff normally, so repeated refreshes yield `unchanged` without churn.
|
|
57
|
+
|
|
58
|
+
## LSP synchronization (opt-in)
|
|
59
|
+
|
|
60
|
+
`LanguageIntelligence` stays a standalone host-activated factory — no LSP server is spawned by construction and nothing is baked into `createCodingTools`/`createAllTools` or any agent assembly. Hosts wire it explicitly:
|
|
61
|
+
|
|
62
|
+
```ts
|
|
63
|
+
const lang = createLanguageIntelligence({ workspaceRoot, servers: { ts: {...} }, policy });
|
|
64
|
+
await lang.syncDocument("src/app.ts"); // full-content didChange, monotonic version
|
|
65
|
+
const delta = await lang.diagnosticDelta({ files: ["src/app.ts", "src/lib.ts"], previous });
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
- `syncDocument(file)` reads the file and sends a full-content `textDocument/didChange` (protocol-valid LSP 3.17; no diff engine). Versions are monotonic per document: didOpen stamps 1, each didChange increments.
|
|
69
|
+
- `diagnosticDelta({ files, previous })` refreshes each changed file: pull diagnostics (`textDocument/diagnostic` with `previousResultId` reuse, `kind: full|unchanged`) when the server advertises `diagnosticProvider`, otherwise the push cache (`textDocument/publishDiagnostics`, which always replaces the full set — publish `[]` to clear). Results are normalized, generation-stamped with the document version, and diffed against `previous`. Stale views (previous generation newer than the refresh) are dropped per file.
|
|
70
|
+
- Refresh is bounded to the requested files (never a whole-workspace pull) and to the standard LSP caps (message bytes, diagnostics/file, pending requests, results/query, timeout, servers).
|
|
71
|
+
|
|
72
|
+
## Related APIs
|
|
73
|
+
|
|
74
|
+
- `docs/work-artifacts-and-review.md`: artifact revisions and approve/reject semantics the manifest composes over.
|
|
75
|
+
- `docs/language-intelligence.md`: full LSP contract.
|
|
76
|
+
- `docs/coding-agent-tools.md`: `coding_check` named checks the manifest summarizes.
|
package/docs/coding-security.md
CHANGED
|
@@ -206,6 +206,8 @@ Containment resolves symlinks and rejects paths outside roots. Command rules are
|
|
|
206
206
|
|
|
207
207
|
Docker sandbox containment—not command regexes—enforces filesystem/network/process boundaries for the reference adapter. Network defaults to none; a custom Docker network still requires a host firewall/proxy for DNS/egress claims. Import rejects symlink escapes, devices, FIFOs, and sockets; export counts entries/bytes and hashes before host retention. Secrets in `secrets` are redacted from adapter errors and never exported as environment metadata. Unified workspace mode reuses existing sandbox/repo/coding hard caps and does not introduce unbounded host↔container sync loops. Host mode and `allowMixedWorkspaceWiring` never claim disposable containment. Durable workflow denial/cancellation is terminal and attributable; approved resume still fails if roots, command rules, read-only mode, or other policy changed while suspended. Cache keys are fixed-size SHA-256 digests of selected identity plus action shape; caches remain process-local, retain at most 1,000 decisions with oldest-entry eviction, and have no default/global mode. Path checks and cache lookup are local; sandbox latency belongs to the supplied adapter and Docker daemon.
|
|
208
208
|
|
|
209
|
+
The protected coding journey (0.2.6, plan 026 Task 7) exercises these boundaries for real at release time: scripts/phase26-coding-journey.test.mjs packs the published packages into a fresh consumer and runs the digest-pinned Docker sandbox, the host Playwright browser, the real forge, durable Postgres checkpoints/leases, and the host PTY adapter (frozen profile) — every missing service records blocked, never a passing skip, and the retained phase26-coding-journey-report.json carries timings/states/ids only (no prompts, source bodies, terminal output, tokens, or browser storage). See docs/release-and-install.md for the operator runbook.
|
|
210
|
+
|
|
209
211
|
The egress proxy is a policy enforcer, not a firewall: it cannot stop a container whose Docker network reaches the internet directly. Egress attestation (`denyDirectEgress: true`) is a claim the host must make true by network topology; the adapter records it as evidence and fails closed when it is absent or malformed. The proxy performs no TLS interception, no DNS rebinding of its own beyond pinning, and no content filtering; audit records contain no secrets. Frozen caps: 32 concurrent connections (hard 256), 64 MiB request/response bytes (hard 1 GiB), 600 s transfer time (hard 1 h), 128 rules (hard 1,024), 5 redirect hops (hard 10).
|
|
210
212
|
|
|
211
213
|
## Related APIs
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# Coding workspaces
|
|
2
|
+
|
|
3
|
+
Ownership-scoped multi-repository and worktree lifecycle (plan 026 Task 3, `@arnilo/prism-coding-agent`). A durable coding workspace correlates task/session/run identity with host repositories and linked worktrees so that resume, cleanup, artifacts, and recovery stay bounded and reconcilable.
|
|
4
|
+
|
|
5
|
+
The lifecycle composes existing bounded primitives only: `CheckpointStore` CAS records in a separate versioned namespace (`prism.coding-agent.workspace.v1`), `LeaseStore` fencing, and cwd-bound `GitOperations` runners. There is no clone manager, Git library, watcher, new database schema, or second task runtime.
|
|
6
|
+
|
|
7
|
+
## Activation
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
import { createCodingWorkspaceLifecycle } from "@arnilo/prism-coding-agent";
|
|
11
|
+
|
|
12
|
+
const workspaces = createCodingWorkspaceLifecycle({
|
|
13
|
+
checkpoints, // CheckpointStore (ownership-scoped)
|
|
14
|
+
leases, // LeaseStore
|
|
15
|
+
ownerId: replicaId, // worker/replica identity
|
|
16
|
+
ownership: { tenantId }, // part of the trust boundary
|
|
17
|
+
repositories: {
|
|
18
|
+
app: { root: "/src/app", git: appGit }, // git must be cwd-bound to root
|
|
19
|
+
api: { root: "/src/api", git: apiGit },
|
|
20
|
+
},
|
|
21
|
+
worktreeRoots: ["/work/prism"], // host-approved linked-worktree roots
|
|
22
|
+
policy: { allowDirtyCleanup: false }, // all cleanup refusals default to refuse
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
const workspace = await workspaces.create({
|
|
26
|
+
taskId: "task-42",
|
|
27
|
+
repositories: [{ repositoryId: "app", branch: "agent/task-42" }],
|
|
28
|
+
});
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Nothing starts on import or construction; worktrees are created only by explicit `create` calls. Repository roots and worktree roots are canonicalized (`realpath`) and containment-checked; the main worktree of every registered repository is immutable through this service.
|
|
32
|
+
|
|
33
|
+
## Record
|
|
34
|
+
|
|
35
|
+
`CodingWorkspaceRecord` (schemaVersion 1) holds: stable `workspaceId` (deterministic from `taskId`) and `taskId`; `ownerId`; frozen state `active | cleaning | closed | unknown`; per-repository legs with repository id, canonical root, credential-free remote fingerprint (sha256 of the redacted remote URL plus default branch — never a URL), default branch, task branch, base/head shas, worktree id/path, repository state `active | removed | unknown`, created timestamp; artifact references only (kind/uri/sha256/bytes, never contents); fencing token; created/updated/cleanup timestamps. Records are bounded (64 KiB default / 256 KiB hard).
|
|
36
|
+
|
|
37
|
+
## Operations
|
|
38
|
+
|
|
39
|
+
- `create({ taskId, repositories, artifactRefs? })` — validates task/repository/branch identity, acquires a lease, captures fingerprints, adds one linked worktree per repository (`git worktree add -b`), locks each with `prism-workspace:<id>` reason, and persists the record with a fencing-token CAS write. Duplicate create with an identical active record returns it as-is with no Git mutation; a conflicting request, a live foreign lease, or a CAS/fence conflict fails with `ERR_PRISM_WORKSPACE_FENCE`. A worktree left behind by a crashed earlier attempt is reused.
|
|
40
|
+
- `get({ taskId })` / `list({ cursor?, limit? })` — bounded reads; malformed or escaped records fail closed.
|
|
41
|
+
- `verify({ taskId })` — resume gate: revalidates repository root containment, worktree containment, worktree presence and head, and the remote/default-branch fingerprint before tools, processes, index results, patches, or artifacts are reused. Any change fails with `ERR_PRISM_WORKSPACE_FINGERPRINT` / `ERR_PRISM_WORKSPACE_PATH_ESCAPE`.
|
|
42
|
+
- `attachArtifacts({ taskId, artifactRefs })` — bounded CAS update of artifact refs (16 default / 64 hard refs).
|
|
43
|
+
- `cleanup({ taskId })` — removes owned linked worktrees and closes the record. Idempotent on `closed`; refuses while another worker cleans (`cleaning`); every mutation takes the lease and writes with a monotonic fencing token.
|
|
44
|
+
- `remove({ taskId })` — deletes the durable record only; never touches Git.
|
|
45
|
+
|
|
46
|
+
## Cleanup refusals
|
|
47
|
+
|
|
48
|
+
Cleanup refuses, unless the host policy explicitly allows the documented action:
|
|
49
|
+
|
|
50
|
+
- dirty worktree — `ERR_PRISM_WORKSPACE_DIRTY` (`allowDirtyCleanup` → forced removal, potential data loss);
|
|
51
|
+
- externally locked worktree — `ERR_PRISM_WORKSPACE_LOCKED` (`allowLockedCleanup`); locks owned by this service (`prism-workspace:<id>` reason) are always released first;
|
|
52
|
+
- missing worktree — `ERR_PRISM_WORKSPACE_UNKNOWN` (`allowMissingCleanup` → claim as removed);
|
|
53
|
+
- unowned path (exists on disk but is not a registered worktree) — `ERR_PRISM_WORKSPACE_UNKNOWN` (`allowUnownedCleanup` → unclaim without touching the foreign directory);
|
|
54
|
+
- mismatched head — `ERR_PRISM_WORKSPACE_FINGERPRINT` (`allowMismatchedCleanup` → forced removal);
|
|
55
|
+
- main-worktree path — always `ERR_PRISM_WORKSPACE_MAIN`, no policy overrides.
|
|
56
|
+
|
|
57
|
+
Partial failure persists state `unknown` with per-repository `unknown`/`removed` legs and remains reconcilable: retrying cleanup converges to `closed`.
|
|
58
|
+
|
|
59
|
+
## Ownership and fencing
|
|
60
|
+
|
|
61
|
+
Ownership scopes are part of the trust boundary: records are read and written under the configured `tenantId`/`accountId`/`userId`, and lease acquisition under another scope fails closed as `ERR_PRISM_WORKSPACE_OWNERSHIP`. Every mutation runs under a `LeaseStore` lease (`tryAcquireLease`/`releaseLease`, TTL 30 s default / 300 s hard); the lease fencing token is stored in the record and each `CheckpointStore` save is a version CAS plus a monotonic fencing-token check, so a worker whose lease lapsed or was fenced out cannot overwrite newer state. Stale workers reject deterministically with `ERR_PRISM_WORKSPACE_FENCE`.
|
|
62
|
+
|
|
63
|
+
## Errors
|
|
64
|
+
|
|
65
|
+
`ERR_PRISM_WORKSPACE_UNKNOWN`, `ERR_PRISM_WORKSPACE_LIMIT`, `ERR_PRISM_WORKSPACE_OWNERSHIP`, `ERR_PRISM_WORKSPACE_FENCE`, `ERR_PRISM_WORKSPACE_DIRTY`, `ERR_PRISM_WORKSPACE_LOCKED`, `ERR_PRISM_WORKSPACE_MAIN`, `ERR_PRISM_WORKSPACE_PATH_ESCAPE`, `ERR_PRISM_WORKSPACE_FINGERPRINT` (`WorkspaceError`).
|
|
66
|
+
|
|
67
|
+
## Caps
|
|
68
|
+
|
|
69
|
+
Repositories per task 4 / 16; worktrees 4 / 16 (git caps); record bytes 65536 / 262144; lease TTL 30000 / 300000 ms; cleanup operations 100 / 1000; artifact refs 16 / 64; artifact uri 2048 bytes; task id 128 bytes; repository id 64 bytes. Cleanup is O(worktrees owned by one task); there is no per-file worktree scan and no global timer.
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# Data classification and field-level redaction
|
|
2
|
+
|
|
3
|
+
## What it does
|
|
4
|
+
|
|
5
|
+
Field-level classification and fail-closed redaction at data boundaries (plan 027 Task 8). An explicit host policy classifies every field of a JSON-like value that crosses a boundary — provider prompt egress, tool dispatch/result persistence, artifact write/read/export, audit hashing, telemetry attributes/events, and export — and returns one of four decisions per field: `allow`, `redact`, `tokenize`, or `deny`. Unknown fields fail closed under the protected default; tenant and legal-hold context is carried through the walk. The policy walk is bounded (depth/keys/string budget, cycle detection, wall-clock budget when configured), rejects unsupported values instead of stringifying guesses, and preserves shape when redaction is required (only touched paths are allocated — untouched subtrees share the input reference).
|
|
6
|
+
|
|
7
|
+
Classification is explicit: labels come from a `labelFor` hint function supplied by the boundary owner. There is no automatic sensitive-data discovery, no global registry, no decorator framework, and no second policy language. Existing hardcoded secret redaction (`createSecretRedactor`) remains in place as defense in depth and runs before the policy pass at egress seams.
|
|
8
|
+
|
|
9
|
+
## When to use it
|
|
10
|
+
|
|
11
|
+
- When a boundary must guarantee that classified fields (secrets, financial data, personal data) never reach a sink unchanged, and unknown fields must be blocked rather than guessed at.
|
|
12
|
+
- When different destinations need different handling of the same payload — e.g. allow a field in the prompt but redact it in telemetry — the destination is part of every decision input.
|
|
13
|
+
- When the protected profile must run fail-closed: the provided `createProtectedFieldPolicy()` denies unknown labels on outbound/persisted boundaries by default.
|
|
14
|
+
- When deep-copy cost at a boundary matters: the sparse-copy walker avoids duplicate serialization and shares pristine subtrees.
|
|
15
|
+
|
|
16
|
+
Compatibility note: existing callers that do not supply a policy are untouched (identity fast path); the protected profile is what enables fail-closed behavior, and protected deployments should supply it at every boundary.
|
|
17
|
+
|
|
18
|
+
## Inputs / request
|
|
19
|
+
|
|
20
|
+
- `applyFieldPolicy(value, policy, options)` — `value` is any JSON-like structure (plain objects, arrays, primitives, `Date`/`RegExp`/buffers pass through; `Map`/`Set` normalize to object/array shapes; functions, bigints, symbols, class instances, and cycles are rejected).
|
|
21
|
+
- `policy` — `(input: FieldPolicyInput) => FieldPolicyDecision`, where the input carries `{path, destination, label?, kind, tenantId?, direction, purpose?}`.
|
|
22
|
+
- `options` — `destination` (required), `direction` (default `"outbound"`), `tenantId`, `purpose`, `labelFor` (explicit key→label hints; no auto-discovery), `onRedact` (provenance hook used by the audit adapter), `maxDepth` (32), `maxKeys` (10,000), `maxChars` (1,000,000), `maxPolicyMs` (only when set), `tokenPrefix` (`tok_`).
|
|
23
|
+
- The protected default is `createProtectedFieldPolicy({ publicLabels, deniedLabels, redactedLabels, tokenizedLabels })`: `public`/structural labels pass, `secret` and `financial` deny, `personal` redacts, `token` tokenizes, and anything unlabeled denies on outbound destinations (`prompt`, `tool`, `artifact`, `audit`, `telemetry`, `export`, `persistence`) while passing inbound.
|
|
24
|
+
|
|
25
|
+
## Outputs / response / events
|
|
26
|
+
|
|
27
|
+
- A tree of the same shape with decisions applied: `deny` replaces the value with `[DENIED]`, `redact` replaces string leaves with `[REDACTED]` while preserving containers, `tokenize` replaces string leaves with a deterministic `tok_<hash>` (stable across runs for the same path+value, safe for audit chains), `allow` keeps the value. Untouched branches share the input reference (sparse copy); the input is never mutated.
|
|
28
|
+
- `onRedact` fires once per transformed field with `{path, reason}`; values are never included.
|
|
29
|
+
- `FieldPolicyError` (code `ERR_PRISM_FIELD_POLICY`) on: policy throw, invalid decision, cyclic reference, unsupported value type, or depth/key/byte/time budget breach. Error messages contain the path and the policy error class — never the value.
|
|
30
|
+
|
|
31
|
+
## Request/response example
|
|
32
|
+
|
|
33
|
+
```ts
|
|
34
|
+
import { applyFieldPolicy, createProtectedFieldPolicy } from "@arnilo/prism";
|
|
35
|
+
|
|
36
|
+
const fieldPolicy = createProtectedFieldPolicy();
|
|
37
|
+
const labelFor = (key: string) =>
|
|
38
|
+
key === "apiKey" ? "secret" : key === "email" ? "personal" : key === "score" ? "public" : undefined;
|
|
39
|
+
|
|
40
|
+
const out = applyFieldPolicy(
|
|
41
|
+
{ score: 1, apiKey: "demo-secret-value", email: "ops@example.test", extra: "unknown" },
|
|
42
|
+
fieldPolicy,
|
|
43
|
+
{ destination: "prompt", direction: "outbound", labelFor },
|
|
44
|
+
);
|
|
45
|
+
// { score: 1, apiKey: "[DENIED]", email: "[REDACTED]", extra: "[DENIED]" }
|
|
46
|
+
|
|
47
|
+
// Audit seam: transformation precedes canonical hashing; only {path, reason} survives.
|
|
48
|
+
const redactor = createAuditFieldRedactor(fieldPolicy, { labelFor });
|
|
49
|
+
// pass redactor as the exporter's `redact` option: createAuditExporter({ redact: redactor, ... })
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Implementation example
|
|
53
|
+
|
|
54
|
+
- Root ownership: the contract lives in `src/field-policy.ts` of `@arnilo/prism` and is exported from the package index (`applyFieldPolicy`, `createProtectedFieldPolicy`, `hookFieldPolicy`-style adapter `createAuditFieldRedactor`, `ALLOW_FIELD_POLICY`, `FieldPolicyError`, `FIELD_POLICY_LIMITS`).
|
|
55
|
+
- Egress seams: `redactMessage`, `redactProviderRequest`, `redactAgentEvent`, `redactSessionEntry`, and `redactRunLedgerRecord` (from `./redaction.js`) take an optional `(fieldPolicy, destination, labelFor)` — secret redaction runs first, then the policy pass. Without a policy the functions are unchanged (identity).
|
|
56
|
+
- Audit export: `createAuditFieldRedactor(fieldPolicy, { tenantId, labelFor, purpose })` produces the structural `AuditRedactionPolicy` the exporter already applies before canonical hashing; the record's denied/redacted/tokenized bytes are exactly what gets hashed and verified.
|
|
57
|
+
- Telemetry: `createOpenTelemetryInstrumentation({ fieldPolicy })` filters or masks exported span attributes and events — `allow` keeps, `redact` masks with `[REDACTED]`, `deny` drops, `tokenize` hashes — and policy errors drop the attribute without ever echoing the value.
|
|
58
|
+
- Postgres persistence: stores persist exactly what the caller-appointed policy authorizes; the protected profile is a caller-supplied boundary control, not a store default (outbox payloads keep their canonical-collision semantics unchanged).
|
|
59
|
+
|
|
60
|
+
## Extension and configuration notes
|
|
61
|
+
|
|
62
|
+
- Label vocabulary is bounded by the boundary owner: `public`, `personal`, `secret`, `financial`, `token` in the protected default; custom profiles override `deniedLabels`/`redactedLabels`/`tokenizedLabels` maps with their own reasons.
|
|
63
|
+
- Legal hold does not broaden view/export permissions: holds are store-level (deletion prevention), and export-boundary policy denies classified/unknown fields regardless of hold flags.
|
|
64
|
+
- The `labelFor` hint function is the only discovery mechanism; it must be supplied per boundary by the owner who knows the shape. The walker never guesses labels from key names.
|
|
65
|
+
- Tokens are deterministic per (path, value) for a given run; they are not reversible by design and are not a pseudonymization system (no k-anonymity or re-identification risk model).
|
|
66
|
+
- Migration guidance: for existing callers, add the policy at the outermost seam (the redaction functions or the audit/telemetry options) and verify on canaries first; there is no global config key that enables classification, so adoption is per-boundary and explicit.
|
|
67
|
+
|
|
68
|
+
## Security and performance notes
|
|
69
|
+
|
|
70
|
+
- Fail-closed guarantees: unknown fields never cross outbound/persisted boundaries under the protected default; policy exceptions, invalid decisions, and budget breaches throw without echoing values; cycles and unsupported types throw instead of stringifying guesses; tenant mismatch can be enforced inside host policy via `tenantId` on every decision input.
|
|
71
|
+
- Secret canaries: the ERP-T9 matrix proves secret/personal/financial canaries never reach prompt, tool, artifact, audit, telemetry, persistence, or export sinks (denied = `[DENIED]`, redacted = `[REDACTED]`, tokenized = `tok_…`, and the canary string appears nowhere in transformed output or provenance lists).
|
|
72
|
+
- Bounds (frozen): depth 32, keys 10,000, string budget 1,000,000 chars, optional wall-clock budget; the walk is recursive with an active-path set — cycles terminate, diamond references are re-walked per branch.
|
|
73
|
+
- Overhead (frozen cap `classificationMaxOverheadPercent = 10`): measured against the pre-existing boundary walk (the secret-redaction walk boundaries already ran before classification existed) on the frozen representative payload sizes, interleaved A/B — prompt 4,164 B → 99.0%, toolArgs 2,114 B → 95.8%, toolResult 9,095 B → 97.1%, artifactMetadata 3,692 B → 99.0%, auditRecord 4,243 B → 99.8%, telemetry 1,760 B → 97.7%, exportPage 10,726 B → 98.0% of the redactor-walk baseline (peak 99.8%, all ≤ 110%). The raw ratio vs native `JSON.stringify` is recorded in the Task 8 evidence (≈1.0–1.3× across fixtures); a JS policy gateway cannot beat a native serializer, so the frozen cap is defined against the walk work the boundary already performed, and this stays the benchmark contract.
|
|
74
|
+
- The telemetry seam drops attributes on policy error rather than failing the whole span; the audit seam maintains redaction provenance `{path, reason}` only — values never enter the hash chain.
|
|
75
|
+
|
|
76
|
+
## Related APIs
|
|
77
|
+
|
|
78
|
+
- `redactMessage` / `redactProviderRequest` / `redactAgentEvent` / `redactSessionEntry` / `redactRunLedgerRecord` — the egress seams that take the optional policy (secret redaction first, then classification).
|
|
79
|
+
- `createAuditFieldRedactor` → the audit-export `redact` hook; see [Signed, hash-chained audit export](audit-export.md).
|
|
80
|
+
- `createOpenTelemetryInstrumentation` in `@arnilo/prism-observability-opentelemetry` — the telemetry `fieldPolicy` option.
|
|
81
|
+
- `createProtectedFieldPolicy`, `ALLOW_FIELD_POLICY`, `FieldPolicyError`, `FIELD_POLICY_LIMITS` — the protected default and limits.
|
|
82
|
+
- The ERP-T9 threat matrix (`src/__tests__/field-policy.test.ts`) and the boundary-drill scripts cover the enforcement evidence.
|