dsh-dlp 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +144 -19
- package/lib/cli.js +307 -0
- package/lib/detectors.js +87 -0
- package/lib/guard.js +9 -6
- package/lib/home.js +37 -0
- package/lib/index.js +31 -6
- package/lib/paths.js +38 -0
- package/lib/policy.js +15 -16
- package/lib/redaction.js +9 -1
- package/lib/results.js +25 -9
- package/lib/schema.js +99 -0
- package/lib/types/cli.d.ts +107 -0
- package/lib/types/detectors.d.ts +66 -0
- package/lib/types/guard.d.ts +6 -4
- package/lib/types/home.d.ts +31 -0
- package/lib/types/paths.d.ts +38 -0
- package/lib/types/policy.d.ts +0 -9
- package/lib/types/results.d.ts +15 -6
- package/lib/types/schema.d.ts +41 -0
- package/lib/types/sink.d.ts +6 -0
- package/package.json +4 -1
package/lib/guard.js
CHANGED
|
@@ -27,7 +27,7 @@
|
|
|
27
27
|
* @module dsh-dlp/guard
|
|
28
28
|
*/
|
|
29
29
|
import { DENY_SEVERITY, scanSync, severityRank } from "./detectors.js";
|
|
30
|
-
import { isEgressCapable, matchPathArgument, pathArguments, pathCandidates } from "./paths.js";
|
|
30
|
+
import { isEgressCapable, matchPathArgument, pathArguments, pathCandidates, rulesForTool } from "./paths.js";
|
|
31
31
|
import { nestedStrings } from "./redaction.js";
|
|
32
32
|
/**
|
|
33
33
|
* Denial text for a credential-path match.
|
|
@@ -57,20 +57,23 @@ function secretArgumentReason(toolName, ruleIds, hashes) {
|
|
|
57
57
|
* Credential paths are denied for every tool, not only readers: a shell that
|
|
58
58
|
* can `cat` a key can also copy it. Only path-typed arguments are tested —
|
|
59
59
|
* running the table over every string matches file content and denies writing
|
|
60
|
-
* a `.gitignore` that mentions `.env`.
|
|
61
|
-
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
60
|
+
* a `.gitignore` that mentions `.env`. The one exception is a `writes-only`
|
|
61
|
+
* rule, which a tool classified read-only is exempt from; that is how
|
|
62
|
+
* `$DSH_HOME` stays readable while every write to it is denied. Argument
|
|
63
|
+
* secrets are denied only for egress-capable tools, because denying a local
|
|
64
|
+
* editor for holding the text it was asked to write would break ordinary work
|
|
65
|
+
* without closing an exfiltration path.
|
|
64
66
|
* @param exec - the pending call as the guard stage sees it.
|
|
65
67
|
* @param policy - the effective policy after the tighten-only merge.
|
|
66
68
|
* @param hasher - mints the keyed hashes quoted in a denial reason.
|
|
67
69
|
* @returns the denial, or `undefined` to abstain.
|
|
68
70
|
*/
|
|
69
71
|
export function evaluateGuard(exec, policy, hasher) {
|
|
72
|
+
const rules = rulesForTool(exec.name, policy.credentialPathRules);
|
|
70
73
|
for (const argument of pathArguments(exec.arguments)) {
|
|
71
74
|
const candidates = argument.shell ? pathCandidates(argument.text) : [argument.text];
|
|
72
75
|
for (const candidate of candidates) {
|
|
73
|
-
const rule = matchPathArgument(candidate,
|
|
76
|
+
const rule = matchPathArgument(candidate, rules);
|
|
74
77
|
if (rule === undefined)
|
|
75
78
|
continue;
|
|
76
79
|
const hash = hasher.hash(candidate);
|
package/lib/home.js
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Where the harness keeps its state, and where this plugin's audit sink lands
|
|
3
|
+
* by default.
|
|
4
|
+
*
|
|
5
|
+
* Its own module so the `dsh-dlp report` command can resolve the sink without
|
|
6
|
+
* importing the plugin: `policy.ts` pulls in `@secretlint/core`, `js-yaml` and
|
|
7
|
+
* the schema library, none of which a reader of a JSONL file needs.
|
|
8
|
+
* @module dsh-dlp/home
|
|
9
|
+
*/
|
|
10
|
+
import { homedir } from 'node:os';
|
|
11
|
+
import { join, resolve } from 'node:path';
|
|
12
|
+
/**
|
|
13
|
+
* File name the bundle patch gives the audit sink under the harness home.
|
|
14
|
+
* `cordis.patch.yml` spells the same name; a deployment that sets `auditLog`
|
|
15
|
+
* itself must tell `dsh-dlp report` where it put it.
|
|
16
|
+
*/
|
|
17
|
+
export const DEFAULT_AUDIT_LOG_NAME = 'dsh-dlp.audit.jsonl';
|
|
18
|
+
/**
|
|
19
|
+
* Resolve the harness home the same way the harness does: `$DSH_HOME` when it
|
|
20
|
+
* is set to something other than whitespace, otherwise `~/.dsh`. Read here
|
|
21
|
+
* rather than through `@deepseek-ai/dsh-home-paths` to keep the plugin's
|
|
22
|
+
* runtime imports to the ones a profile is guaranteed to resolve.
|
|
23
|
+
* @param env - environment consulted for `DSH_HOME`; defaults to `process.env`.
|
|
24
|
+
* @returns the absolute harness home.
|
|
25
|
+
*/
|
|
26
|
+
export function resolveDshHome(env = process.env) {
|
|
27
|
+
const configured = env['DSH_HOME'];
|
|
28
|
+
return resolve(configured !== undefined && configured.trim().length > 0 ? configured : join(homedir(), '.dsh'));
|
|
29
|
+
}
|
|
30
|
+
/**
|
|
31
|
+
* Where the audit sink sits when the deployment did not name one.
|
|
32
|
+
* @param env - environment consulted for `DSH_HOME`; defaults to `process.env`.
|
|
33
|
+
* @returns the absolute path the bundle patch configures.
|
|
34
|
+
*/
|
|
35
|
+
export function defaultAuditLog(env = process.env) {
|
|
36
|
+
return join(resolveDshHome(env), DEFAULT_AUDIT_LOG_NAME);
|
|
37
|
+
}
|
package/lib/index.js
CHANGED
|
@@ -69,6 +69,21 @@ export function loadOrCreateKey(path) {
|
|
|
69
69
|
writeFileSync(path, created, { mode: 0o600 });
|
|
70
70
|
return created;
|
|
71
71
|
}
|
|
72
|
+
/**
|
|
73
|
+
* Report a plugin fault on both the deployment's logger and `process.stderr`.
|
|
74
|
+
*
|
|
75
|
+
* `ctx.logger`'s default exporter is an in-memory 1000-entry ring buffer and
|
|
76
|
+
* no shipped bundle mounts a console exporter, so a message that goes only to
|
|
77
|
+
* the logger is invisible on a stock install — which is what an invalid policy
|
|
78
|
+
* file and a failed audit write were. `process.stderr` is what the headless
|
|
79
|
+
* runner itself writes to.
|
|
80
|
+
* @param ctx - the plugin's context, used for its logger.
|
|
81
|
+
* @param message - the whole line to report; never carries a matched value.
|
|
82
|
+
*/
|
|
83
|
+
function report(ctx, message) {
|
|
84
|
+
ctx.logger.error(message);
|
|
85
|
+
process.stderr.write(`${message}\n`);
|
|
86
|
+
}
|
|
72
87
|
/**
|
|
73
88
|
* Load the repo-local policy tier, if the deployment named one.
|
|
74
89
|
*
|
|
@@ -77,7 +92,7 @@ export function loadOrCreateKey(path) {
|
|
|
77
92
|
* well-formed: the recommended `policyFile` is workspace-relative, so failing
|
|
78
93
|
* the mount would refuse to start `dsh` in every repository without one, and
|
|
79
94
|
* would let a hostile repository disable the plugin by shipping a broken file.
|
|
80
|
-
* @param ctx - the plugin's context, used only
|
|
95
|
+
* @param ctx - the plugin's context, used only to report a bad file.
|
|
81
96
|
* @param policyFile - the configured path, or `undefined` when the deployment named none.
|
|
82
97
|
* @returns the validated policy, or `undefined` when there is none to apply.
|
|
83
98
|
*/
|
|
@@ -91,7 +106,7 @@ function loadConfiguredPolicy(ctx, policyFile) {
|
|
|
91
106
|
case 'loaded':
|
|
92
107
|
return load.policy;
|
|
93
108
|
case 'invalid':
|
|
94
|
-
ctx
|
|
109
|
+
report(ctx, `dsh-dlp: ignoring the repo-local policy at ${policyFile}: ${load.problem}`);
|
|
95
110
|
return undefined;
|
|
96
111
|
/* v8 ignore next 4 -- unreachable while `RepoPolicyLoad` stays closed; the arm exists so adding a variant fails the build. */
|
|
97
112
|
default: {
|
|
@@ -110,7 +125,7 @@ export function apply(ctx, config) {
|
|
|
110
125
|
const hasher = new SpanHasher(loadOrCreateKey(config.redactionKeyFile));
|
|
111
126
|
const correlator = new CallCorrelator();
|
|
112
127
|
const sink = new AuditSink(config.auditLog, (error) => {
|
|
113
|
-
ctx
|
|
128
|
+
report(ctx, `dsh-dlp: audit sink write failed: ${String(error)}`);
|
|
114
129
|
});
|
|
115
130
|
/** Identity every audit record carries; the session-log envelope carries none of it. */
|
|
116
131
|
const identity = (exec) => {
|
|
@@ -169,11 +184,20 @@ export function apply(ctx, config) {
|
|
|
169
184
|
}
|
|
170
185
|
if (policy.resultRedaction) {
|
|
171
186
|
ctx.on('tools/post-execute', async (exec, result, next) => {
|
|
172
|
-
|
|
187
|
+
// The live definition, so a schema the deployment's own tool declares is
|
|
188
|
+
// read from the registry rather than assumed. `exec.agent` is the scope
|
|
189
|
+
// key: a scoped tool shadows a global one of the same name.
|
|
190
|
+
const outputSchema = ctx.tools.get(exec.name, exec.agent)?.output.schema;
|
|
191
|
+
const redacted = await redactDecision(await next(), result, policy, hasher, outputSchema);
|
|
192
|
+
const indicators = Object.keys(redacted.indicators).length > 0
|
|
193
|
+
? { unicode: redacted.indicators }
|
|
194
|
+
: {};
|
|
173
195
|
// A truncated scan is recorded even with nothing found: without a record
|
|
174
196
|
// an operator cannot tell "this result was clean" from "this result was
|
|
175
|
-
// only partly examined".
|
|
176
|
-
|
|
197
|
+
// only partly examined". An invisible-character run is recorded on the
|
|
198
|
+
// same terms, because the `report` classes are never replaced and the
|
|
199
|
+
// count is the only trace they leave.
|
|
200
|
+
if (redacted.spans.length > 0 || redacted.truncatedScan || Object.keys(indicators).length > 0) {
|
|
177
201
|
sink.write({
|
|
178
202
|
v: RECORD_VERSION,
|
|
179
203
|
time: new Date().toISOString(),
|
|
@@ -182,6 +206,7 @@ export function apply(ctx, config) {
|
|
|
182
206
|
...identity(exec),
|
|
183
207
|
spans: redacted.spans,
|
|
184
208
|
...redacted.truncatedScan ? { truncatedScan: true } : {},
|
|
209
|
+
...indicators,
|
|
185
210
|
});
|
|
186
211
|
}
|
|
187
212
|
return redacted.decision;
|
package/lib/paths.js
CHANGED
|
@@ -225,3 +225,41 @@ export function isEgressCapable(toolName, extraEgressTools = new Set()) {
|
|
|
225
225
|
return true;
|
|
226
226
|
return !LOCAL_TOOLS.has(toolName);
|
|
227
227
|
}
|
|
228
|
+
/**
|
|
229
|
+
* Tools that can only look: they query the filesystem, the language server,
|
|
230
|
+
* the session store or a running job, and have no operation that changes
|
|
231
|
+
* anything.
|
|
232
|
+
*
|
|
233
|
+
* A `writes-only` rule is lifted for these names and for no others. Every
|
|
234
|
+
* shell, `run_code`, every editor, every `mcp__*` tool and any tool this build
|
|
235
|
+
* has never heard of stays on the deny side, so a new tool is denied until it
|
|
236
|
+
* is classified — the same default as {@link LOCAL_TOOLS}, in the same
|
|
237
|
+
* direction.
|
|
238
|
+
*/
|
|
239
|
+
export const READ_ONLY_TOOLS = new Set([
|
|
240
|
+
'read', 'read_image', 'glob', 'grep', 'lsp',
|
|
241
|
+
'session_search', 'session_trace',
|
|
242
|
+
'session_event_read', 'session_event_search', 'session_event_trace',
|
|
243
|
+
'list_agents', 'job_list', 'job_output',
|
|
244
|
+
'terminal_list', 'terminal_read',
|
|
245
|
+
'get_goal',
|
|
246
|
+
]);
|
|
247
|
+
/**
|
|
248
|
+
* Whether a tool is known to be incapable of changing anything.
|
|
249
|
+
* @param toolName - the executing tool's registered name.
|
|
250
|
+
* @returns `true` only for a name in {@link READ_ONLY_TOOLS}.
|
|
251
|
+
*/
|
|
252
|
+
export function isReadOnlyTool(toolName) {
|
|
253
|
+
return READ_ONLY_TOOLS.has(toolName);
|
|
254
|
+
}
|
|
255
|
+
/**
|
|
256
|
+
* The credential-path rules one tool is judged against.
|
|
257
|
+
* @param toolName - the executing tool's registered name.
|
|
258
|
+
* @param rules - the effective rule table.
|
|
259
|
+
* @returns every rule, minus the `writes-only` ones for a read-only tool.
|
|
260
|
+
*/
|
|
261
|
+
export function rulesForTool(toolName, rules) {
|
|
262
|
+
if (!isReadOnlyTool(toolName))
|
|
263
|
+
return rules;
|
|
264
|
+
return rules.filter(rule => rule.enforcement !== 'writes-only');
|
|
265
|
+
}
|
package/lib/policy.js
CHANGED
|
@@ -15,11 +15,11 @@
|
|
|
15
15
|
* @module dsh-dlp/policy
|
|
16
16
|
*/
|
|
17
17
|
import { readFileSync } from 'node:fs';
|
|
18
|
-
import {
|
|
19
|
-
import { join, resolve } from 'node:path';
|
|
18
|
+
import { resolve } from 'node:path';
|
|
20
19
|
import { JSON_SCHEMA, load } from 'js-yaml';
|
|
21
20
|
import z from '@deepseek-ai/schemastery';
|
|
22
21
|
import { SYNC_RULES, severityRank } from "./detectors.js";
|
|
22
|
+
import { resolveDshHome } from "./home.js";
|
|
23
23
|
import { CREDENTIAL_PATH_RULES } from "./paths.js";
|
|
24
24
|
export const Config = z.object({
|
|
25
25
|
auditLog: z.string().required(),
|
|
@@ -216,36 +216,35 @@ function escapePattern(literal) {
|
|
|
216
216
|
return literal.replace(/[.*+?^${}()|[\]\\]/g, String.raw `\$&`);
|
|
217
217
|
}
|
|
218
218
|
/**
|
|
219
|
-
* Deny rules protecting this plugin's own state.
|
|
219
|
+
* Deny rules protecting this plugin's own state and the harness home.
|
|
220
220
|
*
|
|
221
221
|
* Every one of these is known at mount: the key file whose bytes make a
|
|
222
222
|
* placeholder hash keyed rather than a bare digest, the append-only sink that
|
|
223
223
|
* is the only evidence a decision happened, and the harness home holding the
|
|
224
224
|
* provider credentials, the session logs and the profiles that decide which
|
|
225
225
|
* plugins load at all.
|
|
226
|
+
*
|
|
227
|
+
* The harness home is split. **Writing** anywhere under it is denied for every
|
|
228
|
+
* tool: a prompt-injected agent editing a profile's `cordis.yml` mounts an
|
|
229
|
+
* arbitrary plugin. **Reading** it is denied only where the contents are
|
|
230
|
+
* credentials — `.credentials.yaml`, `.env` and `*.key` are already in the
|
|
231
|
+
* built-in table, and the session logs get a rule here. Everything else under
|
|
232
|
+
* that directory is the installed plugin tree and the profile manifests, which
|
|
233
|
+
* a user debugging a plugin has every reason to read; a blanket read denial
|
|
234
|
+
* there was unoverridable and was the plugin's most likely uninstall reason.
|
|
226
235
|
* @param config - the deployment-controlled configuration.
|
|
227
236
|
* @param dshHome - the resolved harness home.
|
|
228
237
|
* @returns rules appended after the built-in table.
|
|
229
238
|
*/
|
|
230
239
|
function selfProtectionRules(config, dshHome) {
|
|
240
|
+
const home = escapePattern(resolve(dshHome));
|
|
231
241
|
return [
|
|
232
242
|
{ id: 'dsh-dlp/path-own-redaction-key', version: 1, pattern: new RegExp(`^${escapePattern(resolve(config.redactionKeyFile))}$`, 'i') },
|
|
233
243
|
{ id: 'dsh-dlp/path-own-audit-log', version: 1, pattern: new RegExp(`^${escapePattern(resolve(config.auditLog))}$`, 'i') },
|
|
234
|
-
{ id: 'dsh-dlp/path-dsh-
|
|
244
|
+
{ id: 'dsh-dlp/path-dsh-sessions', version: 1, pattern: new RegExp(`^${home}/sessions(/|$)`, 'i') },
|
|
245
|
+
{ id: 'dsh-dlp/path-dsh-home', version: 2, enforcement: 'writes-only', pattern: new RegExp(`^${home}(/|$)`, 'i') },
|
|
235
246
|
];
|
|
236
247
|
}
|
|
237
|
-
/**
|
|
238
|
-
* Resolve the harness home the same way the harness does: `$DSH_HOME` when it
|
|
239
|
-
* is set to something other than whitespace, otherwise `~/.dsh`. Read here
|
|
240
|
-
* rather than through `@deepseek-ai/dsh-home-paths` to keep the plugin's
|
|
241
|
-
* runtime imports to the ones a profile is guaranteed to resolve.
|
|
242
|
-
* @param env - environment consulted for `DSH_HOME`; defaults to `process.env`.
|
|
243
|
-
* @returns the absolute harness home.
|
|
244
|
-
*/
|
|
245
|
-
export function resolveDshHome(env = process.env) {
|
|
246
|
-
const configured = env['DSH_HOME'];
|
|
247
|
-
return resolve(configured !== undefined && configured.trim().length > 0 ? configured : join(homedir(), '.dsh'));
|
|
248
|
-
}
|
|
249
248
|
/**
|
|
250
249
|
* Merge the deployment config with an optional repo-local policy.
|
|
251
250
|
* @param config - the deployment-controlled configuration.
|
package/lib/redaction.js
CHANGED
|
@@ -80,7 +80,15 @@ function expand(text, start, end) {
|
|
|
80
80
|
/** Detections merged into non-overlapping regions, each attributed to its strictest rule. */
|
|
81
81
|
function mergeSpans(text, detections) {
|
|
82
82
|
const expanded = detections
|
|
83
|
-
.map(
|
|
83
|
+
.map(({ ruleId, ruleVersion, severity, exact, start, end }) => ({
|
|
84
|
+
ruleId,
|
|
85
|
+
ruleVersion,
|
|
86
|
+
severity,
|
|
87
|
+
// An exact detection covers precisely what must go: widening an
|
|
88
|
+
// invisible character to its delimiters would delete the visible word
|
|
89
|
+
// around it.
|
|
90
|
+
...exact === true ? { start, end } : expand(text, start, end),
|
|
91
|
+
}))
|
|
84
92
|
.sort((left, right) => left.start - right.start || left.end - right.end);
|
|
85
93
|
const merged = [];
|
|
86
94
|
for (const candidate of expanded) {
|
package/lib/results.js
CHANGED
|
@@ -10,9 +10,10 @@
|
|
|
10
10
|
* whole rule set applies here and not in the guard.
|
|
11
11
|
* @module dsh-dlp/results
|
|
12
12
|
*/
|
|
13
|
-
import { DENY_SEVERITY, scanAll, scanSync, severityRank, } from "./detectors.js";
|
|
13
|
+
import { DENY_SEVERITY, countUnicodeIndicators, scanAll, scanSync, severityRank, } from "./detectors.js";
|
|
14
14
|
import { isEgressCapable } from "./paths.js";
|
|
15
15
|
import { nestedStrings, redactContent, redactJson } from "./redaction.js";
|
|
16
|
+
import { redactionBreaksSchema } from "./schema.js";
|
|
16
17
|
/**
|
|
17
18
|
* Separator the strings of one result are rendered with before the
|
|
18
19
|
* cross-string scan. A newline is what the reader of a tool result sees
|
|
@@ -38,7 +39,7 @@ const RENDER_SEPARATOR = '\n';
|
|
|
38
39
|
* is scanned depend on the same accident.
|
|
39
40
|
* @param strings - every string that will be redacted, in render order.
|
|
40
41
|
* @param policy - the effective policy.
|
|
41
|
-
* @returns a memoized lookup and whether tier 2 saw less than the whole rendering.
|
|
42
|
+
* @returns a memoized lookup, the invisible-character counts, and whether tier 2 saw less than the whole rendering.
|
|
42
43
|
*/
|
|
43
44
|
async function prepareScan(strings, policy) {
|
|
44
45
|
const rendered = strings.join(RENDER_SEPARATOR);
|
|
@@ -69,7 +70,7 @@ async function prepareScan(strings, policy) {
|
|
|
69
70
|
}
|
|
70
71
|
}
|
|
71
72
|
/* v8 ignore next -- every string handed to the walkers was collected for this memo. */
|
|
72
|
-
return { scan: text => memo.get(text) ?? [], truncated };
|
|
73
|
+
return { scan: text => memo.get(text) ?? [], truncated, indicators: countUnicodeIndicators(rendered) };
|
|
73
74
|
}
|
|
74
75
|
/** Text carried by a content array's text blocks. */
|
|
75
76
|
function contentStrings(blocks) {
|
|
@@ -123,10 +124,11 @@ function withheldFeedback(spans) {
|
|
|
123
124
|
* carries a secret, or a value that still scans dirty after redaction.
|
|
124
125
|
* Blocking replaces the whole result, which is the only way to drop `meta`.
|
|
125
126
|
*
|
|
126
|
-
* Replacing the value
|
|
127
|
-
*
|
|
128
|
-
*
|
|
129
|
-
*
|
|
127
|
+
* Replacing the value is re-validated by the registry against the tool's
|
|
128
|
+
* `output.schema`, and a schema that pins the redacted string rejects it. That
|
|
129
|
+
* surfaces as a `ToolOutputError` naming a validation failure, which tells the
|
|
130
|
+
* model nothing it can act on, so the schema is checked here first and the
|
|
131
|
+
* result is withheld with this plugin's own explanation instead.
|
|
130
132
|
*
|
|
131
133
|
* A downstream `accept{content}` over a dirty value is overruled by the value
|
|
132
134
|
* arm, which discards that listener's presentation choice. Keeping it would
|
|
@@ -136,9 +138,10 @@ function withheldFeedback(spans) {
|
|
|
136
138
|
* @param result - the dispatch outcome the waterfall was called with.
|
|
137
139
|
* @param policy - the effective policy.
|
|
138
140
|
* @param hasher - mints each span's keyed hash.
|
|
141
|
+
* @param outputSchema - the executing tool's declared output schema, when one could be resolved.
|
|
139
142
|
* @returns the decision to return, the spans replaced, and scan completeness.
|
|
140
143
|
*/
|
|
141
|
-
export async function redactDecision(decision, result, policy, hasher) {
|
|
144
|
+
export async function redactDecision(decision, result, policy, hasher, outputSchema) {
|
|
142
145
|
if (decision.kind === 'block') {
|
|
143
146
|
const prepared = await prepareScan(contentStrings(decision.feedback), policy);
|
|
144
147
|
const redacted = redactContent(decision.feedback, prepared.scan, hasher);
|
|
@@ -146,6 +149,7 @@ export async function redactDecision(decision, result, policy, hasher) {
|
|
|
146
149
|
decision: redacted.changed ? { ...decision, feedback: redacted.content } : decision,
|
|
147
150
|
spans: redacted.spans,
|
|
148
151
|
truncatedScan: prepared.truncated,
|
|
152
|
+
indicators: prepared.indicators,
|
|
149
153
|
};
|
|
150
154
|
}
|
|
151
155
|
const replacedValue = Object.hasOwn(decision, 'value') ? decision.value : undefined;
|
|
@@ -164,6 +168,14 @@ export async function redactDecision(decision, result, policy, hasher) {
|
|
|
164
168
|
const dirty = (strings) => strings.some(text => prepared.scan(text).length > 0);
|
|
165
169
|
if (value !== undefined && dirty(nestedStrings(value))) {
|
|
166
170
|
const redacted = redactJson(value, prepared.scan, hasher);
|
|
171
|
+
if (redactionBreaksSchema(outputSchema, value, redacted.value)) {
|
|
172
|
+
return {
|
|
173
|
+
decision: { kind: 'block', feedback: withheldFeedback(redacted.spans) },
|
|
174
|
+
spans: redacted.spans,
|
|
175
|
+
truncatedScan: prepared.truncated,
|
|
176
|
+
indicators: prepared.indicators,
|
|
177
|
+
};
|
|
178
|
+
}
|
|
167
179
|
const remaining = nestedStrings(redacted.value);
|
|
168
180
|
const residual = await prepareScan(remaining, policy);
|
|
169
181
|
if (remaining.some(text => residual.scan(text).length > 0)) {
|
|
@@ -171,12 +183,14 @@ export async function redactDecision(decision, result, policy, hasher) {
|
|
|
171
183
|
decision: { kind: 'block', feedback: withheldFeedback(redacted.spans) },
|
|
172
184
|
spans: redacted.spans,
|
|
173
185
|
truncatedScan: prepared.truncated || residual.truncated,
|
|
186
|
+
indicators: prepared.indicators,
|
|
174
187
|
};
|
|
175
188
|
}
|
|
176
189
|
return {
|
|
177
190
|
decision: { kind: 'accept', value: redacted.value, ...contexts },
|
|
178
191
|
spans: redacted.spans,
|
|
179
192
|
truncatedScan: prepared.truncated,
|
|
193
|
+
indicators: prepared.indicators,
|
|
180
194
|
};
|
|
181
195
|
}
|
|
182
196
|
// The value is clean, so the durable result is clean unless `meta` — which
|
|
@@ -187,17 +201,19 @@ export async function redactDecision(decision, result, policy, hasher) {
|
|
|
187
201
|
decision: { kind: 'block', feedback: withheldFeedback(spans) },
|
|
188
202
|
spans,
|
|
189
203
|
truncatedScan: prepared.truncated,
|
|
204
|
+
indicators: prepared.indicators,
|
|
190
205
|
};
|
|
191
206
|
}
|
|
192
207
|
const blocks = replacedContent ?? result.content;
|
|
193
208
|
const redacted = redactContent(blocks, prepared.scan, hasher);
|
|
194
209
|
if (!redacted.changed) {
|
|
195
|
-
return { decision, spans: [], truncatedScan: prepared.truncated };
|
|
210
|
+
return { decision, spans: [], truncatedScan: prepared.truncated, indicators: prepared.indicators };
|
|
196
211
|
}
|
|
197
212
|
return {
|
|
198
213
|
decision: { kind: 'accept', content: redacted.content, ...contexts },
|
|
199
214
|
spans: redacted.spans,
|
|
200
215
|
truncatedScan: prepared.truncated,
|
|
216
|
+
indicators: prepared.indicators,
|
|
201
217
|
};
|
|
202
218
|
}
|
|
203
219
|
/**
|
package/lib/schema.js
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Whether a redacted value still satisfies the tool's declared `output.schema`.
|
|
3
|
+
*
|
|
4
|
+
* Replacing a canonical value makes the registry re-validate it, so a schema
|
|
5
|
+
* that pins the redacted string turns a redaction into a `ToolOutputError` the
|
|
6
|
+
* model cannot act on. Asking the question first lets the listener withhold
|
|
7
|
+
* the result with its own explanation instead.
|
|
8
|
+
*
|
|
9
|
+
* This is a re-implementation of the harness's own check rather than a call
|
|
10
|
+
* into it: every harness type this package uses is imported with `import
|
|
11
|
+
* type`, so nothing from `@deepseek-ai/dsh-*` is emitted as a runtime import
|
|
12
|
+
* and the plugin resolves from a profile directory that has none of them
|
|
13
|
+
* installed. The enforced subset is small — `type`, `oneOf`, `properties`,
|
|
14
|
+
* `required`, `additionalProperties`, `items`, `enum`, `const` — and the
|
|
15
|
+
* caller guards against any disagreement by checking the *original* value
|
|
16
|
+
* first: a value this module rejects before redaction means the answer cannot
|
|
17
|
+
* be trusted, and the redaction proceeds as it did before.
|
|
18
|
+
* @module dsh-dlp/schema
|
|
19
|
+
*/
|
|
20
|
+
/** Whether a value is a plain JSON object rather than an array or a null. */
|
|
21
|
+
function isRecord(value) {
|
|
22
|
+
return typeof value === 'object' && value !== null && !Array.isArray(value);
|
|
23
|
+
}
|
|
24
|
+
/** Whether a scalar passes the node's `enum` and `const` constraints. */
|
|
25
|
+
function allowsScalar(node, value) {
|
|
26
|
+
if (node.enum !== undefined && !node.enum.includes(value))
|
|
27
|
+
return false;
|
|
28
|
+
return node.const === undefined || value === node.const;
|
|
29
|
+
}
|
|
30
|
+
/** Whether every declared property of an object node is satisfied. */
|
|
31
|
+
function satisfiesObject(node, value) {
|
|
32
|
+
const properties = node.properties ?? {};
|
|
33
|
+
for (const key of node.required ?? []) {
|
|
34
|
+
if (!Object.hasOwn(value, key) || value[key] === undefined)
|
|
35
|
+
return false;
|
|
36
|
+
}
|
|
37
|
+
if (node.additionalProperties === false) {
|
|
38
|
+
for (const key of Object.keys(value)) {
|
|
39
|
+
if (!Object.hasOwn(properties, key))
|
|
40
|
+
return false;
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
return Object.entries(properties).every(([key, child]) => !Object.hasOwn(value, key) || value[key] === undefined || satisfiesJsonSchema(child, value[key]));
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* Whether one value satisfies one schema node.
|
|
47
|
+
* @param node - a node of the tool's declared output schema.
|
|
48
|
+
* @param value - the candidate value.
|
|
49
|
+
* @returns `true` when the registry's own validation would accept it.
|
|
50
|
+
*/
|
|
51
|
+
export function satisfiesJsonSchema(node, value) {
|
|
52
|
+
if (node.oneOf !== undefined) {
|
|
53
|
+
return node.oneOf.filter(branch => satisfiesJsonSchema(branch, value)).length === 1;
|
|
54
|
+
}
|
|
55
|
+
switch (node.type) {
|
|
56
|
+
case undefined:
|
|
57
|
+
return true;
|
|
58
|
+
case 'object':
|
|
59
|
+
return isRecord(value) && satisfiesObject(node, value);
|
|
60
|
+
case 'array': {
|
|
61
|
+
const items = node.items;
|
|
62
|
+
return Array.isArray(value) && (items === undefined || value.every(item => satisfiesJsonSchema(items, item)));
|
|
63
|
+
}
|
|
64
|
+
case 'string':
|
|
65
|
+
return typeof value === 'string' && allowsScalar(node, value);
|
|
66
|
+
case 'number':
|
|
67
|
+
return typeof value === 'number' && Number.isFinite(value) && allowsScalar(node, value);
|
|
68
|
+
case 'integer':
|
|
69
|
+
return typeof value === 'number' && Number.isInteger(value) && allowsScalar(node, value);
|
|
70
|
+
case 'boolean':
|
|
71
|
+
return typeof value === 'boolean' && allowsScalar(node, value);
|
|
72
|
+
case 'null':
|
|
73
|
+
return value === null && allowsScalar(node, value);
|
|
74
|
+
/* v8 ignore next 4 -- unreachable while `JsonSchemaType` stays closed; the arm exists so adding a type fails the build. */
|
|
75
|
+
default: {
|
|
76
|
+
const unhandled = node.type;
|
|
77
|
+
throw new TypeError(`dsh-dlp: unhandled JSON schema type ${JSON.stringify(unhandled)}`);
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Whether replacing a value with its redacted copy would fail the tool's
|
|
83
|
+
* output validation.
|
|
84
|
+
*
|
|
85
|
+
* A schema this module already rejects for the original value is one it does
|
|
86
|
+
* not model correctly, so the answer is `false` and the registry decides — the
|
|
87
|
+
* check can withhold a result, and it must never do so on its own confusion.
|
|
88
|
+
* @param schema - the tool's declared output schema, when one could be resolved.
|
|
89
|
+
* @param original - the value the tool produced.
|
|
90
|
+
* @param redacted - the value the redaction pass produced.
|
|
91
|
+
* @returns `true` only when the original validates and the redacted one does not.
|
|
92
|
+
*/
|
|
93
|
+
export function redactionBreaksSchema(schema, original, redacted) {
|
|
94
|
+
if (schema === undefined)
|
|
95
|
+
return false;
|
|
96
|
+
if (!satisfiesJsonSchema(schema, original))
|
|
97
|
+
return false;
|
|
98
|
+
return !satisfiesJsonSchema(schema, redacted);
|
|
99
|
+
}
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* `dsh-dlp report` — read this plugin's audit JSONL and say what it decided.
|
|
4
|
+
*
|
|
5
|
+
* The sink is the only evidence a decision happened, and nothing read it: a
|
|
6
|
+
* user could not answer "what did this block today?". This command reads the
|
|
7
|
+
* file directly and imports nothing from the harness, so it runs wherever the
|
|
8
|
+
* package is installed, with no profile and no `dsh` on the path.
|
|
9
|
+
*
|
|
10
|
+
* The file is a durable boundary — written by an older version of this
|
|
11
|
+
* package, appended to under crash — so every line is parsed defensively and a
|
|
12
|
+
* line that is not a record is counted rather than trusted.
|
|
13
|
+
* @module dsh-dlp/cli
|
|
14
|
+
*/
|
|
15
|
+
/** One line of the audit file, after the fields this command reads are checked. */
|
|
16
|
+
export interface ReportRecord {
|
|
17
|
+
/** Epoch milliseconds parsed from the record's ISO time. */
|
|
18
|
+
readonly time: number;
|
|
19
|
+
readonly kind: string;
|
|
20
|
+
readonly tool?: string;
|
|
21
|
+
readonly sessionId?: string;
|
|
22
|
+
/** Rule ids named by the record's spans, without repeats. */
|
|
23
|
+
readonly ruleIds: readonly string[];
|
|
24
|
+
/** Invisible-character runs by rule id. */
|
|
25
|
+
readonly unicode: Readonly<Record<string, number>>;
|
|
26
|
+
}
|
|
27
|
+
/**
|
|
28
|
+
* Parse one JSONL line into the fields this command reports on.
|
|
29
|
+
* @param line - one line of the audit file.
|
|
30
|
+
* @returns the record, or `undefined` when the line is not one.
|
|
31
|
+
*/
|
|
32
|
+
export declare function parseRecord(line: string): ReportRecord | undefined;
|
|
33
|
+
/** What `report` was asked for. */
|
|
34
|
+
export interface ReportOptions {
|
|
35
|
+
readonly log: string;
|
|
36
|
+
/** Epoch milliseconds; records before it are left out. */
|
|
37
|
+
readonly since?: number;
|
|
38
|
+
readonly session?: string;
|
|
39
|
+
/** Keep only the decisions that let the call through. */
|
|
40
|
+
readonly wouldHave: boolean;
|
|
41
|
+
}
|
|
42
|
+
/** The outcome of reading the command line. */
|
|
43
|
+
export type Invocation = {
|
|
44
|
+
readonly kind: 'report';
|
|
45
|
+
readonly options: ReportOptions;
|
|
46
|
+
} | {
|
|
47
|
+
readonly kind: 'help';
|
|
48
|
+
} | {
|
|
49
|
+
readonly kind: 'error';
|
|
50
|
+
readonly message: string;
|
|
51
|
+
};
|
|
52
|
+
/** Text printed for `--help` and alongside a usage error. */
|
|
53
|
+
export declare const USAGE: string;
|
|
54
|
+
/**
|
|
55
|
+
* Read `--since`: an ISO timestamp, or a span back from now.
|
|
56
|
+
* @param value - the argument as written.
|
|
57
|
+
* @param now - epoch milliseconds a relative span counts back from.
|
|
58
|
+
* @returns epoch milliseconds, or `undefined` when the value is neither.
|
|
59
|
+
*/
|
|
60
|
+
export declare function parseSince(value: string, now: number): number | undefined;
|
|
61
|
+
/**
|
|
62
|
+
* Read the command line.
|
|
63
|
+
* @param argv - arguments after the program name.
|
|
64
|
+
* @param env - environment used for the default sink path.
|
|
65
|
+
* @param now - epoch milliseconds a relative `--since` counts back from.
|
|
66
|
+
* @returns what to run, or the usage error to print.
|
|
67
|
+
*/
|
|
68
|
+
export declare function parseArguments(argv: readonly string[], env: NodeJS.ProcessEnv, now: number): Invocation;
|
|
69
|
+
/** Outcome of reading the audit file. */
|
|
70
|
+
export type AuditFileRead =
|
|
71
|
+
/** No file at that path: nothing has been recorded, or the sink is elsewhere. */
|
|
72
|
+
{
|
|
73
|
+
readonly kind: 'absent';
|
|
74
|
+
} | {
|
|
75
|
+
readonly kind: 'unreadable';
|
|
76
|
+
readonly problem: string;
|
|
77
|
+
} | {
|
|
78
|
+
readonly kind: 'read';
|
|
79
|
+
readonly records: readonly ReportRecord[];
|
|
80
|
+
/** Lines that were not records; a torn final append is the expected cause. */
|
|
81
|
+
readonly unreadable: number;
|
|
82
|
+
};
|
|
83
|
+
/**
|
|
84
|
+
* Read and parse the audit file.
|
|
85
|
+
* @param path - the file to read.
|
|
86
|
+
* @returns its records, its absence, or the problem to print.
|
|
87
|
+
*/
|
|
88
|
+
export declare function readAuditFile(path: string): AuditFileRead;
|
|
89
|
+
/**
|
|
90
|
+
* Render the report.
|
|
91
|
+
* @param records - every record the file yielded.
|
|
92
|
+
* @param unreadable - how many of its lines were not records.
|
|
93
|
+
* @param options - the filters the invocation asked for.
|
|
94
|
+
* @returns the lines to print.
|
|
95
|
+
*/
|
|
96
|
+
export declare function formatReport(records: readonly ReportRecord[], unreadable: number, options: ReportOptions): string[];
|
|
97
|
+
/**
|
|
98
|
+
* Run one invocation.
|
|
99
|
+
* @param argv - arguments after the program name.
|
|
100
|
+
* @param write - receives each line of output.
|
|
101
|
+
* @param fail - receives each line of error output.
|
|
102
|
+
* @param env - environment used for the default sink path.
|
|
103
|
+
* @param now - epoch milliseconds a relative `--since` counts back from.
|
|
104
|
+
* @returns the process exit code.
|
|
105
|
+
*/
|
|
106
|
+
export declare function main(argv: readonly string[], write: (line: string) => void, fail: (line: string) => void, env?: NodeJS.ProcessEnv, now?: number): number;
|
|
107
|
+
//# sourceMappingURL=cli.d.ts.map
|