@argszero/cordis-plugin-sandbox-grant-advisor 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +205 -76
- package/cordis.patch.yml +18 -0
- package/lib/advice.js +130 -18
- package/lib/index.js +208 -51
- package/lib/mode.js +135 -0
- package/lib/signature.js +56 -5
- package/lib/state.js +121 -44
- package/lib/types/advice.d.ts +82 -20
- package/lib/types/index.d.ts +81 -36
- package/lib/types/mode.d.ts +87 -0
- package/lib/types/signature.d.ts +68 -5
- package/lib/types/state.d.ts +84 -34
- package/package.json +4 -3
package/lib/state.js
CHANGED
|
@@ -6,21 +6,52 @@
|
|
|
6
6
|
* failing call carries `exec.agent`, one agent owns one session, and a WeakMap
|
|
7
7
|
* keyed by the agent object lets a finished session's state be collected.
|
|
8
8
|
*
|
|
9
|
-
*
|
|
10
|
-
* feeding on itself:
|
|
9
|
+
* ## One record per family
|
|
11
10
|
*
|
|
12
|
-
*
|
|
11
|
+
* This plugin now recognizes two unrelated environment failures — a workspace
|
|
12
|
+
* that cannot be provisioned (`acl-provisioning`) and a persistent shell that
|
|
13
|
+
* cannot start (`pty-startup`). They are different diagnoses with different
|
|
14
|
+
* remedies, so their bookkeeping is kept apart under one agent
|
|
15
|
+
* ({@link AgentState.families}): an agent that hits both is told about both,
|
|
16
|
+
* and an agent that has already been told about one is still told about the
|
|
17
|
+
* other. Sharing one "already advised" flag would silently swallow the second
|
|
18
|
+
* diagnosis, which is the failure mode this split exists to prevent.
|
|
19
|
+
*
|
|
20
|
+
* Three counters, three meanings — keeping them apart is what stops the plugin
|
|
21
|
+
* from feeding on itself:
|
|
22
|
+
*
|
|
23
|
+
* - `observations` counts *failures of this environment in this family*, i.e.
|
|
13
24
|
* tool results the environment itself produced. A call this plugin denied is
|
|
14
|
-
* not one of them, even though its denial text quotes the
|
|
25
|
+
* not one of them, even though its denial text quotes the producer's line.
|
|
15
26
|
* - `failingKeys` holds the call identities (tool + canonical arguments) that
|
|
16
27
|
* have already failed this way. The fail-fast half may only refuse a call it
|
|
17
28
|
* has *watched fail* — never a call it merely recognizes as similar.
|
|
18
|
-
*
|
|
19
|
-
* `
|
|
20
|
-
* succeeds (see `observeSuccess`), not carried for the whole session.
|
|
29
|
+
* - `denials` is spent per *episode*: it is re-armed when a watched call finally
|
|
30
|
+
* succeeds (see `observeSuccess`), not carried for the whole session.
|
|
21
31
|
*
|
|
22
32
|
* @module
|
|
23
33
|
*/
|
|
34
|
+
/** The state of an agent this plugin has never observed. */
|
|
35
|
+
export function emptyState() {
|
|
36
|
+
return { families: {}, withheld: false };
|
|
37
|
+
}
|
|
38
|
+
/** The record for one family, if this agent has one. */
|
|
39
|
+
function familyOf(state, family) {
|
|
40
|
+
return state?.families[family];
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* Replace one family's record, leaving the others (and the withheld flag) alone.
|
|
44
|
+
* @param state - the agent's current state, or undefined on first sight.
|
|
45
|
+
* @param family - the family being updated.
|
|
46
|
+
* @param record - the family's new record.
|
|
47
|
+
* @returns the updated state.
|
|
48
|
+
*/
|
|
49
|
+
function withFamily(state, family, record) {
|
|
50
|
+
return {
|
|
51
|
+
families: { ...state?.families, [family]: record },
|
|
52
|
+
withheld: state?.withheld ?? false,
|
|
53
|
+
};
|
|
54
|
+
}
|
|
24
55
|
/**
|
|
25
56
|
* Canonicalize a parsed argument value into a stable string.
|
|
26
57
|
*
|
|
@@ -63,46 +94,91 @@ export function callKey(name, args) {
|
|
|
63
94
|
return `${name}(${canonicalize(args)})`;
|
|
64
95
|
}
|
|
65
96
|
/**
|
|
66
|
-
* Record one observed
|
|
97
|
+
* Record one observed failure in one family.
|
|
67
98
|
* @param state - the agent's current state, or undefined on first sight.
|
|
99
|
+
* @param family - the family the failure belongs to.
|
|
68
100
|
* @param failure - the recognized failure.
|
|
69
101
|
* @param key - the identity of the call that failed.
|
|
70
102
|
* @returns the updated state.
|
|
71
103
|
*/
|
|
72
|
-
export function observe(state, failure, key) {
|
|
73
|
-
const
|
|
104
|
+
export function observe(state, family, failure, key) {
|
|
105
|
+
const previous = familyOf(state, family);
|
|
106
|
+
const failingKeys = new Set(previous?.failingKeys ?? []);
|
|
74
107
|
failingKeys.add(key);
|
|
75
|
-
return {
|
|
76
|
-
observations: (
|
|
108
|
+
return withFamily(state, family, {
|
|
109
|
+
observations: (previous?.observations ?? 0) + 1,
|
|
77
110
|
last: failure,
|
|
78
111
|
failingKeys,
|
|
79
|
-
denials:
|
|
80
|
-
advised:
|
|
112
|
+
denials: previous?.denials ?? 0,
|
|
113
|
+
advised: previous?.advised ?? false,
|
|
114
|
+
});
|
|
115
|
+
}
|
|
116
|
+
/**
|
|
117
|
+
* Whether this family's durable advisory has already been delivered.
|
|
118
|
+
* @param state - the agent's current state, or undefined.
|
|
119
|
+
* @param family - the family in question.
|
|
120
|
+
* @returns true when the advice is already in the session.
|
|
121
|
+
*/
|
|
122
|
+
export function advisedOf(state, family) {
|
|
123
|
+
return familyOf(state, family)?.advised ?? false;
|
|
124
|
+
}
|
|
125
|
+
/**
|
|
126
|
+
* Mark one family's durable advisory as delivered.
|
|
127
|
+
* @param state - the agent's current state.
|
|
128
|
+
* @param family - the family that was advised.
|
|
129
|
+
* @returns the updated state.
|
|
130
|
+
*/
|
|
131
|
+
export function recordAdvice(state, family) {
|
|
132
|
+
const previous = familyOf(state, family);
|
|
133
|
+
if (previous === undefined)
|
|
134
|
+
return state;
|
|
135
|
+
return withFamily(state, family, { ...previous, advised: true });
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* Record that a recognized failure was withheld from the model.
|
|
139
|
+
* @param state - the agent's current state, or undefined on first sight.
|
|
140
|
+
* @returns the updated state.
|
|
141
|
+
*/
|
|
142
|
+
export function recordWithheld(state) {
|
|
143
|
+
return {
|
|
144
|
+
families: { ...state?.families },
|
|
145
|
+
withheld: true,
|
|
81
146
|
};
|
|
82
147
|
}
|
|
83
148
|
/**
|
|
84
149
|
* Record that a call carrying the same identity as a previously failing one
|
|
85
|
-
* succeeded
|
|
86
|
-
*
|
|
87
|
-
*
|
|
150
|
+
* succeeded — in **every** family that was watching that identity.
|
|
151
|
+
*
|
|
152
|
+
* The environment worked at least once for that call, so the entry stops
|
|
153
|
+
* justifying a denial — and is dropped rather than kept, so a later failure
|
|
154
|
+
* re-earns it. A success is evidence about the environment, not about one
|
|
155
|
+
* diagnosis: the same call cannot have started working for one family's reason
|
|
156
|
+
* and not the other's.
|
|
88
157
|
*
|
|
89
|
-
*
|
|
90
|
-
* consequence: without it the budget is per agent for the whole
|
|
91
|
-
* makes the entry above unobservable — past `maxDenials` this
|
|
92
|
-
* nothing ever again, so clearing the key would change no
|
|
93
|
-
* bound reads as "at most `maxDenials` refusals per
|
|
94
|
-
* environment that breaks, is repaired and breaks
|
|
95
|
-
* while a session can always make progress by
|
|
158
|
+
* Each affected family's denial budget is re-armed at the same moment, and only
|
|
159
|
+
* then. Measured consequence: without it the budget is per agent for the whole
|
|
160
|
+
* session, which makes the entry above unobservable — past `maxDenials` this
|
|
161
|
+
* plugin refuses nothing ever again, so clearing the key would change no
|
|
162
|
+
* decision. With it, the bound reads as "at most `maxDenials` refusals per
|
|
163
|
+
* episode of brokenness": an environment that breaks, is repaired and breaks
|
|
164
|
+
* again may be refused again, while a session can always make progress by
|
|
165
|
+
* spending the budget.
|
|
96
166
|
* @param state - the agent's current state.
|
|
97
167
|
* @param key - the identity of the call that just succeeded.
|
|
98
|
-
* @returns the updated state, unchanged when
|
|
168
|
+
* @returns the updated state, unchanged when no family was watching the key.
|
|
99
169
|
*/
|
|
100
170
|
export function observeSuccess(state, key) {
|
|
101
|
-
|
|
171
|
+
const watched = Object.entries(state.families)
|
|
172
|
+
.filter(([, record]) => record.failingKeys.has(key));
|
|
173
|
+
if (watched.length === 0)
|
|
102
174
|
return state;
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
175
|
+
let next = state;
|
|
176
|
+
for (const [family, record] of watched) {
|
|
177
|
+
const failingKeys = new Set(record.failingKeys);
|
|
178
|
+
failingKeys.delete(key);
|
|
179
|
+
next = withFamily(next, family, { ...record, failingKeys, denials: 0 });
|
|
180
|
+
}
|
|
181
|
+
return next;
|
|
106
182
|
}
|
|
107
183
|
/**
|
|
108
184
|
* Whether a call may be refused before dispatch.
|
|
@@ -112,34 +188,35 @@ export function observeSuccess(state, key) {
|
|
|
112
188
|
* fail. The second condition is what keeps the fail-fast half from blocking a
|
|
113
189
|
* workaround: a different command, or the same command under a different policy
|
|
114
190
|
* after the user changed configuration, has no key here.
|
|
191
|
+
*
|
|
192
|
+
* Only the `acl-provisioning` family ever reaches this question; see the
|
|
193
|
+
* `denialText` doc for why the blocking half does not extend to `pty-startup`.
|
|
115
194
|
* @param state - the agent's current state, or undefined.
|
|
195
|
+
* @param family - the family whose threshold is being asked about.
|
|
116
196
|
* @param key - the identity of the call about to dispatch.
|
|
117
197
|
* @param enforceAfter - the configured threshold; 0 disables the half entirely.
|
|
118
198
|
* @param maxDenials - the configured denial budget.
|
|
119
199
|
* @returns whether to deny.
|
|
120
200
|
*/
|
|
121
|
-
export function shouldDeny(state, key, enforceAfter, maxDenials) {
|
|
122
|
-
|
|
201
|
+
export function shouldDeny(state, family, key, enforceAfter, maxDenials) {
|
|
202
|
+
const record = familyOf(state, family);
|
|
203
|
+
if (enforceAfter === 0 || record === undefined)
|
|
123
204
|
return false;
|
|
124
|
-
if (
|
|
205
|
+
if (record.observations < enforceAfter)
|
|
125
206
|
return false;
|
|
126
|
-
if (
|
|
207
|
+
if (record.denials >= maxDenials)
|
|
127
208
|
return false;
|
|
128
|
-
return
|
|
209
|
+
return record.failingKeys.has(key);
|
|
129
210
|
}
|
|
130
211
|
/**
|
|
131
212
|
* Spend one denial.
|
|
132
213
|
* @param state - the agent's current state.
|
|
214
|
+
* @param family - the family being denied.
|
|
133
215
|
* @returns the updated state.
|
|
134
216
|
*/
|
|
135
|
-
export function recordDenial(state) {
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
* @param state - the agent's current state.
|
|
141
|
-
* @returns the updated state.
|
|
142
|
-
*/
|
|
143
|
-
export function recordAdvice(state) {
|
|
144
|
-
return { ...state, advised: true };
|
|
217
|
+
export function recordDenial(state, family) {
|
|
218
|
+
const previous = familyOf(state, family);
|
|
219
|
+
if (previous === undefined)
|
|
220
|
+
return state;
|
|
221
|
+
return withFamily(state, family, { ...previous, denials: previous.denials + 1 });
|
|
145
222
|
}
|
package/lib/types/advice.d.ts
CHANGED
|
@@ -1,38 +1,100 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* What the model — and through it the user — is told about a
|
|
3
|
-
* failure, and what is deliberately withheld.
|
|
2
|
+
* What the model — and through it the user — is told about a recognized
|
|
3
|
+
* environment failure, and what is deliberately withheld.
|
|
4
4
|
*
|
|
5
5
|
* The text is assembled here as pure functions so every sentence can be pinned
|
|
6
|
-
* by a test.
|
|
6
|
+
* by a test, one family at a time. The two families are shaped by the same two
|
|
7
|
+
* questions, and they answer them differently:
|
|
7
8
|
*
|
|
8
|
-
* - **
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
* the
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
9
|
+
* - **The ACL failure** (`acl-provisioning`) *is* fixable by the caller, so its
|
|
10
|
+
* advice names the right the caller is missing and gives the command.
|
|
11
|
+
* The reported failures are `ERROR_ACCESS_DENIED` from a *merged* DACL + SACL
|
|
12
|
+
* write; the missing right is `WRITE_OWNER` on the directory — an object right
|
|
13
|
+
* the caller can grant itself with `icacls`, unelevated. It is **not**
|
|
14
|
+
* `SeSecurityPrivilege`, the token privilege the reports naturally reach for;
|
|
15
|
+
* `whoami /priv` cannot show the difference, and elevation is the wrong lever.
|
|
16
|
+
* - **The persistent-shell failure** (`pty-startup`) is *not* fixable by the
|
|
17
|
+
* caller — least of all by the model, which has no shell to run anything in.
|
|
18
|
+
* So its advice says so and stops: the remedy is a user-side preset choice,
|
|
19
|
+
* and the model's instruction is to stop retrying and use its file tools.
|
|
20
|
+
* Handing the model a command here would be advice to run something that
|
|
21
|
+
* cannot run, and naming a one-shot shell tool would be advice to call a tool
|
|
22
|
+
* the failing composition does not mount.
|
|
23
|
+
*
|
|
24
|
+
* Both give a **discriminator, not just a remedy**: applying a fix without
|
|
25
|
+
* confirming the cause teaches nothing when the fix does not work. For the ACL
|
|
26
|
+
* family that is `icacls <dir>`, looking for an ACE that names the caller's own
|
|
27
|
+
* SID and grants `(F)` — which separates "Modify-only directory" from "the
|
|
28
|
+
* documented prerequisite is wrong", the open question upstream. For the PTY
|
|
29
|
+
* family it is the **effective sandbox mode**, which is why that advisory is
|
|
30
|
+
* only ever built with the mode the call actually ran under.
|
|
19
31
|
*
|
|
20
32
|
* @module
|
|
21
33
|
*/
|
|
22
|
-
import type { ProvisioningFailure } from './signature.js';
|
|
23
|
-
|
|
24
|
-
|
|
34
|
+
import type { ProvisioningFailure, RecognizedFailure } from './signature.js';
|
|
35
|
+
import type { SandboxModeName } from './mode.js';
|
|
36
|
+
/** The upstream threads the ACL advisory is a stopgap for. */
|
|
37
|
+
export declare const ACL_DISCUSSIONS = "#7538 / #7622 / #7646";
|
|
38
|
+
/** The upstream thread the persistent-shell advisory is a stopgap for. */
|
|
39
|
+
export declare const PTY_DISCUSSIONS = "#7638";
|
|
25
40
|
/** The documented prerequisite, quoted from the backend's README. */
|
|
26
41
|
export declare const PREREQUISITE = "granted directories must be caller-owned and grant `WRITE_OWNER`";
|
|
42
|
+
/**
|
|
43
|
+
* Where a user's own preset changes actually live.
|
|
44
|
+
*
|
|
45
|
+
* This is deliberately **not** the legacy `$DSH_HOME/.agent-presets/<id>/`
|
|
46
|
+
* directory: that shape predates declarative presets and **nothing reads it any
|
|
47
|
+
* more** (the registry "neither scans directories nor accepts preset paths").
|
|
48
|
+
* A preset is a `@deepseek-ai/dsh-agent-preset` row, and changing one means
|
|
49
|
+
* overriding or inserting that row in a patch layer, which is what this path
|
|
50
|
+
* names. Advising a folder the harness stopped reading would be the same defect
|
|
51
|
+
* this plugin exists to answer — a remedy that does not work, delivered
|
|
52
|
+
* confidently.
|
|
53
|
+
*/
|
|
54
|
+
export declare const PROFILE_PATCH = "$DSH_HOME/profiles/<profile>/cordis.patch.yml";
|
|
55
|
+
/** The machine-wide patch layer, for a change that should hold in every profile. */
|
|
56
|
+
export declare const GLOBAL_PATCH = "$DSH_HOME/cordis.patch.yml";
|
|
57
|
+
/** The row id the shipped `minimal` preset is declared under. */
|
|
58
|
+
export declare const MINIMAL_PRESET_ROW = "preset-minimal";
|
|
59
|
+
/** The one-shot shell tool the `standard` preset mounts on Windows. */
|
|
60
|
+
export declare const ONE_SHOT_SHELL = "@deepseek-ai/dsh-tool-pwsh";
|
|
61
|
+
/** What the caller knows about the failing call, beyond the failure text. */
|
|
62
|
+
export interface AdvisoryContext {
|
|
63
|
+
/** URL quoted in place of the discussions list; optional. */
|
|
64
|
+
readonly href?: string;
|
|
65
|
+
/** The tool whose call failed, quoted back so the advice is about that call. */
|
|
66
|
+
readonly tool?: string;
|
|
67
|
+
/**
|
|
68
|
+
* The sandbox mode the failing call ran under. Required by the
|
|
69
|
+
* `pty-startup` family — the whole diagnosis is the mode — and unused by the
|
|
70
|
+
* ACL family.
|
|
71
|
+
*/
|
|
72
|
+
readonly mode?: SandboxModeName;
|
|
73
|
+
}
|
|
27
74
|
/**
|
|
28
75
|
* Build the advisory attached to the failing tool result.
|
|
76
|
+
*
|
|
77
|
+
* The family decides everything: one function so a caller does not have to
|
|
78
|
+
* remember which family needs which fact, and so the mode requirement of the
|
|
79
|
+
* PTY family is enforced by construction rather than by convention.
|
|
29
80
|
* @param failure - the recognized failure.
|
|
30
|
-
* @param
|
|
31
|
-
* @returns the user-role notice text, with
|
|
81
|
+
* @param context - what the caller knows about the failing call.
|
|
82
|
+
* @returns the user-role notice text, with any remedy ready to paste.
|
|
83
|
+
* @throws when a PTY failure is advised without its resolved sandbox mode.
|
|
32
84
|
*/
|
|
33
|
-
export declare function advisoryText(failure:
|
|
85
|
+
export declare function advisoryText(failure: RecognizedFailure, context?: AdvisoryContext): string;
|
|
34
86
|
/**
|
|
35
87
|
* Build the pre-dispatch denial for the optional fail-fast half.
|
|
88
|
+
*
|
|
89
|
+
* The blocking half is deliberately **ACL-only**, and this function's parameter
|
|
90
|
+
* type is where that is enforced. The PTY family gets an advisory and nothing
|
|
91
|
+
* else, for a reason that is about the remedy rather than about the failure:
|
|
92
|
+
* the ACL remedy is a command the user can run *while the session continues*,
|
|
93
|
+
* so refusing further identical calls cannot make the session unfinishable —
|
|
94
|
+
* spending the budget always lets the call through, and a repaired environment
|
|
95
|
+
* is discovered by exactly that. The PTY remedy is a preset swap, which happens
|
|
96
|
+
* between sessions; refusing calls could only pad a session that is already
|
|
97
|
+
* unable to do the thing being refused.
|
|
36
98
|
* @param failure - the recognized failure.
|
|
37
99
|
* @param observed - how many provisioning failures this agent has produced.
|
|
38
100
|
* @param denial - this denial's 1-based ordinal.
|
package/lib/types/index.d.ts
CHANGED
|
@@ -1,23 +1,38 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `sandbox-grant-advisor`: turn
|
|
3
|
-
*
|
|
4
|
-
* transcript — can act on.
|
|
2
|
+
* `sandbox-grant-advisor`: turn an environment failure that has no path forward
|
|
3
|
+
* into a diagnosis the model — and the user reading the transcript — can act on.
|
|
5
4
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
5
|
+
* ## The two failures it recognizes
|
|
6
|
+
*
|
|
7
|
+
* **Workspace provisioning (Windows ACL).** Three reports of one signature
|
|
8
|
+
* (`#7538`, `#7622`, `#7646`) describe the same shape: the host-side write grant
|
|
9
|
+
* for a sandboxed workspace cannot be applied, every sandboxed command then
|
|
10
|
+
* fails identically **before it runs**, and the error text is a bare Win32 line:
|
|
10
11
|
*
|
|
11
12
|
* SetNamedSecurityInfoW failed (Win32 5): grantWrite(D:\ws)
|
|
12
13
|
*
|
|
13
14
|
* The grant is materialized lazily on the first confined call and nothing is
|
|
14
15
|
* cached when it throws, so the failure repeats per command rather than once
|
|
15
16
|
* (850 calls / 39 sessions in `#7622`; 52,588 output tokens with no output in
|
|
16
|
-
* `#7538`). The
|
|
17
|
-
*
|
|
18
|
-
* the caller `WRITE_OWNER` — never reaches the user, so sessions escape into
|
|
17
|
+
* `#7538`). The remedy the backend documents — the directory must grant the
|
|
18
|
+
* caller `WRITE_OWNER` — never reaches the user, so sessions escape into
|
|
19
19
|
* `danger-full-access` or die on the model's output cap.
|
|
20
20
|
*
|
|
21
|
+
* **Persistent shell startup (#7638).** With the `minimal` preset on Windows the
|
|
22
|
+
* only shell tool is a persistent PTY (`dsh-terminal-bash` +
|
|
23
|
+
* `dsh-tool-pwsh-persistent`), and under a *confining* sandbox mode every call
|
|
24
|
+
* fails instantly with
|
|
25
|
+
*
|
|
26
|
+
* PTY shell exited during startup
|
|
27
|
+
*
|
|
28
|
+
* — the backend cannot create the pseudo-console inside the sandbox, so the
|
|
29
|
+
* child exits before its first prompt. Retrying never helps, the message points
|
|
30
|
+
* at no cause, and because `minimal` mounts no fallback shell tool the session
|
|
31
|
+
* has no command execution left at all. The reporter's own three-arm control
|
|
32
|
+
* makes the sandbox mode the discriminator: minimal × confining fails, minimal ×
|
|
33
|
+
* `danger-full-access` succeeds, `standard` (one-shot shell) × confining
|
|
34
|
+
* succeeds.
|
|
35
|
+
*
|
|
21
36
|
* ## Where it acts, and why there
|
|
22
37
|
*
|
|
23
38
|
* One listener on the public `tools/post-execute` waterfall
|
|
@@ -28,44 +43,65 @@
|
|
|
28
43
|
* to the model in the same step (`PostToolDecision`'s `additionalContexts`,
|
|
29
44
|
* a durable user-role message).
|
|
30
45
|
*
|
|
31
|
-
* `ctx.sandbox.confine(argv, policy, signal)` sees the failure too,
|
|
32
|
-
* do this: its signature carries no agent, so a wrapper could detect
|
|
33
|
-
* condition and never deliver a word about it to the session that is stuck.
|
|
46
|
+
* `ctx.sandbox.confine(argv, policy, signal)` sees the confinement failure too,
|
|
47
|
+
* and cannot do this: its signature carries no agent, so a wrapper could detect
|
|
48
|
+
* the condition and never deliver a word about it to the session that is stuck.
|
|
49
|
+
*
|
|
50
|
+
* The PTY family needs one fact the failure text does not carry — the effective
|
|
51
|
+
* sandbox mode — and takes it from `ctx.sandboxPolicy.resolve({ session })`:
|
|
52
|
+
* the same resolver the terminal layer calls before spawning, with the same
|
|
53
|
+
* session. See `src/mode.ts` for why that lookup is guarded rather than
|
|
54
|
+
* imported, and what happens when it cannot answer.
|
|
34
55
|
*
|
|
35
56
|
* ## What it does
|
|
36
57
|
*
|
|
37
|
-
* 1. **One durable advisory per agent.** On the first recognized
|
|
38
|
-
* failure, the failing tool result is enriched with a user-role
|
|
39
|
-
* names the missing right (`WRITE_OWNER` on the
|
|
40
|
-
* `SeSecurityPrivilege`), gives the unelevated one-line
|
|
41
|
-
* gives the discriminator that separates a Modify-only
|
|
42
|
-
* wrong prerequisite.
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
* the
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
*
|
|
58
|
+
* 1. **One durable advisory per agent, per family.** On the first recognized
|
|
59
|
+
* failure of a family, the failing tool result is enriched with a user-role
|
|
60
|
+
* notice. For the ACL family it names the missing right (`WRITE_OWNER` on the
|
|
61
|
+
* directory, not `SeSecurityPrivilege`), gives the unelevated one-line
|
|
62
|
+
* `icacls` remedy, and gives the discriminator that separates a Modify-only
|
|
63
|
+
* directory from a wrong prerequisite. For the PTY family it names the
|
|
64
|
+
* combination that fails (persistent PTY × a confining mode), states the
|
|
65
|
+
* resolved mode, says plainly that no command can fix it, and hands the
|
|
66
|
+
* user-side preset choice over. Both ride `additionalContexts`, so the model
|
|
67
|
+
* sees the diagnosis beside the failure rather than only in a log it never
|
|
68
|
+
* reads.
|
|
69
|
+
* 2. **A bounded fail-fast, ACL family only.** With `enforceAfter` set, a call
|
|
70
|
+
* this plugin has *watched fail* this way is refused at `tools/pre-execute`
|
|
71
|
+
* once the environment has failed at least that many times. It is off by
|
|
72
|
+
* default: the useful signal here is the diagnosis, and a plugin that blocks
|
|
73
|
+
* command execution for a reason it merely recognizes is a risk, not a
|
|
74
|
+
* feature. See the README for why the blocking half is deliberately narrow
|
|
75
|
+
* and why it does not cover the PTY family.
|
|
76
|
+
* 3. **A disclosure when it withholds.** The PTY advisory is only sent when the
|
|
77
|
+
* resolved mode actually confines; if the mode is not confining, or cannot be
|
|
78
|
+
* resolved at all, the failure is left exactly as it was **and the host log
|
|
79
|
+
* says so once**. Silence alone would make "the sandbox is not the cause" and
|
|
80
|
+
* "this plugin could not tell" indistinguishable from the outside.
|
|
50
81
|
*
|
|
51
82
|
* ## Honest boundaries
|
|
52
83
|
*
|
|
53
84
|
* - **The Windows path cannot be witnessed on macOS**, where this plugin was
|
|
54
85
|
* built and tested. What is tested is the decision layer: classification,
|
|
55
|
-
* once-per-agent delivery, the
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
86
|
+
* once-per-agent-per-family delivery, the sandbox-mode gate and its
|
|
87
|
+
* fail-closed behaviour, the fail-fast budget, and the wiring to the real
|
|
88
|
+
* `ToolRuntime` — against synthetic results carrying the producers' exact
|
|
89
|
+
* error shapes, with the formats taken from
|
|
90
|
+
* `packages/subprocess/win32-process/src/errors.ts` and
|
|
91
|
+
* `packages/terminal/terminal-bash/src/{index,session}.ts`.
|
|
59
92
|
* - **It does not repair anything.** No ACL is written, no privilege is
|
|
60
|
-
* requested, nothing is elevated
|
|
93
|
+
* requested, nothing is elevated, no preset is installed and no mode is
|
|
94
|
+
* changed: both remedies are the user's to apply.
|
|
61
95
|
* - **It complements, rather than replaces, `repeat-guard-escalation`.** That
|
|
62
96
|
* guard keys on *call identity* (identical arguments retried); this one keys
|
|
63
97
|
* on the *environment signature*, which is how several different commands can
|
|
64
98
|
* share one cause. They can be mounted together.
|
|
65
|
-
* - **The real fix is upstream
|
|
66
|
-
* condition at the site that knows it (`grantWrite` computes
|
|
99
|
+
* - **The real fix is upstream**, in both families: the ACL failure should name
|
|
100
|
+
* the outstanding condition at the site that knows it (`grantWrite` computes
|
|
67
101
|
* `hasExactGrant`/`hasExactDeny`/`hasExactLabel` and discards which was
|
|
68
|
-
* false)
|
|
102
|
+
* false), and the PTY startup path should either report "this sandbox mode is
|
|
103
|
+
* incompatible with the PTY backend" or fall back to a one-shot shell. This
|
|
104
|
+
* plugin is the stopgap.
|
|
69
105
|
*
|
|
70
106
|
* @module @argszero/cordis-plugin-sandbox-grant-advisor
|
|
71
107
|
*/
|
|
@@ -87,12 +123,21 @@ export declare const SOURCE_KIND = "sandbox-grant-advisor";
|
|
|
87
123
|
export declare const DEFAULT_ENFORCE_AFTER = 0;
|
|
88
124
|
/** Default denial budget once the blocking half is enabled. */
|
|
89
125
|
export declare const DEFAULT_MAX_DENIALS = 2;
|
|
126
|
+
/**
|
|
127
|
+
* The family the optional blocking half applies to.
|
|
128
|
+
*
|
|
129
|
+
* The ACL remedy is a command the user can run while the session continues; the
|
|
130
|
+
* PTY remedy is a preset swap between sessions. Refusing calls is only useful
|
|
131
|
+
* in the first case — see `denialText` in `src/advice.ts`.
|
|
132
|
+
*/
|
|
133
|
+
export declare const ENFORCED_FAMILY = "acl-provisioning";
|
|
90
134
|
/** Configures what is watched and whether the blocking half runs. */
|
|
91
135
|
export interface Config {
|
|
92
136
|
/**
|
|
93
|
-
*
|
|
137
|
+
* ACL provisioning failures after which an identical, already-failing call is
|
|
94
138
|
* denied before dispatch. `0` (the default) disables the half entirely; the
|
|
95
|
-
* advisory half is unaffected and always on.
|
|
139
|
+
* advisory half is unaffected and always on. The blocking half does not apply
|
|
140
|
+
* to the persistent-shell family.
|
|
96
141
|
*/
|
|
97
142
|
enforceAfter?: number;
|
|
98
143
|
/**
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Which sandbox mode a call ran under, and whether that mode confines.
|
|
3
|
+
*
|
|
4
|
+
* The persistent-shell diagnosis is only true under a **confining** mode. The
|
|
5
|
+
* terminal backend hands the shell argv straight through when the resolved mode
|
|
6
|
+
* is `danger-full-access` and confines it otherwise
|
|
7
|
+
* (`packages/terminal/terminal-bash/src/index.ts`:
|
|
8
|
+
* `if (policy.mode === 'danger-full-access') return argv`), so the same
|
|
9
|
+
* `PTY shell exited during startup` under `danger-full-access` is a different
|
|
10
|
+
* story — a broken or missing shell — and this plugin must stay silent about it
|
|
11
|
+
* rather than assert a sandbox cause it cannot support. The report that frames
|
|
12
|
+
* this family (#7638) says the same thing from the other side: its author's
|
|
13
|
+
* three-arm control shows the mode is the discriminator (minimal × confining
|
|
14
|
+
* fails, minimal × `danger-full-access` succeeds, standard × confining
|
|
15
|
+
* succeeds).
|
|
16
|
+
*
|
|
17
|
+
* The answer is taken from `ctx.sandboxPolicy.resolve({ session })` — the same
|
|
18
|
+
* resolver the terminal layer itself calls before spawning, with the same
|
|
19
|
+
* session — so what is quoted in the advisory is the policy that actually
|
|
20
|
+
* governed the failing call, not a guess reconstructed from configuration.
|
|
21
|
+
*
|
|
22
|
+
* ## Why this is a guarded lookup instead of an import
|
|
23
|
+
*
|
|
24
|
+
* `@deepseek-ai/dsh-sandbox-policy` is **optional** in this plugin's world: a
|
|
25
|
+
* composition may simply not mount the service, and this plugin must degrade to
|
|
26
|
+
* silence rather than fail to load. Two consequences shape this module:
|
|
27
|
+
*
|
|
28
|
+
* - A declared peer dependency is a claim about versions, and this package's
|
|
29
|
+
* packaging guard refuses both an import that is not declared and a
|
|
30
|
+
* declaration that is not imported. A *type-only* import would therefore turn
|
|
31
|
+
* an optional integration into a mandatory claim on every line the peer range
|
|
32
|
+
* admits — and a range that admits a line nobody ran is exactly the defect
|
|
33
|
+
* that guard exists to prevent.
|
|
34
|
+
* - What is left is the consumer-side capability guard: look the service up,
|
|
35
|
+
* check the shape of the answer instead of trusting it, and **fail closed**
|
|
36
|
+
* (`{ ok: false }`): an unresolvable mode withholds the advisory, and the
|
|
37
|
+
* caller discloses that withholding on the host side. An unresolvable mode is
|
|
38
|
+
* emphatically *not* an invitation to fall back to the deployment default —
|
|
39
|
+
* a session that overrode its mode to `danger-full-access` would then be
|
|
40
|
+
* diagnosed as if it were confined.
|
|
41
|
+
*
|
|
42
|
+
* The call site this module depends on is stable across every line the peer
|
|
43
|
+
* range claims: `resolve(request?: SandboxPolicyRequest): SandboxExecutionPolicy`
|
|
44
|
+
* is declared at the same position of `lib/types/index.d.ts` in every published
|
|
45
|
+
* build from `0.1.2-rc.1` to `0.1.7-rc.1`, and `Agent.session` is present on
|
|
46
|
+
* the same span of `@deepseek-ai/dsh-agent`.
|
|
47
|
+
*
|
|
48
|
+
* @module
|
|
49
|
+
*/
|
|
50
|
+
import type { Context } from '@deepseek-ai/cordis';
|
|
51
|
+
import type { Agent } from '@deepseek-ai/dsh-agent';
|
|
52
|
+
/** The three modes the harness resolves. */
|
|
53
|
+
export type SandboxModeName = 'read-only' | 'workspace-write' | 'danger-full-access';
|
|
54
|
+
/**
|
|
55
|
+
* Whether a mode confines the process it is asked to spawn.
|
|
56
|
+
* @param mode - the resolved mode.
|
|
57
|
+
* @returns true for every mode except `danger-full-access`.
|
|
58
|
+
*/
|
|
59
|
+
export declare function confines(mode: SandboxModeName): boolean;
|
|
60
|
+
/** The outcome of resolving one agent's effective sandbox mode. */
|
|
61
|
+
export type ModeResolution =
|
|
62
|
+
/**
|
|
63
|
+
* The mode the failing call ran under. `danger-full-access` is reported too:
|
|
64
|
+
* it is a real answer, and the caller's job (not this module's) is to decide
|
|
65
|
+
* that a non-confining mode is not this plugin's story.
|
|
66
|
+
*/
|
|
67
|
+
{
|
|
68
|
+
readonly ok: true;
|
|
69
|
+
readonly mode: SandboxModeName;
|
|
70
|
+
}
|
|
71
|
+
/** No answer was available, and why — the host side says so out loud. */
|
|
72
|
+
| {
|
|
73
|
+
readonly ok: false;
|
|
74
|
+
readonly withheld: string;
|
|
75
|
+
};
|
|
76
|
+
/**
|
|
77
|
+
* Resolve the effective sandbox mode for one agent's call.
|
|
78
|
+
*
|
|
79
|
+
* The service is looked up through `ctx.get` — the documented optional lookup —
|
|
80
|
+
* and the resolver is invoked with the agent's own session, so a session that
|
|
81
|
+
* logged a `sandbox/mode` override is answered with that override rather than
|
|
82
|
+
* with the deployment default.
|
|
83
|
+
* @param ctx - the plugin's context.
|
|
84
|
+
* @param agent - the agent whose call failed.
|
|
85
|
+
* @returns the mode, or the reason it could not be resolved.
|
|
86
|
+
*/
|
|
87
|
+
export declare function resolveSandboxMode(ctx: Context, agent: Agent): ModeResolution;
|