@nexrall/code-core 1.4.65 → 1.4.66
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/agent/agentTypes.d.ts +2 -2
- package/dist/agent/agentTypes.d.ts.map +1 -1
- package/dist/agent/agentTypes.js +7 -143
- package/dist/agent/loop.d.ts +1 -1
- package/dist/agent/loop.js +6 -6
- package/dist/agent/memory.d.ts +11 -0
- package/dist/agent/memory.d.ts.map +1 -1
- package/dist/agent/memory.js +23 -3
- package/dist/agent/planMode.d.ts +14 -0
- package/dist/agent/planMode.d.ts.map +1 -1
- package/dist/agent/planMode.js +145 -17
- package/dist/agent/securityLint.js +2 -2
- package/dist/api/client.d.ts +1 -1
- package/dist/permissions/destructive.d.ts +9 -7
- package/dist/permissions/destructive.d.ts.map +1 -1
- package/dist/permissions/destructive.js +55 -8
- package/dist/permissions/destructiveTokens.d.ts +29 -0
- package/dist/permissions/destructiveTokens.d.ts.map +1 -0
- package/dist/permissions/destructiveTokens.js +469 -0
- package/dist/permissions/modePolicy.d.ts +12 -8
- package/dist/permissions/modePolicy.d.ts.map +1 -1
- package/dist/permissions/modePolicy.js +14 -10
- package/dist/types.d.ts +2 -2
- package/package.json +8 -7
package/README.md
CHANGED
|
@@ -122,9 +122,9 @@ const cmds = loadSlashCommands(process.cwd());
|
|
|
122
122
|
const review = findSlashCommand(cmds, 'review');
|
|
123
123
|
const prompt = expandCommand(review, 'main..feature-branch', process.cwd());
|
|
124
124
|
|
|
125
|
-
// Built-in '
|
|
125
|
+
// Built-in 'explorer' agent (read-only, usable as subagent_type: 'explorer')
|
|
126
126
|
const agents = loadAgentTypes(process.cwd());
|
|
127
|
-
const
|
|
127
|
+
const explorer = findAgentType(agents, 'explorer');
|
|
128
128
|
```
|
|
129
129
|
|
|
130
130
|
## Checkpoint / rewind
|
|
@@ -19,7 +19,7 @@ export interface AgentType {
|
|
|
19
19
|
* When true, this agent's write tools may only target TEST files.
|
|
20
20
|
*
|
|
21
21
|
* Needed because a tool allowlist is all-or-nothing per tool: granting
|
|
22
|
-
* `edit_file` grants it for every path.
|
|
22
|
+
* `edit_file` grants it for every path. A user-defined test-writer agent must be able to
|
|
23
23
|
* write tests while being unable to "fix" production source to make a test
|
|
24
24
|
* pass — the single most common way a test-writing agent destroys signal — and
|
|
25
25
|
* a prompt instruction alone cannot guarantee that. Enforced in loop.ts's
|
|
@@ -53,7 +53,7 @@ export interface AgentType {
|
|
|
53
53
|
* delegation prompt instead.
|
|
54
54
|
*
|
|
55
55
|
* Only for agents that neither write files nor make judgement calls about conventions
|
|
56
|
-
* — a reviewer or test-writer needs the project's rules, an explorer does not.
|
|
56
|
+
* — a reviewer or test-writer needs the project's rules, an explorer or planner does not.
|
|
57
57
|
*/
|
|
58
58
|
lightPrompt?: boolean;
|
|
59
59
|
/** Tools this agent may NOT use, on top of (or instead of) an allowlist. Claude Code's `disallowedTools`. */
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"agentTypes.d.ts","sourceRoot":"","sources":["../../src/agent/agentTypes.ts"],"names":[],"mappings":"AAoCA,MAAM,WAAW,SAAS;IACxB,IAAI,EAAE,MAAM,CAAC;IACb,WAAW,EAAE,MAAM,CAAC;IACpB,6EAA6E;IAC7E,KAAK,CAAC,EAAE,MAAM,EAAE,CAAC;IACjB,iDAAiD;IACjD,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,qEAAqE;IACrE,MAAM,EAAE,MAAM,CAAC;IACf;;;;;;OAMG;IACH,MAAM,EAAE,SAAS,GAAG,QAAQ,GAAG,SAAS,GAAG,QAAQ,GAAG,cAAc,CAAC;IACrE;;;;;;;;;OASG;IACH,aAAa,CAAC,EAAE,OAAO,CAAC;IACxB;;;;;;;;;;;;;;OAcG;IACH,MAAM,CAAC,EAAE,gBAAgB,CAAC;IAC1B;;;;;;;;;;;;OAYG;IACH,WAAW,CAAC,EAAE,OAAO,CAAC;IACtB,6GAA6G;IAC7G,eAAe,CAAC,EAAE,MAAM,EAAE,CAAC;IAC3B,0FAA0F;IAC1F,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,8GAA8G;IAC9G,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,qFAAqF;IACrF,UAAU,CAAC,EAAE,OAAO,CAAC;IACrB,+FAA+F;IAC/F,SAAS,CAAC,EAAE,UAAU,CAAC;IACvB,gHAAgH;IAChH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,mCAAmC;IACnC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,iGAAiG;IACjG,MAAM,CAAC,EAAE,MAAM,EAAE,CAAC;IAClB,+GAA+G;IAC/G,UAAU,CAAC,EAAE,MAAM,EAAE,CAAC;IACtB;;;;OAIG;IACH,KAAK,CAAC,EAAE,UAAU,CAAC;CACpB;AAED,MAAM,WAAW,YAAY;IAAG,IAAI,EAAE,SAAS,CAAC;IAAC,OAAO,EAAE,MAAM,CAAC;IAAC,UAAU,CAAC,EAAE,MAAM,CAAA;CAAE;AACvF,MAAM,WAAW,cAAc;IAAG,OAAO,CAAC,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,YAAY,EAAE,CAAA;CAAE;AAC3E,MAAM,WAAW,UAAU;IACzB,UAAU,CAAC,EAAE,cAAc,EAAE,CAAC;IAC9B,WAAW,CAAC,EAAE,cAAc,EAAE,CAAC;IAC/B,uFAAuF;IACvF,IAAI,CAAC,EAAE,YAAY,EAAE,CAAC;CACvB;AAED,qDAAqD;AACrD,MAAM,MAAM,gBAAgB,GAAG,SAAS,GAAG,MAAM,GAAG,OAAO,CAAC;
|
|
1
|
+
{"version":3,"file":"agentTypes.d.ts","sourceRoot":"","sources":["../../src/agent/agentTypes.ts"],"names":[],"mappings":"AAoCA,MAAM,WAAW,SAAS;IACxB,IAAI,EAAE,MAAM,CAAC;IACb,WAAW,EAAE,MAAM,CAAC;IACpB,6EAA6E;IAC7E,KAAK,CAAC,EAAE,MAAM,EAAE,CAAC;IACjB,iDAAiD;IACjD,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,qEAAqE;IACrE,MAAM,EAAE,MAAM,CAAC;IACf;;;;;;OAMG;IACH,MAAM,EAAE,SAAS,GAAG,QAAQ,GAAG,SAAS,GAAG,QAAQ,GAAG,cAAc,CAAC;IACrE;;;;;;;;;OASG;IACH,aAAa,CAAC,EAAE,OAAO,CAAC;IACxB;;;;;;;;;;;;;;OAcG;IACH,MAAM,CAAC,EAAE,gBAAgB,CAAC;IAC1B;;;;;;;;;;;;OAYG;IACH,WAAW,CAAC,EAAE,OAAO,CAAC;IACtB,6GAA6G;IAC7G,eAAe,CAAC,EAAE,MAAM,EAAE,CAAC;IAC3B,0FAA0F;IAC1F,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,8GAA8G;IAC9G,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,qFAAqF;IACrF,UAAU,CAAC,EAAE,OAAO,CAAC;IACrB,+FAA+F;IAC/F,SAAS,CAAC,EAAE,UAAU,CAAC;IACvB,gHAAgH;IAChH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,mCAAmC;IACnC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,iGAAiG;IACjG,MAAM,CAAC,EAAE,MAAM,EAAE,CAAC;IAClB,+GAA+G;IAC/G,UAAU,CAAC,EAAE,MAAM,EAAE,CAAC;IACtB;;;;OAIG;IACH,KAAK,CAAC,EAAE,UAAU,CAAC;CACpB;AAED,MAAM,WAAW,YAAY;IAAG,IAAI,EAAE,SAAS,CAAC;IAAC,OAAO,EAAE,MAAM,CAAC;IAAC,UAAU,CAAC,EAAE,MAAM,CAAA;CAAE;AACvF,MAAM,WAAW,cAAc;IAAG,OAAO,CAAC,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,YAAY,EAAE,CAAA;CAAE;AAC3E,MAAM,WAAW,UAAU;IACzB,UAAU,CAAC,EAAE,cAAc,EAAE,CAAC;IAC9B,WAAW,CAAC,EAAE,cAAc,EAAE,CAAC;IAC/B,uFAAuF;IACvF,IAAI,CAAC,EAAE,YAAY,EAAE,CAAC;CACvB;AAED,qDAAqD;AACrD,MAAM,MAAM,gBAAgB,GAAG,SAAS,GAAG,MAAM,GAAG,OAAO,CAAC;AAoP5D,wBAAgB,iBAAiB,CAAC,CAAC,EAAE,MAAM,GAAG,MAAM,CAOnD;AAgBD,eAAO,MAAM,YAAY,8IAKf,CAAC;AAEX,2MAA2M;AAC3M,eAAO,MAAM,kBAAkB,4CAA6C,CAAC;AAC7E,MAAM,MAAM,SAAS,GAAG,CAAC,OAAO,kBAAkB,CAAC,CAAC,MAAM,CAAC,CAAC;AAE5D;;;;;;GAMG;AACH,MAAM,WAAW,YAAY;IAC3B,qEAAqE;IACrE,IAAI,EAAE,MAAM,CAAC;IACb,gDAAgD;IAChD,KAAK,EAAE,MAAM,CAAC;IACd,OAAO,EAAE,MAAM,CAAC;CACjB;AAOD;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,UAAU,CAAC,CAAC,EAAE,MAAM,GAAG,SAAS,GAAG,MAAM,GAAG,SAAS,CASpE;AAyOD;;;;;GAKG;AACH,wBAAgB,eAAe,CAAC,GAAG,EAAE,MAAM,GAAG;IAAE,KAAK,CAAC,EAAE,UAAU,CAAC;IAAC,KAAK,CAAC,EAAE,MAAM,CAAA;CAAE,CAuCnF;AAqBD;;;;;;;;GAQG;AACH,wBAAgB,cAAc,CAAC,OAAO,EAAE,MAAM,GAAG,SAAS,EAAE,CAE3D;AAED;;;;;;;;;;;;;;;;;GAiBG;AACH,wBAAgB,0BAA0B,CACxC,OAAO,EAAE,MAAM,EACf,KAAK,GAAE,SAAS,EAAO,GACtB;IAAE,KAAK,EAAE,SAAS,EAAE,CAAC;IAAC,QAAQ,EAAE,YAAY,EAAE,CAAA;CAAE,CAuBlD;AAED,yFAAyF;AACzF,wBAAgB,cAAc,IAAI,MAAM,EAAE,CAEzC;AAED,2FAA2F;AAC3F,wBAAgB,aAAa,IAAI,SAAS,EAAE,CAE3C;AACD;;;;;;;;;;;;;;GAcG;AACH,wBAAgB,eAAe,CAAC,KAAK,EAAE,SAAS,EAAE,GAAG,MAAM,CAY1D;AASD,wBAAgB,aAAa,CAAC,KAAK,EAAE,SAAS,EAAE,EAAE,IAAI,EAAE,MAAM,GAAG,SAAS,GAAG,SAAS,GAAG,SAAS,CAIjG"}
|
package/dist/agent/agentTypes.js
CHANGED
|
@@ -65,7 +65,7 @@ function parseMemoryScope(v) {
|
|
|
65
65
|
// capability on whichever client it runs.
|
|
66
66
|
//
|
|
67
67
|
// This exists because it was previously inlined per agent, and the one agent that
|
|
68
|
-
// had it
|
|
68
|
+
// had it listed five VS Code-only tools, leaving that agent with just five
|
|
69
69
|
// usable tools when run from the CLI — a silent capability gap that is very easy
|
|
70
70
|
// to reintroduce by copy-pasting a list.
|
|
71
71
|
const READ_ONLY_TOOLS = [
|
|
@@ -111,20 +111,8 @@ const READ_ONLY_TOOLS = [
|
|
|
111
111
|
* only way to stop it being "helpfully" re-added.
|
|
112
112
|
*/
|
|
113
113
|
// web_search is a client-side tool now (executor.ts webSearch), so it must be granted
|
|
114
|
-
// explicitly — planner
|
|
114
|
+
// explicitly — planner was told to use it and then refused.
|
|
115
115
|
const RESEARCH_TOOLS = [...READ_ONLY_TOOLS, 'fetch_url', 'web_search'];
|
|
116
|
-
/** Read-only + the write tools, for agents that produce code. */
|
|
117
|
-
const WRITE_TOOLS = [
|
|
118
|
-
...READ_ONLY_TOOLS,
|
|
119
|
-
'write_file', 'edit_file', 'multi_edit', 'create_directory', 'move_file', 'copy_file',
|
|
120
|
-
// notebook_edit is a write tool like any other, and allowsTestOnlyWrite already
|
|
121
|
-
// knows its shape (`source` is cell CONTENT, not a path). Omitting it just meant
|
|
122
|
-
// test-writer silently could not touch notebooks.
|
|
123
|
-
'notebook_edit',
|
|
124
|
-
// Office document generation — same write-access tier as write_file, just a
|
|
125
|
-
// structured content shape instead of raw text (see tools/executor.ts).
|
|
126
|
-
'write_docx', 'write_xlsx', 'write_pptx',
|
|
127
|
-
];
|
|
128
116
|
const BUILTIN_AGENTS = [
|
|
129
117
|
{
|
|
130
118
|
// The catch-all, matching Claude Code's `general-purpose`.
|
|
@@ -165,66 +153,6 @@ const BUILTIN_AGENTS = [
|
|
|
165
153
|
'(with file paths), what you ran and its outcome, and anything you deliberately left undone.',
|
|
166
154
|
source: 'builtin',
|
|
167
155
|
},
|
|
168
|
-
{
|
|
169
|
-
name: 'reviewer',
|
|
170
|
-
description: 'Read-only code reviewer — finds correctness bugs, edge cases, and security issues in a diff or file set. Cannot modify files.',
|
|
171
|
-
tools: READ_ONLY_TOOLS,
|
|
172
|
-
source: 'builtin',
|
|
173
|
-
prompt: [
|
|
174
|
-
'You are a meticulous senior code reviewer. You NEVER modify files — you only read, search, and report.',
|
|
175
|
-
'',
|
|
176
|
-
'Method:',
|
|
177
|
-
'1. Read the full context around every change you are asked to review; never judge a hunk in isolation.',
|
|
178
|
-
'2. Hunt specifically for: correctness bugs, unhandled edge cases (empty/null/unicode/concurrency/timezone),',
|
|
179
|
-
' security issues (injection, path traversal, secrets in code, unsafe deserialization), breaking API',
|
|
180
|
-
' changes (search for callers first), silent behaviour changes, and swallowed errors.',
|
|
181
|
-
'3. Verify test coverage: are the changed paths tested? Were assertions weakened or tests deleted?',
|
|
182
|
-
'4. Only use bash for read-only commands (git diff/log/show, grep, test runs). Never run mutating commands.',
|
|
183
|
-
'',
|
|
184
|
-
'Report format: 🔴 Critical / 🟡 Warning / 🟢 Suggestion, each with file:line and a concrete fix,',
|
|
185
|
-
'then a final verdict (APPROVE or REQUEST CHANGES) with a one-paragraph rationale.',
|
|
186
|
-
].join('\n'),
|
|
187
|
-
},
|
|
188
|
-
// Promoted from the security-audit plugin to a builtin.
|
|
189
|
-
//
|
|
190
|
-
// Leaving it plugin-only was indefensible next to `reviewer` being builtin:
|
|
191
|
-
// reviewer's own prompt already tells it to look for security issues, so
|
|
192
|
-
// security IS treated as default work — yet the specialist agent for it was
|
|
193
|
-
// invisible unless the user happened to know the plugin existed. For an agent
|
|
194
|
-
// that WRITES code, "you only get a security review if you knew to install
|
|
195
|
-
// something" is the wrong default.
|
|
196
|
-
{
|
|
197
|
-
name: 'security-auditor',
|
|
198
|
-
description: 'Read-only security auditor — hunts injection, authz, secrets, and validation flaws in a path or diff. Cannot modify files.',
|
|
199
|
-
tools: READ_ONLY_TOOLS,
|
|
200
|
-
// No `model` override — inherits the session's, same as general-purpose.
|
|
201
|
-
// A previous version forced 'pro' (Claude Opus 5) here, which silently
|
|
202
|
-
// billed Anthropic even for a session running entirely on OpenAI/DeepSeek/
|
|
203
|
-
// Qwen. Consistency with the main conversation's provider matters more
|
|
204
|
-
// than defaulting every specialist to a fixed tier.
|
|
205
|
-
source: 'builtin',
|
|
206
|
-
prompt: [
|
|
207
|
-
'You are a security auditor. You find real, exploitable flaws — not style issues.',
|
|
208
|
-
'',
|
|
209
|
-
'Method:',
|
|
210
|
-
'1. Map the attack surface FIRST: entry points (HTTP routes, message handlers, CLI args, file/network',
|
|
211
|
-
' input, deserialization), then trace user-controlled data inward to where it is used.',
|
|
212
|
-
'2. For each finding: file:line, the flaw class, a one-line exploit scenario, and the concrete fix.',
|
|
213
|
-
'3. Grade severity honestly: Critical = remote compromise or data breach; High = auth bypass/IDOR;',
|
|
214
|
-
' Medium = needs unusual preconditions; Low = hardening.',
|
|
215
|
-
'',
|
|
216
|
-
'Classes worth the most attention, in order: injection (SQL/command/template/prototype), broken',
|
|
217
|
-
'authz (missing ownership checks, IDOR, trusting client-supplied ids), secrets committed to source,',
|
|
218
|
-
'path traversal, SSRF, unsafe deserialization, missing rate limits on expensive or auth endpoints,',
|
|
219
|
-
'and crypto misuse (hand-rolled comparison, predictable randomness).',
|
|
220
|
-
'',
|
|
221
|
-
'Hard rules:',
|
|
222
|
-
'- READ-ONLY: never modify, create or delete files. bash only for read-only inspection.',
|
|
223
|
-
'- NEVER print a discovered secret\'s value. Report its location and advise rotation.',
|
|
224
|
-
'- Distinguish EXPLOITABLE from theoretical, and say which one each finding is.',
|
|
225
|
-
'- "No issues found in scope X" is a valid, useful result. Do not pad the report to look thorough.',
|
|
226
|
-
].join('\n'),
|
|
227
|
-
},
|
|
228
156
|
// The gap Claude Code fills with its built-in `Explore`: read-heavy codebase
|
|
229
157
|
// search that would otherwise flood the parent's context. Defaults to the
|
|
230
158
|
// cheapest model on purpose — "find every caller of X" has no need of a
|
|
@@ -265,10 +193,14 @@ const BUILTIN_AGENTS = [
|
|
|
265
193
|
// quietly become "start editing".
|
|
266
194
|
{
|
|
267
195
|
name: 'planner',
|
|
196
|
+
// Lean prompt, as Claude Code's Plan builtin: it researches and reports, and the MAIN
|
|
197
|
+
// agent (which has nexrall.md and the conversation) judges the plan against project
|
|
198
|
+
// rules. Conventions that must bind the plan belong in the delegation prompt.
|
|
199
|
+
lightPrompt: true,
|
|
268
200
|
description: 'Read-only planning agent — researches a change and returns a concrete step-by-step implementation plan with risks and affected files. Cannot modify files.',
|
|
269
201
|
tools: RESEARCH_TOOLS,
|
|
270
202
|
// No `model` override — inherits the session's, for the same reason as
|
|
271
|
-
// explorer
|
|
203
|
+
// explorer above: a fixed alias silently billed Anthropic
|
|
272
204
|
// regardless of the user's chosen provider.
|
|
273
205
|
source: 'builtin',
|
|
274
206
|
prompt: [
|
|
@@ -289,74 +221,6 @@ const BUILTIN_AGENTS = [
|
|
|
289
221
|
'- Anything genuinely ambiguous, stated as an open question rather than a silent assumption.',
|
|
290
222
|
].join('\n'),
|
|
291
223
|
},
|
|
292
|
-
// Promoted from the test-gen plugin. Needs write access — it produces test
|
|
293
|
-
// files — but is deliberately forbidden from touching source, because "make the
|
|
294
|
-
// tests pass" is the single most common way an agent destroys signal.
|
|
295
|
-
{
|
|
296
|
-
name: 'test-writer',
|
|
297
|
-
description: 'Writes tests that follow the project\'s existing conventions. May create/edit TEST files only — never production source.',
|
|
298
|
-
tools: WRITE_TOOLS,
|
|
299
|
-
// Enforced, not merely requested: the permission gate refuses a write whose
|
|
300
|
-
// path is not a test file. Without this the allowlist would grant edit_file
|
|
301
|
-
// for every path and the rule below would be a suggestion the model is free
|
|
302
|
-
// to rationalise its way past.
|
|
303
|
-
testFilesOnly: true,
|
|
304
|
-
source: 'builtin',
|
|
305
|
-
prompt: [
|
|
306
|
-
'You write tests. You may create and edit TEST files only.',
|
|
307
|
-
'',
|
|
308
|
-
'Hard rules — these are the ways test-writing agents destroy value, so they are non-negotiable:',
|
|
309
|
-
'- NEVER modify production source to make a test pass. If the code looks wrong, REPORT it and stop.',
|
|
310
|
-
'- NEVER weaken, delete or skip an existing assertion or test.',
|
|
311
|
-
'- A test that cannot fail is worse than no test. Every test must be able to fail for one clear reason.',
|
|
312
|
-
'',
|
|
313
|
-
'Method:',
|
|
314
|
-
'1. Read the existing tests FIRST and copy their conventions exactly — runner, file naming, layout,',
|
|
315
|
-
' assertion style, fixture/helper patterns. Never introduce a new framework.',
|
|
316
|
-
'2. Test observable behaviour and the contract, not private internals.',
|
|
317
|
-
'3. Cover the boring-but-real cases: empty input, null/undefined, unicode and non-BMP characters,',
|
|
318
|
-
' boundaries, error paths, concurrency where it applies.',
|
|
319
|
-
'4. No sleeps or wall-clock dependence — those produce the flaky tests that get deleted later.',
|
|
320
|
-
'5. RUN the tests you wrote and report the real output. Never claim a test passes without running it.',
|
|
321
|
-
].join('\n'),
|
|
322
|
-
},
|
|
323
|
-
// The DevOps gap — answered with a READ-ONLY advisor, not an operator.
|
|
324
|
-
//
|
|
325
|
-
// A "DevOps agent" with write/apply access is a genuinely different risk class
|
|
326
|
-
// from the others here: its mistakes are `kubectl delete`, a bad `terraform
|
|
327
|
-
// apply`, a broken deploy pipeline — often not revertible and affecting
|
|
328
|
-
// production rather than a working tree. So this one diagnoses and proposes a
|
|
329
|
-
// diff; a human applies it. That asymmetry is the whole design.
|
|
330
|
-
{
|
|
331
|
-
name: 'devops-advisor',
|
|
332
|
-
description: 'Read-only CI/CD, container, and infrastructure advisor — diagnoses pipelines, Dockerfiles, and k8s manifests and proposes concrete fixes as a diff. Never applies changes.',
|
|
333
|
-
tools: RESEARCH_TOOLS,
|
|
334
|
-
// No `model` override — inherits the session's, for the same reason as the
|
|
335
|
-
// other specialists above.
|
|
336
|
-
source: 'builtin',
|
|
337
|
-
prompt: [
|
|
338
|
-
'You are an infrastructure and delivery advisor. You DIAGNOSE and PROPOSE. You never apply changes.',
|
|
339
|
-
'',
|
|
340
|
-
'Hard rules:',
|
|
341
|
-
'- READ-ONLY, and stricter than the other read-only agents: bash is for INSPECTION only',
|
|
342
|
-
' (git log/diff, cat, grep, `kubectl get/describe`, `docker images`, `terraform plan`).',
|
|
343
|
-
' NEVER run anything that mutates infrastructure — no apply/delete/scale/rollout/restart/push,',
|
|
344
|
-
' no `terraform apply`, no `helm upgrade`. If a fix needs such a command, WRITE IT OUT for a human.',
|
|
345
|
-
'- Never print secret values from env files, k8s Secrets or CI variables. Reference them by name.',
|
|
346
|
-
'',
|
|
347
|
-
'Method:',
|
|
348
|
-
'1. Read what actually exists — workflow files, Dockerfiles, manifests, kustomize overlays, the',
|
|
349
|
-
' deploy scripts — before drawing any conclusion. Never reason from what a stack "usually" looks like.',
|
|
350
|
-
'2. Follow the real path a change takes to production, and name the step that is broken or missing.',
|
|
351
|
-
'3. Check the failure modes that bite hardest: CI path filters that skip files a workload actually',
|
|
352
|
-
' needs, image tags that do not match what is deployed, missing health probes, absent resource',
|
|
353
|
-
' limits, secrets baked into images, ports/timeouts inconsistent between proxy and app, and',
|
|
354
|
-
' migrations that must run before the new image is live.',
|
|
355
|
-
'',
|
|
356
|
-
'Output: the diagnosis, the evidence (file:line or command output), the proposed change as a diff or',
|
|
357
|
-
'exact file content, and the command a human should run to apply and verify it.',
|
|
358
|
-
].join('\n'),
|
|
359
|
-
},
|
|
360
224
|
];
|
|
361
225
|
/**
|
|
362
226
|
* Every tool name a client may offer, for validating a hand-written allowlist.
|
package/dist/agent/loop.d.ts
CHANGED
|
@@ -330,7 +330,7 @@ export declare const WRITE_TOOL_NAMES: Set<string>;
|
|
|
330
330
|
* May an agent restricted to `testFilesOnly` perform this tool call?
|
|
331
331
|
*
|
|
332
332
|
* A tool allowlist is all-or-nothing per tool: granting `edit_file` grants it for
|
|
333
|
-
* every path in the repo.
|
|
333
|
+
* every path in the repo. A user-defined test-writer agent needs write access to produce
|
|
334
334
|
* tests, but must NOT be able to "fix" production source so a failing test goes
|
|
335
335
|
* green — the single most common way a test-writing agent destroys the signal it
|
|
336
336
|
* was asked to create. Its prompt says so; this makes it a refusal rather than a
|
package/dist/agent/loop.js
CHANGED
|
@@ -1529,7 +1529,7 @@ started) {
|
|
|
1529
1529
|
// Composed from the PROJECT's instructions (`_projectNexrallMd`), never from the
|
|
1530
1530
|
// parent's own composite prompt: a grandchild used to inherit its parent's role
|
|
1531
1531
|
// ("# Sub-agent role: general-purpose … make the changes") on top of its own
|
|
1532
|
-
// ("
|
|
1532
|
+
// ("explorer … you NEVER modify files"), plus the parent's private memory notes.
|
|
1533
1533
|
const projectMd = options._projectNexrallMd ?? options.nexrallMd ?? '';
|
|
1534
1534
|
const inheritedMd = agent?.lightPrompt ? '' : projectMd;
|
|
1535
1535
|
// `skills:` — preload those skills' instructions (raw body: no !`cmd` expansion runs
|
|
@@ -1567,8 +1567,8 @@ started) {
|
|
|
1567
1567
|
// ── The child's allowlist is INTERSECTED with the parent's ──────────────────
|
|
1568
1568
|
//
|
|
1569
1569
|
// A child's own definition can only ever NARROW what its parent had, never widen it.
|
|
1570
|
-
// Without this, nesting is a privilege-escalation ladder:
|
|
1571
|
-
// has no write_file, but `general-purpose` declares no `tools:` at all (= full access),
|
|
1570
|
+
// Without this, nesting is a privilege-escalation ladder: a read-only agent (planner, or any
|
|
1571
|
+
// user-defined reviewer) has no write_file, but `general-purpose` declares no `tools:` at all (= full access),
|
|
1572
1572
|
// so a read-only agent could delegate to an unrestricted one and edit the repo through
|
|
1573
1573
|
// it. The user's "this agent cannot write" would silently mean "cannot write directly".
|
|
1574
1574
|
//
|
|
@@ -1645,7 +1645,7 @@ started) {
|
|
|
1645
1645
|
// Path-scoped write restriction — see allowsTestOnlyWrite for the reasoning and its
|
|
1646
1646
|
// known limit. Applies when THIS agent declares it OR any ancestor did: like the tool
|
|
1647
1647
|
// allowlist above, a restriction can only ever be narrowed by nesting, never shed.
|
|
1648
|
-
// Without the inherited half,
|
|
1648
|
+
// Without the inherited half, a test-only agent could delegate to an unrestricted agent and
|
|
1649
1649
|
// have production source written on its behalf.
|
|
1650
1650
|
if (mcpAllow && req.tool.includes('__') && !mcpAllow.has(req.tool.split('__')[0])) {
|
|
1651
1651
|
throw new ToolNotAllowedError(`The "${agent?.name ?? 'general-purpose'}" sub-agent may only use tools from these MCP servers: ` +
|
|
@@ -2312,7 +2312,7 @@ exports.WRITE_TOOL_NAMES = new Set(['write_file', 'edit_file', 'multi_edit', 'de
|
|
|
2312
2312
|
* May an agent restricted to `testFilesOnly` perform this tool call?
|
|
2313
2313
|
*
|
|
2314
2314
|
* A tool allowlist is all-or-nothing per tool: granting `edit_file` grants it for
|
|
2315
|
-
* every path in the repo.
|
|
2315
|
+
* every path in the repo. A user-defined test-writer agent needs write access to produce
|
|
2316
2316
|
* tests, but must NOT be able to "fix" production source so a failing test goes
|
|
2317
2317
|
* green — the single most common way a test-writing agent destroys the signal it
|
|
2318
2318
|
* was asked to create. Its prompt says so; this makes it a refusal rather than a
|
|
@@ -3203,7 +3203,7 @@ async function runAgentLoop(initialMessages, options) {
|
|
|
3203
3203
|
hasOwnAgentStore: !!options._agentMemory,
|
|
3204
3204
|
// Same derivation, same reason, for the network. The base prompt's
|
|
3205
3205
|
// "Web search" section tells the run to reach the network, but
|
|
3206
|
-
// READ_ONLY_TOOLS (explorer,
|
|
3206
|
+
// READ_ONLY_TOOLS (explorer, planner, ...) contains neither
|
|
3207
3207
|
// `fetch_url` nor `web_search`. MEASURED on the 195-trial 2026-08-10
|
|
3208
3208
|
// deepseek-v4-pro benchmark: 29 of 51 `fetch_url` failures and 3 of 14
|
|
3209
3209
|
// `web_search` failures were the permission gate refusing a sub-agent
|
package/dist/agent/memory.d.ts
CHANGED
|
@@ -48,6 +48,17 @@ export declare function memoryStats(scope: MemoryScope, workDir?: string): {
|
|
|
48
48
|
entries: number;
|
|
49
49
|
archived: number;
|
|
50
50
|
};
|
|
51
|
+
/**
|
|
52
|
+
* Keep only the lines of a consolidation reply that are real memory entries.
|
|
53
|
+
*
|
|
54
|
+
* The prompt says "Output ONLY the bullet lines", but models do not always comply: a reply
|
|
55
|
+
* that opens with "I'll read the full, untruncated memory text before consolidating…" (seen
|
|
56
|
+
* in a user's global.md and a project file) used to be saved verbatim as an ACTIVE fact, then
|
|
57
|
+
* injected into every later session as ground truth. An entry is a line of the exact shape
|
|
58
|
+
* memory_write produces — `- [YYYY-MM-DD] …` — so anything else (preamble, "Here are the
|
|
59
|
+
* merged bullets:", a code fence, a trailing remark) is dropped rather than persisted.
|
|
60
|
+
*/
|
|
61
|
+
export declare function extractConsolidatedEntries(reply: string): string[];
|
|
51
62
|
/**
|
|
52
63
|
* Summarize the OLDEST half of a memory file's ACTIVE entries into a few consolidated
|
|
53
64
|
* bullets via an LLM call, keeping the most recent entries verbatim — same "compact,
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"memory.d.ts","sourceRoot":"","sources":["../../src/agent/memory.ts"],"names":[],"mappings":"AAiEA,MAAM,MAAM,WAAW,GAAG,SAAS,GAAG,QAAQ,CAAC;AAE/C;;;sDAGsD;AACtD,wBAAgB,cAAc,CAAC,KAAK,EAAE,WAAW,EAAE,OAAO,CAAC,EAAE,MAAM,GAAG,MAAM,CAI3E;AAED,eAAO,MAAM,sBAAsB,MAAM,CAAC;AAC1C,eAAO,MAAM,gBAAgB,QAAS,CAAC;AACvC,eAAO,MAAM,4BAA4B,QAAS,CAAC;AAMnD,eAAO,MAAM,qBAAqB,QAAmC,CAAC;AAKtE,eAAO,MAAM,wBAAwB,QAAmC,CAAC;AAoFzE,MAAM,WAAW,iBAAiB;IAChC,EAAE,EAAE,OAAO,CAAC;IACZ,OAAO,EAAE,OAAO,CAAC;IACjB,KAAK,EAAE,WAAW,CAAC;IACnB,IAAI,EAAE,MAAM,CAAC;IACb,gFAAgF;IAChF,UAAU,CAAC,EAAE,OAAO,CAAC;CACtB;AAED,MAAM,WAAW,kBAAkB;IACjC;;;;kCAI8B;IAC9B,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB;;6DAEyD;IACzD,MAAM,CAAC,EAAE,MAAM,CAAC;CACjB;AAED,wBAAsB,WAAW,CAC/B,OAAO,EAAE,MAAM,EACf,KAAK,EAAE,WAAW,EAClB,OAAO,CAAC,EAAE,MAAM,EAChB,IAAI,GAAE,kBAAuB,GAC5B,OAAO,CAAC,iBAAiB,CAAC,CA2D5B;AAED,MAAM,WAAW,iBAAiB;IAChC;;;iDAG6C;IAC7C,eAAe,CAAC,EAAE,OAAO,CAAC;CAC3B;AAED,wBAAgB,UAAU,CAAC,KAAK,EAAE,WAAW,EAAE,OAAO,CAAC,EAAE,MAAM,EAAE,IAAI,GAAE,iBAAsB,GAAG,MAAM,CAKrG;AAED;gFACgF;AAChF,wBAAgB,aAAa,CAAC,OAAO,CAAC,EAAE,MAAM,EAAE,IAAI,GAAE,iBAAsB,GAAG,MAAM,CASpF;AAED,wBAAgB,WAAW,CAAC,KAAK,EAAE,WAAW,EAAE,OAAO,CAAC,EAAE,MAAM,GAAG,IAAI,CAGtE;AAED,wBAAgB,WAAW,CACzB,KAAK,EAAE,WAAW,EAClB,OAAO,CAAC,EAAE,MAAM,GACf;IAAE,IAAI,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,MAAM,CAAC;IAAC,OAAO,EAAE,MAAM,CAAC;IAAC,QAAQ,EAAE,MAAM,CAAA;CAAE,CAUpE;AAED;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,wBAAsB,qBAAqB,CACzC,KAAK,EAAE,WAAW,EAClB,OAAO,EAAE,MAAM,GAAG,SAAS,EAC3B,SAAS,EAAE,CAAC,MAAM,EAAE,MAAM,KAAK,OAAO,CAAC,MAAM,CAAC,GAC7C,OAAO,CAAC,OAAO,CAAC,
|
|
1
|
+
{"version":3,"file":"memory.d.ts","sourceRoot":"","sources":["../../src/agent/memory.ts"],"names":[],"mappings":"AAiEA,MAAM,MAAM,WAAW,GAAG,SAAS,GAAG,QAAQ,CAAC;AAE/C;;;sDAGsD;AACtD,wBAAgB,cAAc,CAAC,KAAK,EAAE,WAAW,EAAE,OAAO,CAAC,EAAE,MAAM,GAAG,MAAM,CAI3E;AAED,eAAO,MAAM,sBAAsB,MAAM,CAAC;AAC1C,eAAO,MAAM,gBAAgB,QAAS,CAAC;AACvC,eAAO,MAAM,4BAA4B,QAAS,CAAC;AAMnD,eAAO,MAAM,qBAAqB,QAAmC,CAAC;AAKtE,eAAO,MAAM,wBAAwB,QAAmC,CAAC;AAoFzE,MAAM,WAAW,iBAAiB;IAChC,EAAE,EAAE,OAAO,CAAC;IACZ,OAAO,EAAE,OAAO,CAAC;IACjB,KAAK,EAAE,WAAW,CAAC;IACnB,IAAI,EAAE,MAAM,CAAC;IACb,gFAAgF;IAChF,UAAU,CAAC,EAAE,OAAO,CAAC;CACtB;AAED,MAAM,WAAW,kBAAkB;IACjC;;;;kCAI8B;IAC9B,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB;;6DAEyD;IACzD,MAAM,CAAC,EAAE,MAAM,CAAC;CACjB;AAED,wBAAsB,WAAW,CAC/B,OAAO,EAAE,MAAM,EACf,KAAK,EAAE,WAAW,EAClB,OAAO,CAAC,EAAE,MAAM,EAChB,IAAI,GAAE,kBAAuB,GAC5B,OAAO,CAAC,iBAAiB,CAAC,CA2D5B;AAED,MAAM,WAAW,iBAAiB;IAChC;;;iDAG6C;IAC7C,eAAe,CAAC,EAAE,OAAO,CAAC;CAC3B;AAED,wBAAgB,UAAU,CAAC,KAAK,EAAE,WAAW,EAAE,OAAO,CAAC,EAAE,MAAM,EAAE,IAAI,GAAE,iBAAsB,GAAG,MAAM,CAKrG;AAED;gFACgF;AAChF,wBAAgB,aAAa,CAAC,OAAO,CAAC,EAAE,MAAM,EAAE,IAAI,GAAE,iBAAsB,GAAG,MAAM,CASpF;AAED,wBAAgB,WAAW,CAAC,KAAK,EAAE,WAAW,EAAE,OAAO,CAAC,EAAE,MAAM,GAAG,IAAI,CAGtE;AAED,wBAAgB,WAAW,CACzB,KAAK,EAAE,WAAW,EAClB,OAAO,CAAC,EAAE,MAAM,GACf;IAAE,IAAI,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,MAAM,CAAC;IAAC,OAAO,EAAE,MAAM,CAAC;IAAC,QAAQ,EAAE,MAAM,CAAA;CAAE,CAUpE;AAED;;;;;;;;;GASG;AACH,wBAAgB,0BAA0B,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,EAAE,CAKlE;AAED;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,wBAAsB,qBAAqB,CACzC,KAAK,EAAE,WAAW,EAClB,OAAO,EAAE,MAAM,GAAG,SAAS,EAC3B,SAAS,EAAE,CAAC,MAAM,EAAE,MAAM,KAAK,OAAO,CAAC,MAAM,CAAC,GAC7C,OAAO,CAAC,OAAO,CAAC,CAyDlB;AAkBD,uEAAuE;AACvE,eAAO,MAAM,uBAAuB,OAAQ,CAAC;AAC7C,mFAAmF;AACnF,eAAO,MAAM,sBAAsB,QAAS,CAAC;AAE7C;;;;;;;;;GASG;AACH,wBAAgB,eAAe,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO,CAErD;AAED;;;;;GAKG;AACH,wBAAgB,eAAe,CAC7B,SAAS,EAAE,MAAM,EACjB,KAAK,EAAE,SAAS,GAAG,MAAM,GAAG,OAAO,EACnC,OAAO,CAAC,EAAE,MAAM,GACf,MAAM,GAAG,IAAI,CAqBf;AAED,iEAAiE;AACjE,wBAAgB,eAAe,CAC7B,SAAS,EAAE,MAAM,EACjB,KAAK,EAAE,SAAS,GAAG,MAAM,GAAG,OAAO,EACnC,OAAO,CAAC,EAAE,MAAM,GACf,MAAM,CAIR;AAED;;;;;;;GAOG;AACH,wBAAsB,gBAAgB,CACpC,SAAS,EAAE,MAAM,EACjB,KAAK,EAAE,SAAS,GAAG,MAAM,GAAG,OAAO,EACnC,OAAO,EAAE,MAAM,EACf,OAAO,CAAC,EAAE,MAAM,GACf,OAAO,CAAC;IAAE,EAAE,EAAE,OAAO,CAAC;IAAC,OAAO,EAAE,OAAO,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,CAAC,CAyB1D;AAED;;;;;GAKG;AACH,wBAAgB,mBAAmB,CACjC,SAAS,EAAE,MAAM,EACjB,KAAK,EAAE,SAAS,GAAG,MAAM,GAAG,OAAO,EACnC,OAAO,CAAC,EAAE,MAAM,GACf,MAAM,CAYR"}
|
package/dist/agent/memory.js
CHANGED
|
@@ -40,6 +40,7 @@ exports.readMemory = readMemory;
|
|
|
40
40
|
exports.readAllMemory = readAllMemory;
|
|
41
41
|
exports.clearMemory = clearMemory;
|
|
42
42
|
exports.memoryStats = memoryStats;
|
|
43
|
+
exports.extractConsolidatedEntries = extractConsolidatedEntries;
|
|
43
44
|
exports.compactMemoryIfNeeded = compactMemoryIfNeeded;
|
|
44
45
|
exports.isSafeAgentName = isSafeAgentName;
|
|
45
46
|
exports.agentMemoryPath = agentMemoryPath;
|
|
@@ -318,6 +319,22 @@ function memoryStats(scope, workDir) {
|
|
|
318
319
|
archived: archived.filter((l) => /^-\s\[\d{4}-\d{2}-\d{2}\]/.test(l)).length,
|
|
319
320
|
};
|
|
320
321
|
}
|
|
322
|
+
/**
|
|
323
|
+
* Keep only the lines of a consolidation reply that are real memory entries.
|
|
324
|
+
*
|
|
325
|
+
* The prompt says "Output ONLY the bullet lines", but models do not always comply: a reply
|
|
326
|
+
* that opens with "I'll read the full, untruncated memory text before consolidating…" (seen
|
|
327
|
+
* in a user's global.md and a project file) used to be saved verbatim as an ACTIVE fact, then
|
|
328
|
+
* injected into every later session as ground truth. An entry is a line of the exact shape
|
|
329
|
+
* memory_write produces — `- [YYYY-MM-DD] …` — so anything else (preamble, "Here are the
|
|
330
|
+
* merged bullets:", a code fence, a trailing remark) is dropped rather than persisted.
|
|
331
|
+
*/
|
|
332
|
+
function extractConsolidatedEntries(reply) {
|
|
333
|
+
return reply
|
|
334
|
+
.split('\n')
|
|
335
|
+
.map((l) => l.trimEnd())
|
|
336
|
+
.filter((l) => /^-\s\[\d{4}-\d{2}-\d{2}\]\s*\S/.test(l));
|
|
337
|
+
}
|
|
321
338
|
/**
|
|
322
339
|
* Summarize the OLDEST half of a memory file's ACTIVE entries into a few consolidated
|
|
323
340
|
* bullets via an LLM call, keeping the most recent entries verbatim — same "compact,
|
|
@@ -370,9 +387,12 @@ async function compactMemoryIfNeeded(scope, workDir, summarize) {
|
|
|
370
387
|
`draws from, and preserve any leading "[src: ...]" provenance tag verbatim when present. ` +
|
|
371
388
|
`Output ONLY the bullet lines, nothing else.\n\n${older.join('\n')}`;
|
|
372
389
|
try {
|
|
373
|
-
const summarized = (await summarize(prompt))
|
|
374
|
-
|
|
375
|
-
|
|
390
|
+
const summarized = extractConsolidatedEntries(await summarize(prompt));
|
|
391
|
+
// Zero valid bullets means the model answered in prose (or refused): treat it
|
|
392
|
+
// exactly like a failed call and keep the original entries. Writing the reply
|
|
393
|
+
// anyway would REPLACE the older half with chatter.
|
|
394
|
+
if (summarized.length > 0) {
|
|
395
|
+
active = [...summarized, ...recent];
|
|
376
396
|
changed = true;
|
|
377
397
|
}
|
|
378
398
|
}
|
package/dist/agent/planMode.d.ts
CHANGED
|
@@ -10,6 +10,20 @@ export interface PlanModeRefusal {
|
|
|
10
10
|
* "Provably" is doing real work: unknown verbs are refused, not guessed at.
|
|
11
11
|
*/
|
|
12
12
|
export declare function isReadOnlyCommand(command: string): boolean;
|
|
13
|
+
/**
|
|
14
|
+
* Is a (possibly piped / chained) command read-only on every segment?
|
|
15
|
+
*
|
|
16
|
+
* This is what the permission GATE uses to skip a prompt in ask/edit mode. Plan mode
|
|
17
|
+
* keeps the stricter whole-shape refusal above (`checkPlanMode`): a temporary lock can
|
|
18
|
+
* afford to say "no pipelines", an approval gate that says it about `git log | head`
|
|
19
|
+
* just trains people to click through prompts.
|
|
20
|
+
*
|
|
21
|
+
* Fail-closed throughout: anything that is not plainly a chain of read-only commands —
|
|
22
|
+
* a write redirect, `$(…)`, backticks, a lone `&`, an unknown verb, a segment split
|
|
23
|
+
* wrongly by a quote — returns false, which means "ask the user". The cost of a false
|
|
24
|
+
* false is one extra keypress; the cost of a false true is `rm` running unasked.
|
|
25
|
+
*/
|
|
26
|
+
export declare function isReadOnlyPipeline(command: string): boolean;
|
|
13
27
|
/**
|
|
14
28
|
* Should this tool call be refused because the session is in plan mode?
|
|
15
29
|
* Returns null when the call is allowed.
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"planMode.d.ts","sourceRoot":"","sources":["../../src/agent/planMode.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"planMode.d.ts","sourceRoot":"","sources":["../../src/agent/planMode.ts"],"names":[],"mappings":"AAgJA,MAAM,WAAW,eAAe;IAC9B,wDAAwD;IACxD,MAAM,EAAE,eAAe,GAAG,oBAAoB,CAAC;IAC/C,gFAAgF;IAChF,OAAO,EAAE,MAAM,CAAC;CACjB;AAqFD;;;;GAIG;AACH,wBAAgB,iBAAiB,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CA8D1D;AAQD;;;;;;;;;;;;GAYG;AACH,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CAO3D;AAED;;;GAGG;AACH,wBAAgB,aAAa,CAC3B,IAAI,EAAE,MAAM,EACZ,KAAK,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,GAC7B,eAAe,GAAG,IAAI,CAoCxB;AAED,oEAAoE;AACpE,eAAO,MAAM,sBAAsB,QAqBvB,CAAC"}
|
package/dist/agent/planMode.js
CHANGED
|
@@ -19,6 +19,7 @@
|
|
|
19
19
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
20
20
|
exports.PLAN_MODE_INSTRUCTIONS = void 0;
|
|
21
21
|
exports.isReadOnlyCommand = isReadOnlyCommand;
|
|
22
|
+
exports.isReadOnlyPipeline = isReadOnlyPipeline;
|
|
22
23
|
exports.checkPlanMode = checkPlanMode;
|
|
23
24
|
/**
|
|
24
25
|
* Tools that cannot alter anything outside the conversation.
|
|
@@ -71,10 +72,31 @@ const READ_ONLY_BASH = new Set([
|
|
|
71
72
|
'grep', 'egrep', 'fgrep', 'rg', 'ag', 'ack', 'find', 'fd', 'locate',
|
|
72
73
|
'diff', 'comm', 'cmp', 'sort', 'uniq', 'cut', 'tr', 'column', 'jq', 'yq',
|
|
73
74
|
'date', 'whoami', 'hostname', 'uname', 'env', 'printenv', 'id', 'groups',
|
|
74
|
-
'ps', '
|
|
75
|
-
|
|
75
|
+
'ps', 'uptime', 'free', 'node', 'python', 'python3',
|
|
76
|
+
// `less`/`more`/`top` are deliberately NOT here: pagers run `!cmd` from inside and
|
|
77
|
+
// `top` never exits. `env` is listed but only bare — see toolFlagsOk.
|
|
78
|
+
'tree', 'nl', 'tac', 'strings', 'md5sum', 'sha256sum',
|
|
76
79
|
'true', 'false', 'test', 'sleep', 'seq', 'expr',
|
|
80
|
+
// Changes this shell's cwd only. Needed so `cd x && ls` is not refused.
|
|
81
|
+
'cd',
|
|
77
82
|
]);
|
|
83
|
+
/**
|
|
84
|
+
* A verb written with a path is only trusted when it lives in a system bin dir.
|
|
85
|
+
* Stripping to the basename (the old behaviour) meant `./ls`, `/tmp/evil/cat` and
|
|
86
|
+
* `build/git` — attacker-written files — were classified by their NAME and waved
|
|
87
|
+
* through as read-only.
|
|
88
|
+
*/
|
|
89
|
+
const SYSTEM_BIN_DIRS = new Set([
|
|
90
|
+
'/bin', '/usr/bin', '/usr/local/bin', '/sbin', '/usr/sbin', '/opt/homebrew/bin',
|
|
91
|
+
]);
|
|
92
|
+
/**
|
|
93
|
+
* Environment assignments that change WHAT a read-only verb runs: the search path,
|
|
94
|
+
* loader, pager/editor/external-diff hooks, and the interpreters' option vars.
|
|
95
|
+
* `GIT_EXTERNAL_DIFF=./x git diff` and `PATH=/tmp ls` run the attacker's program
|
|
96
|
+
* while every word the classifier looks at is innocent.
|
|
97
|
+
*/
|
|
98
|
+
const UNSAFE_ENV = /^(PATH|HOME|SHELL|IFS|ENV|BASH_ENV|PAGER|MANPAGER|EDITOR|VISUAL|BROWSER|LD_\w*|DYLD_\w*|GIT_\w*|NODE_\w*|NPM_\w*|npm_\w*|PYTHON\w*|BASH\w*|RUBY\w*|PERL\w*|LESS\w*|XDG_\w*|\w*OPTS?)=/i;
|
|
99
|
+
const GH_NOUNS = new Set(['pr', 'issue', 'run', 'repo', 'release', 'workflow', 'gist', 'label']);
|
|
78
100
|
/**
|
|
79
101
|
* Subcommands that are read-only for tools whose safety depends on the verb.
|
|
80
102
|
*
|
|
@@ -117,6 +139,83 @@ const READ_ONLY_NESTED = {
|
|
|
117
139
|
* `|` is included: `cat x | tee y` writes, and so does `... | sh`.
|
|
118
140
|
*/
|
|
119
141
|
const SHELL_CONTROL = /[\n\r;&|`]|\$\(|>>|>|<\(/;
|
|
142
|
+
const isShort = (f) => f.startsWith('-') && !f.startsWith('--');
|
|
143
|
+
const flagName = (f) => f.split('=')[0];
|
|
144
|
+
/**
|
|
145
|
+
* Flags of an allowlisted VERB that turn it into a writer or an executor. The verb
|
|
146
|
+
* list says `sort` is read-only; `sort -o f` writes f. Each entry below is a real way
|
|
147
|
+
* to mutate through a verb the list trusts — the list alone was the hole.
|
|
148
|
+
*/
|
|
149
|
+
function toolFlagsOk(verb, args) {
|
|
150
|
+
const flags = args.filter((a) => a.startsWith('-'));
|
|
151
|
+
const pos = args.filter((a) => !a.startsWith('-'));
|
|
152
|
+
switch (verb) {
|
|
153
|
+
case 'env': return args.length === 0; // `env CMD` RUNS CMD
|
|
154
|
+
case 'sort':
|
|
155
|
+
return !flags.some((f) => (isShort(f) && /^-[A-Za-z]*o/.test(f)) || f.startsWith('--output') || f.startsWith('--compress-program'));
|
|
156
|
+
case 'uniq': return pos.length <= 1; // second positional is an OUTPUT file
|
|
157
|
+
case 'find':
|
|
158
|
+
return !args.some((a) => ['-exec', '-execdir', '-delete', '-ok', '-okdir', '-fprint', '-fprint0', '-fprintf', '-fls'].includes(a));
|
|
159
|
+
case 'fd':
|
|
160
|
+
return !flags.some((f) => f.startsWith('--exec') || (isShort(f) && /^-[A-Za-z]*[xX]/.test(f)));
|
|
161
|
+
case 'rg': return !flags.some((f) => flagName(f) === '--pre' || f.startsWith('--hostname-bin'));
|
|
162
|
+
case 'ag':
|
|
163
|
+
case 'ack': return !flags.some((f) => f.startsWith('--pager'));
|
|
164
|
+
case 'yq': return !flags.some((f) => (isShort(f) && /^-[A-Za-z]*i/.test(f)) || f.startsWith('--inplace'));
|
|
165
|
+
case 'tree': return !flags.some((f) => isShort(f) && /^-[A-Za-z]*o/.test(f));
|
|
166
|
+
case 'date': return !flags.some((f) => (isShort(f) && /^-[A-Za-z]*s/.test(f)) || f.startsWith('--set'));
|
|
167
|
+
case 'hostname': return pos.length === 0; // `hostname NAME` renames the machine
|
|
168
|
+
case 'file': return !flags.some((f) => (isShort(f) && /^-[A-Za-z]*C/.test(f)) || f.startsWith('--compile'));
|
|
169
|
+
default: return true;
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
/** Every flag must be a known read-only one; anything unfamiliar fails closed. */
|
|
173
|
+
function flagsWithin(flags, shorts, longs) {
|
|
174
|
+
return flags.every((f) => {
|
|
175
|
+
if (isShort(f))
|
|
176
|
+
return [...f.slice(1)].every((c) => shorts.includes(c) || /\d/.test(c));
|
|
177
|
+
return longs.includes(flagName(f));
|
|
178
|
+
});
|
|
179
|
+
}
|
|
180
|
+
const LISTISH = ['-l', '--list', '--contains', '--no-contains', '--merged', '--no-merged', '--points-at'];
|
|
181
|
+
/**
|
|
182
|
+
* Same idea one level down: `git branch` lists, `git branch newbranch` creates,
|
|
183
|
+
* `git branch -D main` deletes — one subcommand, three effects.
|
|
184
|
+
*/
|
|
185
|
+
function subcommandFlagsOk(verb, sub, args) {
|
|
186
|
+
const flags = args.filter((a) => a.startsWith('-'));
|
|
187
|
+
const pos = args.filter((a) => !a.startsWith('-')); // pos[0] === sub
|
|
188
|
+
const listish = flags.some((f) => LISTISH.includes(flagName(f)) || (isShort(f) && f.includes('l')));
|
|
189
|
+
if (verb === 'git') {
|
|
190
|
+
if (flags.some((f) => f.startsWith('--output') || f === '--ext-diff' || f === '-O' || f.startsWith('--open-files-in-pager')))
|
|
191
|
+
return false;
|
|
192
|
+
switch (sub) {
|
|
193
|
+
case 'branch':
|
|
194
|
+
return flagsWithin(flags, 'arvlq', ['--all', '--remotes', '--verbose', '--list', '--show-current', '--contains', '--no-contains', '--merged', '--no-merged', '--points-at', '--sort', '--format', '--column', '--no-column', '--color', '--no-color', '--abbrev', '--no-abbrev', '--ignore-case', '--omit-empty'])
|
|
195
|
+
&& (pos.length <= 1 || listish);
|
|
196
|
+
case 'tag':
|
|
197
|
+
return flagsWithin(flags, 'lnv', ['--list', '--contains', '--no-contains', '--merged', '--no-merged', '--points-at', '--sort', '--format', '--column', '--no-column', '--color', '--no-color', '--verify', '--ignore-case', '--omit-empty'])
|
|
198
|
+
&& (pos.length <= 1 || listish || flags.some((f) => f === '-v' || f === '--verify'));
|
|
199
|
+
case 'remote':
|
|
200
|
+
return flagsWithin(flags, 'vn', ['--verbose']) && (pos[1] === undefined || pos[1] === 'show' || pos[1] === 'get-url');
|
|
201
|
+
case 'config':
|
|
202
|
+
return flagsWithin(flags, 'lz', ['--get', '--get-all', '--get-regexp', '--list', '--show-origin', '--show-scope', '--global', '--local', '--system', '--worktree', '--name-only', '--null', '--includes', '--no-includes', '--type'])
|
|
203
|
+
&& pos.length <= 2; // `config` itself + at most one key
|
|
204
|
+
case 'reflog': return pos[1] !== 'expire' && pos[1] !== 'delete';
|
|
205
|
+
default: return true;
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
if (sub === 'config' && (verb === 'npm' || verb === 'pnpm' || verb === 'yarn')) {
|
|
209
|
+
return pos[1] === undefined || pos[1] === 'get' || pos[1] === 'list' || pos[1] === 'ls';
|
|
210
|
+
}
|
|
211
|
+
if (verb === 'go') {
|
|
212
|
+
return !flags.some((f) => {
|
|
213
|
+
const n = f.replace(/^--?/, '-');
|
|
214
|
+
return n === '-w' || n === '-u' || n.startsWith('-vettool') || n.startsWith('-toolexec') || n === '-exec';
|
|
215
|
+
});
|
|
216
|
+
}
|
|
217
|
+
return true;
|
|
218
|
+
}
|
|
120
219
|
function firstWords(command) {
|
|
121
220
|
return command.trim().split(/\s+/).filter(Boolean);
|
|
122
221
|
}
|
|
@@ -137,15 +236,21 @@ function isReadOnlyCommand(command) {
|
|
|
137
236
|
return false;
|
|
138
237
|
// `VAR=x cmd` — skip leading environment assignments to find the real verb.
|
|
139
238
|
let i = 0;
|
|
140
|
-
while (i < words.length && /^[A-Za-z_][A-Za-z0-9_]*=/.test(words[i]))
|
|
239
|
+
while (i < words.length && /^[A-Za-z_][A-Za-z0-9_]*=/.test(words[i])) {
|
|
240
|
+
if (UNSAFE_ENV.test(words[i]))
|
|
241
|
+
return false;
|
|
141
242
|
i++;
|
|
243
|
+
}
|
|
142
244
|
if (i >= words.length)
|
|
143
245
|
return false;
|
|
144
246
|
let verb = words[i];
|
|
145
|
-
//
|
|
247
|
+
// /usr/bin/git → git, but ONLY for a system bin dir (see SYSTEM_BIN_DIRS).
|
|
146
248
|
const slash = verb.lastIndexOf('/');
|
|
147
|
-
if (slash >= 0)
|
|
249
|
+
if (slash >= 0) {
|
|
250
|
+
if (!SYSTEM_BIN_DIRS.has(verb.slice(0, slash)))
|
|
251
|
+
return false;
|
|
148
252
|
verb = verb.slice(slash + 1);
|
|
253
|
+
}
|
|
149
254
|
// `sudo anything` is refused outright regardless of the verb behind it: plan
|
|
150
255
|
// mode is a promise about this machine, and privilege escalation is exactly
|
|
151
256
|
// where a mistaken allow is least recoverable.
|
|
@@ -159,16 +264,12 @@ function isReadOnlyCommand(command) {
|
|
|
159
264
|
// Bare `git` / `npm` just prints help — harmless.
|
|
160
265
|
return true;
|
|
161
266
|
}
|
|
162
|
-
if (subs.has(sub))
|
|
163
|
-
|
|
164
|
-
if (verb === 'git' && sub === 'config') {
|
|
165
|
-
const args = words.slice(i + 2).filter((w) => !w.startsWith('-'));
|
|
166
|
-
return args.length <= 1;
|
|
167
|
-
}
|
|
168
|
-
return true;
|
|
169
|
-
}
|
|
267
|
+
if (subs.has(sub))
|
|
268
|
+
return subcommandFlagsOk(verb, sub, words.slice(i + 1));
|
|
170
269
|
const nested = READ_ONLY_NESTED[verb];
|
|
171
|
-
|
|
270
|
+
// `gh api list -X DELETE` has "list" in second position too — the noun must be a
|
|
271
|
+
// real resource noun, and `api` (arbitrary HTTP verbs) is not one.
|
|
272
|
+
if (nested && rest[1] && nested.has(rest[1]) && (verb !== 'gh' || GH_NOUNS.has(sub)))
|
|
172
273
|
return true;
|
|
173
274
|
return false;
|
|
174
275
|
}
|
|
@@ -179,15 +280,42 @@ function isReadOnlyCommand(command) {
|
|
|
179
280
|
if (verb === 'node' || verb === 'python' || verb === 'python3') {
|
|
180
281
|
return words.slice(i + 1).every((w) => w === '--version' || w === '-V' || w === '-v');
|
|
181
282
|
}
|
|
182
|
-
|
|
183
|
-
if (verb === 'find' && words.some((w) => w === '-exec' || w === '-execdir' || w === '-delete')) {
|
|
283
|
+
if (!toolFlagsOk(verb, words.slice(i + 1)))
|
|
184
284
|
return false;
|
|
185
|
-
}
|
|
186
285
|
// Reading a file with `test`/`stat` is fine, but `tee` is not in the list at
|
|
187
286
|
// all, and `echo`/`printf` are only safe because redirection is already
|
|
188
287
|
// rejected by SHELL_CONTROL above.
|
|
189
288
|
return true;
|
|
190
289
|
}
|
|
290
|
+
/**
|
|
291
|
+
* Output redirections that cannot write anywhere interesting. Without this `ls 2>&1`
|
|
292
|
+
* and `cmd 2>/dev/null` — extremely common — would all count as "redirection".
|
|
293
|
+
*/
|
|
294
|
+
const HARMLESS_REDIRECT = /\s*\d?>\s*&\d\b|\s*\d?>>?\s*\/dev\/null(?=\s|$)/g;
|
|
295
|
+
/**
|
|
296
|
+
* Is a (possibly piped / chained) command read-only on every segment?
|
|
297
|
+
*
|
|
298
|
+
* This is what the permission GATE uses to skip a prompt in ask/edit mode. Plan mode
|
|
299
|
+
* keeps the stricter whole-shape refusal above (`checkPlanMode`): a temporary lock can
|
|
300
|
+
* afford to say "no pipelines", an approval gate that says it about `git log | head`
|
|
301
|
+
* just trains people to click through prompts.
|
|
302
|
+
*
|
|
303
|
+
* Fail-closed throughout: anything that is not plainly a chain of read-only commands —
|
|
304
|
+
* a write redirect, `$(…)`, backticks, a lone `&`, an unknown verb, a segment split
|
|
305
|
+
* wrongly by a quote — returns false, which means "ask the user". The cost of a false
|
|
306
|
+
* false is one extra keypress; the cost of a false true is `rm` running unasked.
|
|
307
|
+
*/
|
|
308
|
+
function isReadOnlyPipeline(command) {
|
|
309
|
+
const cmd = command.replace(HARMLESS_REDIRECT, ' ').trim();
|
|
310
|
+
if (!cmd)
|
|
311
|
+
return false;
|
|
312
|
+
if (/[>`]|\$\(|<\(/.test(cmd))
|
|
313
|
+
return false;
|
|
314
|
+
const segs = cmd.split(/\|\||&&|[;|\n\r]/).map((x) => x.trim()).filter(Boolean);
|
|
315
|
+
if (!segs.length)
|
|
316
|
+
return false;
|
|
317
|
+
return segs.every((seg) => !seg.includes('&') && isReadOnlyCommand(seg));
|
|
318
|
+
}
|
|
191
319
|
/**
|
|
192
320
|
* Should this tool call be refused because the session is in plan mode?
|
|
193
321
|
* Returns null when the call is allowed.
|