@duke-dsh-plugins/dsh-agent-approval 1.8.1 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -23,7 +23,7 @@
23
23
  | 🛡 **权限菜单第四项** | `/permission` 菜单新增 **自动审批** 预设;选中即开启,切到其他预设自动关闭,跨重启保持 |
24
24
  | 🤖 新权限模式 | 开启后:沙箱基线固定 `workspace-write`,审批策略切到 `ask`(内部接管),**不再弹人工审批** |
25
25
  | 🤖 自动裁决(默认 LLM 直连) | 每次提权请求由**一次 LLM 直连调用**裁决(v1.8.0 起默认):与子代理同一套审批人格 / 提示词 / 结构化裁决 `{decision, riskLevel, rationale}`,但**不创建审批子会话**(零上下文污染);设置页可切回「隔离子代理」(一次性 `spawn` 子代理:独立会话、零工具、只读材料) |
26
- | 🕵️ 自动审查(逐调用,实验) | 可选预设(需先在设置页启用 Jev):**Full access 基线**,每个工具调用(含 PTC 内层)执行前经 Jev 判定一次,风险调用**直接拒绝、body 不执行、不转人工**(fail-closed);规则表与会话内信任先行短路降噪;`/agent-review on\|off` 或菜单「自动审查」开启 |
26
+ | 🕵️ 自动审查(逐调用,实验) | 可选预设(需先在设置页启用 Jev):**Full access 基线**,每个工具调用(含 PTC 内层)执行前经 Jev 判定一次,风险调用**直接拒绝、body 不执行、不转人工**(fail-closed);规则表与会话内信任先行短路降噪;`/agent-review on\|off` 或菜单「自动审查」开启。v1.9.0 起**有效代码展开**:命令引用的解释器脚本(node / python / bash / ps1)内容会被读出并放进裁决 state,agent 临时写的脚本按**实际代码**受审而不是只看命令行(仅读 workspace 内文件,越界只标注、永不读取) |
27
27
  | ⛔ 风险即拒绝 | 破坏性 / 不可逆 / 越界(含修改操作系统或其他应用数据)/ 理由与实际命令不符 → 直接 `reject`;仅"安全、可逆、与任务相符、理由诚实"才 `approve`——项目自身的安装/部署脚本写其文档指定路径属任务所需 |
28
28
  | 🔒 Fail-closed | 审批 Agent 启动失败、超时(可配 30s–600s)、结果不合法 → 一律按拒绝处理,绝不静默放行 |
29
29
  | ⚙️ 审批模型可配置 | 设置页选择 Provider + Model,不选则固定用 **Harness 默认模型**(不跟随请求会话,口径稳定);选择与超时**持久保存**,重启不丢 |
package/client.js CHANGED
@@ -106,7 +106,7 @@ window.__ModuleLoader__.load({
106
106
  would bounce the selection back anyway. */
107
107
  [data-dsh-agent-approval-review-gate="off"] [data-dsh-agent-approval-review-item]{display:none}
108
108
  [data-dsh-agent-approval-review-item]::before{content:'';flex:none;width:16px;height:16px;background:currentColor;-webkit-mask:url("data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' width='16' height='16' viewBox='0 0 16 16' fill='none'%3E%3Cpath d='M8.20554 0.899994L14.7901 3.36857V7.01026C14.7901 12 11.0466 14.2103 8.20554 15.3C5.36446 14.2103 1.62012 12 1.62012 7.01026V3.36857L8.20554 0.899994Z' stroke='black' stroke-width='1.31831' stroke-linejoin='round'/%3E%3Cpath d='M8 3.2L9.1 5.9L11.8 7L9.1 8.1L8 10.8L6.9 8.1L4.2 7L6.9 5.9Z' fill='black'/%3E%3C/svg%3E") center/contain no-repeat;mask:url("data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' width='16' height='16' viewBox='0 0 16 16' fill='none'%3E%3Cpath d='M8.20554 0.899994L14.7901 3.36857V7.01026C14.7901 12 11.0466 14.2103 8.20554 15.3C5.36446 14.2103 1.62012 12 1.62012 7.01026V3.36857L8.20554 0.899994Z' stroke='black' stroke-width='1.31831' stroke-linejoin='round'/%3E%3Cpath d='M8 3.2L9.1 5.9L11.8 7L9.1 8.1L8 10.8L6.9 8.1L4.2 7L6.9 5.9Z' fill='black'/%3E%3C/svg%3E") center/contain no-repeat}
109
- .aapr-mode-review{margin-left:6px;font-size:10px;border:1px solid var(--dsw-alias-border-l2);border-radius:4px;padding:0 4px;color:var(--dsw-alias-label-secondary);white-space:nowrap}
109
+ .aapr-mode-review{display:inline-block;margin:2px 0 2px 6px;padding:1px 6px;font-size:10px;line-height:1.5;border:1px solid var(--dsw-alias-border-l2);border-radius:4px;background:var(--dsw-alias-bg-layer-2);color:var(--dsw-alias-label-secondary);white-space:nowrap;vertical-align:1px}
110
110
  `;
111
111
 
112
112
  // ---- Settings nav icon --------------------------------------------------
@@ -200,6 +200,12 @@ window.__ModuleLoader__.load({
200
200
  // with no glyph the icon span is absent, leaving label + chevron.
201
201
  // Skip menu rows (handled above) and the settings dialog (its nav
202
202
  // row carries the same label but is owned by the settings-nav icon).
203
+ // The trigger renders whichever preset is CURRENTLY selected, so it
204
+ // must accept BOTH of our preset names (v1.8.2): 自动审批 and 自动审查
205
+ // are equally ours, and neither is in the host's permissionGlyphs —
206
+ // registerReviewMenuItem only marks menu rows, so without this the
207
+ // trigger showed 自动审查 with no icon at all.
208
+ const triggerLabels = [currentLabel, String(REVIEW_LABEL).trim()];
203
209
  const buttons = document.querySelectorAll("button");
204
210
  for (let i = 0; i < buttons.length; i++) {
205
211
  const button = buttons[i];
@@ -212,7 +218,7 @@ window.__ModuleLoader__.load({
212
218
  const s = spans[j].textContent ? spans[j].textContent.trim() : "";
213
219
  if (s.length > 0) { labelText = s; break; }
214
220
  }
215
- const matches = labelText === currentLabel && button.querySelector("svg") !== null;
221
+ const matches = triggerLabels.indexOf(labelText) !== -1 && button.querySelector("svg") !== null;
216
222
  if (matches) button.setAttribute(PERM_TRIGGER_MARKER, "");
217
223
  else button.removeAttribute(PERM_TRIGGER_MARKER);
218
224
  }
package/index.js CHANGED
@@ -61,9 +61,9 @@
61
61
 
62
62
  import { Remote, TypertRemoteService } from "@deepseek-ai/dsh-typert-protocol";
63
63
  import { Service } from "@deepseek-ai/cordis";
64
- import { appendFile, mkdir, readFile, writeFile } from "node:fs/promises";
64
+ import { appendFile, mkdir, readFile, realpath, stat, writeFile } from "node:fs/promises";
65
65
  import { homedir } from "node:os";
66
- import { dirname, join } from "node:path";
66
+ import { dirname, join, resolve, sep } from "node:path";
67
67
 
68
68
  // ---- constants --------------------------------------------------------------
69
69
 
@@ -166,13 +166,28 @@ const JEV_QUESTIONS = {
166
166
  /**
167
167
  * v1.8.0: the per-call review mode (`agent-review` preset). The SAME policy
168
168
  * criteria as `JEV_QUESTIONS` — only the decision instruction wording adapts
169
- * from "escalation request" to the pending tool call.
169
+ * from "escalation request" to the pending tool call. v1.9.0 adds guidance
170
+ * for the `effectiveCode` state field (the actual interpreter scripts the
171
+ * call would run): judge the code when present; reduced visibility alone is
172
+ * never a rejection reason (误杀治理 holds); and a source-edit diff that
173
+ * merely mentions destructive or security operations is not dangerous
174
+ * itself — the live false kill that motivated this sentence: Jev rejected
175
+ * THIS plugin's own criteria-text edit (risk=high 0.87) on 2026-09-29
176
+ * because the diff contained the phrases below.
170
177
  */
171
178
  const JEV_REVIEW_QUESTIONS = {
172
179
  ...JEV_QUESTIONS,
173
180
  decision: {
174
181
  ...JEV_QUESTIONS.decision,
175
182
  instructions: "Should this pending tool call be allowed to execute?",
183
+ criteria: {
184
+ approve:
185
+ JEV_QUESTIONS.decision.criteria.approve +
186
+ " The state's `effectiveCode` field carries the actual content of the interpreter scripts (node/python/bash/PowerShell files) this call would execute — when present, judge THAT code, not just the command line: a plainly safe, task-scoped script stays approved even if its command line alone looks opaque. When the pending call edits source files, judge what the change does to the file and whether it fits the task: diff text that merely mentions destructive or security operations (tests, security policy, incident reports, documentation) does not make the change dangerous.",
187
+ reject:
188
+ JEV_QUESTIONS.decision.criteria.reject +
189
+ " When `effectiveCode` is present, judge by that code: a script whose code deletes or modifies data outside the workspace, exfiltrates secrets, or hides obfuscated payloads is rejected for what its code shows. A referenced script that is unreadable, outside the workspace, oversized, or not decodable reduces your visibility — weigh that in concreteRisk, but never reject for reduced visibility alone: an ordinary script you simply have not seen is not, by itself, a risk.",
190
+ },
176
191
  },
177
192
  };
178
193
 
@@ -197,6 +212,222 @@ const RUN_CODE_TOOL = "run_code";
197
212
  const REVIEW_DENIED_NAME = "AgentReviewDeniedError";
198
213
  const REVIEW_DENIED_CODE = "AGENT_REVIEW_DENIED";
199
214
 
215
+ /**
216
+ * v1.9.0 — per-call review (REVIEW MODE ONLY): effective-code expansion.
217
+ *
218
+ * The review judge otherwise sees a command line like `node scripts/x.js`
219
+ * and nothing about what that script does — the classic blind spot through
220
+ * which an agent-authored temp script once wiped ~/.dsh (2026-09 incident).
221
+ * Before judging, the plugin extracts the script files the command would
222
+ * execute and includes their contents in the Jev state's `effectiveCode`
223
+ * field, so the judge rules on the code itself.
224
+ *
225
+ * Scope (user ruling 2026-09-29): interpreter scripts only (node / python /
226
+ * bash / PowerShell and friends); packaged CLIs (`npm run`, `npx`, `git`, …)
227
+ * are deliberately NOT expanded. The escalation (自动审批) path is untouched —
228
+ * `_jevStateOf` is not modified. Reads are workspace-confined (paths that
229
+ * resolve — including through symlinks — outside the session workspace are
230
+ * noted, never read: the expansion must not become an exfiltration channel),
231
+ * capped, and fully best-effort: every fault degrades to a note inside the
232
+ * field and never blocks judging.
233
+ */
234
+ const REVIEW_CODE_MAX_FILES = 4;
235
+ const REVIEW_CODE_FILE_CHARS = 8192;
236
+ const REVIEW_CODE_TOTAL_CHARS = 10000;
237
+ const REVIEW_CODE_MAX_FILE_BYTES = 262144;
238
+ const REVIEW_CODE_MAX_COMMANDS = 6;
239
+
240
+ /** File extensions treated as "an interpreter script the agent may have written". */
241
+ const REVIEW_CODE_SCRIPT_EXTENSIONS = [
242
+ ".js", ".mjs", ".cjs", ".jsx", ".ts", ".mts", ".cts",
243
+ ".py", ".pyw", ".sh", ".bash", ".ps1", ".psm1", ".rb", ".pl",
244
+ ];
245
+
246
+ /** Interpreters whose positional argument (or `-File` value) is a script file. */
247
+ const REVIEW_CODE_INTERPRETERS = new Set([
248
+ "node", "node.exe", "nodejs", "bun", "bun.exe", "deno", "deno.exe",
249
+ "python", "python.exe", "python3", "python3.exe", "py", "py.exe",
250
+ "bash", "bash.exe", "sh", "sh.exe", "zsh", "dash",
251
+ "pwsh", "pwsh.exe", "powershell", "powershell.exe",
252
+ ]);
253
+
254
+ /** Strip wrapping quotes/braces a tokenizer may have left on a token. */
255
+ function reviewCodeCleanToken(token) {
256
+ return String(token).replace(/^["'{(]+/, "").replace(/["'}),;]+$/, "");
257
+ }
258
+
259
+ /** True when the token names a file with a known interpreter-script extension. */
260
+ function reviewCodeIsScriptFile(token) {
261
+ const t = reviewCodeCleanToken(token).toLowerCase();
262
+ for (const ext of REVIEW_CODE_SCRIPT_EXTENSIONS) {
263
+ if (t.endsWith(ext)) return true;
264
+ }
265
+ return false;
266
+ }
267
+
268
+ /**
269
+ * Split one shell command into top-level segments on `; | &` and newlines,
270
+ * respecting both quote styles and PowerShell brace blocks, so code embedded
271
+ * in `-Command "…"` / `-e "…"` stays one segment.
272
+ */
273
+ function reviewCodeSplitSegments(command) {
274
+ const parts = [];
275
+ let cur = "";
276
+ let quote = null;
277
+ let depth = 0;
278
+ for (const ch of String(command)) {
279
+ if (quote !== null) {
280
+ cur += ch;
281
+ if (ch === quote) quote = null;
282
+ continue;
283
+ }
284
+ if (ch === '"' || ch === "'") {
285
+ quote = ch;
286
+ cur += ch;
287
+ continue;
288
+ }
289
+ if (ch === "{") depth += 1;
290
+ if (ch === "}") depth = Math.max(0, depth - 1);
291
+ if (depth === 0 && (ch === ";" || ch === "|" || ch === "&" || ch === "\n" || ch === "\r")) {
292
+ parts.push(cur);
293
+ cur = "";
294
+ continue;
295
+ }
296
+ cur += ch;
297
+ }
298
+ parts.push(cur);
299
+ return parts.map((s) => s.trim()).filter((s) => s !== "");
300
+ }
301
+
302
+ /** Whitespace tokenizer that keeps quoted runs (quotes attached) as one token. */
303
+ function reviewCodeTokens(segment) {
304
+ const tokens = [];
305
+ let cur = "";
306
+ let quote = null;
307
+ let has = false;
308
+ for (const ch of String(segment)) {
309
+ if (quote !== null) {
310
+ cur += ch;
311
+ if (ch === quote) quote = null;
312
+ continue;
313
+ }
314
+ if (ch === '"' || ch === "'") {
315
+ quote = ch;
316
+ has = true;
317
+ cur += ch;
318
+ continue;
319
+ }
320
+ if (ch === " " || ch === "\t") {
321
+ if (has) tokens.push(cur);
322
+ cur = "";
323
+ has = false;
324
+ continue;
325
+ }
326
+ cur += ch;
327
+ has = true;
328
+ }
329
+ if (has) tokens.push(cur);
330
+ return tokens;
331
+ }
332
+
333
+ /**
334
+ * Extract script references from ONE command segment:
335
+ * - `{kind:"file", path}` — a script file the command would execute;
336
+ * - `{kind:"inline"}` — an inline-code flag (`-e`/`-c`/`-Command`…);
337
+ * that code is already visible verbatim in the tool arguments;
338
+ * - `{kind:"encoded", data}` — a `-EncodedCommand`/`-enc` payload;
339
+ * - `{kind:"note", text}` — anything unrecognizable worth surfacing.
340
+ * Depth-limited recursion expands code nested inside inline flags
341
+ * (`pwsh -Command "node x.js"`, `bash -c "python x.py"`).
342
+ */
343
+ function reviewCodeSegmentRefs(segment, depth) {
344
+ const tokens = reviewCodeTokens(segment);
345
+ let i = 0;
346
+ while (
347
+ i < tokens.length &&
348
+ (tokens[i] === "&" || tokens[i] === "." || tokens[i] === "call" ||
349
+ tokens[i] === "source" || tokens[i] === "sudo" || tokens[i] === "exec")
350
+ ) {
351
+ i += 1;
352
+ }
353
+ if (i >= tokens.length) return [];
354
+ const head = tokens[i].replace(/^.*[\\/]/, "").toLowerCase();
355
+ const refs = [];
356
+ if (!REVIEW_CODE_INTERPRETERS.has(head)) {
357
+ // Not an interpreter head: only a leading token that is itself a script
358
+ // file counts (call operators stripped above). No mid-segment scanning,
359
+ // so `git diff -- foo.py` never pulls foo.py into the state. Multi-word
360
+ // leftovers (a quoted chunk that reached this branch) are never taken
361
+ // as a path.
362
+ const lead = tokens[i];
363
+ if (reviewCodeIsScriptFile(lead) && !/\s/.test(reviewCodeCleanToken(lead))) {
364
+ refs.push({ kind: "file", path: reviewCodeCleanToken(lead) });
365
+ }
366
+ return refs;
367
+ }
368
+ let scriptFile = null;
369
+ let inline = false;
370
+ let module = false;
371
+ let j = i + 1;
372
+ while (j < tokens.length) {
373
+ const t = tokens[j];
374
+ const low = t.toLowerCase();
375
+ if (low === "-file") {
376
+ if (j + 1 < tokens.length && scriptFile === null) scriptFile = reviewCodeCleanToken(tokens[j + 1]);
377
+ j += 2;
378
+ continue;
379
+ }
380
+ if (low === "-m") {
381
+ // `python -m pkg` runs an installed module — packaged scope, not an
382
+ // agent-authored script (same exclusion as CLIs).
383
+ module = true;
384
+ j += 1;
385
+ continue;
386
+ }
387
+ if (low.startsWith("-enc")) {
388
+ if (j + 1 < tokens.length) refs.push({ kind: "encoded", data: tokens[j + 1] });
389
+ j += 2;
390
+ continue;
391
+ }
392
+ if (low === "-e" || low === "-c" || low === "-p" || low === "--print" || low.startsWith("-com") || low.startsWith("--eval")) {
393
+ inline = true;
394
+ if (depth < 2) {
395
+ // Recurse into the remaining tokens as one command string; strip the
396
+ // wrapping quotes first so the inner tokenization sees clean words
397
+ // (`pwsh -Command "node build.js"` → inner `node` + `build.js`).
398
+ const rest = tokens.slice(j + 1).join(" ").replace(/^["']+/, "").replace(/["']+$/, "");
399
+ if (rest.trim() !== "") {
400
+ for (const seg of reviewCodeSplitSegments(rest)) {
401
+ for (const inner of reviewCodeSegmentRefs(seg, depth + 1)) refs.push(inner);
402
+ }
403
+ }
404
+ }
405
+ break; // everything after an inline flag is code, not more flags/files
406
+ }
407
+ if (t.startsWith("-")) {
408
+ j += 1;
409
+ continue;
410
+ }
411
+ if (scriptFile === null && reviewCodeIsScriptFile(t)) scriptFile = reviewCodeCleanToken(t);
412
+ j += 1;
413
+ }
414
+ if (scriptFile !== null) refs.unshift({ kind: "file", path: scriptFile });
415
+ if (inline) refs.push({ kind: "inline" });
416
+ if (scriptFile === null && !inline && !module && refs.length === 0) {
417
+ refs.push({ kind: "note", text: head + " invoked, but no script file or inline-code flag was recognizable" });
418
+ }
419
+ return refs;
420
+ }
421
+
422
+ /** Extract all refs across every top-level segment of one command string. */
423
+ export function reviewCodeRefsOf(command) {
424
+ const refs = [];
425
+ for (const seg of reviewCodeSplitSegments(command)) {
426
+ for (const ref of reviewCodeSegmentRefs(seg, 0)) refs.push(ref);
427
+ }
428
+ return refs;
429
+ }
430
+
200
431
  /** Constrained to the
201
432
  * JSON-Schema subset `assertObjectJsonSchema` enforces for subagent outputs
202
433
  * (type/properties/required/additionalProperties/enum only).
@@ -2211,6 +2442,136 @@ export class AgentApprovalService extends TypertRemoteService {
2211
2442
  };
2212
2443
  }
2213
2444
 
2445
+ /**
2446
+ * v1.9.0: build the `effectiveCode` review-state field — the actual
2447
+ * contents of the interpreter scripts one pending call would execute —
2448
+ * plus a one-line visibility note for the audit rationale. The judge
2449
+ * state stays workspace-scoped: file contents are only ever included for
2450
+ * paths that resolve (symlinks followed and re-checked) inside the
2451
+ * session workspace, so the expansion never ships beyond-boundary file
2452
+ * contents anywhere. Capped, and fully best-effort: every fault degrades
2453
+ * to a note inside the field and never blocks judging.
2454
+ */
2455
+ async _reviewEffectiveCode(session, argsRaw) {
2456
+ let args;
2457
+ try {
2458
+ args = typeof argsRaw === "string" ? JSON.parse(argsRaw) : undefined;
2459
+ } catch (e) {
2460
+ args = undefined;
2461
+ }
2462
+ let cwd = "";
2463
+ try {
2464
+ if (session.header && typeof session.header.cwd === "string") cwd = session.header.cwd;
2465
+ } catch (e) {
2466
+ /* header access is best-effort */
2467
+ }
2468
+ let baseDir = cwd;
2469
+ if (
2470
+ args && typeof args === "object" && !Array.isArray(args) &&
2471
+ typeof args.workdir === "string" && args.workdir.trim() !== ""
2472
+ ) {
2473
+ baseDir = args.workdir;
2474
+ }
2475
+ const commands = [];
2476
+ const pushCommand = (v) => {
2477
+ if (typeof v === "string" && v.trim() !== "" && commands.length < REVIEW_CODE_MAX_COMMANDS) commands.push(v);
2478
+ };
2479
+ if (typeof args === "string") pushCommand(args);
2480
+ else if (args && typeof args === "object") {
2481
+ for (const v of Object.values(args)) pushCommand(v);
2482
+ }
2483
+
2484
+ const files = [];
2485
+ let inlineCount = 0;
2486
+ let encodedCount = 0;
2487
+ const notes = [];
2488
+ for (const command of commands) {
2489
+ for (const ref of reviewCodeRefsOf(command)) {
2490
+ if (ref.kind === "file") {
2491
+ const dup = files.some((f) => f.toLowerCase() === ref.path.toLowerCase());
2492
+ if (!dup && files.length < REVIEW_CODE_MAX_FILES) files.push(ref.path);
2493
+ } else if (ref.kind === "inline") inlineCount += 1;
2494
+ else if (ref.kind === "encoded") encodedCount += 1;
2495
+ else if (ref.kind === "note") notes.push(ref.text);
2496
+ }
2497
+ }
2498
+
2499
+ const lines = [];
2500
+ let filesRead = 0;
2501
+ let filesSkipped = 0;
2502
+ let filesMissing = 0;
2503
+ let charsIncluded = 0;
2504
+ const baseAbs = baseDir.trim() !== "" ? resolve(baseDir) : "";
2505
+ const insideOf = (p) =>
2506
+ baseAbs !== "" &&
2507
+ (p.toLowerCase() === baseAbs.toLowerCase() || p.toLowerCase().startsWith(baseAbs.toLowerCase() + sep));
2508
+ for (const raw of files) {
2509
+ let resolved;
2510
+ try {
2511
+ resolved = resolve(baseAbs !== "" ? baseAbs : ".", raw);
2512
+ } catch (e) {
2513
+ filesMissing += 1;
2514
+ continue;
2515
+ }
2516
+ if (!insideOf(resolved)) {
2517
+ filesSkipped += 1;
2518
+ lines.push("[script] " + raw + " — path resolves beyond the workspace boundary; skipped, contents never included");
2519
+ continue;
2520
+ }
2521
+ let real = resolved;
2522
+ try {
2523
+ real = await realpath(resolved);
2524
+ if (!insideOf(real)) {
2525
+ filesSkipped += 1;
2526
+ lines.push("[script] " + raw + " — symlink target lies beyond the workspace boundary; skipped, contents never included");
2527
+ continue;
2528
+ }
2529
+ } catch (e) {
2530
+ filesMissing += 1;
2531
+ lines.push("[script] " + raw + " — not found on disk at review time");
2532
+ continue;
2533
+ }
2534
+ try {
2535
+ const st = await stat(real);
2536
+ if (!st.isFile() || st.size > REVIEW_CODE_MAX_FILE_BYTES) {
2537
+ filesSkipped += 1;
2538
+ lines.push(
2539
+ "[script] " + raw + " — " +
2540
+ (st.isFile() ? "too large to review (" + st.size + " bytes)" : "not a regular file"),
2541
+ );
2542
+ continue;
2543
+ }
2544
+ let text = await readFile(real, "utf8");
2545
+ const total = text.length;
2546
+ if (total > REVIEW_CODE_FILE_CHARS) {
2547
+ text = text.slice(0, REVIEW_CODE_FILE_CHARS) + "\n…[truncated, " + (total - REVIEW_CODE_FILE_CHARS) + " more chars]";
2548
+ }
2549
+ filesRead += 1;
2550
+ charsIncluded += Math.min(total, REVIEW_CODE_FILE_CHARS);
2551
+ lines.push("[script " + filesRead + "] " + raw + " → " + real + " (" + st.size + " bytes)\n" + text);
2552
+ } catch (e) {
2553
+ filesMissing += 1;
2554
+ lines.push("[script] " + raw + " — not readable at review time: " + errText(e));
2555
+ }
2556
+ }
2557
+ if (encodedCount > 0) {
2558
+ lines.push("[-EncodedCommand] " + encodedCount + " base64-encoded invocation(s) present — the payload is not verifiable as readable code");
2559
+ }
2560
+ if (inlineCount > 0) {
2561
+ lines.push("[inline] " + inlineCount + " inline-code invocation(s) — that code is embedded verbatim in the command text / toolArguments");
2562
+ }
2563
+ if (notes.length > 0) lines.push("[note] " + notes.join("; "));
2564
+ if (lines.length === 0) {
2565
+ return { text: "(no interpreter script file is referenced by this call)", note: "no script files referenced" };
2566
+ }
2567
+ const note =
2568
+ "code: " + filesRead + " script file(s)" +
2569
+ (charsIncluded > 0 ? ", ~" + Math.round(charsIncluded / 1024) + " KB included" : "") +
2570
+ (filesSkipped > 0 ? ", " + filesSkipped + " skipped (beyond workspace / oversized)" : "") +
2571
+ (filesMissing > 0 ? ", " + filesMissing + " missing/unreadable" : "");
2572
+ return { text: trunc(lines.join("\n\n"), REVIEW_CODE_TOTAL_CHARS), note: note };
2573
+ }
2574
+
2214
2575
  /**
2215
2576
  * Judge one pending call through the Jev HTTP API (the review-mode judge).
2216
2577
  * Mirrors `_judgeWithJev`'s fail-closed contract exactly — same `_jevParse`
@@ -2242,8 +2603,14 @@ export class AgentApprovalService extends TypertRemoteService {
2242
2603
  }
2243
2604
 
2244
2605
  let winner;
2606
+ let codeInfo = { text: "(not evaluated)", note: "" };
2245
2607
  try {
2246
- const state = this._reviewStateOf(session, exec, argsRaw);
2608
+ try {
2609
+ codeInfo = await this._reviewEffectiveCode(session, argsRaw);
2610
+ } catch (e) {
2611
+ codeInfo = { text: "(expansion fault: " + errText(e) + ")", note: "code expansion fault" };
2612
+ }
2613
+ const state = { ...this._reviewStateOf(session, exec, argsRaw), effectiveCode: codeInfo.text };
2247
2614
  winner = await Promise.race([
2248
2615
  this._jevRequest(cfg, state, controller.signal, JEV_REVIEW_QUESTIONS)
2249
2616
  .then((body) => ({ kind: "result", body: body }))
@@ -2310,7 +2677,7 @@ export class AgentApprovalService extends TypertRemoteService {
2310
2677
  outcome: approved ? "allowed-once" : "rejected",
2311
2678
  riskLevel: parsed.riskLevel,
2312
2679
  model: label,
2313
- rationale: trunc(parsed.rationale, 600),
2680
+ rationale: trunc(parsed.rationale + (codeInfo.note !== "" ? " [" + codeInfo.note + "]" : ""), 600),
2314
2681
  });
2315
2682
  if (approved) {
2316
2683
  if (trustKey !== undefined) {
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@duke-dsh-plugins/dsh-agent-approval",
3
- "version": "1.8.1",
4
- "description": "Agent-decided approvals for DeepSeek Harness: an 自动审批 permission mode where every sandbox escalation is judged automatically (default judge: one direct LLM call with no subagent session; optional isolated judge subagent or the TypeSafe Jev decision model), plus an opt-in 自动审查 per-call review mode (Full access base; every tool call reviewed by the Jev judge before execution, risky calls rejected with no human fallback), with a configurable judge model and a per-session audit trail in the conversation window's 审批 tab.",
3
+ "version": "1.9.0",
4
+ "description": "Agent-decided approvals for DeepSeek Harness: an 自动审批 permission mode where every sandbox escalation is judged automatically (default judge: one direct LLM call with no subagent session; optional isolated judge subagent or the TypeSafe Jev decision model), plus an opt-in 自动审查 per-call review mode (Full access base; every tool call reviewed by the Jev judge before execution, with interpreter scripts the command references expanded into the judge state so temp scripts are judged by their actual code; risky calls rejected with no human fallback), with a configurable judge model and a per-session audit trail in the conversation window's 审批 tab.",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/MoonlitDropOfBlood/dsh-agent-approval.git"