@duke-dsh-plugins/dsh-agent-approval 1.8.2 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -10
- package/client.js +176 -172
- package/index.js +478 -228
- package/package.json +2 -2
- package/typert.host.js +13 -5
package/index.js
CHANGED
|
@@ -31,11 +31,14 @@
|
|
|
31
31
|
* `callId`) plus the asker's stated reason. A rejection must name the
|
|
32
32
|
* concrete, credible risk the operation creates (destructive /
|
|
33
33
|
* irreversible / out-of-scope / dishonest); vague unease is approved.
|
|
34
|
-
* v1.
|
|
35
|
-
*
|
|
36
|
-
*
|
|
37
|
-
*
|
|
38
|
-
*
|
|
34
|
+
* v1.10.0: the judge is ALWAYS an LLM — either one direct stream call or
|
|
35
|
+
* the spawn child. The TypeSafe Jev "System One" decision model (v1.6.0
|
|
36
|
+
* through v1.9.x could judge escalations too, as the synthetic provider
|
|
37
|
+
* id `typesafe`) is NO LONGER selectable here: its calibrated-but-shallow
|
|
38
|
+
* risk judgement is a weaker safety net than an LLM judge on the
|
|
39
|
+
* escalation path, which is exactly the path a human approval would have
|
|
40
|
+
* guarded. Jev now backs the separate per-call review mode ONLY
|
|
41
|
+
* (`_reviewWithJev`), configured independently from this judge.
|
|
39
42
|
*
|
|
40
43
|
* 3. FAIL CLOSED — any infrastructure fault, timeout, malformed verdict, or
|
|
41
44
|
* cancellation maps to the fail-closed approval outcomes
|
|
@@ -61,9 +64,9 @@
|
|
|
61
64
|
|
|
62
65
|
import { Remote, TypertRemoteService } from "@deepseek-ai/dsh-typert-protocol";
|
|
63
66
|
import { Service } from "@deepseek-ai/cordis";
|
|
64
|
-
import { appendFile, mkdir, readFile, writeFile } from "node:fs/promises";
|
|
67
|
+
import { appendFile, mkdir, readFile, realpath, stat, writeFile } from "node:fs/promises";
|
|
65
68
|
import { homedir } from "node:os";
|
|
66
|
-
import { dirname, join } from "node:path";
|
|
69
|
+
import { dirname, join, resolve, sep } from "node:path";
|
|
67
70
|
|
|
68
71
|
// ---- constants --------------------------------------------------------------
|
|
69
72
|
|
|
@@ -110,29 +113,35 @@ const DATA_DIR = join(process.env.DSH_HOME || join(homedir(), ".dsh"), "agent-ap
|
|
|
110
113
|
const CONFIG_FILE = join(DATA_DIR, "config.json");
|
|
111
114
|
|
|
112
115
|
/**
|
|
113
|
-
* v1.6.0
|
|
114
|
-
* model (https://api.typesafe.ai/v1/systemone)
|
|
115
|
-
* it answers typed questions (Choice / Score / Noul) over
|
|
116
|
-
* calibrated probability distributions in ~70–500ms
|
|
117
|
-
*
|
|
118
|
-
*
|
|
119
|
-
*
|
|
120
|
-
*
|
|
121
|
-
*
|
|
122
|
-
* confidence
|
|
123
|
-
* grant, and (below the gate) not a recorded rejection either.
|
|
116
|
+
* v1.6.0 (removed from the escalation judge in v1.10.0): the TypeSafe Jev
|
|
117
|
+
* "System One" decision model (https://api.typesafe.ai/v1/systemone). It does
|
|
118
|
+
* not generate text — it answers typed questions (Choice / Score / Noul) over
|
|
119
|
+
* one `state` with calibrated probability distributions in ~70–500ms, and it
|
|
120
|
+
* cannot appear in `llm.listProviders()` (not a chat route), hence the direct
|
|
121
|
+
* HTTP call. It now judges the PER-CALL REVIEW mode only (`_reviewWithJev`),
|
|
122
|
+
* configured independently of the approval judge. Fail-closed is preserved end
|
|
123
|
+
* to end there as well: any transport fault, non-200, malformed answer, or a
|
|
124
|
+
* confidence below the configured gate denies the call (or, on the low
|
|
125
|
+
* confidence branch, records `unavailable` without a verdict).
|
|
124
126
|
*/
|
|
125
|
-
const JEV_PROVIDER = "typesafe";
|
|
126
127
|
const JEV_DEFAULT_MODEL = "jev-latest";
|
|
127
128
|
const JEV_DEFAULT_ENDPOINT = "https://api.typesafe.ai/v1/systemone";
|
|
128
129
|
const JEV_DEFAULT_CONFIDENCE = 0.5;
|
|
130
|
+
/**
|
|
131
|
+
* The retired escalation-judge provider id (v1.6.0–v1.9.x). Kept ONLY to
|
|
132
|
+
* recognize and drop it from persisted configs: a config.json still carrying
|
|
133
|
+
* `model.provider: "typesafe"` migrates to the harness default route, and
|
|
134
|
+
* `setModel` treats it as "clear the override". Never used to route a judge.
|
|
135
|
+
*/
|
|
136
|
+
const JEV_LEGACY_PROVIDER = "typesafe";
|
|
129
137
|
|
|
130
138
|
/**
|
|
131
|
-
* The typed questions sent to Jev
|
|
132
|
-
* (
|
|
133
|
-
*
|
|
134
|
-
*
|
|
135
|
-
*
|
|
139
|
+
* The typed questions sent to Jev — the shared policy base for
|
|
140
|
+
* `JEV_REVIEW_QUESTIONS` (the only sender since v1.10.0). Policy lives in the
|
|
141
|
+
* criteria descriptions (Jev reads instructions literally and injects domain
|
|
142
|
+
* knowledge only through state + criteria); the wording mirrors the LLM
|
|
143
|
+
* judge's APPROVE conditions / REJECT list / 误杀治理 so both judge to the
|
|
144
|
+
* same standard. `decision` and `riskLevel` are Choices (discrete options +
|
|
136
145
|
* probabilities + confidence); `concreteRisk` is a Noul probe whose
|
|
137
146
|
* probability is folded into the synthesized audit rationale.
|
|
138
147
|
*/
|
|
@@ -164,15 +173,31 @@ const JEV_QUESTIONS = {
|
|
|
164
173
|
};
|
|
165
174
|
|
|
166
175
|
/**
|
|
167
|
-
* v1.8.0: the per-call review mode (`agent-review` preset)
|
|
168
|
-
* criteria as `JEV_QUESTIONS` — only the
|
|
169
|
-
* from "escalation request" to the
|
|
176
|
+
* v1.8.0: the per-call review mode (`agent-review` preset) — since v1.10.0
|
|
177
|
+
* the ONLY Jev caller. The SAME policy criteria as `JEV_QUESTIONS` — only the
|
|
178
|
+
* decision instruction wording adapts from "escalation request" to the
|
|
179
|
+
* pending tool call. v1.9.0 adds guidance
|
|
180
|
+
* for the `effectiveCode` state field (the actual interpreter scripts the
|
|
181
|
+
* call would run): judge the code when present; reduced visibility alone is
|
|
182
|
+
* never a rejection reason (误杀治理 holds); and a source-edit diff that
|
|
183
|
+
* merely mentions destructive or security operations is not dangerous
|
|
184
|
+
* itself — the live false kill that motivated this sentence: Jev rejected
|
|
185
|
+
* THIS plugin's own criteria-text edit (risk=high 0.87) on 2026-09-29
|
|
186
|
+
* because the diff contained the phrases below.
|
|
170
187
|
*/
|
|
171
188
|
const JEV_REVIEW_QUESTIONS = {
|
|
172
189
|
...JEV_QUESTIONS,
|
|
173
190
|
decision: {
|
|
174
191
|
...JEV_QUESTIONS.decision,
|
|
175
192
|
instructions: "Should this pending tool call be allowed to execute?",
|
|
193
|
+
criteria: {
|
|
194
|
+
approve:
|
|
195
|
+
JEV_QUESTIONS.decision.criteria.approve +
|
|
196
|
+
" The state's `effectiveCode` field carries the actual content of the interpreter scripts (node/python/bash/PowerShell files) this call would execute — when present, judge THAT code, not just the command line: a plainly safe, task-scoped script stays approved even if its command line alone looks opaque. When the pending call edits source files, judge what the change does to the file and whether it fits the task: diff text that merely mentions destructive or security operations (tests, security policy, incident reports, documentation) does not make the change dangerous.",
|
|
197
|
+
reject:
|
|
198
|
+
JEV_QUESTIONS.decision.criteria.reject +
|
|
199
|
+
" When `effectiveCode` is present, judge by that code: a script whose code deletes or modifies data outside the workspace, exfiltrates secrets, or hides obfuscated payloads is rejected for what its code shows. A referenced script that is unreadable, outside the workspace, oversized, or not decodable reduces your visibility — weigh that in concreteRisk, but never reject for reduced visibility alone: an ordinary script you simply have not seen is not, by itself, a risk.",
|
|
200
|
+
},
|
|
176
201
|
},
|
|
177
202
|
};
|
|
178
203
|
|
|
@@ -197,6 +222,222 @@ const RUN_CODE_TOOL = "run_code";
|
|
|
197
222
|
const REVIEW_DENIED_NAME = "AgentReviewDeniedError";
|
|
198
223
|
const REVIEW_DENIED_CODE = "AGENT_REVIEW_DENIED";
|
|
199
224
|
|
|
225
|
+
/**
|
|
226
|
+
* v1.9.0 — per-call review (REVIEW MODE ONLY): effective-code expansion.
|
|
227
|
+
*
|
|
228
|
+
* The review judge otherwise sees a command line like `node scripts/x.js`
|
|
229
|
+
* and nothing about what that script does — the classic blind spot through
|
|
230
|
+
* which an agent-authored temp script once wiped ~/.dsh (2026-09 incident).
|
|
231
|
+
* Before judging, the plugin extracts the script files the command would
|
|
232
|
+
* execute and includes their contents in the Jev state's `effectiveCode`
|
|
233
|
+
* field, so the judge rules on the code itself.
|
|
234
|
+
*
|
|
235
|
+
* Scope (user ruling 2026-09-29): interpreter scripts only (node / python /
|
|
236
|
+
* bash / PowerShell and friends); packaged CLIs (`npm run`, `npx`, `git`, …)
|
|
237
|
+
* are deliberately NOT expanded. The escalation (自动审批) path is untouched —
|
|
238
|
+
* `_jevStateOf` is not modified. Reads are workspace-confined (paths that
|
|
239
|
+
* resolve — including through symlinks — outside the session workspace are
|
|
240
|
+
* noted, never read: the expansion must not become an exfiltration channel),
|
|
241
|
+
* capped, and fully best-effort: every fault degrades to a note inside the
|
|
242
|
+
* field and never blocks judging.
|
|
243
|
+
*/
|
|
244
|
+
const REVIEW_CODE_MAX_FILES = 4;
|
|
245
|
+
const REVIEW_CODE_FILE_CHARS = 8192;
|
|
246
|
+
const REVIEW_CODE_TOTAL_CHARS = 10000;
|
|
247
|
+
const REVIEW_CODE_MAX_FILE_BYTES = 262144;
|
|
248
|
+
const REVIEW_CODE_MAX_COMMANDS = 6;
|
|
249
|
+
|
|
250
|
+
/** File extensions treated as "an interpreter script the agent may have written". */
|
|
251
|
+
const REVIEW_CODE_SCRIPT_EXTENSIONS = [
|
|
252
|
+
".js", ".mjs", ".cjs", ".jsx", ".ts", ".mts", ".cts",
|
|
253
|
+
".py", ".pyw", ".sh", ".bash", ".ps1", ".psm1", ".rb", ".pl",
|
|
254
|
+
];
|
|
255
|
+
|
|
256
|
+
/** Interpreters whose positional argument (or `-File` value) is a script file. */
|
|
257
|
+
const REVIEW_CODE_INTERPRETERS = new Set([
|
|
258
|
+
"node", "node.exe", "nodejs", "bun", "bun.exe", "deno", "deno.exe",
|
|
259
|
+
"python", "python.exe", "python3", "python3.exe", "py", "py.exe",
|
|
260
|
+
"bash", "bash.exe", "sh", "sh.exe", "zsh", "dash",
|
|
261
|
+
"pwsh", "pwsh.exe", "powershell", "powershell.exe",
|
|
262
|
+
]);
|
|
263
|
+
|
|
264
|
+
/** Strip wrapping quotes/braces a tokenizer may have left on a token. */
|
|
265
|
+
function reviewCodeCleanToken(token) {
|
|
266
|
+
return String(token).replace(/^["'{(]+/, "").replace(/["'}),;]+$/, "");
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/** True when the token names a file with a known interpreter-script extension. */
|
|
270
|
+
function reviewCodeIsScriptFile(token) {
|
|
271
|
+
const t = reviewCodeCleanToken(token).toLowerCase();
|
|
272
|
+
for (const ext of REVIEW_CODE_SCRIPT_EXTENSIONS) {
|
|
273
|
+
if (t.endsWith(ext)) return true;
|
|
274
|
+
}
|
|
275
|
+
return false;
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
/**
|
|
279
|
+
* Split one shell command into top-level segments on `; | &` and newlines,
|
|
280
|
+
* respecting both quote styles and PowerShell brace blocks, so code embedded
|
|
281
|
+
* in `-Command "…"` / `-e "…"` stays one segment.
|
|
282
|
+
*/
|
|
283
|
+
function reviewCodeSplitSegments(command) {
|
|
284
|
+
const parts = [];
|
|
285
|
+
let cur = "";
|
|
286
|
+
let quote = null;
|
|
287
|
+
let depth = 0;
|
|
288
|
+
for (const ch of String(command)) {
|
|
289
|
+
if (quote !== null) {
|
|
290
|
+
cur += ch;
|
|
291
|
+
if (ch === quote) quote = null;
|
|
292
|
+
continue;
|
|
293
|
+
}
|
|
294
|
+
if (ch === '"' || ch === "'") {
|
|
295
|
+
quote = ch;
|
|
296
|
+
cur += ch;
|
|
297
|
+
continue;
|
|
298
|
+
}
|
|
299
|
+
if (ch === "{") depth += 1;
|
|
300
|
+
if (ch === "}") depth = Math.max(0, depth - 1);
|
|
301
|
+
if (depth === 0 && (ch === ";" || ch === "|" || ch === "&" || ch === "\n" || ch === "\r")) {
|
|
302
|
+
parts.push(cur);
|
|
303
|
+
cur = "";
|
|
304
|
+
continue;
|
|
305
|
+
}
|
|
306
|
+
cur += ch;
|
|
307
|
+
}
|
|
308
|
+
parts.push(cur);
|
|
309
|
+
return parts.map((s) => s.trim()).filter((s) => s !== "");
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
/** Whitespace tokenizer that keeps quoted runs (quotes attached) as one token. */
|
|
313
|
+
function reviewCodeTokens(segment) {
|
|
314
|
+
const tokens = [];
|
|
315
|
+
let cur = "";
|
|
316
|
+
let quote = null;
|
|
317
|
+
let has = false;
|
|
318
|
+
for (const ch of String(segment)) {
|
|
319
|
+
if (quote !== null) {
|
|
320
|
+
cur += ch;
|
|
321
|
+
if (ch === quote) quote = null;
|
|
322
|
+
continue;
|
|
323
|
+
}
|
|
324
|
+
if (ch === '"' || ch === "'") {
|
|
325
|
+
quote = ch;
|
|
326
|
+
has = true;
|
|
327
|
+
cur += ch;
|
|
328
|
+
continue;
|
|
329
|
+
}
|
|
330
|
+
if (ch === " " || ch === "\t") {
|
|
331
|
+
if (has) tokens.push(cur);
|
|
332
|
+
cur = "";
|
|
333
|
+
has = false;
|
|
334
|
+
continue;
|
|
335
|
+
}
|
|
336
|
+
cur += ch;
|
|
337
|
+
has = true;
|
|
338
|
+
}
|
|
339
|
+
if (has) tokens.push(cur);
|
|
340
|
+
return tokens;
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
/**
|
|
344
|
+
* Extract script references from ONE command segment:
|
|
345
|
+
* - `{kind:"file", path}` — a script file the command would execute;
|
|
346
|
+
* - `{kind:"inline"}` — an inline-code flag (`-e`/`-c`/`-Command`…);
|
|
347
|
+
* that code is already visible verbatim in the tool arguments;
|
|
348
|
+
* - `{kind:"encoded", data}` — a `-EncodedCommand`/`-enc` payload;
|
|
349
|
+
* - `{kind:"note", text}` — anything unrecognizable worth surfacing.
|
|
350
|
+
* Depth-limited recursion expands code nested inside inline flags
|
|
351
|
+
* (`pwsh -Command "node x.js"`, `bash -c "python x.py"`).
|
|
352
|
+
*/
|
|
353
|
+
function reviewCodeSegmentRefs(segment, depth) {
|
|
354
|
+
const tokens = reviewCodeTokens(segment);
|
|
355
|
+
let i = 0;
|
|
356
|
+
while (
|
|
357
|
+
i < tokens.length &&
|
|
358
|
+
(tokens[i] === "&" || tokens[i] === "." || tokens[i] === "call" ||
|
|
359
|
+
tokens[i] === "source" || tokens[i] === "sudo" || tokens[i] === "exec")
|
|
360
|
+
) {
|
|
361
|
+
i += 1;
|
|
362
|
+
}
|
|
363
|
+
if (i >= tokens.length) return [];
|
|
364
|
+
const head = tokens[i].replace(/^.*[\\/]/, "").toLowerCase();
|
|
365
|
+
const refs = [];
|
|
366
|
+
if (!REVIEW_CODE_INTERPRETERS.has(head)) {
|
|
367
|
+
// Not an interpreter head: only a leading token that is itself a script
|
|
368
|
+
// file counts (call operators stripped above). No mid-segment scanning,
|
|
369
|
+
// so `git diff -- foo.py` never pulls foo.py into the state. Multi-word
|
|
370
|
+
// leftovers (a quoted chunk that reached this branch) are never taken
|
|
371
|
+
// as a path.
|
|
372
|
+
const lead = tokens[i];
|
|
373
|
+
if (reviewCodeIsScriptFile(lead) && !/\s/.test(reviewCodeCleanToken(lead))) {
|
|
374
|
+
refs.push({ kind: "file", path: reviewCodeCleanToken(lead) });
|
|
375
|
+
}
|
|
376
|
+
return refs;
|
|
377
|
+
}
|
|
378
|
+
let scriptFile = null;
|
|
379
|
+
let inline = false;
|
|
380
|
+
let module = false;
|
|
381
|
+
let j = i + 1;
|
|
382
|
+
while (j < tokens.length) {
|
|
383
|
+
const t = tokens[j];
|
|
384
|
+
const low = t.toLowerCase();
|
|
385
|
+
if (low === "-file") {
|
|
386
|
+
if (j + 1 < tokens.length && scriptFile === null) scriptFile = reviewCodeCleanToken(tokens[j + 1]);
|
|
387
|
+
j += 2;
|
|
388
|
+
continue;
|
|
389
|
+
}
|
|
390
|
+
if (low === "-m") {
|
|
391
|
+
// `python -m pkg` runs an installed module — packaged scope, not an
|
|
392
|
+
// agent-authored script (same exclusion as CLIs).
|
|
393
|
+
module = true;
|
|
394
|
+
j += 1;
|
|
395
|
+
continue;
|
|
396
|
+
}
|
|
397
|
+
if (low.startsWith("-enc")) {
|
|
398
|
+
if (j + 1 < tokens.length) refs.push({ kind: "encoded", data: tokens[j + 1] });
|
|
399
|
+
j += 2;
|
|
400
|
+
continue;
|
|
401
|
+
}
|
|
402
|
+
if (low === "-e" || low === "-c" || low === "-p" || low === "--print" || low.startsWith("-com") || low.startsWith("--eval")) {
|
|
403
|
+
inline = true;
|
|
404
|
+
if (depth < 2) {
|
|
405
|
+
// Recurse into the remaining tokens as one command string; strip the
|
|
406
|
+
// wrapping quotes first so the inner tokenization sees clean words
|
|
407
|
+
// (`pwsh -Command "node build.js"` → inner `node` + `build.js`).
|
|
408
|
+
const rest = tokens.slice(j + 1).join(" ").replace(/^["']+/, "").replace(/["']+$/, "");
|
|
409
|
+
if (rest.trim() !== "") {
|
|
410
|
+
for (const seg of reviewCodeSplitSegments(rest)) {
|
|
411
|
+
for (const inner of reviewCodeSegmentRefs(seg, depth + 1)) refs.push(inner);
|
|
412
|
+
}
|
|
413
|
+
}
|
|
414
|
+
}
|
|
415
|
+
break; // everything after an inline flag is code, not more flags/files
|
|
416
|
+
}
|
|
417
|
+
if (t.startsWith("-")) {
|
|
418
|
+
j += 1;
|
|
419
|
+
continue;
|
|
420
|
+
}
|
|
421
|
+
if (scriptFile === null && reviewCodeIsScriptFile(t)) scriptFile = reviewCodeCleanToken(t);
|
|
422
|
+
j += 1;
|
|
423
|
+
}
|
|
424
|
+
if (scriptFile !== null) refs.unshift({ kind: "file", path: scriptFile });
|
|
425
|
+
if (inline) refs.push({ kind: "inline" });
|
|
426
|
+
if (scriptFile === null && !inline && !module && refs.length === 0) {
|
|
427
|
+
refs.push({ kind: "note", text: head + " invoked, but no script file or inline-code flag was recognizable" });
|
|
428
|
+
}
|
|
429
|
+
return refs;
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
/** Extract all refs across every top-level segment of one command string. */
|
|
433
|
+
export function reviewCodeRefsOf(command) {
|
|
434
|
+
const refs = [];
|
|
435
|
+
for (const seg of reviewCodeSplitSegments(command)) {
|
|
436
|
+
for (const ref of reviewCodeSegmentRefs(seg, 0)) refs.push(ref);
|
|
437
|
+
}
|
|
438
|
+
return refs;
|
|
439
|
+
}
|
|
440
|
+
|
|
200
441
|
/** Constrained to the
|
|
201
442
|
* JSON-Schema subset `assertObjectJsonSchema` enforces for subagent outputs
|
|
202
443
|
* (type/properties/required/additionalProperties/enum only).
|
|
@@ -378,10 +619,13 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
378
619
|
*/
|
|
379
620
|
this._reviewDefault = false;
|
|
380
621
|
/**
|
|
381
|
-
* TypeSafe Jev direct backend settings
|
|
382
|
-
* the
|
|
383
|
-
*
|
|
384
|
-
*
|
|
622
|
+
* TypeSafe Jev direct backend settings — v1.10.0: the INDEPENDENT
|
|
623
|
+
* configuration of the 自动审查 (per-call review) mode, no longer part of
|
|
624
|
+
* the approval judge's provider selection. Configuring it (a resolvable
|
|
625
|
+
* key) is what opens the review gate; see `_jevGateOk`. The API key lives
|
|
626
|
+
* in plaintext on this machine only (config.json, same trust domain as
|
|
627
|
+
* the rest of the settings); an empty key falls back to the
|
|
628
|
+
* TYPESAFE_API_KEY env var.
|
|
385
629
|
*/
|
|
386
630
|
this._jev = {
|
|
387
631
|
apiKey: "",
|
|
@@ -645,7 +889,7 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
645
889
|
return "自动审批 is already ON for this session — switch modes through the /permission menu";
|
|
646
890
|
}
|
|
647
891
|
if (!this._jevGateOk()) {
|
|
648
|
-
return "自动审查 requires the Jev judge:
|
|
892
|
+
return "自动审查 requires the Jev judge: configure a Jev API key in Settings → 自动审批 → 「自动审查」first";
|
|
649
893
|
}
|
|
650
894
|
this._enableCore(session, agent, "review");
|
|
651
895
|
if (this._presetRegistered(REVIEW_PRESET_NAME)) {
|
|
@@ -676,13 +920,14 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
676
920
|
}
|
|
677
921
|
|
|
678
922
|
/**
|
|
679
|
-
* v1.8.0 gate for the per-call review mode
|
|
680
|
-
*
|
|
681
|
-
*
|
|
682
|
-
*
|
|
923
|
+
* v1.8.0, re-based in v1.10.0: the gate for the per-call review mode. The
|
|
924
|
+
* Jev backend is the review mode's own, independent configuration — the
|
|
925
|
+
* gate is simply "a Jev API key resolves" (config.json or
|
|
926
|
+
* TYPESAFE_API_KEY). It no longer requires switching the APPROVAL judge to
|
|
927
|
+
* Jev (that option is gone as of v1.10.0), and it never reads `_model`.
|
|
683
928
|
*/
|
|
684
929
|
_jevGateOk() {
|
|
685
|
-
return this.
|
|
930
|
+
return this._jevEffective().key !== "";
|
|
686
931
|
}
|
|
687
932
|
|
|
688
933
|
/**
|
|
@@ -707,7 +952,7 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
707
952
|
durationMs: 0,
|
|
708
953
|
childSessionId: "",
|
|
709
954
|
rationale:
|
|
710
|
-
"自动审查 requires the Jev judge (Settings →
|
|
955
|
+
"自动审查 requires the Jev judge (Settings → 自动审批 → 「自动审查」卡片: configure a Jev API key); falling back to the 自动审批 preset (fail closed)",
|
|
711
956
|
mode: "review",
|
|
712
957
|
});
|
|
713
958
|
const presets = this.ctx.get("permissionPresets");
|
|
@@ -1057,7 +1302,13 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
1057
1302
|
typeof cfg.model.provider === "string" &&
|
|
1058
1303
|
typeof cfg.model.model === "string"
|
|
1059
1304
|
) {
|
|
1060
|
-
|
|
1305
|
+
// v1.10.0 migration: a persisted `typesafe` judge route (v1.6.0–
|
|
1306
|
+
// v1.9.x) is no longer a valid judge — fall back to the harness
|
|
1307
|
+
// default route instead of keeping a dead provider on the wire.
|
|
1308
|
+
this._model =
|
|
1309
|
+
cfg.model.provider === JEV_LEGACY_PROVIDER
|
|
1310
|
+
? { provider: "", model: "" }
|
|
1311
|
+
: { provider: cfg.model.provider, model: cfg.model.model };
|
|
1061
1312
|
}
|
|
1062
1313
|
if (cfg.judgeMode === JUDGE_MODE_LLM || cfg.judgeMode === JUDGE_MODE_SUBAGENT) {
|
|
1063
1314
|
this._judgeMode = cfg.judgeMode;
|
|
@@ -1258,10 +1509,6 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
1258
1509
|
* records display — "p/m" = selected, "default(p/m)" = harness default.
|
|
1259
1510
|
*/
|
|
1260
1511
|
_judgeRoute() {
|
|
1261
|
-
if (this._model.provider === JEV_PROVIDER) {
|
|
1262
|
-
const model = this._jevEffective().model;
|
|
1263
|
-
return { provider: JEV_PROVIDER, model: model, label: "jev(" + model + ")" };
|
|
1264
|
-
}
|
|
1265
1512
|
if (this._model.provider !== "" && this._model.model !== "") {
|
|
1266
1513
|
return {
|
|
1267
1514
|
provider: this._model.provider,
|
|
@@ -1327,13 +1574,11 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
1327
1574
|
return "allowed-once";
|
|
1328
1575
|
}
|
|
1329
1576
|
|
|
1330
|
-
// 3. The judge
|
|
1331
|
-
//
|
|
1332
|
-
// direct ctx.llm.stream() call (no subagent either —
|
|
1333
|
-
// judgeMode === "subagent" spawns the judge child
|
|
1334
|
-
|
|
1335
|
-
return this._judgeWithJev(session, req, argsRaw, base, trustKey);
|
|
1336
|
-
}
|
|
1577
|
+
// 3. The judge — always an LLM (v1.10.0: Jev is no longer selectable
|
|
1578
|
+
// here; it judges the per-call review mode only). The DEFAULT "llm"
|
|
1579
|
+
// mode is one direct ctx.llm.stream() call (no subagent either —
|
|
1580
|
+
// v1.8.0); only judgeMode === "subagent" spawns the judge child
|
|
1581
|
+
// through `spawn`.
|
|
1337
1582
|
if (this._judgeMode !== JUDGE_MODE_SUBAGENT) {
|
|
1338
1583
|
return this._judgeWithLlmStream(session, req, argsRaw, base, trustKey);
|
|
1339
1584
|
}
|
|
@@ -1472,7 +1717,7 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
1472
1717
|
* exactly — same persona, same `_judgePrompt`, same VERDICT_SCHEMA verdict
|
|
1473
1718
|
* contract — only the invocation differs: no subagent session is created
|
|
1474
1719
|
* (zero judge-side context pollution; `childSessionId` stays empty).
|
|
1475
|
-
*
|
|
1720
|
+
* Fail-closed contract:
|
|
1476
1721
|
* - no concrete route / llm fault / non-'stop' finish / malformed verdict
|
|
1477
1722
|
* / timeout → `unavailable`
|
|
1478
1723
|
* - request cancelled mid-flight → `cancelled`
|
|
@@ -1738,13 +1983,12 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
1738
1983
|
return { decision: value.decision, riskLevel: value.riskLevel, rationale: value.rationale };
|
|
1739
1984
|
}
|
|
1740
1985
|
|
|
1741
|
-
// ---- the TypeSafe Jev direct backend
|
|
1986
|
+
// ---- the TypeSafe Jev direct backend (自动审查 only, since v1.10.0) ---------
|
|
1742
1987
|
|
|
1743
1988
|
/**
|
|
1744
1989
|
* Effective Jev settings with env fallback and clamping applied. The key
|
|
1745
1990
|
* may come from config.json or the TYPESAFE_API_KEY environment variable;
|
|
1746
|
-
* an absent key keeps the
|
|
1747
|
-
* `unavailable` (fail closed) until one is configured.
|
|
1991
|
+
* an absent key keeps the review gate closed, so 自动审查 cannot be armed.
|
|
1748
1992
|
*/
|
|
1749
1993
|
_jevEffective() {
|
|
1750
1994
|
const key = String(this._jev.apiKey || process.env.TYPESAFE_API_KEY || "").trim();
|
|
@@ -1792,110 +2036,12 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
1792
2036
|
}
|
|
1793
2037
|
|
|
1794
2038
|
/**
|
|
1795
|
-
*
|
|
1796
|
-
*
|
|
1797
|
-
*
|
|
1798
|
-
*
|
|
1799
|
-
*
|
|
1800
|
-
* - request cancelled mid-flight → `cancelled`
|
|
1801
|
-
* - overall timeout (the same `this._timeoutMs` budget) → `unavailable`
|
|
1802
|
-
* - confidence below the configured gate → `unavailable` (the model is
|
|
1803
|
-
* not sure enough to decide: never a grant, and not a recorded
|
|
1804
|
-
* rejection either — the v1.4.0 误杀治理 applies symmetrically)
|
|
1805
|
-
* Jev does not generate text, so the audit rationale is synthesized from
|
|
1806
|
-
* the returned distributions; the served model version (`body.model`,
|
|
1807
|
-
* which resolves aliases like jev-latest) is what the audit displays.
|
|
2039
|
+
* The single POST to the System One endpoint; resolves the parsed body.
|
|
2040
|
+
* `questions` defaults to the per-call review set (the only caller since
|
|
2041
|
+
* v1.10.0). `state` leaves this machine by design — that is the review
|
|
2042
|
+
* judge's whole point (see the workspace-confined code expansion in
|
|
2043
|
+
* `_reviewEffectiveCode`).
|
|
1808
2044
|
*/
|
|
1809
|
-
async _judgeWithJev(session, req, argsRaw, base, trustKey) {
|
|
1810
|
-
const cfg = this._jevEffective();
|
|
1811
|
-
if (cfg.key === "") {
|
|
1812
|
-
this._record(session, {
|
|
1813
|
-
...base,
|
|
1814
|
-
outcome: "unavailable",
|
|
1815
|
-
riskLevel: "-",
|
|
1816
|
-
model: "jev(" + cfg.model + ")",
|
|
1817
|
-
rationale: "Jev backend selected but no API key configured (Settings → 自动审批, or the TYPESAFE_API_KEY environment variable)",
|
|
1818
|
-
});
|
|
1819
|
-
return "unavailable";
|
|
1820
|
-
}
|
|
1821
|
-
if (typeof fetch !== "function") {
|
|
1822
|
-
this._record(session, { ...base, outcome: "unavailable", riskLevel: "-", model: "jev(" + cfg.model + ")", rationale: "fetch is unavailable in this runtime" });
|
|
1823
|
-
return "unavailable";
|
|
1824
|
-
}
|
|
1825
|
-
|
|
1826
|
-
const startedAt = Date.now();
|
|
1827
|
-
const controller = new AbortController();
|
|
1828
|
-
const signal = req.signal;
|
|
1829
|
-
const onAbort = () => controller.abort();
|
|
1830
|
-
if (signal && typeof signal.addEventListener === "function") {
|
|
1831
|
-
signal.addEventListener("abort", onAbort, { once: true });
|
|
1832
|
-
}
|
|
1833
|
-
|
|
1834
|
-
let winner;
|
|
1835
|
-
try {
|
|
1836
|
-
const state = this._jevStateOf(session, req, argsRaw);
|
|
1837
|
-
winner = await Promise.race([
|
|
1838
|
-
this._jevRequest(cfg, state, controller.signal)
|
|
1839
|
-
.then((body) => ({ kind: "result", body: body }))
|
|
1840
|
-
.catch((error) => ({
|
|
1841
|
-
kind: "fault",
|
|
1842
|
-
error: error,
|
|
1843
|
-
aborted: error && error.name === "AbortError",
|
|
1844
|
-
})),
|
|
1845
|
-
(signal
|
|
1846
|
-
? new Promise((resolve) => {
|
|
1847
|
-
if (signal.aborted) {
|
|
1848
|
-
resolve(true);
|
|
1849
|
-
return;
|
|
1850
|
-
}
|
|
1851
|
-
signal.addEventListener("abort", () => resolve(true), { once: true });
|
|
1852
|
-
})
|
|
1853
|
-
: Promise.resolve(false)
|
|
1854
|
-
).then((v) => ({ kind: "aborted", aborted: v })),
|
|
1855
|
-
this.ctx.timeout(this._timeoutMs).then(() => ({ kind: "timeout" })),
|
|
1856
|
-
]);
|
|
1857
|
-
} finally {
|
|
1858
|
-
if (signal && typeof signal.removeEventListener === "function") {
|
|
1859
|
-
signal.removeEventListener("abort", onAbort);
|
|
1860
|
-
}
|
|
1861
|
-
// Whether we lost the race to timeout/cancel or the request already
|
|
1862
|
-
// settled, closing the transport is always safe.
|
|
1863
|
-
try {
|
|
1864
|
-
controller.abort();
|
|
1865
|
-
} catch (e) {
|
|
1866
|
-
/* controller abort never blocks the outcome */
|
|
1867
|
-
}
|
|
1868
|
-
}
|
|
1869
|
-
const durationMs = Date.now() - startedAt;
|
|
1870
|
-
|
|
1871
|
-
if (winner.kind === "result") {
|
|
1872
|
-
return this._jevVerdict(session, winner.body, cfg, base, trustKey, durationMs);
|
|
1873
|
-
}
|
|
1874
|
-
if (winner.kind === "aborted") {
|
|
1875
|
-
this._record(session, { ...base, outcome: "cancelled", riskLevel: "-", model: "jev(" + cfg.model + ")", rationale: "request cancelled while Jev was judging" });
|
|
1876
|
-
return "cancelled";
|
|
1877
|
-
}
|
|
1878
|
-
if (winner.kind === "timeout") {
|
|
1879
|
-
this._record(session, {
|
|
1880
|
-
...base,
|
|
1881
|
-
outcome: "unavailable",
|
|
1882
|
-
riskLevel: "-",
|
|
1883
|
-
model: "jev(" + cfg.model + ")",
|
|
1884
|
-
rationale: "Jev request timed out after " + String(this._timeoutMs) + "ms (fail closed)",
|
|
1885
|
-
});
|
|
1886
|
-
return "unavailable";
|
|
1887
|
-
}
|
|
1888
|
-
if (winner.aborted) {
|
|
1889
|
-
this._record(session, { ...base, outcome: "cancelled", riskLevel: "-", model: "jev(" + cfg.model + ")", rationale: "request cancelled while Jev was judging" });
|
|
1890
|
-
return "cancelled";
|
|
1891
|
-
}
|
|
1892
|
-
this._record(session, { ...base, outcome: "unavailable", riskLevel: "-", model: "jev(" + cfg.model + ")", rationale: "Jev request failed: " + errText(winner.error) });
|
|
1893
|
-
return "unavailable";
|
|
1894
|
-
}
|
|
1895
|
-
|
|
1896
|
-
/** The single POST to the System One endpoint; resolves the parsed body.
|
|
1897
|
-
* `questions` defaults to the escalation set; the review path passes
|
|
1898
|
-
* `JEV_REVIEW_QUESTIONS`. */
|
|
1899
2045
|
async _jevRequest(cfg, state, abortSignal, questions) {
|
|
1900
2046
|
const response = await fetch(cfg.endpoint, {
|
|
1901
2047
|
method: "POST",
|
|
@@ -1906,7 +2052,7 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
1906
2052
|
body: JSON.stringify({
|
|
1907
2053
|
state: state,
|
|
1908
2054
|
model: cfg.model,
|
|
1909
|
-
questions: questions === undefined ?
|
|
2055
|
+
questions: questions === undefined ? JEV_REVIEW_QUESTIONS : questions,
|
|
1910
2056
|
}),
|
|
1911
2057
|
signal: abortSignal,
|
|
1912
2058
|
});
|
|
@@ -1925,9 +2071,9 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
1925
2071
|
}
|
|
1926
2072
|
|
|
1927
2073
|
/**
|
|
1928
|
-
* Parse + gate one Jev response into a normalized verdict, shared by
|
|
1929
|
-
* escalation path
|
|
1930
|
-
*
|
|
2074
|
+
* Parse + gate one Jev response into a normalized verdict, shared by every
|
|
2075
|
+
* review call (the escalation path stopped using Jev in v1.10.0) so the
|
|
2076
|
+
* review judge always holds to the same standard:
|
|
1931
2077
|
* - `{ kind: "malformed", served }` — any missing/out-of-shape answer
|
|
1932
2078
|
* - `{ kind: "low-confidence", served, choice, riskLevel, confidence, gate }`
|
|
1933
2079
|
* - `{ kind: "verdict", served, choice, riskLevel, rationale }`
|
|
@@ -2000,56 +2146,6 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
2000
2146
|
return { kind: "verdict", served: served, choice: choice, riskLevel: riskChoice, rationale: rationale };
|
|
2001
2147
|
}
|
|
2002
2148
|
|
|
2003
|
-
/**
|
|
2004
|
-
* Map a Jev response to the same outcomes the subagent path produces.
|
|
2005
|
-
* Returns the waterfall outcome string; records the audit line itself.
|
|
2006
|
-
*/
|
|
2007
|
-
_jevVerdict(session, body, cfg, base, trustKey, durationMs) {
|
|
2008
|
-
const parsed = this._jevParse(body, cfg);
|
|
2009
|
-
const label = "jev(" + parsed.served + ")";
|
|
2010
|
-
base.durationMs = durationMs;
|
|
2011
|
-
|
|
2012
|
-
if (parsed.kind === "malformed") {
|
|
2013
|
-
this._record(session, {
|
|
2014
|
-
...base,
|
|
2015
|
-
outcome: "unavailable",
|
|
2016
|
-
riskLevel: "-",
|
|
2017
|
-
model: label,
|
|
2018
|
-
rationale: "Jev returned no valid verdict shape (decision/riskLevel/concreteRisk incomplete)",
|
|
2019
|
-
});
|
|
2020
|
-
return "unavailable";
|
|
2021
|
-
}
|
|
2022
|
-
if (parsed.kind === "low-confidence") {
|
|
2023
|
-
this._record(session, {
|
|
2024
|
-
...base,
|
|
2025
|
-
outcome: "unavailable",
|
|
2026
|
-
riskLevel: parsed.riskLevel,
|
|
2027
|
-
model: label,
|
|
2028
|
-
rationale:
|
|
2029
|
-
"Jev confidence " + parsed.confidence.toFixed(2) + " is below the gate " + parsed.gate.toFixed(2) + " (decision draft: " + parsed.choice + ") — fail closed",
|
|
2030
|
-
});
|
|
2031
|
-
return "unavailable";
|
|
2032
|
-
}
|
|
2033
|
-
|
|
2034
|
-
const approved = parsed.choice === "approve";
|
|
2035
|
-
this._record(session, {
|
|
2036
|
-
...base,
|
|
2037
|
-
outcome: approved ? "allowed-once" : "rejected",
|
|
2038
|
-
riskLevel: parsed.riskLevel,
|
|
2039
|
-
model: label,
|
|
2040
|
-
rationale: trunc(parsed.rationale, 600),
|
|
2041
|
-
});
|
|
2042
|
-
if (approved && trustKey !== undefined) {
|
|
2043
|
-
let set = this._trusted.get(session.id);
|
|
2044
|
-
if (set === undefined) {
|
|
2045
|
-
set = new Set();
|
|
2046
|
-
this._trusted.set(session.id, set);
|
|
2047
|
-
}
|
|
2048
|
-
set.add(trustKey);
|
|
2049
|
-
}
|
|
2050
|
-
return approved ? "allowed-once" : "rejected";
|
|
2051
|
-
}
|
|
2052
|
-
|
|
2053
2149
|
// ---- v1.8.0 per-call review mode (agent-review) ----------------------------
|
|
2054
2150
|
|
|
2055
2151
|
/**
|
|
@@ -2212,12 +2308,142 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
2212
2308
|
}
|
|
2213
2309
|
|
|
2214
2310
|
/**
|
|
2215
|
-
*
|
|
2216
|
-
*
|
|
2217
|
-
*
|
|
2218
|
-
*
|
|
2219
|
-
*
|
|
2220
|
-
*
|
|
2311
|
+
* v1.9.0: build the `effectiveCode` review-state field — the actual
|
|
2312
|
+
* contents of the interpreter scripts one pending call would execute —
|
|
2313
|
+
* plus a one-line visibility note for the audit rationale. The judge
|
|
2314
|
+
* state stays workspace-scoped: file contents are only ever included for
|
|
2315
|
+
* paths that resolve (symlinks followed and re-checked) inside the
|
|
2316
|
+
* session workspace, so the expansion never ships beyond-boundary file
|
|
2317
|
+
* contents anywhere. Capped, and fully best-effort: every fault degrades
|
|
2318
|
+
* to a note inside the field and never blocks judging.
|
|
2319
|
+
*/
|
|
2320
|
+
async _reviewEffectiveCode(session, argsRaw) {
|
|
2321
|
+
let args;
|
|
2322
|
+
try {
|
|
2323
|
+
args = typeof argsRaw === "string" ? JSON.parse(argsRaw) : undefined;
|
|
2324
|
+
} catch (e) {
|
|
2325
|
+
args = undefined;
|
|
2326
|
+
}
|
|
2327
|
+
let cwd = "";
|
|
2328
|
+
try {
|
|
2329
|
+
if (session.header && typeof session.header.cwd === "string") cwd = session.header.cwd;
|
|
2330
|
+
} catch (e) {
|
|
2331
|
+
/* header access is best-effort */
|
|
2332
|
+
}
|
|
2333
|
+
let baseDir = cwd;
|
|
2334
|
+
if (
|
|
2335
|
+
args && typeof args === "object" && !Array.isArray(args) &&
|
|
2336
|
+
typeof args.workdir === "string" && args.workdir.trim() !== ""
|
|
2337
|
+
) {
|
|
2338
|
+
baseDir = args.workdir;
|
|
2339
|
+
}
|
|
2340
|
+
const commands = [];
|
|
2341
|
+
const pushCommand = (v) => {
|
|
2342
|
+
if (typeof v === "string" && v.trim() !== "" && commands.length < REVIEW_CODE_MAX_COMMANDS) commands.push(v);
|
|
2343
|
+
};
|
|
2344
|
+
if (typeof args === "string") pushCommand(args);
|
|
2345
|
+
else if (args && typeof args === "object") {
|
|
2346
|
+
for (const v of Object.values(args)) pushCommand(v);
|
|
2347
|
+
}
|
|
2348
|
+
|
|
2349
|
+
const files = [];
|
|
2350
|
+
let inlineCount = 0;
|
|
2351
|
+
let encodedCount = 0;
|
|
2352
|
+
const notes = [];
|
|
2353
|
+
for (const command of commands) {
|
|
2354
|
+
for (const ref of reviewCodeRefsOf(command)) {
|
|
2355
|
+
if (ref.kind === "file") {
|
|
2356
|
+
const dup = files.some((f) => f.toLowerCase() === ref.path.toLowerCase());
|
|
2357
|
+
if (!dup && files.length < REVIEW_CODE_MAX_FILES) files.push(ref.path);
|
|
2358
|
+
} else if (ref.kind === "inline") inlineCount += 1;
|
|
2359
|
+
else if (ref.kind === "encoded") encodedCount += 1;
|
|
2360
|
+
else if (ref.kind === "note") notes.push(ref.text);
|
|
2361
|
+
}
|
|
2362
|
+
}
|
|
2363
|
+
|
|
2364
|
+
const lines = [];
|
|
2365
|
+
let filesRead = 0;
|
|
2366
|
+
let filesSkipped = 0;
|
|
2367
|
+
let filesMissing = 0;
|
|
2368
|
+
let charsIncluded = 0;
|
|
2369
|
+
const baseAbs = baseDir.trim() !== "" ? resolve(baseDir) : "";
|
|
2370
|
+
const insideOf = (p) =>
|
|
2371
|
+
baseAbs !== "" &&
|
|
2372
|
+
(p.toLowerCase() === baseAbs.toLowerCase() || p.toLowerCase().startsWith(baseAbs.toLowerCase() + sep));
|
|
2373
|
+
for (const raw of files) {
|
|
2374
|
+
let resolved;
|
|
2375
|
+
try {
|
|
2376
|
+
resolved = resolve(baseAbs !== "" ? baseAbs : ".", raw);
|
|
2377
|
+
} catch (e) {
|
|
2378
|
+
filesMissing += 1;
|
|
2379
|
+
continue;
|
|
2380
|
+
}
|
|
2381
|
+
if (!insideOf(resolved)) {
|
|
2382
|
+
filesSkipped += 1;
|
|
2383
|
+
lines.push("[script] " + raw + " — path resolves beyond the workspace boundary; skipped, contents never included");
|
|
2384
|
+
continue;
|
|
2385
|
+
}
|
|
2386
|
+
let real = resolved;
|
|
2387
|
+
try {
|
|
2388
|
+
real = await realpath(resolved);
|
|
2389
|
+
if (!insideOf(real)) {
|
|
2390
|
+
filesSkipped += 1;
|
|
2391
|
+
lines.push("[script] " + raw + " — symlink target lies beyond the workspace boundary; skipped, contents never included");
|
|
2392
|
+
continue;
|
|
2393
|
+
}
|
|
2394
|
+
} catch (e) {
|
|
2395
|
+
filesMissing += 1;
|
|
2396
|
+
lines.push("[script] " + raw + " — not found on disk at review time");
|
|
2397
|
+
continue;
|
|
2398
|
+
}
|
|
2399
|
+
try {
|
|
2400
|
+
const st = await stat(real);
|
|
2401
|
+
if (!st.isFile() || st.size > REVIEW_CODE_MAX_FILE_BYTES) {
|
|
2402
|
+
filesSkipped += 1;
|
|
2403
|
+
lines.push(
|
|
2404
|
+
"[script] " + raw + " — " +
|
|
2405
|
+
(st.isFile() ? "too large to review (" + st.size + " bytes)" : "not a regular file"),
|
|
2406
|
+
);
|
|
2407
|
+
continue;
|
|
2408
|
+
}
|
|
2409
|
+
let text = await readFile(real, "utf8");
|
|
2410
|
+
const total = text.length;
|
|
2411
|
+
if (total > REVIEW_CODE_FILE_CHARS) {
|
|
2412
|
+
text = text.slice(0, REVIEW_CODE_FILE_CHARS) + "\n…[truncated, " + (total - REVIEW_CODE_FILE_CHARS) + " more chars]";
|
|
2413
|
+
}
|
|
2414
|
+
filesRead += 1;
|
|
2415
|
+
charsIncluded += Math.min(total, REVIEW_CODE_FILE_CHARS);
|
|
2416
|
+
lines.push("[script " + filesRead + "] " + raw + " → " + real + " (" + st.size + " bytes)\n" + text);
|
|
2417
|
+
} catch (e) {
|
|
2418
|
+
filesMissing += 1;
|
|
2419
|
+
lines.push("[script] " + raw + " — not readable at review time: " + errText(e));
|
|
2420
|
+
}
|
|
2421
|
+
}
|
|
2422
|
+
if (encodedCount > 0) {
|
|
2423
|
+
lines.push("[-EncodedCommand] " + encodedCount + " base64-encoded invocation(s) present — the payload is not verifiable as readable code");
|
|
2424
|
+
}
|
|
2425
|
+
if (inlineCount > 0) {
|
|
2426
|
+
lines.push("[inline] " + inlineCount + " inline-code invocation(s) — that code is embedded verbatim in the command text / toolArguments");
|
|
2427
|
+
}
|
|
2428
|
+
if (notes.length > 0) lines.push("[note] " + notes.join("; "));
|
|
2429
|
+
if (lines.length === 0) {
|
|
2430
|
+
return { text: "(no interpreter script file is referenced by this call)", note: "no script files referenced" };
|
|
2431
|
+
}
|
|
2432
|
+
const note =
|
|
2433
|
+
"code: " + filesRead + " script file(s)" +
|
|
2434
|
+
(charsIncluded > 0 ? ", ~" + Math.round(charsIncluded / 1024) + " KB included" : "") +
|
|
2435
|
+
(filesSkipped > 0 ? ", " + filesSkipped + " skipped (beyond workspace / oversized)" : "") +
|
|
2436
|
+
(filesMissing > 0 ? ", " + filesMissing + " missing/unreadable" : "");
|
|
2437
|
+
return { text: trunc(lines.join("\n\n"), REVIEW_CODE_TOTAL_CHARS), note: note };
|
|
2438
|
+
}
|
|
2439
|
+
|
|
2440
|
+
/**
|
|
2441
|
+
* Judge one pending call through the Jev HTTP API — the review mode's only
|
|
2442
|
+
* judge, and the only remaining Jev caller since v1.10.0. Fail-closed
|
|
2443
|
+
* through `_jevParse` (same validation, same confidence gate) but every
|
|
2444
|
+
* outcome is FINAL: rejections, low confidence, timeouts and faults all
|
|
2445
|
+
* deny the call (no human fallback). Returns `undefined` to allow,
|
|
2446
|
+
* otherwise a pre-execute decision.
|
|
2221
2447
|
*/
|
|
2222
2448
|
async _reviewWithJev(session, exec, argsRaw, base, trustKey) {
|
|
2223
2449
|
const cfg = this._jevEffective();
|
|
@@ -2242,8 +2468,14 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
2242
2468
|
}
|
|
2243
2469
|
|
|
2244
2470
|
let winner;
|
|
2471
|
+
let codeInfo = { text: "(not evaluated)", note: "" };
|
|
2245
2472
|
try {
|
|
2246
|
-
|
|
2473
|
+
try {
|
|
2474
|
+
codeInfo = await this._reviewEffectiveCode(session, argsRaw);
|
|
2475
|
+
} catch (e) {
|
|
2476
|
+
codeInfo = { text: "(expansion fault: " + errText(e) + ")", note: "code expansion fault" };
|
|
2477
|
+
}
|
|
2478
|
+
const state = { ...this._reviewStateOf(session, exec, argsRaw), effectiveCode: codeInfo.text };
|
|
2247
2479
|
winner = await Promise.race([
|
|
2248
2480
|
this._jevRequest(cfg, state, controller.signal, JEV_REVIEW_QUESTIONS)
|
|
2249
2481
|
.then((body) => ({ kind: "result", body: body }))
|
|
@@ -2310,7 +2542,7 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
2310
2542
|
outcome: approved ? "allowed-once" : "rejected",
|
|
2311
2543
|
riskLevel: parsed.riskLevel,
|
|
2312
2544
|
model: label,
|
|
2313
|
-
rationale: trunc(parsed.rationale, 600),
|
|
2545
|
+
rationale: trunc(parsed.rationale + (codeInfo.note !== "" ? " [" + codeInfo.note + "]" : ""), 600),
|
|
2314
2546
|
});
|
|
2315
2547
|
if (approved) {
|
|
2316
2548
|
if (trustKey !== undefined) {
|
|
@@ -2409,15 +2641,18 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
2409
2641
|
|
|
2410
2642
|
/**
|
|
2411
2643
|
* Set the judge model override. Empty strings clear it (the judge then runs
|
|
2412
|
-
* on the harness default route, never the requester's).
|
|
2644
|
+
* on the harness default route, never the requester's). v1.10.0: Jev is no
|
|
2645
|
+
* longer a judge provider — a legacy `typesafe` selection migrates to the
|
|
2646
|
+
* harness default route. Persisted.
|
|
2413
2647
|
*/
|
|
2414
2648
|
async setModel(request) {
|
|
2415
2649
|
const provider = request && typeof request.provider === "string" ? request.provider : "";
|
|
2416
2650
|
const model = request && typeof request.model === "string" ? request.model : "";
|
|
2417
|
-
if (provider ===
|
|
2418
|
-
//
|
|
2419
|
-
//
|
|
2420
|
-
|
|
2651
|
+
if (provider === JEV_LEGACY_PROVIDER) {
|
|
2652
|
+
// v1.6.0–v1.9.x routed the escalation judge to Jev. That option is gone
|
|
2653
|
+
// (Jev judges the per-call review mode only), so an old client still
|
|
2654
|
+
// sending it clears the override instead of resurrecting a dead route.
|
|
2655
|
+
this._model = { provider: "", model: "" };
|
|
2421
2656
|
} else {
|
|
2422
2657
|
this._model =
|
|
2423
2658
|
provider !== "" && model !== "" ? { provider, model } : { provider: "", model: "" };
|
|
@@ -2461,20 +2696,35 @@ export class AgentApprovalService extends TypertRemoteService {
|
|
|
2461
2696
|
}
|
|
2462
2697
|
|
|
2463
2698
|
/**
|
|
2464
|
-
* Set the TypeSafe Jev backend settings (only provided fields change)
|
|
2465
|
-
*
|
|
2466
|
-
*
|
|
2699
|
+
* Set the TypeSafe Jev backend settings (only provided fields change) — the
|
|
2700
|
+
* 自动审查 mode's own, independent configuration (v1.10.0: Jev is not a
|
|
2701
|
+
* judge provider, so nothing here touches the approval route). Saving a
|
|
2702
|
+
* resolvable key OPENS the review gate, and opening the gate for the first
|
|
2703
|
+
* time turns 自动审查 on for new sessions (`_reviewDefault`) — "配置了 Jev
|
|
2704
|
+
* 就启用自动审查". The same card switches it back off. `confidence` is the
|
|
2705
|
+
* gate below which Jev's answer is not trusted and the call is denied;
|
|
2706
|
+
* clamped to [0.01, 0.99]. Persisted.
|
|
2467
2707
|
*/
|
|
2468
2708
|
async setJevConfig(request) {
|
|
2469
2709
|
const r = request && typeof request === "object" ? request : {};
|
|
2710
|
+
const gateBefore = this._jevGateOk();
|
|
2470
2711
|
if (typeof r.apiKey === "string") this._jev.apiKey = r.apiKey.trim();
|
|
2471
2712
|
if (typeof r.endpoint === "string") this._jev.endpoint = r.endpoint.trim();
|
|
2472
2713
|
if (typeof r.model === "string") this._jev.model = r.model.trim();
|
|
2473
2714
|
if (typeof r.confidence === "number" && Number.isFinite(r.confidence)) {
|
|
2474
2715
|
this._jev.confidence = Math.min(0.99, Math.max(0.01, r.confidence));
|
|
2475
2716
|
}
|
|
2717
|
+
const gateAfter = this._jevGateOk();
|
|
2718
|
+
if (gateAfter && !gateBefore) this._reviewDefault = true;
|
|
2476
2719
|
this._persistConfig();
|
|
2477
|
-
return {
|
|
2720
|
+
return {
|
|
2721
|
+
ok: true,
|
|
2722
|
+
value: {
|
|
2723
|
+
jev: this._jevShape(),
|
|
2724
|
+
reviewAvailable: gateAfter,
|
|
2725
|
+
reviewDefault: this._reviewDefault,
|
|
2726
|
+
},
|
|
2727
|
+
};
|
|
2478
2728
|
}
|
|
2479
2729
|
|
|
2480
2730
|
/** Set the judge timeout (clamped to [MIN, MAX] milliseconds). Persisted. */
|