@pi-in-go/pigpen-jev 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CREDITS.md +22 -0
- package/LICENSE +22 -0
- package/README.md +237 -0
- package/extensions/jev/ask.go +166 -0
- package/extensions/jev/ask_test.go +218 -0
- package/extensions/jev/backend.go +128 -0
- package/extensions/jev/bench_test.go +64 -0
- package/extensions/jev/boundaries_test.go +159 -0
- package/extensions/jev/command.go +224 -0
- package/extensions/jev/commands_test.go +214 -0
- package/extensions/jev/config.go +450 -0
- package/extensions/jev/errors_test.go +191 -0
- package/extensions/jev/extension.go +391 -0
- package/extensions/jev/fakehost_test.go +548 -0
- package/extensions/jev/gate.go +125 -0
- package/extensions/jev/gate_test.go +610 -0
- package/extensions/jev/gatekey_test.go +24 -0
- package/extensions/jev/go.mod +9 -0
- package/extensions/jev/go.sum +2 -0
- package/extensions/jev/go.work +10 -0
- package/extensions/jev/helpers_test.go +404 -0
- package/extensions/jev/memo.go +88 -0
- package/extensions/jev/output.go +89 -0
- package/extensions/jev/output_test.go +187 -0
- package/extensions/jev/ownmodel_test.go +118 -0
- package/extensions/jev/render.go +136 -0
- package/extensions/jev/review_test.go +310 -0
- package/extensions/jev/source_test.go +57 -0
- package/extensions/jev/text.go +174 -0
- package/extensions/jev/trust_test.go +335 -0
- package/extensions/jev/types.go +227 -0
- package/libs/typesafe/CONTRACT.md +125 -0
- package/libs/typesafe/CREDITS.md +37 -0
- package/libs/typesafe/LICENSE +23 -0
- package/libs/typesafe/README.md +19 -0
- package/libs/typesafe/go.mod +3 -0
- package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
- package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
- package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
- package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
- package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
- package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
- package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
- package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
- package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
- package/libs/typesafe/libraries/ownmodel/run.go +288 -0
- package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
- package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
- package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
- package/libs/typesafe/libraries/typesafe/answers.go +268 -0
- package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
- package/libs/typesafe/libraries/typesafe/batch.go +80 -0
- package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
- package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
- package/libs/typesafe/libraries/typesafe/client.go +561 -0
- package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
- package/libs/typesafe/libraries/typesafe/doc.go +27 -0
- package/libs/typesafe/libraries/typesafe/entry.go +142 -0
- package/libs/typesafe/libraries/typesafe/env.go +11 -0
- package/libs/typesafe/libraries/typesafe/errors.go +310 -0
- package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
- package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
- package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
- package/libs/typesafe/libraries/typesafe/logging.go +160 -0
- package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
- package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
- package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
- package/libs/typesafe/libraries/typesafe/questions.go +490 -0
- package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
- package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
- package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
- package/libs/typesafe/libraries/typesafe/retry.go +350 -0
- package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
- package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
- package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
- package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
- package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
- package/libs/typesafe/libraries/typesafe/version.go +10 -0
- package/libs/typesafe/package.json +37 -0
- package/libs/typesafe/provenance.json +49 -0
- package/package.json +42 -0
- package/port/PORT.md +107 -0
- package/port/e2e/gate-and-output.py +35 -0
- package/port/e2e/jev-ask.py +36 -0
- package/port/e2e/model-switch.py +44 -0
- package/port/e2e/off-by-default.py +34 -0
- package/port/gen-scenarios.py +103 -0
- package/port/golden/cache-identical-calls.jsonl +30 -0
- package/port/golden/clear.jsonl +22 -0
- package/port/golden/commands.jsonl +43 -0
- package/port/golden/enforce-accept.jsonl +23 -0
- package/port/golden/enforce-decline.jsonl +22 -0
- package/port/golden/jev-ask.jsonl +20 -0
- package/port/golden/output-advice.jsonl +23 -0
- package/port/golden/output-leak.jsonl +24 -0
- package/port/golden/output-low-confidence.jsonl +22 -0
- package/port/golden/shadow-flagged.jsonl +23 -0
- package/port/golden/unjudged-tools.jsonl +19 -0
- package/port/golden/write-elision.jsonl +21 -0
- package/port/mutate-unit.py +63 -0
- package/port/mutations.json +578 -0
- package/port/oracle/LICENSE +21 -0
- package/port/oracle/README.md +181 -0
- package/port/oracle/SHA256SUMS +8 -0
- package/port/oracle/package.json +43 -0
- package/port/oracle/src/client.ts +409 -0
- package/port/oracle/src/config.ts +363 -0
- package/port/oracle/src/gate.ts +229 -0
- package/port/oracle/src/index.ts +649 -0
- package/port/oracle/src/output.ts +163 -0
- package/port/red-run.log +309 -0
- package/port/scenarios/cache-identical-calls.json +71 -0
- package/port/scenarios/clear.json +61 -0
- package/port/scenarios/commands.json +119 -0
- package/port/scenarios/enforce-accept.json +66 -0
- package/port/scenarios/enforce-decline.json +57 -0
- package/port/scenarios/jev-ask.json +83 -0
- package/port/scenarios/output-advice.json +61 -0
- package/port/scenarios/output-leak.json +61 -0
- package/port/scenarios/output-low-confidence.json +61 -0
- package/port/scenarios/shadow-flagged.json +61 -0
- package/port/scenarios/unjudged-tools.json +55 -0
- package/port/scenarios/write-elision.json +53 -0
- package/provenance.json +18 -0
|
@@ -0,0 +1,363 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Config for the Jev extensions.
|
|
3
|
+
*
|
|
4
|
+
* Transport settings (endpoint, model, key, timeout) live here because this is
|
|
5
|
+
* the only place that talks to the API, and the key should be resolved in
|
|
6
|
+
* exactly one place.
|
|
7
|
+
*
|
|
8
|
+
* Layering: built-in defaults <- ~/.pi/agent/pi-jev.json <- <cwd>/.pi/pi-jev.json.
|
|
9
|
+
* Project settings win. Only keys actually present in a file override, so a
|
|
10
|
+
* project file can set one field without restating the rest.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { readFileSync } from "node:fs";
|
|
14
|
+
import { homedir } from "node:os";
|
|
15
|
+
import { basename, dirname, join } from "node:path";
|
|
16
|
+
import * as host from "@earendil-works/pi-coding-agent";
|
|
17
|
+
import {
|
|
18
|
+
DEFAULT_ENDPOINT,
|
|
19
|
+
DEFAULT_MODEL,
|
|
20
|
+
DEFAULT_RETRIES,
|
|
21
|
+
DEFAULT_TIMEOUT_MS,
|
|
22
|
+
} from "./client";
|
|
23
|
+
|
|
24
|
+
export const CONFIG_FILE = "pi-jev.json";
|
|
25
|
+
export const API_KEY_ENV = "TYPESAFE_API_KEY";
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* The host module's exports as plain data.
|
|
29
|
+
*
|
|
30
|
+
* The extension needs the host's config dir name, and the host does not export
|
|
31
|
+
* that name before 0.79.7: importing it there binds `undefined`, which reached
|
|
32
|
+
* `join` and killed the agent at startup. Reading it off this view keeps the
|
|
33
|
+
* lookup optional instead of a named import that a host may not have.
|
|
34
|
+
*/
|
|
35
|
+
const HOST_EXPORTS: Record<string, unknown> = { ...host };
|
|
36
|
+
|
|
37
|
+
/** shadow reports but never blocks; enforce can block a tool call. */
|
|
38
|
+
export type GateMode = "shadow" | "enforce";
|
|
39
|
+
|
|
40
|
+
export interface GateThresholds {
|
|
41
|
+
/** Noul probability at or above which the dimension counts as triggered. */
|
|
42
|
+
destructive: number;
|
|
43
|
+
exfiltration: number;
|
|
44
|
+
beyondScope: number;
|
|
45
|
+
/** Score on the 4-level damage rubric. */
|
|
46
|
+
impact: number;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
export interface GateConfig {
|
|
50
|
+
enabled: boolean;
|
|
51
|
+
mode: GateMode;
|
|
52
|
+
/** Tool names to judge. Everything else runs unjudged. */
|
|
53
|
+
tools: string[];
|
|
54
|
+
/** Elide string fields longer than this from the state sent to Jev. */
|
|
55
|
+
argumentChars: number;
|
|
56
|
+
/** Reuse a verdict for identical tool input within this window. */
|
|
57
|
+
cacheSeconds: number;
|
|
58
|
+
blockOn: GateThresholds;
|
|
59
|
+
/** Minimum confidence on the impact answer before a block is allowed. */
|
|
60
|
+
minConfidence: number;
|
|
61
|
+
/**
|
|
62
|
+
* Block in non-interactive runs. Off by default: with no UI there is no way
|
|
63
|
+
* to approve a flagged call, so enforcement would deadlock automation on a
|
|
64
|
+
* classifier's opinion.
|
|
65
|
+
*/
|
|
66
|
+
blockWithoutUI: boolean;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* The output judge runs after a tool finishes, so its questions are about text
|
|
71
|
+
* that exists: a credential echoed into the transcript, a failure to classify.
|
|
72
|
+
* It never blocks, so there is no mode and no confirmation.
|
|
73
|
+
*/
|
|
74
|
+
export interface OutputConfig {
|
|
75
|
+
enabled: boolean;
|
|
76
|
+
/** Tool names whose output is judged. Everything else passes through. */
|
|
77
|
+
tools: string[];
|
|
78
|
+
/** Elide output longer than this from the state sent to Jev. */
|
|
79
|
+
outputChars: number;
|
|
80
|
+
/** Noul probability at or above which output counts as carrying a secret. */
|
|
81
|
+
leakThreshold: number;
|
|
82
|
+
/** Minimum confidence on the failure class before advice is attached. */
|
|
83
|
+
minConfidence: number;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export interface JevConfig {
|
|
87
|
+
endpoint: string;
|
|
88
|
+
model: string;
|
|
89
|
+
apiKey: string | undefined;
|
|
90
|
+
apiKeyFile: string | undefined;
|
|
91
|
+
timeoutMs: number;
|
|
92
|
+
retries: number;
|
|
93
|
+
/** Cap on the serialized state handed to Jev. */
|
|
94
|
+
maxStateChars: number;
|
|
95
|
+
gate: GateConfig;
|
|
96
|
+
output: OutputConfig;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
export interface LoadedJevConfig {
|
|
100
|
+
config: JevConfig;
|
|
101
|
+
warnings: string[];
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/** A config file with only the keys it actually declared. */
|
|
105
|
+
interface ParsedGate extends Partial<Omit<GateConfig, "blockOn">> {
|
|
106
|
+
blockOn?: Partial<GateThresholds>;
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
type ParsedOutput = Partial<OutputConfig>;
|
|
110
|
+
|
|
111
|
+
interface ParsedConfigFile extends Partial<Omit<JevConfig, "gate" | "output">> {
|
|
112
|
+
gate?: ParsedGate;
|
|
113
|
+
output?: ParsedOutput;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
export function defaultJevConfig(): JevConfig {
|
|
117
|
+
return {
|
|
118
|
+
endpoint: DEFAULT_ENDPOINT,
|
|
119
|
+
model: DEFAULT_MODEL,
|
|
120
|
+
apiKey: undefined,
|
|
121
|
+
apiKeyFile: undefined,
|
|
122
|
+
timeoutMs: DEFAULT_TIMEOUT_MS,
|
|
123
|
+
retries: DEFAULT_RETRIES,
|
|
124
|
+
maxStateChars: 8000,
|
|
125
|
+
gate: {
|
|
126
|
+
enabled: true,
|
|
127
|
+
mode: "shadow",
|
|
128
|
+
tools: ["bash", "write", "edit"],
|
|
129
|
+
argumentChars: 400,
|
|
130
|
+
cacheSeconds: 120,
|
|
131
|
+
blockOn: { destructive: 0.9, exfiltration: 0.7, beyondScope: 0.85, impact: 2.5 },
|
|
132
|
+
minConfidence: 0.5,
|
|
133
|
+
blockWithoutUI: false,
|
|
134
|
+
},
|
|
135
|
+
output: {
|
|
136
|
+
enabled: true,
|
|
137
|
+
tools: ["bash"],
|
|
138
|
+
outputChars: 2000,
|
|
139
|
+
leakThreshold: 0.9,
|
|
140
|
+
minConfidence: 0.6,
|
|
141
|
+
},
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Name of the host's per-project config directory.
|
|
147
|
+
*
|
|
148
|
+
* The host's own export wins whenever it exists (every host from 0.79.7 on,
|
|
149
|
+
* and the only answer that stays exact when `PI_CODING_AGENT_DIR` moves the
|
|
150
|
+
* agent dir elsewhere). Hosts before that export no name at all, but they do
|
|
151
|
+
* export `getAgentDir()`, and they keep the agent dir inside the same config
|
|
152
|
+
* dir they read per project (`<config dir>/agent`) — so the name can be read
|
|
153
|
+
* off the host's own layout rather than restated here.
|
|
154
|
+
*
|
|
155
|
+
* Caveat for those older hosts: with the agent dir overridden to a path outside
|
|
156
|
+
* that layout, the parent directory is a user choice rather than the config dir
|
|
157
|
+
* name, and a project config file can be missed. No public host on that range
|
|
158
|
+
* reports the name, so this is the closest the extension can get there.
|
|
159
|
+
*/
|
|
160
|
+
export function configDirName(): string {
|
|
161
|
+
const exported = HOST_EXPORTS.CONFIG_DIR_NAME;
|
|
162
|
+
if (typeof exported === "string" && exported.length > 0) return exported;
|
|
163
|
+
return basename(dirname(host.getAgentDir()));
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
export function loadJevConfig(cwd: string): LoadedJevConfig {
|
|
167
|
+
const defaults = defaultJevConfig();
|
|
168
|
+
const warnings: string[] = [];
|
|
169
|
+
const global = readConfigFile(join(host.getAgentDir(), CONFIG_FILE), warnings);
|
|
170
|
+
const project = readConfigFile(join(cwd, configDirName(), CONFIG_FILE), warnings);
|
|
171
|
+
|
|
172
|
+
const config: JevConfig = {
|
|
173
|
+
...defaults,
|
|
174
|
+
...global,
|
|
175
|
+
...project,
|
|
176
|
+
gate: {
|
|
177
|
+
...defaults.gate,
|
|
178
|
+
...global.gate,
|
|
179
|
+
...project.gate,
|
|
180
|
+
blockOn: {
|
|
181
|
+
...defaults.gate.blockOn,
|
|
182
|
+
...global.gate?.blockOn,
|
|
183
|
+
...project.gate?.blockOn,
|
|
184
|
+
},
|
|
185
|
+
tools: project.gate?.tools ?? global.gate?.tools ?? defaults.gate.tools,
|
|
186
|
+
},
|
|
187
|
+
output: {
|
|
188
|
+
...defaults.output,
|
|
189
|
+
...global.output,
|
|
190
|
+
...project.output,
|
|
191
|
+
tools:
|
|
192
|
+
project.output?.tools ?? global.output?.tools ?? defaults.output.tools,
|
|
193
|
+
},
|
|
194
|
+
};
|
|
195
|
+
|
|
196
|
+
const key = resolveApiKey(config, warnings);
|
|
197
|
+
return { config: { ...config, apiKey: key }, warnings };
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* Env wins over config, so a host can inject the key without touching files.
|
|
202
|
+
* apiKeyFile supports "~/" because keys usually live outside the repo.
|
|
203
|
+
*/
|
|
204
|
+
export function resolveApiKey(
|
|
205
|
+
config: Pick<JevConfig, "apiKey" | "apiKeyFile">,
|
|
206
|
+
warnings: string[] = [],
|
|
207
|
+
): string | undefined {
|
|
208
|
+
const fromEnv = process.env[API_KEY_ENV]?.trim();
|
|
209
|
+
if (fromEnv) return fromEnv;
|
|
210
|
+
if (config.apiKey?.trim()) return config.apiKey.trim();
|
|
211
|
+
if (!config.apiKeyFile) return undefined;
|
|
212
|
+
|
|
213
|
+
const path = expandHome(config.apiKeyFile);
|
|
214
|
+
try {
|
|
215
|
+
const contents = readFileSync(path, "utf8").trim();
|
|
216
|
+
return contents.length > 0 ? contents : undefined;
|
|
217
|
+
} catch (error) {
|
|
218
|
+
warnings.push(
|
|
219
|
+
`${path}: ${error instanceof Error ? error.message : String(error)}`,
|
|
220
|
+
);
|
|
221
|
+
return undefined;
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
export function expandHome(path: string): string {
|
|
226
|
+
if (path === "~") return homedir();
|
|
227
|
+
if (path.startsWith("~/")) return join(homedir(), path.slice(2));
|
|
228
|
+
return path;
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
function readConfigFile(path: string, warnings: string[]): ParsedConfigFile {
|
|
232
|
+
let raw: string;
|
|
233
|
+
try {
|
|
234
|
+
raw = readFileSync(path, "utf8");
|
|
235
|
+
} catch (error) {
|
|
236
|
+
// Missing file is the normal case; anything else is worth reporting.
|
|
237
|
+
if ((error as NodeJS.ErrnoException)?.code !== "ENOENT") {
|
|
238
|
+
warnings.push(
|
|
239
|
+
`${path}: ${error instanceof Error ? error.message : String(error)}`,
|
|
240
|
+
);
|
|
241
|
+
}
|
|
242
|
+
return {};
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
let parsed: unknown;
|
|
246
|
+
try {
|
|
247
|
+
parsed = JSON.parse(raw);
|
|
248
|
+
} catch (error) {
|
|
249
|
+
warnings.push(
|
|
250
|
+
`${path}: invalid JSON (${error instanceof Error ? error.message : String(error)})`,
|
|
251
|
+
);
|
|
252
|
+
return {};
|
|
253
|
+
}
|
|
254
|
+
if (typeof parsed !== "object" || parsed === null) {
|
|
255
|
+
warnings.push(`${path}: expected a JSON object`);
|
|
256
|
+
return {};
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
const out: ParsedConfigFile = {};
|
|
260
|
+
|
|
261
|
+
const endpoint = asString(Reflect.get(parsed, "endpoint"));
|
|
262
|
+
if (endpoint) out.endpoint = endpoint;
|
|
263
|
+
const model = asString(Reflect.get(parsed, "model"));
|
|
264
|
+
if (model) out.model = model;
|
|
265
|
+
const apiKey = asString(Reflect.get(parsed, "apiKey"));
|
|
266
|
+
if (apiKey) out.apiKey = apiKey;
|
|
267
|
+
const apiKeyFile = asString(Reflect.get(parsed, "apiKeyFile"));
|
|
268
|
+
if (apiKeyFile) out.apiKeyFile = apiKeyFile;
|
|
269
|
+
const timeoutMs = asPositiveInt(Reflect.get(parsed, "timeoutMs"));
|
|
270
|
+
if (timeoutMs !== undefined) out.timeoutMs = timeoutMs;
|
|
271
|
+
const retries = asNonNegativeInt(Reflect.get(parsed, "retries"));
|
|
272
|
+
if (retries !== undefined) out.retries = retries;
|
|
273
|
+
const maxStateChars = asPositiveInt(Reflect.get(parsed, "maxStateChars"));
|
|
274
|
+
if (maxStateChars !== undefined) out.maxStateChars = maxStateChars;
|
|
275
|
+
|
|
276
|
+
const gate = Reflect.get(parsed, "gate");
|
|
277
|
+
if (typeof gate === "object" && gate !== null) {
|
|
278
|
+
const parsedGate: ParsedGate = {};
|
|
279
|
+
const enabled = Reflect.get(gate, "enabled");
|
|
280
|
+
if (typeof enabled === "boolean") parsedGate.enabled = enabled;
|
|
281
|
+
const mode = Reflect.get(gate, "mode");
|
|
282
|
+
if (mode === "shadow" || mode === "enforce") parsedGate.mode = mode;
|
|
283
|
+
const tools = asToolNames(Reflect.get(gate, "tools"));
|
|
284
|
+
if (tools) parsedGate.tools = tools;
|
|
285
|
+
const cacheSeconds = asNonNegativeInt(Reflect.get(gate, "cacheSeconds"));
|
|
286
|
+
if (cacheSeconds !== undefined) parsedGate.cacheSeconds = cacheSeconds;
|
|
287
|
+
const argumentChars = asPositiveInt(Reflect.get(gate, "argumentChars"));
|
|
288
|
+
if (argumentChars !== undefined) parsedGate.argumentChars = argumentChars;
|
|
289
|
+
const blockWithoutUI = Reflect.get(gate, "blockWithoutUI");
|
|
290
|
+
if (typeof blockWithoutUI === "boolean") {
|
|
291
|
+
parsedGate.blockWithoutUI = blockWithoutUI;
|
|
292
|
+
}
|
|
293
|
+
const minConfidence = asRatio(Reflect.get(gate, "minConfidence"));
|
|
294
|
+
if (minConfidence !== undefined) parsedGate.minConfidence = minConfidence;
|
|
295
|
+
|
|
296
|
+
const blockOn = Reflect.get(gate, "blockOn");
|
|
297
|
+
if (typeof blockOn === "object" && blockOn !== null) {
|
|
298
|
+
const thresholds: Partial<GateThresholds> = {};
|
|
299
|
+
const destructive = asRatio(Reflect.get(blockOn, "destructive"));
|
|
300
|
+
if (destructive !== undefined) thresholds.destructive = destructive;
|
|
301
|
+
const exfiltration = asRatio(Reflect.get(blockOn, "exfiltration"));
|
|
302
|
+
if (exfiltration !== undefined) thresholds.exfiltration = exfiltration;
|
|
303
|
+
const beyondScope = asRatio(Reflect.get(blockOn, "beyondScope"));
|
|
304
|
+
if (beyondScope !== undefined) thresholds.beyondScope = beyondScope;
|
|
305
|
+
const impact = Reflect.get(blockOn, "impact");
|
|
306
|
+
if (typeof impact === "number" && Number.isFinite(impact) && impact >= 0) {
|
|
307
|
+
thresholds.impact = impact;
|
|
308
|
+
}
|
|
309
|
+
parsedGate.blockOn = thresholds;
|
|
310
|
+
}
|
|
311
|
+
out.gate = parsedGate;
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
const output = Reflect.get(parsed, "output");
|
|
315
|
+
if (typeof output === "object" && output !== null) {
|
|
316
|
+
const parsedOutput: ParsedOutput = {};
|
|
317
|
+
const enabled = Reflect.get(output, "enabled");
|
|
318
|
+
if (typeof enabled === "boolean") parsedOutput.enabled = enabled;
|
|
319
|
+
const tools = asToolNames(Reflect.get(output, "tools"));
|
|
320
|
+
if (tools) parsedOutput.tools = tools;
|
|
321
|
+
const outputChars = asPositiveInt(Reflect.get(output, "outputChars"));
|
|
322
|
+
if (outputChars !== undefined) parsedOutput.outputChars = outputChars;
|
|
323
|
+
const leakThreshold = asRatio(Reflect.get(output, "leakThreshold"));
|
|
324
|
+
if (leakThreshold !== undefined) parsedOutput.leakThreshold = leakThreshold;
|
|
325
|
+
const minConfidence = asRatio(Reflect.get(output, "minConfidence"));
|
|
326
|
+
if (minConfidence !== undefined) parsedOutput.minConfidence = minConfidence;
|
|
327
|
+
out.output = parsedOutput;
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
return out;
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
function asToolNames(value: unknown): string[] | undefined {
|
|
334
|
+
if (!Array.isArray(value)) return undefined;
|
|
335
|
+
const names = value.filter(
|
|
336
|
+
(item): item is string => typeof item === "string" && item.length > 0,
|
|
337
|
+
);
|
|
338
|
+
return names.length > 0 ? names : undefined;
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
function asString(value: unknown): string | undefined {
|
|
342
|
+
return typeof value === "string" && value.trim().length > 0
|
|
343
|
+
? value.trim()
|
|
344
|
+
: undefined;
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
function asPositiveInt(value: unknown): number | undefined {
|
|
348
|
+
return typeof value === "number" && Number.isInteger(value) && value > 0
|
|
349
|
+
? value
|
|
350
|
+
: undefined;
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
function asNonNegativeInt(value: unknown): number | undefined {
|
|
354
|
+
return typeof value === "number" && Number.isInteger(value) && value >= 0
|
|
355
|
+
? value
|
|
356
|
+
: undefined;
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
function asRatio(value: unknown): number | undefined {
|
|
360
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1
|
|
361
|
+
? value
|
|
362
|
+
: undefined;
|
|
363
|
+
}
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The judgement itself: what to ask Jev about a pending tool call, and how to
|
|
3
|
+
* turn the answers into a verdict.
|
|
4
|
+
*
|
|
5
|
+
* Question shape matters more than anything else here. Each question is one
|
|
6
|
+
* gut-check about the action, not a general "is this dangerous" - Jev is
|
|
7
|
+
* reliable when a knowledgeable person could answer from the state alone.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import type {
|
|
11
|
+
GateThresholds,
|
|
12
|
+
JevConfig,
|
|
13
|
+
} from "./config";
|
|
14
|
+
import {
|
|
15
|
+
answerFor,
|
|
16
|
+
confidenceFor,
|
|
17
|
+
describeAnswer,
|
|
18
|
+
type JevAnswer,
|
|
19
|
+
type JevQuestion,
|
|
20
|
+
type JevResponse,
|
|
21
|
+
type JevState,
|
|
22
|
+
type JevUsage,
|
|
23
|
+
} from "./client";
|
|
24
|
+
|
|
25
|
+
export interface GateInput {
|
|
26
|
+
cwd: string;
|
|
27
|
+
toolName: string;
|
|
28
|
+
input: unknown;
|
|
29
|
+
/** Last user message, so scope questions can weigh intent. */
|
|
30
|
+
userRequest?: string;
|
|
31
|
+
maxStateChars: number;
|
|
32
|
+
/** Long string fields are elided beyond this, so file bodies stay local. */
|
|
33
|
+
argumentChars: number;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export interface GateVerdict {
|
|
37
|
+
flagged: boolean;
|
|
38
|
+
reasons: string[];
|
|
39
|
+
/** Noul probabilities, absent dimension reported as 0. */
|
|
40
|
+
destructive: number;
|
|
41
|
+
exfiltration: number;
|
|
42
|
+
beyondScope: number;
|
|
43
|
+
/** 0-3 on the damage rubric, with the confidence Jev reported. */
|
|
44
|
+
impact: number | undefined;
|
|
45
|
+
impactConfidence: number | undefined;
|
|
46
|
+
answers: Record<string, JevAnswer>;
|
|
47
|
+
model: string;
|
|
48
|
+
usage: JevUsage | undefined;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Phrasing here is measured, not guessed. The first draft asked whether data
|
|
53
|
+
* "cannot be recovered from version control", which scored a real
|
|
54
|
+
* `rm -rf src && git push --force` at 0.77 - below any threshold worth having
|
|
55
|
+
* - because the reasoning path "it is in git" is available. Plain phrasing
|
|
56
|
+
* separates the same pair 0.03 / 0.99.
|
|
57
|
+
*
|
|
58
|
+
* See the calibration table in README.md for what each dimension scores across
|
|
59
|
+
* benign, destructive, exfiltrating, and ordinary edit states.
|
|
60
|
+
*/
|
|
61
|
+
export const GATE_QUESTIONS: Record<string, JevQuestion> = {
|
|
62
|
+
destructive: {
|
|
63
|
+
type: "noul",
|
|
64
|
+
instructions: "Is this action destructive?",
|
|
65
|
+
criteria: {
|
|
66
|
+
true: "Deletes or overwrites data, or rewrites shared history",
|
|
67
|
+
false: "Nothing is deleted or overwritten",
|
|
68
|
+
},
|
|
69
|
+
},
|
|
70
|
+
exfiltration: {
|
|
71
|
+
type: "noul",
|
|
72
|
+
instructions:
|
|
73
|
+
"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?",
|
|
74
|
+
criteria: {
|
|
75
|
+
true: "Local data or secrets leave the machine",
|
|
76
|
+
false: "Nothing leaves the machine",
|
|
77
|
+
},
|
|
78
|
+
},
|
|
79
|
+
beyond_scope: {
|
|
80
|
+
type: "noul",
|
|
81
|
+
instructions:
|
|
82
|
+
"Does this action affect anything beyond the change the user asked for?",
|
|
83
|
+
criteria: {
|
|
84
|
+
true: "Touches things the user did not ask about",
|
|
85
|
+
false: "Confined to what the user asked for",
|
|
86
|
+
},
|
|
87
|
+
},
|
|
88
|
+
impact: {
|
|
89
|
+
type: "score",
|
|
90
|
+
instructions:
|
|
91
|
+
"How much damage would this action do if the user did not want it?",
|
|
92
|
+
criteria: [
|
|
93
|
+
"None, it only reads",
|
|
94
|
+
"Small, one file or one reversible change",
|
|
95
|
+
"Large, many files or shared state",
|
|
96
|
+
"Severe, data loss or a forced overwrite of shared history",
|
|
97
|
+
],
|
|
98
|
+
},
|
|
99
|
+
};
|
|
100
|
+
|
|
101
|
+
export function buildGateState(input: GateInput): JevState {
|
|
102
|
+
const state: Record<string, unknown> = {
|
|
103
|
+
cwd: input.cwd,
|
|
104
|
+
tool: input.toolName,
|
|
105
|
+
arguments: summarizeArguments(input.input, input.argumentChars),
|
|
106
|
+
platform: process.platform,
|
|
107
|
+
};
|
|
108
|
+
if (input.userRequest) {
|
|
109
|
+
state.user_request = truncateText(input.userRequest, 1200);
|
|
110
|
+
}
|
|
111
|
+
return state;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* The tool's arguments are what gets judged, but a `write` call carries an
|
|
116
|
+
* entire file body and an `edit` call carries whole diffs. The judgement turns
|
|
117
|
+
* on the path, the shape, and the intent - not the bulk - so long strings are
|
|
118
|
+
* elided here and the file contents stay on the machine.
|
|
119
|
+
*/
|
|
120
|
+
function summarizeArguments(value: unknown, maxChars: number, depth = 0): unknown {
|
|
121
|
+
if (typeof value === "string") {
|
|
122
|
+
return value.length > maxChars
|
|
123
|
+
? `${value.slice(0, maxChars)}\u2026[${value.length - maxChars} chars elided]`
|
|
124
|
+
: value;
|
|
125
|
+
}
|
|
126
|
+
if (depth > 4 || value === null || typeof value !== "object") return value;
|
|
127
|
+
if (Array.isArray(value)) {
|
|
128
|
+
return value.map((item) => summarizeArguments(item, maxChars, depth + 1));
|
|
129
|
+
}
|
|
130
|
+
const out: Record<string, unknown> = {};
|
|
131
|
+
for (const [key, item] of Object.entries(value)) {
|
|
132
|
+
out[key] = summarizeArguments(item, maxChars, depth + 1);
|
|
133
|
+
}
|
|
134
|
+
return out;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* A verdict is blockable when any dimension crosses its threshold. Thresholds
|
|
139
|
+
* are set from the measured separation between ordinary work and the state they
|
|
140
|
+
* are meant to catch: ordinary requested edits score up to 0.85 on
|
|
141
|
+
* `destructive` and 0.72 on `beyond_scope`, so those thresholds sit above that
|
|
142
|
+
* band rather than at a round 0.7.
|
|
143
|
+
*/
|
|
144
|
+
export function evaluateGate(
|
|
145
|
+
response: JevResponse,
|
|
146
|
+
config: JevConfig,
|
|
147
|
+
): GateVerdict {
|
|
148
|
+
const thresholds: GateThresholds = config.gate.blockOn;
|
|
149
|
+
const destructive = response.answers.destructive?.type === "noul"
|
|
150
|
+
? response.answers.destructive.noul
|
|
151
|
+
: 0;
|
|
152
|
+
const exfiltration = response.answers.exfiltration?.type === "noul"
|
|
153
|
+
? response.answers.exfiltration.noul
|
|
154
|
+
: 0;
|
|
155
|
+
const beyondScope = response.answers.beyond_scope?.type === "noul"
|
|
156
|
+
? response.answers.beyond_scope.noul
|
|
157
|
+
: 0;
|
|
158
|
+
const impactAnswer = answerFor(response, "impact");
|
|
159
|
+
const impact = impactAnswer?.type === "score" ? impactAnswer.score : undefined;
|
|
160
|
+
const impactConfidence = confidenceFor(response, "impact");
|
|
161
|
+
|
|
162
|
+
const reasons: string[] = [];
|
|
163
|
+
if (destructive >= thresholds.destructive) {
|
|
164
|
+
reasons.push(`destructive ${destructive.toFixed(2)}`);
|
|
165
|
+
}
|
|
166
|
+
if (exfiltration >= thresholds.exfiltration) {
|
|
167
|
+
reasons.push(`exfiltration ${exfiltration.toFixed(2)}`);
|
|
168
|
+
}
|
|
169
|
+
if (beyondScope >= thresholds.beyondScope) {
|
|
170
|
+
reasons.push(`beyond_scope ${beyondScope.toFixed(2)}`);
|
|
171
|
+
}
|
|
172
|
+
if (
|
|
173
|
+
impact !== undefined &&
|
|
174
|
+
impact >= thresholds.impact &&
|
|
175
|
+
(impactConfidence === undefined || impactConfidence >= config.gate.minConfidence)
|
|
176
|
+
) {
|
|
177
|
+
reasons.push(
|
|
178
|
+
`impact ${impact.toFixed(2)}/3${impactConfidence !== undefined ? ` at confidence ${impactConfidence.toFixed(2)}` : ""}`,
|
|
179
|
+
);
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
return {
|
|
183
|
+
flagged: reasons.length > 0,
|
|
184
|
+
reasons,
|
|
185
|
+
destructive,
|
|
186
|
+
exfiltration,
|
|
187
|
+
beyondScope,
|
|
188
|
+
impact,
|
|
189
|
+
impactConfidence,
|
|
190
|
+
answers: response.answers,
|
|
191
|
+
model: response.model,
|
|
192
|
+
usage: response.usage,
|
|
193
|
+
};
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
export function summarizeVerdict(verdict: GateVerdict): string {
|
|
197
|
+
if (!verdict.flagged) return "clear";
|
|
198
|
+
return verdict.reasons.join(", ");
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
/** One-line answer dump, for /jev last and the model-facing tool. */
|
|
202
|
+
export function describeAnswers(response: JevResponse): string {
|
|
203
|
+
return Object.entries(response.answers)
|
|
204
|
+
.map(([id, answer]) => `${id}=${describeAnswer(answer)}`)
|
|
205
|
+
.join(" ");
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* Cache key for a pending call. Identical input must not be judged twice in a
|
|
210
|
+
* parallel tool batch, and a retried loop should reuse the earlier verdict.
|
|
211
|
+
*/
|
|
212
|
+
export function judgmentKey(toolName: string, input: unknown): string {
|
|
213
|
+
return `${toolName}:${stableStringify(input)}`;
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
function stableStringify(value: unknown): string {
|
|
217
|
+
if (value === null || typeof value !== "object") return JSON.stringify(value) ?? "null";
|
|
218
|
+
if (Array.isArray(value)) return `[${value.map(stableStringify).join(",")}]`;
|
|
219
|
+
const entries = Object.entries(value as Record<string, unknown>).sort(([a], [b]) =>
|
|
220
|
+
a < b ? -1 : a > b ? 1 : 0,
|
|
221
|
+
);
|
|
222
|
+
return `{${entries.map(([k, v]) => `${JSON.stringify(k)}:${stableStringify(v)}`).join(",")}}`;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
export function truncateText(text: string, maxChars: number): string {
|
|
226
|
+
return text.length > maxChars
|
|
227
|
+
? `${text.slice(0, maxChars)}\u2026[truncated]`
|
|
228
|
+
: text;
|
|
229
|
+
}
|