@pi-in-go/pigpen-jev 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/CREDITS.md +22 -0
  2. package/LICENSE +22 -0
  3. package/README.md +237 -0
  4. package/extensions/jev/ask.go +166 -0
  5. package/extensions/jev/ask_test.go +218 -0
  6. package/extensions/jev/backend.go +128 -0
  7. package/extensions/jev/bench_test.go +64 -0
  8. package/extensions/jev/boundaries_test.go +159 -0
  9. package/extensions/jev/command.go +224 -0
  10. package/extensions/jev/commands_test.go +214 -0
  11. package/extensions/jev/config.go +450 -0
  12. package/extensions/jev/errors_test.go +191 -0
  13. package/extensions/jev/extension.go +391 -0
  14. package/extensions/jev/fakehost_test.go +548 -0
  15. package/extensions/jev/gate.go +125 -0
  16. package/extensions/jev/gate_test.go +610 -0
  17. package/extensions/jev/gatekey_test.go +24 -0
  18. package/extensions/jev/go.mod +9 -0
  19. package/extensions/jev/go.sum +2 -0
  20. package/extensions/jev/go.work +10 -0
  21. package/extensions/jev/helpers_test.go +404 -0
  22. package/extensions/jev/memo.go +88 -0
  23. package/extensions/jev/output.go +89 -0
  24. package/extensions/jev/output_test.go +187 -0
  25. package/extensions/jev/ownmodel_test.go +118 -0
  26. package/extensions/jev/render.go +136 -0
  27. package/extensions/jev/review_test.go +310 -0
  28. package/extensions/jev/source_test.go +57 -0
  29. package/extensions/jev/text.go +174 -0
  30. package/extensions/jev/trust_test.go +335 -0
  31. package/extensions/jev/types.go +227 -0
  32. package/libs/typesafe/CONTRACT.md +125 -0
  33. package/libs/typesafe/CREDITS.md +37 -0
  34. package/libs/typesafe/LICENSE +23 -0
  35. package/libs/typesafe/README.md +19 -0
  36. package/libs/typesafe/go.mod +3 -0
  37. package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
  38. package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
  39. package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
  40. package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
  41. package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
  42. package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
  43. package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
  44. package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
  45. package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
  46. package/libs/typesafe/libraries/ownmodel/run.go +288 -0
  47. package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
  48. package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
  49. package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
  50. package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
  51. package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
  52. package/libs/typesafe/libraries/typesafe/answers.go +268 -0
  53. package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
  54. package/libs/typesafe/libraries/typesafe/batch.go +80 -0
  55. package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
  56. package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
  57. package/libs/typesafe/libraries/typesafe/client.go +561 -0
  58. package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
  59. package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
  60. package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
  61. package/libs/typesafe/libraries/typesafe/doc.go +27 -0
  62. package/libs/typesafe/libraries/typesafe/entry.go +142 -0
  63. package/libs/typesafe/libraries/typesafe/env.go +11 -0
  64. package/libs/typesafe/libraries/typesafe/errors.go +310 -0
  65. package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
  66. package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
  67. package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
  68. package/libs/typesafe/libraries/typesafe/logging.go +160 -0
  69. package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
  70. package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
  71. package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
  72. package/libs/typesafe/libraries/typesafe/questions.go +490 -0
  73. package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
  74. package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
  75. package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
  76. package/libs/typesafe/libraries/typesafe/retry.go +350 -0
  77. package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
  78. package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
  79. package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
  80. package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
  81. package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
  82. package/libs/typesafe/libraries/typesafe/version.go +10 -0
  83. package/libs/typesafe/package.json +37 -0
  84. package/libs/typesafe/provenance.json +49 -0
  85. package/package.json +42 -0
  86. package/port/PORT.md +107 -0
  87. package/port/e2e/gate-and-output.py +35 -0
  88. package/port/e2e/jev-ask.py +36 -0
  89. package/port/e2e/model-switch.py +44 -0
  90. package/port/e2e/off-by-default.py +34 -0
  91. package/port/gen-scenarios.py +103 -0
  92. package/port/golden/cache-identical-calls.jsonl +30 -0
  93. package/port/golden/clear.jsonl +22 -0
  94. package/port/golden/commands.jsonl +43 -0
  95. package/port/golden/enforce-accept.jsonl +23 -0
  96. package/port/golden/enforce-decline.jsonl +22 -0
  97. package/port/golden/jev-ask.jsonl +20 -0
  98. package/port/golden/output-advice.jsonl +23 -0
  99. package/port/golden/output-leak.jsonl +24 -0
  100. package/port/golden/output-low-confidence.jsonl +22 -0
  101. package/port/golden/shadow-flagged.jsonl +23 -0
  102. package/port/golden/unjudged-tools.jsonl +19 -0
  103. package/port/golden/write-elision.jsonl +21 -0
  104. package/port/mutate-unit.py +63 -0
  105. package/port/mutations.json +578 -0
  106. package/port/oracle/LICENSE +21 -0
  107. package/port/oracle/README.md +181 -0
  108. package/port/oracle/SHA256SUMS +8 -0
  109. package/port/oracle/package.json +43 -0
  110. package/port/oracle/src/client.ts +409 -0
  111. package/port/oracle/src/config.ts +363 -0
  112. package/port/oracle/src/gate.ts +229 -0
  113. package/port/oracle/src/index.ts +649 -0
  114. package/port/oracle/src/output.ts +163 -0
  115. package/port/red-run.log +309 -0
  116. package/port/scenarios/cache-identical-calls.json +71 -0
  117. package/port/scenarios/clear.json +61 -0
  118. package/port/scenarios/commands.json +119 -0
  119. package/port/scenarios/enforce-accept.json +66 -0
  120. package/port/scenarios/enforce-decline.json +57 -0
  121. package/port/scenarios/jev-ask.json +83 -0
  122. package/port/scenarios/output-advice.json +61 -0
  123. package/port/scenarios/output-leak.json +61 -0
  124. package/port/scenarios/output-low-confidence.json +61 -0
  125. package/port/scenarios/shadow-flagged.json +61 -0
  126. package/port/scenarios/unjudged-tools.json +55 -0
  127. package/port/scenarios/write-elision.json +53 -0
  128. package/provenance.json +18 -0
@@ -0,0 +1,649 @@
1
+ import { StringEnum } from "@earendil-works/pi-ai";
2
+ import type {
3
+ ExtensionAPI,
4
+ ExtensionContext,
5
+ } from "@earendil-works/pi-coding-agent";
6
+ import { Type } from "typebox";
7
+ import {
8
+ askJev,
9
+ describeAnswer,
10
+ JevError,
11
+ redact,
12
+ rememberSecret,
13
+ type JevQuestion,
14
+ type JevResponse,
15
+ } from "./client";
16
+ import {
17
+ API_KEY_ENV,
18
+ loadJevConfig,
19
+ type JevConfig,
20
+ type LoadedJevConfig,
21
+ } from "./config";
22
+ import {
23
+ buildGateState,
24
+ describeAnswers,
25
+ evaluateGate,
26
+ GATE_QUESTIONS,
27
+ judgmentKey,
28
+ summarizeVerdict,
29
+ type GateVerdict,
30
+ } from "./gate";
31
+ import {
32
+ buildOutputState,
33
+ evaluateOutput,
34
+ OUTPUT_QUESTIONS,
35
+ outputKey,
36
+ type OutputVerdict,
37
+ } from "./output";
38
+
39
+ /**
40
+ * pi-jev - TypeSafe Jev as a decision layer for pi.
41
+ *
42
+ * Two jobs, both decided by typed questions rather than by prompt text:
43
+ *
44
+ * 1. Gate. Before a mutating tool runs, ask Jev whether the action is
45
+ * destructive, exfiltrating, or outside scope, plus how reversible it is.
46
+ * Shadow mode (default) reports. Enforce mode asks the user to confirm.
47
+ * 2. Output judge. After a tool runs, ask whether its output carries a
48
+ * secret and what kind of failure it reports. Never blocks; it appends a
49
+ * line to the tool result the model reads.
50
+ * 3. jev_ask. A model-facing tool for decisions that should be typed and
51
+ * calibrated instead of written: classification, relevance, yes/no checks.
52
+ *
53
+ * Fails open. An API outage must never stop the agent from working, so every
54
+ * error path returns "no verdict" instead of a block.
55
+ */
56
+
57
+ const STATUS_KEY = "jev";
58
+ const ERROR_NOTIFY_INTERVAL_MS = 60_000;
59
+ /** Identical output is judged once per window, keyed by tool and output hash. */
60
+ const OUTPUT_CACHE_SECONDS = 120;
61
+
62
+ const QuestionParam = Type.Object({
63
+ id: Type.String({
64
+ description: "Short key for this question. The answer comes back under it.",
65
+ }),
66
+ type: StringEnum(["noul", "choice", "score"] as const, {
67
+ description:
68
+ "noul = yes/no probability, choice = pick one option, score = value on a rubric",
69
+ }),
70
+ instructions: Type.String({
71
+ description:
72
+ "The one thing to judge. One specific, well-scoped gut-check per question.",
73
+ }),
74
+ options: Type.Optional(
75
+ Type.Array(
76
+ Type.Object({
77
+ name: Type.String({ description: "Option key" }),
78
+ description: Type.Optional(
79
+ Type.String({ description: "When this option applies" }),
80
+ ),
81
+ }),
82
+ { description: "choice only: the options to choose between" },
83
+ ),
84
+ ),
85
+ levels: Type.Optional(
86
+ Type.Array(Type.String(), {
87
+ description: "score only: ordered rubric levels, lowest first, at least two",
88
+ }),
89
+ ),
90
+ });
91
+
92
+ const AskParams = Type.Object({
93
+ state: Type.String({
94
+ description:
95
+ "The text to judge: tool output, a diff, a message, a document excerpt.",
96
+ }),
97
+ questions: Type.Array(QuestionParam, {
98
+ description:
99
+ "One or more questions. All are evaluated in parallel against the same state.",
100
+ }),
101
+ });
102
+
103
+ export default function jevExtension(pi: ExtensionAPI): void {
104
+ let loaded: LoadedJevConfig = loadJevConfig(process.cwd());
105
+ let config: JevConfig = loaded.config;
106
+ rememberSecret(config.apiKey);
107
+
108
+ let gateOn = config.gate.enabled;
109
+ let outputOn = config.output.enabled;
110
+ let mode = config.gate.mode;
111
+
112
+ const cache = new Map<string, { at: number; verdict: GateVerdict }>();
113
+ const inflight = new Map<string, Promise<GateVerdict | undefined>>();
114
+ const outputCache = new Map<string, { at: number; verdict: OutputVerdict }>();
115
+ const outputInflight = new Map<string, Promise<OutputVerdict | undefined>>();
116
+ let last: { tool: string; verdict: GateVerdict; at: number } | undefined;
117
+ let lastOutput: { tool: string; verdict: OutputVerdict; at: number } | undefined;
118
+ let lastErrorAt = 0;
119
+ let missingKeyWarned = false;
120
+
121
+ function reload(cwd: string): void {
122
+ loaded = loadJevConfig(cwd);
123
+ config = loaded.config;
124
+ rememberSecret(config.apiKey);
125
+ gateOn = config.gate.enabled;
126
+ outputOn = config.output.enabled;
127
+ mode = config.gate.mode;
128
+ }
129
+
130
+ pi.on("session_start", async (_event, ctx) => {
131
+ reload(ctx.cwd);
132
+ for (const warning of loaded.warnings) {
133
+ ctx.ui.notify(`pi-jev: ${redact(warning)}`, "warning");
134
+ }
135
+ if (!config.apiKey) {
136
+ if (!missingKeyWarned) {
137
+ missingKeyWarned = true;
138
+ ctx.ui.notify(
139
+ `pi-jev: no key. Set ${API_KEY_ENV} or apiKeyFile in pi-jev.json; the gate is inactive until then.`,
140
+ "warning",
141
+ );
142
+ }
143
+ ctx.ui.setStatus(STATUS_KEY, undefined);
144
+ return;
145
+ }
146
+ ctx.ui.setStatus(
147
+ STATUS_KEY,
148
+ `jev: ${gateOn ? mode : "off"}${outputOn ? "" : " (out off)"}`,
149
+ );
150
+ });
151
+
152
+ pi.on("tool_call", async (event, ctx) => {
153
+ if (!gateOn || !config.apiKey) return;
154
+ if (!config.gate.tools.includes(event.toolName)) return;
155
+
156
+ const key = judgmentKey(event.toolName, event.input);
157
+ const verdict = await verdictFor(key, event, ctx);
158
+ if (!verdict) return;
159
+
160
+ last = { tool: event.toolName, verdict, at: Date.now() };
161
+ ctx.ui.setStatus(
162
+ STATUS_KEY,
163
+ verdict.flagged
164
+ ? `jev: ${summarizeVerdict(verdict)}`
165
+ : `jev: clear (${mode})`,
166
+ );
167
+ if (!verdict.flagged) return;
168
+
169
+ const reason = `pi-jev: ${summarizeVerdict(verdict)}`;
170
+ if (mode === "shadow") {
171
+ ctx.ui.notify(
172
+ `jev shadow: ${event.toolName} - ${summarizeVerdict(verdict)}`,
173
+ "warning",
174
+ );
175
+ return;
176
+ }
177
+
178
+ if (!ctx.hasUI) {
179
+ if (config.gate.blockWithoutUI) return { block: true, reason };
180
+ // No UI means no way to approve a flagged call. Degrade to shadow
181
+ // rather than deadlocking a headless run on a classifier's opinion.
182
+ ctx.ui.notify(
183
+ `jev: ${event.toolName} - ${summarizeVerdict(verdict)} (headless: not blocking; set gate.blockWithoutUI to block)`,
184
+ "warning",
185
+ );
186
+ return;
187
+ }
188
+ const allow = await ctx.ui.confirm(
189
+ "Jev flagged this tool call",
190
+ `${event.toolName}\n${summarizeVerdict(verdict)}\n\nRun it anyway?`,
191
+ );
192
+ return allow ? undefined : { block: true, reason: `${reason} (declined)` };
193
+ });
194
+
195
+ pi.on("tool_result", async (event, ctx) => {
196
+ if (!outputOn || !config.apiKey) return;
197
+ if (!config.output.tools.includes(event.toolName)) return;
198
+ const text = contentText(event.content);
199
+ if (!text.trim()) return;
200
+
201
+ const verdict = await outputVerdictFor(
202
+ outputKey(event.toolName, text),
203
+ {
204
+ toolName: event.toolName,
205
+ input: event.input,
206
+ output: text,
207
+ isError: event.isError,
208
+ },
209
+ ctx,
210
+ );
211
+ if (!verdict?.notice) return;
212
+
213
+ lastOutput = { tool: event.toolName, verdict, at: Date.now() };
214
+ ctx.ui.setStatus(STATUS_KEY, `jev: ${verdict.kind} (${event.toolName})`);
215
+ if (verdict.kind === "leak") {
216
+ ctx.ui.notify(
217
+ `jev: ${event.toolName} output may carry a secret (${verdict.leaksSecret.toFixed(2)})`,
218
+ "warning",
219
+ );
220
+ }
221
+ // The model reads the tool result, so the notice rides with it.
222
+ return {
223
+ content: [
224
+ ...event.content,
225
+ { type: "text" as const, text: `[pi-jev] ${verdict.notice}` },
226
+ ],
227
+ };
228
+ });
229
+
230
+ async function outputVerdictFor(
231
+ key: string,
232
+ event: { toolName: string; input: unknown; output: string; isError: boolean },
233
+ ctx: ExtensionContext,
234
+ ): Promise<OutputVerdict | undefined> {
235
+ const cached = outputCache.get(key);
236
+ if (
237
+ cached &&
238
+ (Date.now() - cached.at) / 1000 <= OUTPUT_CACHE_SECONDS
239
+ ) {
240
+ return cached.verdict;
241
+ }
242
+ const pending = outputInflight.get(key);
243
+ if (pending) return pending;
244
+
245
+ const promise = judgeOutput(event, ctx).finally(() =>
246
+ outputInflight.delete(key),
247
+ );
248
+ outputInflight.set(key, promise);
249
+ const verdict = await promise;
250
+ if (verdict) {
251
+ outputCache.set(key, { at: Date.now(), verdict });
252
+ if (outputCache.size > 64) outputCache.clear();
253
+ }
254
+ return verdict;
255
+ }
256
+
257
+ async function judgeOutput(
258
+ event: { toolName: string; input: unknown; output: string; isError: boolean },
259
+ ctx: ExtensionContext,
260
+ ): Promise<OutputVerdict | undefined> {
261
+ const apiKey = config.apiKey;
262
+ if (!apiKey) return undefined;
263
+ try {
264
+ const response = await askJev({
265
+ state: buildOutputState({
266
+ cwd: ctx.cwd,
267
+ toolName: event.toolName,
268
+ input: event.input,
269
+ output: event.output,
270
+ isError: event.isError,
271
+ outputChars: config.output.outputChars,
272
+ }),
273
+ questions: OUTPUT_QUESTIONS,
274
+ apiKey,
275
+ model: config.model,
276
+ endpoint: config.endpoint,
277
+ timeoutMs: config.timeoutMs,
278
+ retries: config.retries,
279
+ signal: ctx.signal,
280
+ });
281
+ return evaluateOutput(response, config);
282
+ } catch (error) {
283
+ notifyError(ctx, error);
284
+ return undefined;
285
+ }
286
+ }
287
+
288
+ async function verdictFor(
289
+ key: string,
290
+ event: { toolName: string; input: unknown },
291
+ ctx: ExtensionContext,
292
+ ): Promise<GateVerdict | undefined> {
293
+ const cached = cache.get(key);
294
+ if (
295
+ cached &&
296
+ (Date.now() - cached.at) / 1000 <= config.gate.cacheSeconds
297
+ ) {
298
+ return cached.verdict;
299
+ }
300
+ const pending = inflight.get(key);
301
+ if (pending) return pending;
302
+
303
+ const promise = judge(event, ctx).finally(() => inflight.delete(key));
304
+ inflight.set(key, promise);
305
+ const verdict = await promise;
306
+ if (verdict) {
307
+ cache.set(key, { at: Date.now(), verdict });
308
+ prune();
309
+ }
310
+ return verdict;
311
+ }
312
+
313
+ async function judge(
314
+ event: { toolName: string; input: unknown },
315
+ ctx: ExtensionContext,
316
+ ): Promise<GateVerdict | undefined> {
317
+ const apiKey = config.apiKey;
318
+ if (!apiKey) return undefined;
319
+ try {
320
+ const response = await askJev({
321
+ state: buildGateState({
322
+ cwd: ctx.cwd,
323
+ toolName: event.toolName,
324
+ input: event.input,
325
+ userRequest: lastUserRequest(ctx),
326
+ maxStateChars: config.maxStateChars,
327
+ argumentChars: config.gate.argumentChars,
328
+ }),
329
+ questions: GATE_QUESTIONS,
330
+ apiKey,
331
+ model: config.model,
332
+ endpoint: config.endpoint,
333
+ timeoutMs: config.timeoutMs,
334
+ retries: config.retries,
335
+ signal: ctx.signal,
336
+ });
337
+ return evaluateGate(response, config);
338
+ } catch (error) {
339
+ notifyError(ctx, error);
340
+ return undefined;
341
+ }
342
+ }
343
+
344
+ function notifyError(ctx: ExtensionContext, error: unknown): void {
345
+ const at = Date.now();
346
+ if (at - lastErrorAt < ERROR_NOTIFY_INTERVAL_MS) return;
347
+ lastErrorAt = at;
348
+ ctx.ui.notify(
349
+ `pi-jev: ${redact(error instanceof Error ? error.message : String(error))} (failing open)`,
350
+ "error",
351
+ );
352
+ }
353
+
354
+ function prune(): void {
355
+ if (cache.size <= 64) return;
356
+ const cutoff = Date.now() - config.gate.cacheSeconds * 1000;
357
+ for (const [key, entry] of cache) {
358
+ if (entry.at < cutoff) cache.delete(key);
359
+ }
360
+ while (cache.size > 64) {
361
+ const oldest = cache.keys().next().value;
362
+ if (oldest === undefined) break;
363
+ cache.delete(oldest);
364
+ }
365
+ }
366
+
367
+ pi.registerCommand("jev", {
368
+ description:
369
+ "TypeSafe Jev: status, mode, on/off, last verdict, last judged output",
370
+ handler: async (args, ctx) => {
371
+ const [sub, value] = args.trim().toLowerCase().split(/\s+/);
372
+ if (sub === "on" || sub === "off") {
373
+ const on = sub === "on";
374
+ gateOn = on;
375
+ outputOn = on;
376
+ ctx.ui.setStatus(STATUS_KEY, on ? `jev: ${mode}` : "jev: off");
377
+ ctx.ui.notify(`pi-jev: gate ${sub}`, "info");
378
+ return;
379
+ }
380
+ if (sub === "mode") {
381
+ if (value !== "shadow" && value !== "enforce") {
382
+ ctx.ui.notify(
383
+ `pi-jev: mode is ${mode} (usage: /jev mode shadow|enforce)`,
384
+ "warning",
385
+ );
386
+ return;
387
+ }
388
+ mode = value;
389
+ ctx.ui.setStatus(
390
+ STATUS_KEY,
391
+ `jev: ${gateOn ? mode : "off"}${outputOn ? "" : " (out off)"}`,
392
+ );
393
+ ctx.ui.notify(
394
+ `pi-jev: ${mode}${mode === "shadow" ? " (reports, never blocks)" : " (asks before running flagged calls)"}`,
395
+ "info",
396
+ );
397
+ return;
398
+ }
399
+ if (sub === "last") {
400
+ if (!last) {
401
+ ctx.ui.notify("pi-jev: no verdicts yet", "info");
402
+ return;
403
+ }
404
+ ctx.ui.notify(
405
+ `pi-jev: ${last.tool} - ${summarizeVerdict(last.verdict)} | ${describeAnswers({ model: last.verdict.model, answers: last.verdict.answers })}`,
406
+ "info",
407
+ );
408
+ return;
409
+ }
410
+ if (sub === "output") {
411
+ if (!lastOutput) {
412
+ ctx.ui.notify("pi-jev: no tool output judged yet", "info");
413
+ return;
414
+ }
415
+ ctx.ui.notify(
416
+ `pi-jev output: ${lastOutput.tool} - ${lastOutput.verdict.kind} | leak ${lastOutput.verdict.leaksSecret.toFixed(2)} | class ${lastOutput.verdict.failureClass ?? "none"} at ${lastOutput.verdict.classConfidence?.toFixed(2) ?? "n/a"}`,
417
+ "info",
418
+ );
419
+ return;
420
+ }
421
+ if (sub === "check") {
422
+ const text = args.trim().slice("check".length).trim();
423
+ if (!text) {
424
+ ctx.ui.notify("pi-jev: usage /jev check <text>", "warning");
425
+ return;
426
+ }
427
+ const apiKey = config.apiKey;
428
+ if (!apiKey) {
429
+ ctx.ui.notify(`pi-jev: no key (${API_KEY_ENV} unset)`, "warning");
430
+ return;
431
+ }
432
+ try {
433
+ const response = await askJev({
434
+ state: text,
435
+ questions: GATE_QUESTIONS,
436
+ apiKey,
437
+ model: config.model,
438
+ endpoint: config.endpoint,
439
+ timeoutMs: config.timeoutMs,
440
+ retries: config.retries,
441
+ signal: ctx.signal,
442
+ });
443
+ const verdict = evaluateGate(response, config);
444
+ last = { tool: "check", verdict, at: Date.now() };
445
+ ctx.ui.notify(
446
+ `pi-jev check: ${summarizeVerdict(verdict)} | ${describeAnswers(response)}`,
447
+ "info",
448
+ );
449
+ } catch (error) {
450
+ ctx.ui.notify(
451
+ `pi-jev: ${redact(error instanceof Error ? error.message : String(error))}`,
452
+ "error",
453
+ );
454
+ }
455
+ return;
456
+ }
457
+
458
+ const key = config.apiKey
459
+ ? config.apiKeyFile
460
+ ? `apiKeyFile ${config.apiKeyFile}`
461
+ : "apiKey (inline)"
462
+ : `missing (${API_KEY_ENV})`;
463
+ ctx.ui.notify(
464
+ `pi-jev: ${gateOn ? "on" : "off"}, out ${outputOn ? "on" : "off"}, mode ${mode}, model ${config.model}, key ${key}, judging ${config.gate.tools.join("/")}, out tools ${config.output.tools.join("/")}, cache ${cache.size}/${outputCache.size}`,
465
+ "info",
466
+ );
467
+ },
468
+ });
469
+
470
+ if (config.apiKey) {
471
+ pi.registerTool({
472
+ name: "jev_ask",
473
+ label: "Jev Ask",
474
+ description:
475
+ "Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.",
476
+ promptSnippet:
477
+ "Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers",
478
+ promptGuidelines: [
479
+ "Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.",
480
+ "Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself.",
481
+ ],
482
+ parameters: AskParams,
483
+ async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
484
+ const apiKey = config.apiKey;
485
+ if (!apiKey) {
486
+ return {
487
+ content: [
488
+ { type: "text" as const, text: `jev_ask: no API key (${API_KEY_ENV} unset)` },
489
+ ],
490
+ details: { ok: false },
491
+ };
492
+ }
493
+ const questions: Record<string, JevQuestion> = {};
494
+ for (const question of params.questions) {
495
+ const built = toQuestion(question);
496
+ if (typeof built === "string") {
497
+ return {
498
+ content: [
499
+ { type: "text" as const, text: `jev_ask: ${built}` },
500
+ ],
501
+ details: { ok: false },
502
+ };
503
+ }
504
+ questions[question.id] = built;
505
+ }
506
+
507
+ try {
508
+ const response = await askJev({
509
+ state: params.state,
510
+ questions,
511
+ apiKey,
512
+ model: config.model,
513
+ endpoint: config.endpoint,
514
+ timeoutMs: config.timeoutMs,
515
+ retries: config.retries,
516
+ signal,
517
+ });
518
+ return {
519
+ content: [
520
+ {
521
+ type: "text" as const,
522
+ text: renderAnswers(response, questions),
523
+ },
524
+ ],
525
+ details: {
526
+ ok: true,
527
+ model: response.model,
528
+ usage: response.usage,
529
+ answers: response.answers,
530
+ },
531
+ };
532
+ } catch (error) {
533
+ return {
534
+ content: [
535
+ {
536
+ type: "text" as const,
537
+ text: `jev_ask: ${redact(error instanceof JevError ? error.message : String(error))}`,
538
+ },
539
+ ],
540
+ details: { ok: false },
541
+ };
542
+ }
543
+ },
544
+ });
545
+ }
546
+ }
547
+
548
+ /** Text of a tool result, so the output judge sees what the model will see. */
549
+ function contentText(content: unknown): string {
550
+ if (typeof content === "string") return content;
551
+ if (!Array.isArray(content)) return "";
552
+ const parts: string[] = [];
553
+ for (const block of content) {
554
+ if (typeof block !== "object" || block === null) continue;
555
+ const text = Reflect.get(block, "text");
556
+ if (typeof text === "string") parts.push(text);
557
+ }
558
+ return parts.join("\n");
559
+ }
560
+
561
+ interface QuestionParamValue {
562
+ id: string;
563
+ type: "noul" | "choice" | "score";
564
+ instructions: string;
565
+ options?: { name: string; description?: string }[];
566
+ levels?: string[];
567
+ }
568
+
569
+ /** Returns a Jev question, or a message explaining why the shape is invalid. */
570
+ function toQuestion(question: QuestionParamValue): JevQuestion | string {
571
+ const { id, instructions } = question;
572
+ if (question.type === "choice") {
573
+ if (!question.options || question.options.length === 0) {
574
+ return `question "${id}": choice needs at least one option`;
575
+ }
576
+ const criteria: Record<string, string | null> = {};
577
+ for (const option of question.options) {
578
+ criteria[option.name] = option.description ?? null;
579
+ }
580
+ return { type: "choice", instructions, criteria };
581
+ }
582
+ if (question.type === "score") {
583
+ if (!question.levels || question.levels.length < 2) {
584
+ return `question "${id}": score needs at least two levels`;
585
+ }
586
+ return { type: "score", instructions, criteria: question.levels };
587
+ }
588
+ return { type: "noul", instructions };
589
+ }
590
+
591
+ function renderAnswers(
592
+ response: JevResponse,
593
+ questions: Record<string, JevQuestion>,
594
+ ): string {
595
+ const lines = [`model ${response.model}`];
596
+ for (const [id, answer] of Object.entries(response.answers)) {
597
+ const question = questions[id];
598
+ const tail = question ? ` <- ${question.instructions}` : "";
599
+ lines.push(
600
+ `${id}: ${answer.type === "choice" ? formatChoice(answer) : describeAnswer(answer)}${tail}`,
601
+ );
602
+ }
603
+ if (response.usage) {
604
+ lines.push(
605
+ `tokens ${response.usage.input_tokens ?? 0} in / ${response.usage.output_tokens ?? 0} out`,
606
+ );
607
+ }
608
+ return lines.join("\n");
609
+ }
610
+
611
+ function formatChoice(answer: {
612
+ choice: string;
613
+ probabilities: Record<string, number>;
614
+ confidence: number;
615
+ }): string {
616
+ const ranked = Object.entries(answer.probabilities)
617
+ .sort(([, a], [, b]) => b - a)
618
+ .map(([option, probability]) => `${option} ${probability.toFixed(2)}`)
619
+ .join(", ");
620
+ return `${answer.choice} (conf ${answer.confidence.toFixed(2)}) [${ranked}]`;
621
+ }
622
+
623
+ /** Latest user text in the branch, so scope questions can weigh intent. */
624
+ function lastUserRequest(ctx: ExtensionContext): string | undefined {
625
+ const entries = ctx.sessionManager.buildContextEntries();
626
+ for (let index = entries.length - 1; index >= 0; index -= 1) {
627
+ const entry = entries[index];
628
+ if (entry?.type !== "message") continue;
629
+ const message = entry.message;
630
+ if (message.role !== "user") continue;
631
+ const text = messageText(message.content);
632
+ if (text) return text;
633
+ }
634
+ return undefined;
635
+ }
636
+
637
+ function messageText(content: unknown): string | undefined {
638
+ if (typeof content === "string") return content.trim() || undefined;
639
+ if (!Array.isArray(content)) return undefined;
640
+ const parts: string[] = [];
641
+ for (const block of content) {
642
+ if (typeof block !== "object" || block === null) continue;
643
+ const type = Reflect.get(block, "type");
644
+ const text = Reflect.get(block, "text");
645
+ if (type === "text" && typeof text === "string") parts.push(text);
646
+ }
647
+ const joined = parts.join("\n").trim();
648
+ return joined || undefined;
649
+ }