@pi-in-go/pigpen-jev 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/CREDITS.md +22 -0
  2. package/LICENSE +22 -0
  3. package/README.md +237 -0
  4. package/extensions/jev/ask.go +166 -0
  5. package/extensions/jev/ask_test.go +218 -0
  6. package/extensions/jev/backend.go +128 -0
  7. package/extensions/jev/bench_test.go +64 -0
  8. package/extensions/jev/boundaries_test.go +159 -0
  9. package/extensions/jev/command.go +224 -0
  10. package/extensions/jev/commands_test.go +214 -0
  11. package/extensions/jev/config.go +450 -0
  12. package/extensions/jev/errors_test.go +191 -0
  13. package/extensions/jev/extension.go +391 -0
  14. package/extensions/jev/fakehost_test.go +548 -0
  15. package/extensions/jev/gate.go +125 -0
  16. package/extensions/jev/gate_test.go +610 -0
  17. package/extensions/jev/gatekey_test.go +24 -0
  18. package/extensions/jev/go.mod +9 -0
  19. package/extensions/jev/go.sum +2 -0
  20. package/extensions/jev/go.work +10 -0
  21. package/extensions/jev/helpers_test.go +404 -0
  22. package/extensions/jev/memo.go +88 -0
  23. package/extensions/jev/output.go +89 -0
  24. package/extensions/jev/output_test.go +187 -0
  25. package/extensions/jev/ownmodel_test.go +118 -0
  26. package/extensions/jev/render.go +136 -0
  27. package/extensions/jev/review_test.go +310 -0
  28. package/extensions/jev/source_test.go +57 -0
  29. package/extensions/jev/text.go +174 -0
  30. package/extensions/jev/trust_test.go +335 -0
  31. package/extensions/jev/types.go +227 -0
  32. package/libs/typesafe/CONTRACT.md +125 -0
  33. package/libs/typesafe/CREDITS.md +37 -0
  34. package/libs/typesafe/LICENSE +23 -0
  35. package/libs/typesafe/README.md +19 -0
  36. package/libs/typesafe/go.mod +3 -0
  37. package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
  38. package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
  39. package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
  40. package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
  41. package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
  42. package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
  43. package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
  44. package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
  45. package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
  46. package/libs/typesafe/libraries/ownmodel/run.go +288 -0
  47. package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
  48. package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
  49. package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
  50. package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
  51. package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
  52. package/libs/typesafe/libraries/typesafe/answers.go +268 -0
  53. package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
  54. package/libs/typesafe/libraries/typesafe/batch.go +80 -0
  55. package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
  56. package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
  57. package/libs/typesafe/libraries/typesafe/client.go +561 -0
  58. package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
  59. package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
  60. package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
  61. package/libs/typesafe/libraries/typesafe/doc.go +27 -0
  62. package/libs/typesafe/libraries/typesafe/entry.go +142 -0
  63. package/libs/typesafe/libraries/typesafe/env.go +11 -0
  64. package/libs/typesafe/libraries/typesafe/errors.go +310 -0
  65. package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
  66. package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
  67. package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
  68. package/libs/typesafe/libraries/typesafe/logging.go +160 -0
  69. package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
  70. package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
  71. package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
  72. package/libs/typesafe/libraries/typesafe/questions.go +490 -0
  73. package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
  74. package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
  75. package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
  76. package/libs/typesafe/libraries/typesafe/retry.go +350 -0
  77. package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
  78. package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
  79. package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
  80. package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
  81. package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
  82. package/libs/typesafe/libraries/typesafe/version.go +10 -0
  83. package/libs/typesafe/package.json +37 -0
  84. package/libs/typesafe/provenance.json +49 -0
  85. package/package.json +42 -0
  86. package/port/PORT.md +107 -0
  87. package/port/e2e/gate-and-output.py +35 -0
  88. package/port/e2e/jev-ask.py +36 -0
  89. package/port/e2e/model-switch.py +44 -0
  90. package/port/e2e/off-by-default.py +34 -0
  91. package/port/gen-scenarios.py +103 -0
  92. package/port/golden/cache-identical-calls.jsonl +30 -0
  93. package/port/golden/clear.jsonl +22 -0
  94. package/port/golden/commands.jsonl +43 -0
  95. package/port/golden/enforce-accept.jsonl +23 -0
  96. package/port/golden/enforce-decline.jsonl +22 -0
  97. package/port/golden/jev-ask.jsonl +20 -0
  98. package/port/golden/output-advice.jsonl +23 -0
  99. package/port/golden/output-leak.jsonl +24 -0
  100. package/port/golden/output-low-confidence.jsonl +22 -0
  101. package/port/golden/shadow-flagged.jsonl +23 -0
  102. package/port/golden/unjudged-tools.jsonl +19 -0
  103. package/port/golden/write-elision.jsonl +21 -0
  104. package/port/mutate-unit.py +63 -0
  105. package/port/mutations.json +578 -0
  106. package/port/oracle/LICENSE +21 -0
  107. package/port/oracle/README.md +181 -0
  108. package/port/oracle/SHA256SUMS +8 -0
  109. package/port/oracle/package.json +43 -0
  110. package/port/oracle/src/client.ts +409 -0
  111. package/port/oracle/src/config.ts +363 -0
  112. package/port/oracle/src/gate.ts +229 -0
  113. package/port/oracle/src/index.ts +649 -0
  114. package/port/oracle/src/output.ts +163 -0
  115. package/port/red-run.log +309 -0
  116. package/port/scenarios/cache-identical-calls.json +71 -0
  117. package/port/scenarios/clear.json +61 -0
  118. package/port/scenarios/commands.json +119 -0
  119. package/port/scenarios/enforce-accept.json +66 -0
  120. package/port/scenarios/enforce-decline.json +57 -0
  121. package/port/scenarios/jev-ask.json +83 -0
  122. package/port/scenarios/output-advice.json +61 -0
  123. package/port/scenarios/output-leak.json +61 -0
  124. package/port/scenarios/output-low-confidence.json +61 -0
  125. package/port/scenarios/shadow-flagged.json +61 -0
  126. package/port/scenarios/unjudged-tools.json +55 -0
  127. package/port/scenarios/write-elision.json +53 -0
  128. package/provenance.json +18 -0
@@ -0,0 +1,404 @@
1
+ package jev_test
2
+
3
+ import (
4
+ "encoding/json"
5
+ "io"
6
+ "net/http"
7
+ "net/http/httptest"
8
+ "os"
9
+ "path/filepath"
10
+ "strings"
11
+ "sync"
12
+ "testing"
13
+ "time"
14
+
15
+ jev "github.com/MichaelKinsy/pigpen/jev"
16
+ )
17
+
18
+ // ---- a fake Jev HTTP endpoint ------------------------------------------------
19
+
20
+ // jevReply is the scripted reply to one request.
21
+ type jevReply struct {
22
+ status int
23
+ body any // marshalled unless it is a string
24
+ delay time.Duration
25
+ }
26
+
27
+ type recordedRequest struct {
28
+ Path string
29
+ Auth string
30
+ Body map[string]any
31
+ Raw string
32
+ }
33
+
34
+ type fakeJev struct {
35
+ srv *httptest.Server
36
+ mu sync.Mutex
37
+ got []recordedRequest
38
+ // next decides the reply for request number i (0-based).
39
+ next func(i int, req recordedRequest) jevReply
40
+ }
41
+
42
+ func newFakeJev(t *testing.T, next func(i int, req recordedRequest) jevReply) *fakeJev {
43
+ t.Helper()
44
+ f := &fakeJev{next: next}
45
+ f.srv = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
46
+ raw, _ := io.ReadAll(r.Body)
47
+ rec := recordedRequest{Path: r.URL.Path, Auth: r.Header.Get("Authorization"), Raw: string(raw)}
48
+ _ = json.Unmarshal(raw, &rec.Body)
49
+ f.mu.Lock()
50
+ i := len(f.got)
51
+ f.got = append(f.got, rec)
52
+ f.mu.Unlock()
53
+ reply := f.next(i, rec)
54
+ if reply.delay > 0 {
55
+ time.Sleep(reply.delay)
56
+ }
57
+ if reply.status == 0 {
58
+ reply.status = 200
59
+ }
60
+ w.WriteHeader(reply.status)
61
+ if s, ok := reply.body.(string); ok {
62
+ _, _ = io.WriteString(w, s)
63
+ return
64
+ }
65
+ _ = json.NewEncoder(w).Encode(reply.body)
66
+ }))
67
+ t.Cleanup(f.srv.Close)
68
+ return f
69
+ }
70
+
71
+ func (f *fakeJev) requests() []recordedRequest {
72
+ f.mu.Lock()
73
+ defer f.mu.Unlock()
74
+ return append([]recordedRequest(nil), f.got...)
75
+ }
76
+
77
+ func (f *fakeJev) count() int { return len(f.requests()) }
78
+
79
+ // always replies with the same body.
80
+ func always(body any) func(int, recordedRequest) jevReply {
81
+ return func(int, recordedRequest) jevReply { return jevReply{body: body} }
82
+ }
83
+
84
+ func noul(v float64) map[string]any { return map[string]any{"type": "noul", "noul": v} }
85
+
86
+ func score(v, conf float64) map[string]any {
87
+ return map[string]any{"type": "score", "score": v, "confidence": conf,
88
+ "legend": map[string]any{"0": "a", "1": "b", "2": "c", "3": "d"}, "probabilities": map[string]any{"0": 0.1, "1": 0.2, "2": 0.3, "3": 0.4}}
89
+ }
90
+
91
+ func choice(name string, conf float64) map[string]any {
92
+ return map[string]any{"type": "choice", "choice": name, "confidence": conf, "probabilities": map[string]any{name: conf}}
93
+ }
94
+
95
+ // gateBody is a Jev response for the four gate questions.
96
+ func gateBody(destructive, exfil, beyond, impact, impactConf float64) map[string]any {
97
+ return map[string]any{"model": "jev-test", "answers": map[string]any{
98
+ "destructive": noul(destructive), "exfiltration": noul(exfil), "beyond_scope": noul(beyond), "impact": score(impact, impactConf),
99
+ }, "usage": map[string]any{"input_tokens": 10, "output_tokens": 2}}
100
+ }
101
+
102
+ func clearGate() map[string]any { return gateBody(0.03, 0.04, 0.4, 0.02, 0.9) }
103
+ func flaggedGate() map[string]any { return gateBody(0.99, 0.79, 0.98, 3.0, 0.91) }
104
+
105
+ func outBody(leak float64, class string, conf float64) map[string]any {
106
+ return map[string]any{"model": "jev-test", "answers": map[string]any{
107
+ "leaks_secret": noul(leak), "failure_class": choice(class, conf),
108
+ }}
109
+ }
110
+
111
+ // ---- environment ---------------------------------------------------------------
112
+
113
+ const testKey = "tsk-test-key-0123456789"
114
+
115
+ type env struct {
116
+ agent string
117
+ cwd string
118
+ }
119
+
120
+ // newEnv points the agent config directory at a temp dir, clears the API key
121
+ // variable and returns the directories. Tests are not parallel: the extension
122
+ // reads process environment like the real one.
123
+ func newEnv(t *testing.T) *env {
124
+ t.Helper()
125
+ agent := t.TempDir()
126
+ t.Setenv("PIG_CODING_AGENT_DIR", agent)
127
+ t.Setenv("PI_CODING_AGENT_DIR", agent)
128
+ t.Setenv("TYPESAFE_API_KEY", "")
129
+ t.Setenv("HOME", t.TempDir())
130
+ return &env{agent: agent, cwd: t.TempDir()}
131
+ }
132
+
133
+ func (e *env) writeGlobal(t *testing.T, cfg map[string]any) {
134
+ t.Helper()
135
+ b, _ := json.Marshal(cfg)
136
+ if err := os.WriteFile(filepath.Join(e.agent, "pi-jev.json"), b, 0o600); err != nil {
137
+ t.Fatal(err)
138
+ }
139
+ }
140
+
141
+ func (e *env) writeProject(t *testing.T, cfg map[string]any) {
142
+ t.Helper()
143
+ dir := filepath.Join(e.cwd, ".pig")
144
+ if err := os.MkdirAll(dir, 0o755); err != nil {
145
+ t.Fatal(err)
146
+ }
147
+ b, _ := json.Marshal(cfg)
148
+ if err := os.WriteFile(filepath.Join(dir, "pi-jev.json"), b, 0o600); err != nil {
149
+ t.Fatal(err)
150
+ }
151
+ }
152
+
153
+ // jevConfig is a global config that opts in to the Jev HTTP backend against srv.
154
+ // display "plain" reproduces the original's wording; the disclosure banner is
155
+ // acknowledged so a trace holds only what the original also produced.
156
+ func jevConfig(srv *fakeJev, extra map[string]any) map[string]any {
157
+ cfg := map[string]any{
158
+ "enabled": true, "acknowledged": true, "display": "plain", "backend": "typesafe",
159
+ "endpoint": srv.srv.URL, "model": "jev-test", "retries": 0, "timeoutMs": 5000,
160
+ }
161
+ for k, v := range extra {
162
+ cfg[k] = v
163
+ }
164
+ return cfg
165
+ }
166
+
167
+ // hostState is what the fake host answers for the calls the extension makes.
168
+ type hostState struct {
169
+ mu sync.Mutex
170
+ confirm bool
171
+ user string // initial user message in the session branch
172
+ entries []map[string]any
173
+ model map[string]any
174
+ completes []map[string]any
175
+ complete func(model, request map[string]any) map[string]any
176
+ confirms []map[string]any
177
+ }
178
+
179
+ func newHostState() *hostState {
180
+ return &hostState{model: map[string]any{"id": "judge-1", "name": "judge-1", "provider": "acme"}}
181
+ }
182
+
183
+ func (s *hostState) onCall(method string, args map[string]any) (map[string]any, string) {
184
+ s.mu.Lock()
185
+ defer s.mu.Unlock()
186
+ switch method {
187
+ case "ui.confirm":
188
+ s.confirms = append(s.confirms, args)
189
+ return map[string]any{"confirmed": s.confirm}, ""
190
+ case "getModelInfo":
191
+ return s.model, ""
192
+ case "getModel":
193
+ return map[string]any{"provider": args["provider"], "id": args["modelId"], "api": "openai-completions"}, ""
194
+ case "getModelAuth":
195
+ return map[string]any{"ok": true, "apiKey": "host-held-key"}, ""
196
+ case "watchSessionLog":
197
+ if len(s.entries) == 0 && s.user != "" {
198
+ s.entries = append(s.entries, userEntry("e1", "", s.user))
199
+ }
200
+ cursor, _ := args["cursor"].(float64)
201
+ if int(cursor) > len(s.entries) {
202
+ cursor = float64(len(s.entries))
203
+ }
204
+ leaf := ""
205
+ if n := len(s.entries); n > 0 {
206
+ leaf, _ = s.entries[n-1]["id"].(string)
207
+ }
208
+ return map[string]any{"entries": s.entries[int(cursor):], "entryCount": len(s.entries), "hasMore": false, "leafId": leaf}, ""
209
+ case "complete":
210
+ s.completes = append(s.completes, args)
211
+ if s.complete != nil {
212
+ m, _ := args["model"].(map[string]any)
213
+ r, _ := args["request"].(map[string]any)
214
+ return s.complete(m, r), ""
215
+ }
216
+ return nil, "no completion scripted"
217
+ }
218
+ return nil, ""
219
+ }
220
+
221
+ func (s *hostState) completeCalls() []map[string]any {
222
+ s.mu.Lock()
223
+ defer s.mu.Unlock()
224
+ return append([]map[string]any(nil), s.completes...)
225
+ }
226
+
227
+ func (s *hostState) confirmCalls() []map[string]any {
228
+ s.mu.Lock()
229
+ defer s.mu.Unlock()
230
+ return append([]map[string]any(nil), s.confirms...)
231
+ }
232
+
233
+ func textMessage(text string) map[string]any {
234
+ return map[string]any{"role": "assistant", "stopReason": "stop", "content": []any{map[string]any{"type": "text", "text": text}}}
235
+ }
236
+
237
+ // start builds the host, fires session_start and returns.
238
+ func start(t *testing.T, e *env, hs *hostState, opts HostOptions) *Host {
239
+ t.Helper()
240
+ opts.Cwd = e.cwd
241
+ if opts.OnCall == nil {
242
+ opts.OnCall = hs.onCall
243
+ }
244
+ h := StartHost(t, jev.Extension(), opts)
245
+ h.Fire("session_start", map[string]any{"reason": "startup"})
246
+ return h
247
+ }
248
+
249
+ // ---- reading what the extension did --------------------------------------------
250
+
251
+ func (h *Host) notifications() []string {
252
+ var out []string
253
+ for _, c := range h.CallsTo("ui.notify") {
254
+ msg, _ := c.Args["message"].(string)
255
+ level, _ := c.Args["level"].(string)
256
+ out = append(out, level+": "+msg)
257
+ }
258
+ return out
259
+ }
260
+
261
+ func (h *Host) statuses() []string {
262
+ var out []string
263
+ for _, c := range h.CallsTo("ui.setStatus") {
264
+ text, _ := c.Args["text"].(string)
265
+ out = append(out, text)
266
+ }
267
+ return out
268
+ }
269
+
270
+ func (h *Host) lastStatus() string {
271
+ s := h.statuses()
272
+ if len(s) == 0 {
273
+ return "<none>"
274
+ }
275
+ return s[len(s)-1]
276
+ }
277
+
278
+ func (h *Host) anyNotification(substr string) bool {
279
+ for _, n := range h.notifications() {
280
+ if strings.Contains(n, substr) {
281
+ return true
282
+ }
283
+ }
284
+ return false
285
+ }
286
+
287
+ // toolCall fires a tool_call event and decodes the {block, reason} result.
288
+ func (h *Host) toolCall(name string, input map[string]any) (block bool, reason string) {
289
+ h.t.Helper()
290
+ raw := h.Fire("tool_call", map[string]any{"toolName": name, "toolCallId": "c1", "input": input})
291
+ if len(raw) == 0 || string(raw) == "null" {
292
+ return false, ""
293
+ }
294
+ var r struct {
295
+ Block bool `json:"block"`
296
+ Reason string `json:"reason"`
297
+ }
298
+ if err := json.Unmarshal(raw, &r); err != nil {
299
+ h.t.Fatalf("tool_call result %s: %v", raw, err)
300
+ }
301
+ return r.Block, r.Reason
302
+ }
303
+
304
+ // toolResult fires tool_result and returns the appended text blocks ("" when the
305
+ // handler left the result alone).
306
+ func (h *Host) toolResult(name string, input map[string]any, content []any, isError bool) (patched []string, none bool) {
307
+ h.t.Helper()
308
+ raw := h.Fire("tool_result", map[string]any{"toolName": name, "toolCallId": "c1", "input": input, "content": content, "isError": isError})
309
+ if len(raw) == 0 || string(raw) == "null" {
310
+ return nil, true
311
+ }
312
+ var r struct {
313
+ Content []struct {
314
+ Type string `json:"type"`
315
+ Text string `json:"text"`
316
+ } `json:"content"`
317
+ }
318
+ if err := json.Unmarshal(raw, &r); err != nil {
319
+ h.t.Fatalf("tool_result result %s: %v", raw, err)
320
+ }
321
+ if r.Content == nil {
322
+ return nil, true
323
+ }
324
+ for _, c := range r.Content {
325
+ patched = append(patched, c.Text)
326
+ }
327
+ return patched, false
328
+ }
329
+
330
+ func text(s string) []any { return []any{map[string]any{"type": "text", "text": s}} }
331
+
332
+ // stateOf returns the decoded "state" of a recorded gate request.
333
+ func stateOf(t *testing.T, r recordedRequest) map[string]any {
334
+ t.Helper()
335
+ st, ok := r.Body["state"].(map[string]any)
336
+ if !ok {
337
+ t.Fatalf("request state is not an object: %s", r.Raw)
338
+ }
339
+ return st
340
+ }
341
+
342
+ func mustContain(t *testing.T, what, got string, subs ...string) {
343
+ t.Helper()
344
+ for _, s := range subs {
345
+ if !strings.Contains(got, s) {
346
+ t.Errorf("%s = %q, want it to contain %q", what, got, s)
347
+ }
348
+ }
349
+ }
350
+
351
+ func userEntry(id, parent, text string) map[string]any {
352
+ e := map[string]any{"type": "message", "id": id, "message": map[string]any{
353
+ "role": "user", "content": []any{map[string]any{"type": "text", "text": text}}}}
354
+ if parent != "" {
355
+ e["parentId"] = parent
356
+ }
357
+ return e
358
+ }
359
+
360
+ // pushUser appends a user message to the session the way the host does after the
361
+ // extension subscribed: a state_update notification with the appended entry.
362
+ func (h *Host) pushUser(hs *hostState, text string) {
363
+ h.t.Helper()
364
+ hs.mu.Lock()
365
+ parent := ""
366
+ if n := len(hs.entries); n > 0 {
367
+ parent, _ = hs.entries[n-1]["id"].(string)
368
+ }
369
+ id := "e" + string(rune('1'+len(hs.entries)))
370
+ entry := userEntry(id, parent, text)
371
+ hs.entries = append(hs.entries, entry)
372
+ count := len(hs.entries)
373
+ hs.mu.Unlock()
374
+ // Make sure the extension has subscribed (first read) before the push.
375
+ session, _ := json.Marshal(map[string]any{"leafId": id, "entriesAppended": []any{entry}, "entryCount": count})
376
+ h.write(map[string]any{"type": "notify", "notify": map[string]any{"method": "state_update", "args": map[string]any{"state": map[string]any{"session": json.RawMessage(session)}}}})
377
+ }
378
+
379
+ // first returns the first recorded request or fails the test (a red run must
380
+ // report a missing request, not panic on an index).
381
+ func (f *fakeJev) first(t *testing.T) recordedRequest {
382
+ t.Helper()
383
+ r := f.requests()
384
+ if len(r) == 0 {
385
+ t.Fatal("the extension made no request")
386
+ }
387
+ return r[0]
388
+ }
389
+
390
+ // adoptDynamicTools tells the fake host about tools the extension registered at
391
+ // run time (registerTool), which the host would otherwise learn from the register frame.
392
+ func (h *Host) adoptDynamicTools() {
393
+ h.mu.Lock()
394
+ defer h.mu.Unlock()
395
+ for _, c := range h.calls {
396
+ if c.Method == "registerTool" {
397
+ if name, _ := c.Args["name"].(string); name != "" {
398
+ h.tools[name] = true
399
+ }
400
+ }
401
+ }
402
+ }
403
+
404
+ func sleepMs(n int) { time.Sleep(time.Duration(n) * time.Millisecond) }
@@ -0,0 +1,88 @@
1
+ package jev
2
+
3
+ import (
4
+ "sync"
5
+ "time"
6
+ )
7
+
8
+ const memoLimit = 64
9
+
10
+ // memo caches responses for a window and lets identical concurrent requests share
11
+ // one in-flight call (sibling tool calls of one assistant message). Errors are
12
+ // never cached. The cache stores the raw response, so a changed threshold applies
13
+ // to it instead of being baked into an old verdict.
14
+ type memo struct {
15
+ mu sync.Mutex
16
+ entries map[string]memoEntry
17
+ order []string
18
+ flights map[string]*flight
19
+ }
20
+
21
+ type memoEntry struct {
22
+ at time.Time
23
+ resp *response
24
+ }
25
+
26
+ type flight struct {
27
+ done chan struct{}
28
+ resp *response
29
+ err error
30
+ }
31
+
32
+ func newMemo() *memo { return &memo{entries: map[string]memoEntry{}, flights: map[string]*flight{}} }
33
+
34
+ func (m *memo) do(key string, ttl time.Duration, now func() time.Time, fn func() (*response, error)) (*response, error) {
35
+ m.mu.Lock()
36
+ if e, ok := m.entries[key]; ok && ttl > 0 && now().Sub(e.at) <= ttl {
37
+ m.mu.Unlock()
38
+ return e.resp, nil
39
+ }
40
+ if f, ok := m.flights[key]; ok {
41
+ m.mu.Unlock()
42
+ <-f.done
43
+ return f.resp, f.err
44
+ }
45
+ f := &flight{done: make(chan struct{})}
46
+ m.flights[key] = f
47
+ m.mu.Unlock()
48
+
49
+ f.resp, f.err = fn()
50
+
51
+ m.mu.Lock()
52
+ delete(m.flights, key)
53
+ if f.err == nil && ttl > 0 {
54
+ if _, ok := m.entries[key]; !ok {
55
+ m.order = append(m.order, key)
56
+ }
57
+ m.entries[key] = memoEntry{now(), f.resp}
58
+ m.prune(ttl, now())
59
+ }
60
+ m.mu.Unlock()
61
+ close(f.done)
62
+ return f.resp, f.err
63
+ }
64
+
65
+ func (m *memo) prune(ttl time.Duration, at time.Time) {
66
+ if len(m.entries) <= memoLimit {
67
+ return
68
+ }
69
+ kept := m.order[:0]
70
+ for _, k := range m.order {
71
+ if at.Sub(m.entries[k].at) > ttl {
72
+ delete(m.entries, k)
73
+ } else {
74
+ kept = append(kept, k)
75
+ }
76
+ }
77
+ m.order = kept
78
+ for len(m.entries) > memoLimit && len(m.order) > 0 {
79
+ delete(m.entries, m.order[0])
80
+ m.order = m.order[1:]
81
+ }
82
+ }
83
+
84
+ func (m *memo) size() int {
85
+ m.mu.Lock()
86
+ defer m.mu.Unlock()
87
+ return len(m.entries)
88
+ }
@@ -0,0 +1,89 @@
1
+ package jev
2
+
3
+ import (
4
+ "fmt"
5
+ "strings"
6
+ )
7
+
8
+ // The output judge asks about text that exists only after the call: a credential
9
+ // echoed into the transcript, and what kind of failure a result reports. The leak
10
+ // question separated 0.92-0.99 from 0.01-0.02 upstream with no overlap; the class
11
+ // answered at 0.88-1.00 when right and 0.42 when unsure, which is why a
12
+ // confidence floor silences the advice.
13
+ var outputQuestions = []question{
14
+ {ID: "leaks_secret", Type: "noul",
15
+ Instructions: "Does `output` contain a secret or credential that must not be written to a session transcript?",
16
+ CriteriaTrue: "A key, token, password, or private key appears in the text", CriteriaFalse: "Only ordinary program output"},
17
+ {ID: "failure_class", Type: "choice", Instructions: "What kind of failure is `output` reporting?",
18
+ Options: []option{
19
+ {"transient", desc("A network or resource hiccup that may succeed on a retry")},
20
+ {"environment", desc("A missing dependency, port, or tool in the local setup")},
21
+ {"code_bug", desc("The code or types are wrong")},
22
+ {"permission", desc("Access was denied by the OS or a server")},
23
+ {"user_error", desc("The command itself was invoked wrongly")},
24
+ {"no_failure", desc("Output reports success or nothing wrong")},
25
+ }},
26
+ }
27
+
28
+ // classAdvice is a table, not a branch: the class names are Jev's, and adding
29
+ // one is a row rather than a code path.
30
+ var classAdvice = map[string]string{
31
+ "transient": "retrying the same command unchanged is reasonable",
32
+ "environment": "fix the environment (missing tool, port, or service) before retrying",
33
+ "code_bug": "fix the code or types; retrying unchanged will not help",
34
+ "permission": "access was denied; change what is being accessed or ask the user",
35
+ "user_error": "the invocation itself was wrong; fix the command",
36
+ }
37
+
38
+ const outputArgChars = 400
39
+
40
+ func outputStateJSON(cwd, tool string, input any, output string, isError bool, outputChars, maxState int) []byte {
41
+ build := func(argChars, outChars int, dropArgs, _ bool) []byte {
42
+ var b strings.Builder
43
+ b.WriteString(`{"cwd":`)
44
+ b.Write(jsonString(cwd))
45
+ b.WriteString(`,"tool":`)
46
+ b.Write(jsonString(tool))
47
+ fmt.Fprintf(&b, `,"is_error":%t,"arguments":`, isError)
48
+ if dropArgs {
49
+ b.Write(jsonString("…[elided to fit maxStateChars]"))
50
+ } else {
51
+ b.Write(jsonValue(summarize(input, argChars, 0)))
52
+ }
53
+ b.WriteString(`,"output":`)
54
+ b.Write(jsonString(elide(output, outChars)))
55
+ b.WriteByte('}')
56
+ return []byte(b.String())
57
+ }
58
+ return fitState(maxState, outputArgChars, outputChars, build)
59
+ }
60
+
61
+ type outputVerdict struct {
62
+ LeaksSecret float64
63
+ FailureClass string
64
+ ClassConfidence float64
65
+ HasClass bool
66
+ Notice string
67
+ Kind string // leak, advice, none
68
+ Resp *response
69
+ }
70
+
71
+ func evaluateOutput(r *response, c config) outputVerdict {
72
+ v := outputVerdict{Kind: "none", Resp: r}
73
+ v.LeaksSecret = r.Answers["leaks_secret"].Noul
74
+ cls := r.Answers["failure_class"]
75
+ v.FailureClass, v.ClassConfidence, v.HasClass = cls.Choice, cls.Confidence, cls.HasConfidence
76
+ switch {
77
+ case v.LeaksSecret >= c.Output.LeakThreshold:
78
+ v.Kind = "leak"
79
+ v.Notice = fmt.Sprintf("Jev flagged this output as containing a secret (%s). Do not repeat the value in a reply, a file, or a command; refer to it by name instead.", toFixed2(v.LeaksSecret))
80
+ case v.FailureClass != "" && classAdvice[v.FailureClass] != "" && (!v.HasClass || v.ClassConfidence >= c.Output.MinConfidence):
81
+ v.Kind = "advice"
82
+ conf := "n/a"
83
+ if v.HasClass {
84
+ conf = toFixed2(v.ClassConfidence)
85
+ }
86
+ v.Notice = fmt.Sprintf("Jev read this as a %s failure (confidence %s): %s.", v.FailureClass, conf, classAdvice[v.FailureClass])
87
+ }
88
+ return v
89
+ }