@pi-in-go/pigpen-jev 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CREDITS.md +22 -0
- package/LICENSE +22 -0
- package/README.md +237 -0
- package/extensions/jev/ask.go +166 -0
- package/extensions/jev/ask_test.go +218 -0
- package/extensions/jev/backend.go +128 -0
- package/extensions/jev/bench_test.go +64 -0
- package/extensions/jev/boundaries_test.go +159 -0
- package/extensions/jev/command.go +224 -0
- package/extensions/jev/commands_test.go +214 -0
- package/extensions/jev/config.go +450 -0
- package/extensions/jev/errors_test.go +191 -0
- package/extensions/jev/extension.go +391 -0
- package/extensions/jev/fakehost_test.go +548 -0
- package/extensions/jev/gate.go +125 -0
- package/extensions/jev/gate_test.go +610 -0
- package/extensions/jev/gatekey_test.go +24 -0
- package/extensions/jev/go.mod +9 -0
- package/extensions/jev/go.sum +2 -0
- package/extensions/jev/go.work +10 -0
- package/extensions/jev/helpers_test.go +404 -0
- package/extensions/jev/memo.go +88 -0
- package/extensions/jev/output.go +89 -0
- package/extensions/jev/output_test.go +187 -0
- package/extensions/jev/ownmodel_test.go +118 -0
- package/extensions/jev/render.go +136 -0
- package/extensions/jev/review_test.go +310 -0
- package/extensions/jev/source_test.go +57 -0
- package/extensions/jev/text.go +174 -0
- package/extensions/jev/trust_test.go +335 -0
- package/extensions/jev/types.go +227 -0
- package/libs/typesafe/CONTRACT.md +125 -0
- package/libs/typesafe/CREDITS.md +37 -0
- package/libs/typesafe/LICENSE +23 -0
- package/libs/typesafe/README.md +19 -0
- package/libs/typesafe/go.mod +3 -0
- package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
- package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
- package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
- package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
- package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
- package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
- package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
- package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
- package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
- package/libs/typesafe/libraries/ownmodel/run.go +288 -0
- package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
- package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
- package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
- package/libs/typesafe/libraries/typesafe/answers.go +268 -0
- package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
- package/libs/typesafe/libraries/typesafe/batch.go +80 -0
- package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
- package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
- package/libs/typesafe/libraries/typesafe/client.go +561 -0
- package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
- package/libs/typesafe/libraries/typesafe/doc.go +27 -0
- package/libs/typesafe/libraries/typesafe/entry.go +142 -0
- package/libs/typesafe/libraries/typesafe/env.go +11 -0
- package/libs/typesafe/libraries/typesafe/errors.go +310 -0
- package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
- package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
- package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
- package/libs/typesafe/libraries/typesafe/logging.go +160 -0
- package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
- package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
- package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
- package/libs/typesafe/libraries/typesafe/questions.go +490 -0
- package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
- package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
- package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
- package/libs/typesafe/libraries/typesafe/retry.go +350 -0
- package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
- package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
- package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
- package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
- package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
- package/libs/typesafe/libraries/typesafe/version.go +10 -0
- package/libs/typesafe/package.json +37 -0
- package/libs/typesafe/provenance.json +49 -0
- package/package.json +42 -0
- package/port/PORT.md +107 -0
- package/port/e2e/gate-and-output.py +35 -0
- package/port/e2e/jev-ask.py +36 -0
- package/port/e2e/model-switch.py +44 -0
- package/port/e2e/off-by-default.py +34 -0
- package/port/gen-scenarios.py +103 -0
- package/port/golden/cache-identical-calls.jsonl +30 -0
- package/port/golden/clear.jsonl +22 -0
- package/port/golden/commands.jsonl +43 -0
- package/port/golden/enforce-accept.jsonl +23 -0
- package/port/golden/enforce-decline.jsonl +22 -0
- package/port/golden/jev-ask.jsonl +20 -0
- package/port/golden/output-advice.jsonl +23 -0
- package/port/golden/output-leak.jsonl +24 -0
- package/port/golden/output-low-confidence.jsonl +22 -0
- package/port/golden/shadow-flagged.jsonl +23 -0
- package/port/golden/unjudged-tools.jsonl +19 -0
- package/port/golden/write-elision.jsonl +21 -0
- package/port/mutate-unit.py +63 -0
- package/port/mutations.json +578 -0
- package/port/oracle/LICENSE +21 -0
- package/port/oracle/README.md +181 -0
- package/port/oracle/SHA256SUMS +8 -0
- package/port/oracle/package.json +43 -0
- package/port/oracle/src/client.ts +409 -0
- package/port/oracle/src/config.ts +363 -0
- package/port/oracle/src/gate.ts +229 -0
- package/port/oracle/src/index.ts +649 -0
- package/port/oracle/src/output.ts +163 -0
- package/port/red-run.log +309 -0
- package/port/scenarios/cache-identical-calls.json +71 -0
- package/port/scenarios/clear.json +61 -0
- package/port/scenarios/commands.json +119 -0
- package/port/scenarios/enforce-accept.json +66 -0
- package/port/scenarios/enforce-decline.json +57 -0
- package/port/scenarios/jev-ask.json +83 -0
- package/port/scenarios/output-advice.json +61 -0
- package/port/scenarios/output-leak.json +61 -0
- package/port/scenarios/output-low-confidence.json +61 -0
- package/port/scenarios/shadow-flagged.json +61 -0
- package/port/scenarios/unjudged-tools.json +55 -0
- package/port/scenarios/write-elision.json +53 -0
- package/provenance.json +18 -0
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
package jev_test
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"strings"
|
|
5
|
+
"testing"
|
|
6
|
+
"time"
|
|
7
|
+
)
|
|
8
|
+
|
|
9
|
+
// Failure policy: every error path fails open (README "Every error path fails
|
|
10
|
+
// open", AGENTS.md "Do not change an error path to block a tool call").
|
|
11
|
+
|
|
12
|
+
func errorHost(t *testing.T, next func(int, recordedRequest) jevReply, extra map[string]any) (*Host, *fakeJev) {
|
|
13
|
+
t.Helper()
|
|
14
|
+
e := newEnv(t)
|
|
15
|
+
t.Setenv("TYPESAFE_API_KEY", testKey)
|
|
16
|
+
srv := newFakeJev(t, next)
|
|
17
|
+
e.writeGlobal(t, jevConfig(srv, extra))
|
|
18
|
+
return start(t, e, newHostState(), HostOptions{}), srv
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
func TestFailOpen_ServerErrorNeverBlocks(t *testing.T) {
|
|
22
|
+
h, _ := errorHost(t, func(int, recordedRequest) jevReply { return jevReply{status: 500, body: "internal"} },
|
|
23
|
+
map[string]any{"gate": map[string]any{"mode": "enforce", "blockWithoutUI": true}})
|
|
24
|
+
if block, _ := h.toolCall("bash", bash("rm -rf /")); block {
|
|
25
|
+
t.Fatal("an unavailable judge blocked a tool call")
|
|
26
|
+
}
|
|
27
|
+
if !h.anyNotification("error: pi-jev: ") || !h.anyNotification("500") || !h.anyNotification("(failing open)") {
|
|
28
|
+
t.Errorf("notifications = %q", h.notifications())
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
func TestFailOpen_OutputJudgeLeavesResultAlone(t *testing.T) {
|
|
33
|
+
h, _ := errorHost(t, func(int, recordedRequest) jevReply { return jevReply{status: 503, body: "x"} }, nil)
|
|
34
|
+
if _, none := h.toolResult("bash", bash("x"), text("out"), false); !none {
|
|
35
|
+
t.Fatal("patched a result with no verdict")
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
func TestErrors_TimeoutFailsOpen(t *testing.T) {
|
|
40
|
+
h, _ := errorHost(t, func(int, recordedRequest) jevReply {
|
|
41
|
+
return jevReply{body: clearGate(), delay: 1500 * time.Millisecond}
|
|
42
|
+
}, map[string]any{"timeoutMs": 200, "retries": 0})
|
|
43
|
+
start := time.Now()
|
|
44
|
+
if block, _ := h.toolCall("bash", bash("ls")); block {
|
|
45
|
+
t.Fatal("blocked")
|
|
46
|
+
}
|
|
47
|
+
if time.Since(start) > 1200*time.Millisecond {
|
|
48
|
+
t.Errorf("did not time out: %v", time.Since(start))
|
|
49
|
+
}
|
|
50
|
+
if !h.anyNotification("timed out") || !h.anyNotification("(failing open)") {
|
|
51
|
+
t.Errorf("notifications = %q", h.notifications())
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
func TestErrors_ReportedAtMostOncePerMinute(t *testing.T) {
|
|
56
|
+
h, _ := errorHost(t, func(int, recordedRequest) jevReply { return jevReply{status: 500, body: "x"} }, nil)
|
|
57
|
+
for i := 0; i < 4; i++ {
|
|
58
|
+
h.toolCall("bash", bash("cmd "+string(rune('a'+i))))
|
|
59
|
+
}
|
|
60
|
+
n := 0
|
|
61
|
+
for _, note := range h.notifications() {
|
|
62
|
+
if strings.Contains(note, "failing open") {
|
|
63
|
+
n++
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
if n != 1 {
|
|
67
|
+
t.Errorf("error notifications = %d, want 1", n)
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
// Twin of the 0.2.2 fix (issue #3): a response that skips a question is an
|
|
72
|
+
// error, not a clear verdict.
|
|
73
|
+
func TestErrors_EmptyAnswersAreNotAClearVerdict(t *testing.T) {
|
|
74
|
+
h, _ := errorHost(t, always(map[string]any{"model": "m", "answers": map[string]any{}}), nil)
|
|
75
|
+
h.toolCall("bash", bash("rm -rf /"))
|
|
76
|
+
if !h.anyNotification(`did not answer "destructive"`) || !h.anyNotification("(failing open)") {
|
|
77
|
+
t.Errorf("notifications = %q", h.notifications())
|
|
78
|
+
}
|
|
79
|
+
if strings.Contains(h.lastStatus(), "clear") {
|
|
80
|
+
t.Errorf("status = %q, an empty answer must not read as clear", h.lastStatus())
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
func TestErrors_WrongAnswerTypeIsAnError(t *testing.T) {
|
|
85
|
+
body := gateBody(0.1, 0.1, 0.1, 1, 0.9)
|
|
86
|
+
body["answers"].(map[string]any)["destructive"] = score(1, 0.9)
|
|
87
|
+
h, _ := errorHost(t, always(body), nil)
|
|
88
|
+
h.toolCall("bash", bash("x"))
|
|
89
|
+
if !h.anyNotification(`"destructive" is not a noul`) {
|
|
90
|
+
t.Errorf("notifications = %q", h.notifications())
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
func TestErrors_MalformedBodiesFailOpenAndNeverRead(t *testing.T) {
|
|
95
|
+
for _, body := range []string{`{"model":"m"}`, `[1]`, `null`, `not json`} {
|
|
96
|
+
h, _ := errorHost(t, always(body), nil)
|
|
97
|
+
h.toolCall("bash", bash("x"))
|
|
98
|
+
if !h.anyNotification("(failing open)") {
|
|
99
|
+
t.Errorf("body %s: notifications = %q", body, h.notifications())
|
|
100
|
+
}
|
|
101
|
+
if strings.Contains(h.lastStatus(), "clear") {
|
|
102
|
+
t.Errorf("body %s read as clear: %q", body, h.lastStatus())
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// C5: the original accepts a probability of 1.7 or -2 and compares it to a threshold.
|
|
108
|
+
func TestCorrection_OutOfRangeProbabilityIsAnError(t *testing.T) {
|
|
109
|
+
for _, v := range []float64{1.7, -0.2} {
|
|
110
|
+
h, _ := errorHost(t, always(gateBody(v, 0.1, 0.1, 1, 0.9)), nil)
|
|
111
|
+
h.toolCall("bash", bash("x"))
|
|
112
|
+
if !h.anyNotification("outside 0 to 1") {
|
|
113
|
+
t.Errorf("noul %v accepted: %q", v, h.notifications())
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
func TestCorrection_ScoreOutsideTheRubricIsAnError(t *testing.T) {
|
|
119
|
+
h, _ := errorHost(t, always(gateBody(0.1, 0.1, 0.1, 9.0, 0.9)), nil)
|
|
120
|
+
h.toolCall("bash", bash("x"))
|
|
121
|
+
if !h.anyNotification(`"impact" score 9 is outside the 4 levels`) {
|
|
122
|
+
t.Errorf("notifications = %q", h.notifications())
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
func TestCorrection_ConfidenceOutsideZeroToOneIsAnError(t *testing.T) {
|
|
127
|
+
h, _ := errorHost(t, always(gateBody(0.1, 0.1, 0.1, 1, 4.0)), nil)
|
|
128
|
+
h.toolCall("bash", bash("x"))
|
|
129
|
+
if !h.anyNotification("outside 0 to 1") {
|
|
130
|
+
t.Errorf("notifications = %q", h.notifications())
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// C6: on an error the original leaves the previous "clear" status on screen.
|
|
135
|
+
func TestCorrection_ErrorStatusIsUnavailableNotStaleClear(t *testing.T) {
|
|
136
|
+
h, _ := errorHost(t, func(i int, _ recordedRequest) jevReply {
|
|
137
|
+
if i == 0 {
|
|
138
|
+
return jevReply{body: clearGate()}
|
|
139
|
+
}
|
|
140
|
+
return jevReply{status: 500, body: "x"}
|
|
141
|
+
}, nil)
|
|
142
|
+
h.toolCall("bash", bash("first"))
|
|
143
|
+
if h.lastStatus() != "jev: clear (shadow)" {
|
|
144
|
+
t.Fatalf("setup status %q", h.lastStatus())
|
|
145
|
+
}
|
|
146
|
+
h.toolCall("bash", bash("second"))
|
|
147
|
+
if got := h.lastStatus(); got != "jev: unavailable (failing open)" {
|
|
148
|
+
t.Errorf("status = %q, want jev: unavailable (failing open)", got)
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
func TestCorrection_OutputJudgeErrorAlsoShowsUnavailable(t *testing.T) {
|
|
153
|
+
h, _ := errorHost(t, func(int, recordedRequest) jevReply { return jevReply{status: 500, body: "x"} }, nil)
|
|
154
|
+
h.toolResult("bash", bash("x"), text("out"), false)
|
|
155
|
+
if got := h.lastStatus(); got != "jev: unavailable (failing open)" {
|
|
156
|
+
t.Errorf("status = %q", got)
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
func TestErrors_ApiKeyIsRedactedFromNotifications(t *testing.T) {
|
|
161
|
+
h, _ := errorHost(t, func(int, recordedRequest) jevReply { return jevReply{status: 400, body: "bad key " + testKey} }, nil)
|
|
162
|
+
h.toolCall("bash", bash("x"))
|
|
163
|
+
for _, n := range h.notifications() {
|
|
164
|
+
if strings.Contains(n, testKey) {
|
|
165
|
+
t.Fatalf("API key leaked into a notification: %q", n)
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
if !h.anyNotification("[redacted]") {
|
|
169
|
+
t.Errorf("notifications = %q", h.notifications())
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
func TestErrors_QuestionsAreValidatedBeforeAnyRequest(t *testing.T) {
|
|
174
|
+
// A choice with no option is refused locally by jev_ask (see ask_test.go); the
|
|
175
|
+
// gate's own questions are fixed, so this checks that no request is made for
|
|
176
|
+
// an unparsable input.
|
|
177
|
+
h, srv := errorHost(t, always(clearGate()), nil)
|
|
178
|
+
h.toolCall("read", map[string]any{})
|
|
179
|
+
if srv.count() != 0 {
|
|
180
|
+
t.Fatal("request for an unjudged tool")
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
func TestCorrection_ChoiceConfidenceOutsideZeroToOneIsAnError(t *testing.T) {
|
|
185
|
+
body := outBody(0.01, "transient", 7.0)
|
|
186
|
+
h, _ := errorHost(t, always(body), nil)
|
|
187
|
+
h.toolResult("bash", bash("x"), text("out"), false)
|
|
188
|
+
if !h.anyNotification("outside 0 to 1") {
|
|
189
|
+
t.Errorf("notifications = %q", h.notifications())
|
|
190
|
+
}
|
|
191
|
+
}
|
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
// Package jev is a Go port of y0usaf/pi-jev 0.2.2 (commit 88e5fb3): a typed
|
|
2
|
+
// decision layer for the coding agent. A gate judges bash, write and edit calls
|
|
3
|
+
// before they run; an output judge reads what bash printed; jev_ask lets the model
|
|
4
|
+
// ask typed questions itself.
|
|
5
|
+
//
|
|
6
|
+
// It differs from the original where the roadmap's code review found defects, and
|
|
7
|
+
// where the owner's rules require it: it judges nothing until the user opts in,
|
|
8
|
+
// says what leaves the machine, only the user's own config chooses the destination
|
|
9
|
+
// of the API key, and the judge is either the TypeSafe API or the model PiG is
|
|
10
|
+
// configured with. Every difference is listed in port/PORT.md with a test.
|
|
11
|
+
package jev
|
|
12
|
+
|
|
13
|
+
import (
|
|
14
|
+
"context"
|
|
15
|
+
"crypto/sha256"
|
|
16
|
+
"encoding/hex"
|
|
17
|
+
"encoding/json"
|
|
18
|
+
"fmt"
|
|
19
|
+
"slices"
|
|
20
|
+
"strings"
|
|
21
|
+
"sync"
|
|
22
|
+
"time"
|
|
23
|
+
|
|
24
|
+
sdk "github.com/MichaelKinsy/PiG/extensions/sdk"
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
const errorNotifyInterval = time.Minute
|
|
28
|
+
|
|
29
|
+
// outputCacheSeconds: identical output is judged once per window.
|
|
30
|
+
const outputCacheSeconds = 120
|
|
31
|
+
|
|
32
|
+
type lastGate struct {
|
|
33
|
+
tool string
|
|
34
|
+
verdict gateVerdict
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
type lastOut struct {
|
|
38
|
+
tool string
|
|
39
|
+
verdict outputVerdict
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
type ext struct {
|
|
43
|
+
mu sync.Mutex
|
|
44
|
+
cfg config
|
|
45
|
+
be *backend
|
|
46
|
+
beEr error
|
|
47
|
+
|
|
48
|
+
on, gateOn, outputOn bool
|
|
49
|
+
mode string
|
|
50
|
+
|
|
51
|
+
secrets map[string]bool
|
|
52
|
+
last *lastGate
|
|
53
|
+
lastOutput *lastOut
|
|
54
|
+
lastFailure string
|
|
55
|
+
lastErrorAt time.Time
|
|
56
|
+
warned map[string]bool
|
|
57
|
+
askReg bool
|
|
58
|
+
|
|
59
|
+
gateMemo, outMemo *memo
|
|
60
|
+
now func() time.Time
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
// Extension returns the Jev extension.
|
|
64
|
+
func Extension() *sdk.Extension {
|
|
65
|
+
e := sdk.New("jev")
|
|
66
|
+
x := &ext{cfg: defaultConfig(), secrets: map[string]bool{}, warned: map[string]bool{}, gateMemo: newMemo(), outMemo: newMemo(), now: time.Now}
|
|
67
|
+
e.OnSessionStart(func(ctx sdk.Context, _ map[string]any) (any, error) { x.sessionStart(ctx); return nil, nil })
|
|
68
|
+
e.OnEvent(sdk.EventModelSelect, func(ctx sdk.Context, _ map[string]any) (any, error) { x.modelSelect(ctx); return nil, nil })
|
|
69
|
+
e.OnEvent(sdk.EventToolCall, x.toolCall)
|
|
70
|
+
e.OnToolResult(x.toolResult)
|
|
71
|
+
e.Command("jev", "Jev: typed judgments of tool calls and output (status, on/off, mode, last, output, check)", x.command)
|
|
72
|
+
return e
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
type snap struct {
|
|
76
|
+
cfg config
|
|
77
|
+
be *backend
|
|
78
|
+
on, gateOn, outputOn bool
|
|
79
|
+
mode string
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
func (x *ext) snapshot() snap {
|
|
83
|
+
x.mu.Lock()
|
|
84
|
+
defer x.mu.Unlock()
|
|
85
|
+
return snap{x.cfg, x.be, x.on, x.gateOn, x.outputOn, x.mode}
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
func (x *ext) redact(s string) string {
|
|
89
|
+
x.mu.Lock()
|
|
90
|
+
defer x.mu.Unlock()
|
|
91
|
+
for secret := range x.secrets {
|
|
92
|
+
s = strings.ReplaceAll(s, secret, "[redacted]")
|
|
93
|
+
}
|
|
94
|
+
return s
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
func (x *ext) remember(secret string) {
|
|
98
|
+
if s := strings.TrimSpace(secret); len(s) >= 8 {
|
|
99
|
+
x.mu.Lock()
|
|
100
|
+
x.secrets[s] = true
|
|
101
|
+
x.mu.Unlock()
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
func (x *ext) setStatus(ctx sdk.Context, glyph, rest string) {
|
|
106
|
+
ctx.SetStatus(statusKey, statusText(x.plain(), glyph, rest))
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
func (x *ext) baseStatus(ctx sdk.Context) {
|
|
110
|
+
s := x.snapshot()
|
|
111
|
+
mode := "off"
|
|
112
|
+
if s.gateOn {
|
|
113
|
+
mode = s.mode
|
|
114
|
+
}
|
|
115
|
+
rest := mode
|
|
116
|
+
if !s.outputOn {
|
|
117
|
+
rest += " (out off)"
|
|
118
|
+
}
|
|
119
|
+
x.setStatus(ctx, "", rest)
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// sessionStart reloads the configuration. Nothing is judged unless the user's own
|
|
123
|
+
// config says "enabled": a key alone is not consent.
|
|
124
|
+
func (x *ext) sessionStart(ctx sdk.Context) {
|
|
125
|
+
l := loadConfig(ctx.ConfigHome(), ctx.Cwd())
|
|
126
|
+
be, err := newBackend(l.Config, ctx)
|
|
127
|
+
x.mu.Lock()
|
|
128
|
+
x.cfg, x.be, x.beEr = l.Config, be, err
|
|
129
|
+
x.on = l.Config.Enabled
|
|
130
|
+
x.gateOn = x.on && l.Config.Gate.Enabled
|
|
131
|
+
x.outputOn = x.on && l.Config.Output.Enabled
|
|
132
|
+
x.mode = l.Config.Gate.Mode
|
|
133
|
+
x.gateMemo, x.outMemo = newMemo(), newMemo()
|
|
134
|
+
x.mu.Unlock()
|
|
135
|
+
x.remember(l.Config.Key)
|
|
136
|
+
|
|
137
|
+
for _, w := range l.Warnings {
|
|
138
|
+
ctx.Notify("pi-jev: "+x.redact(w), "warning")
|
|
139
|
+
}
|
|
140
|
+
if !l.Config.Enabled {
|
|
141
|
+
return
|
|
142
|
+
}
|
|
143
|
+
if err != nil {
|
|
144
|
+
x.warnOnce(ctx, err.Error())
|
|
145
|
+
return
|
|
146
|
+
}
|
|
147
|
+
x.baseStatus(ctx)
|
|
148
|
+
if !l.Config.Acknowledged {
|
|
149
|
+
ctx.Notify(startupDisclosure(l.Config, be.Destination()), "warning")
|
|
150
|
+
}
|
|
151
|
+
x.registerAsk(ctx)
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// modelSelect keeps the default judge on the model the session uses. The
|
|
155
|
+
// disclosure names that judge as the provider that already receives the
|
|
156
|
+
// conversation, so after a switch (say, to a local model) judging must not keep
|
|
157
|
+
// sending content to the previous provider. A model named in the user's own
|
|
158
|
+
// config stays. Cached verdicts belong to the previous judge and are dropped.
|
|
159
|
+
func (x *ext) modelSelect(ctx sdk.Context) {
|
|
160
|
+
x.mu.Lock()
|
|
161
|
+
cfg, old := x.cfg, x.be
|
|
162
|
+
x.mu.Unlock()
|
|
163
|
+
if cfg.Backend != backendModel || cfg.Model != "" {
|
|
164
|
+
return
|
|
165
|
+
}
|
|
166
|
+
be, err := newBackend(cfg, ctx)
|
|
167
|
+
if err == nil && old != nil && be.Destination() == old.Destination() {
|
|
168
|
+
return
|
|
169
|
+
}
|
|
170
|
+
x.mu.Lock()
|
|
171
|
+
x.be, x.beEr = be, err
|
|
172
|
+
x.gateMemo, x.outMemo = newMemo(), newMemo()
|
|
173
|
+
on := x.on
|
|
174
|
+
x.mu.Unlock()
|
|
175
|
+
if !on {
|
|
176
|
+
return
|
|
177
|
+
}
|
|
178
|
+
if err != nil {
|
|
179
|
+
x.warnOnce(ctx, err.Error())
|
|
180
|
+
return
|
|
181
|
+
}
|
|
182
|
+
x.baseStatus(ctx)
|
|
183
|
+
x.registerAsk(ctx)
|
|
184
|
+
ctx.Notify("pi-jev: now judging with "+be.Destination(), "info")
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
// warnOnce reports a setup problem once per distinct message. The missing-key text
|
|
188
|
+
// is the original's.
|
|
189
|
+
func (x *ext) warnOnce(ctx sdk.Context, reason string) {
|
|
190
|
+
x.mu.Lock()
|
|
191
|
+
seen := x.warned[reason]
|
|
192
|
+
x.warned[reason] = true
|
|
193
|
+
x.mu.Unlock()
|
|
194
|
+
if seen {
|
|
195
|
+
return
|
|
196
|
+
}
|
|
197
|
+
if strings.HasPrefix(reason, "no key.") {
|
|
198
|
+
reason = fmt.Sprintf("no key. Set %s or apiKeyFile in pi-jev.json; the gate is inactive until then.", apiKeyEnv)
|
|
199
|
+
}
|
|
200
|
+
ctx.Notify("pi-jev: "+x.redact(reason), "warning")
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
// goContext turns the request's cancellation into a context.Context.
|
|
204
|
+
func goContext(c sdk.Context) (context.Context, context.CancelFunc) {
|
|
205
|
+
ctx, cancel := context.WithCancel(context.Background())
|
|
206
|
+
if done := c.Done(); done != nil {
|
|
207
|
+
go func() {
|
|
208
|
+
select {
|
|
209
|
+
case <-done:
|
|
210
|
+
cancel()
|
|
211
|
+
case <-ctx.Done():
|
|
212
|
+
}
|
|
213
|
+
}()
|
|
214
|
+
}
|
|
215
|
+
return ctx, cancel
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
// lastUserRequest is the latest user text in the branch, so scope questions can
|
|
219
|
+
// weigh intent. An unreadable session means no request, never a failed judgment.
|
|
220
|
+
func lastUserRequest(ctx sdk.Context) string {
|
|
221
|
+
branch, err := ctx.GetBranch()
|
|
222
|
+
if err != nil {
|
|
223
|
+
return ""
|
|
224
|
+
}
|
|
225
|
+
for i := len(branch) - 1; i >= 0; i-- {
|
|
226
|
+
if branch[i].Role != "user" {
|
|
227
|
+
continue
|
|
228
|
+
}
|
|
229
|
+
if t := strings.TrimSpace(branch[i].Content); t != "" {
|
|
230
|
+
return t
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
return ""
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
// gateKey is the memo key of one pending call: the same call asked again within CacheSeconds reuses the answer.
|
|
237
|
+
// The key is a digest, not the call's arguments: a write of a large file would otherwise sit in the memo, in full,
|
|
238
|
+
// for the whole cache window.
|
|
239
|
+
func gateKey(tool string, input any, cwd, user string) string {
|
|
240
|
+
h := sha256.New()
|
|
241
|
+
h.Write([]byte(tool))
|
|
242
|
+
h.Write([]byte{0})
|
|
243
|
+
enc := json.NewEncoder(h) // straight into the digest: a large argument is never held as a second copy
|
|
244
|
+
enc.SetEscapeHTML(false)
|
|
245
|
+
if err := enc.Encode(input); err != nil {
|
|
246
|
+
h.Write([]byte("null"))
|
|
247
|
+
}
|
|
248
|
+
h.Write([]byte{0})
|
|
249
|
+
h.Write([]byte(cwd))
|
|
250
|
+
h.Write([]byte{0})
|
|
251
|
+
h.Write([]byte(truncateText(user, userRequestChars)))
|
|
252
|
+
return hex.EncodeToString(h.Sum(nil))
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
// failOpen records an unavailable judge and tells the user at most once a minute.
|
|
256
|
+
// The tool call proceeds: an outage must never stop the agent (the original's
|
|
257
|
+
// policy, kept), and the status says so instead of keeping an earlier "clear".
|
|
258
|
+
func (x *ext) failOpen(ctx sdk.Context, err error) {
|
|
259
|
+
msg := x.redact(err.Error())
|
|
260
|
+
x.mu.Lock()
|
|
261
|
+
x.lastFailure = msg
|
|
262
|
+
quiet := x.now().Sub(x.lastErrorAt) < errorNotifyInterval
|
|
263
|
+
if !quiet {
|
|
264
|
+
x.lastErrorAt = x.now()
|
|
265
|
+
}
|
|
266
|
+
c := x.cfg
|
|
267
|
+
x.mu.Unlock()
|
|
268
|
+
x.setStatus(ctx, "✗", "unavailable (failing open)")
|
|
269
|
+
if !quiet {
|
|
270
|
+
ctx.Notify(x.failureNote(msg, c), "error")
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
func (x *ext) toolCall(ctx sdk.Context, data map[string]any) (any, error) {
|
|
275
|
+
s := x.snapshot()
|
|
276
|
+
tool, _ := data["toolName"].(string)
|
|
277
|
+
if !s.gateOn || s.be == nil || !slices.Contains(s.cfg.Gate.Tools, tool) {
|
|
278
|
+
return nil, nil
|
|
279
|
+
}
|
|
280
|
+
input := data["input"]
|
|
281
|
+
user := lastUserRequest(ctx)
|
|
282
|
+
key := gateKey(tool, input, ctx.Cwd(), user)
|
|
283
|
+
resp, err := x.gateMemo.do(key, time.Duration(s.cfg.Gate.CacheSeconds)*time.Second, x.now, func() (*response, error) {
|
|
284
|
+
gctx, cancel := goContext(ctx)
|
|
285
|
+
defer cancel()
|
|
286
|
+
state := gateStateJSON(ctx.Cwd(), tool, input, user, s.cfg.Gate.ArgumentChars, s.cfg.MaxStateChars)
|
|
287
|
+
r, err := s.be.Ask(gctx, ctx, state, false, gateQuestions)
|
|
288
|
+
return r, err
|
|
289
|
+
})
|
|
290
|
+
if err != nil {
|
|
291
|
+
x.failOpen(ctx, err)
|
|
292
|
+
return nil, nil
|
|
293
|
+
}
|
|
294
|
+
v := evaluateGate(resp, s.cfg)
|
|
295
|
+
x.mu.Lock()
|
|
296
|
+
x.last = &lastGate{tool, v}
|
|
297
|
+
x.mu.Unlock()
|
|
298
|
+
if v.Flagged {
|
|
299
|
+
x.setStatus(ctx, "⚠", v.summary())
|
|
300
|
+
} else {
|
|
301
|
+
x.setStatus(ctx, "✓", fmt.Sprintf("clear (%s)", s.mode))
|
|
302
|
+
}
|
|
303
|
+
if !v.Flagged {
|
|
304
|
+
return nil, nil
|
|
305
|
+
}
|
|
306
|
+
reason := "pi-jev: " + v.summary()
|
|
307
|
+
if s.mode == "shadow" {
|
|
308
|
+
ctx.Notify(x.shadowNote(tool, v, s.cfg), "warning")
|
|
309
|
+
return nil, nil
|
|
310
|
+
}
|
|
311
|
+
if !ctx.HasUI() {
|
|
312
|
+
if s.cfg.Gate.BlockWithoutUI {
|
|
313
|
+
return map[string]any{"block": true, "reason": reason}, nil
|
|
314
|
+
}
|
|
315
|
+
// No UI means no way to approve a flagged call. Degrade to a warning rather
|
|
316
|
+
// than deadlocking a headless run on a classifier's opinion.
|
|
317
|
+
ctx.Notify(x.headlessNote(tool, v, s.cfg), "warning")
|
|
318
|
+
return nil, nil
|
|
319
|
+
}
|
|
320
|
+
allow, cerr := ctx.Confirm("Jev flagged this tool call", x.confirmMessage(tool, input, v, s.cfg, s.be.Destination()))
|
|
321
|
+
if cerr != nil {
|
|
322
|
+
// The judge answered and flagged the call; only the user's answer is missing.
|
|
323
|
+
// Fail-open covers an unavailable judge, not a missing approval: the original's
|
|
324
|
+
// handler throws here and Pi blocks the call (agent-session.ts
|
|
325
|
+
// _installAgentToolHooks, "Extension failed, blocking execution").
|
|
326
|
+
return map[string]any{"block": true, "reason": fmt.Sprintf("%s (not confirmed: %s)", reason, x.redact(cerr.Error()))}, nil
|
|
327
|
+
}
|
|
328
|
+
if allow {
|
|
329
|
+
return nil, nil
|
|
330
|
+
}
|
|
331
|
+
return map[string]any{"block": true, "reason": reason + " (declined)"}, nil
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
// contentText is the text of a tool result: the blocks' text joined by newlines.
|
|
335
|
+
func contentText(content any) string {
|
|
336
|
+
if s, ok := content.(string); ok {
|
|
337
|
+
return s
|
|
338
|
+
}
|
|
339
|
+
blocks, _ := content.([]any)
|
|
340
|
+
var parts []string
|
|
341
|
+
for _, b := range blocks {
|
|
342
|
+
if m, ok := b.(map[string]any); ok {
|
|
343
|
+
if t, ok := m["text"].(string); ok {
|
|
344
|
+
parts = append(parts, t)
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
}
|
|
348
|
+
return strings.Join(parts, "\n")
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
func (x *ext) toolResult(ctx sdk.Context, data map[string]any) (any, error) {
|
|
352
|
+
s := x.snapshot()
|
|
353
|
+
tool, _ := data["toolName"].(string)
|
|
354
|
+
if !s.outputOn || s.be == nil || !slices.Contains(s.cfg.Output.Tools, tool) {
|
|
355
|
+
return nil, nil
|
|
356
|
+
}
|
|
357
|
+
out := contentText(data["content"])
|
|
358
|
+
if strings.TrimSpace(out) == "" {
|
|
359
|
+
return nil, nil
|
|
360
|
+
}
|
|
361
|
+
input, isErr := data["input"], data["isError"] == true
|
|
362
|
+
key := tool + "\x00" + out // as the original: identical output costs one judgment
|
|
363
|
+
resp, err := x.outMemo.do(key, outputCacheSeconds*time.Second, x.now, func() (*response, error) {
|
|
364
|
+
gctx, cancel := goContext(ctx)
|
|
365
|
+
defer cancel()
|
|
366
|
+
state := outputStateJSON(ctx.Cwd(), tool, input, out, isErr, s.cfg.Output.OutputChars, s.cfg.MaxStateChars)
|
|
367
|
+
return s.be.Ask(gctx, ctx, state, false, outputQuestions)
|
|
368
|
+
})
|
|
369
|
+
if err != nil {
|
|
370
|
+
x.failOpen(ctx, err)
|
|
371
|
+
return nil, nil
|
|
372
|
+
}
|
|
373
|
+
v := evaluateOutput(resp, s.cfg)
|
|
374
|
+
x.mu.Lock()
|
|
375
|
+
x.lastOutput = &lastOut{tool, v}
|
|
376
|
+
x.mu.Unlock()
|
|
377
|
+
if v.Notice == "" {
|
|
378
|
+
return nil, nil
|
|
379
|
+
}
|
|
380
|
+
x.setStatus(ctx, map[string]string{"leak": "⚠", "advice": "ℹ"}[v.Kind], fmt.Sprintf("%s (%s)", v.Kind, tool))
|
|
381
|
+
if v.Kind == "leak" {
|
|
382
|
+
ctx.Notify(x.leakNote(tool, v.LeaksSecret, s.cfg), "warning")
|
|
383
|
+
}
|
|
384
|
+
// The model reads the tool result, so the notice rides with it.
|
|
385
|
+
content, _ := data["content"].([]any)
|
|
386
|
+
if _, isString := data["content"].(string); isString {
|
|
387
|
+
content = []any{map[string]any{"type": "text", "text": data["content"]}}
|
|
388
|
+
}
|
|
389
|
+
patched := append(append([]any{}, content...), map[string]any{"type": "text", "text": "[pi-jev] " + v.Notice})
|
|
390
|
+
return map[string]any{"content": patched}, nil
|
|
391
|
+
}
|