@pi-in-go/pigpen-jev 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CREDITS.md +22 -0
- package/LICENSE +22 -0
- package/README.md +237 -0
- package/extensions/jev/ask.go +166 -0
- package/extensions/jev/ask_test.go +218 -0
- package/extensions/jev/backend.go +128 -0
- package/extensions/jev/bench_test.go +64 -0
- package/extensions/jev/boundaries_test.go +159 -0
- package/extensions/jev/command.go +224 -0
- package/extensions/jev/commands_test.go +214 -0
- package/extensions/jev/config.go +450 -0
- package/extensions/jev/errors_test.go +191 -0
- package/extensions/jev/extension.go +391 -0
- package/extensions/jev/fakehost_test.go +548 -0
- package/extensions/jev/gate.go +125 -0
- package/extensions/jev/gate_test.go +610 -0
- package/extensions/jev/gatekey_test.go +24 -0
- package/extensions/jev/go.mod +9 -0
- package/extensions/jev/go.sum +2 -0
- package/extensions/jev/go.work +10 -0
- package/extensions/jev/helpers_test.go +404 -0
- package/extensions/jev/memo.go +88 -0
- package/extensions/jev/output.go +89 -0
- package/extensions/jev/output_test.go +187 -0
- package/extensions/jev/ownmodel_test.go +118 -0
- package/extensions/jev/render.go +136 -0
- package/extensions/jev/review_test.go +310 -0
- package/extensions/jev/source_test.go +57 -0
- package/extensions/jev/text.go +174 -0
- package/extensions/jev/trust_test.go +335 -0
- package/extensions/jev/types.go +227 -0
- package/libs/typesafe/CONTRACT.md +125 -0
- package/libs/typesafe/CREDITS.md +37 -0
- package/libs/typesafe/LICENSE +23 -0
- package/libs/typesafe/README.md +19 -0
- package/libs/typesafe/go.mod +3 -0
- package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
- package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
- package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
- package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
- package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
- package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
- package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
- package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
- package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
- package/libs/typesafe/libraries/ownmodel/run.go +288 -0
- package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
- package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
- package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
- package/libs/typesafe/libraries/typesafe/answers.go +268 -0
- package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
- package/libs/typesafe/libraries/typesafe/batch.go +80 -0
- package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
- package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
- package/libs/typesafe/libraries/typesafe/client.go +561 -0
- package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
- package/libs/typesafe/libraries/typesafe/doc.go +27 -0
- package/libs/typesafe/libraries/typesafe/entry.go +142 -0
- package/libs/typesafe/libraries/typesafe/env.go +11 -0
- package/libs/typesafe/libraries/typesafe/errors.go +310 -0
- package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
- package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
- package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
- package/libs/typesafe/libraries/typesafe/logging.go +160 -0
- package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
- package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
- package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
- package/libs/typesafe/libraries/typesafe/questions.go +490 -0
- package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
- package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
- package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
- package/libs/typesafe/libraries/typesafe/retry.go +350 -0
- package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
- package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
- package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
- package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
- package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
- package/libs/typesafe/libraries/typesafe/version.go +10 -0
- package/libs/typesafe/package.json +37 -0
- package/libs/typesafe/provenance.json +49 -0
- package/package.json +42 -0
- package/port/PORT.md +107 -0
- package/port/e2e/gate-and-output.py +35 -0
- package/port/e2e/jev-ask.py +36 -0
- package/port/e2e/model-switch.py +44 -0
- package/port/e2e/off-by-default.py +34 -0
- package/port/gen-scenarios.py +103 -0
- package/port/golden/cache-identical-calls.jsonl +30 -0
- package/port/golden/clear.jsonl +22 -0
- package/port/golden/commands.jsonl +43 -0
- package/port/golden/enforce-accept.jsonl +23 -0
- package/port/golden/enforce-decline.jsonl +22 -0
- package/port/golden/jev-ask.jsonl +20 -0
- package/port/golden/output-advice.jsonl +23 -0
- package/port/golden/output-leak.jsonl +24 -0
- package/port/golden/output-low-confidence.jsonl +22 -0
- package/port/golden/shadow-flagged.jsonl +23 -0
- package/port/golden/unjudged-tools.jsonl +19 -0
- package/port/golden/write-elision.jsonl +21 -0
- package/port/mutate-unit.py +63 -0
- package/port/mutations.json +578 -0
- package/port/oracle/LICENSE +21 -0
- package/port/oracle/README.md +181 -0
- package/port/oracle/SHA256SUMS +8 -0
- package/port/oracle/package.json +43 -0
- package/port/oracle/src/client.ts +409 -0
- package/port/oracle/src/config.ts +363 -0
- package/port/oracle/src/gate.ts +229 -0
- package/port/oracle/src/index.ts +649 -0
- package/port/oracle/src/output.ts +163 -0
- package/port/red-run.log +309 -0
- package/port/scenarios/cache-identical-calls.json +71 -0
- package/port/scenarios/clear.json +61 -0
- package/port/scenarios/commands.json +119 -0
- package/port/scenarios/enforce-accept.json +66 -0
- package/port/scenarios/enforce-decline.json +57 -0
- package/port/scenarios/jev-ask.json +83 -0
- package/port/scenarios/output-advice.json +61 -0
- package/port/scenarios/output-leak.json +61 -0
- package/port/scenarios/output-low-confidence.json +61 -0
- package/port/scenarios/shadow-flagged.json +61 -0
- package/port/scenarios/unjudged-tools.json +55 -0
- package/port/scenarios/write-elision.json +53 -0
- package/provenance.json +18 -0
package/CREDITS.md
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# Credits
|
|
2
|
+
|
|
3
|
+
`extensions/jev` is a Go port of **pi-jev** (`@y0usaf/pi-jev` 0.2.2), https://github.com/y0usaf/pi-jev,
|
|
4
|
+
by **y0usaf** (MIT, Copyright (c) 2026 y0usaf). The idea, the three surfaces (tool-call gate, output
|
|
5
|
+
judge, `jev_ask`), the questions and their measured phrasing, the thresholds and their calibration
|
|
6
|
+
tables, the wording of every notice, the `/jev` command and the fail-open policy are y0usaf's.
|
|
7
|
+
|
|
8
|
+
- Original: `src/client.ts`, `src/config.ts`, `src/gate.ts`, `src/output.ts`, `src/index.ts`
|
|
9
|
+
- Pinned commit: `88e5fb3888948e7065110d47cdf6ac57abb71ba4` (release 0.2.2, "failing visibly on a
|
|
10
|
+
response that skips a question"). The npm `latest` is still 0.2.0; this is the reviewed Git commit.
|
|
11
|
+
- The unmodified sources are kept in [`port/oracle/src/`](port/oracle/src) as the equivalence oracle, with
|
|
12
|
+
the upstream license at [`port/oracle/LICENSE`](port/oracle/LICENSE), `package.json` and README.
|
|
13
|
+
|
|
14
|
+
[TypeSafe](https://docs.typesafe.ai) defines Jev (its typed decision model: Noul, Choice, Score).
|
|
15
|
+
The HTTP protocol and the own-model backend come from the shared Go client
|
|
16
|
+
[`components/typesafe`](../typesafe/CREDITS.md) (ports of TypeSafe's `typesafe-sdk-js` and
|
|
17
|
+
`system-one-adapter-python`), which this Package uses; it does not carry a client of its own.
|
|
18
|
+
|
|
19
|
+
The Go code, the tests and the scenarios were written for this Package by Michael Kinsy. What was
|
|
20
|
+
modified relative to the original, and why, is the list of corrections in
|
|
21
|
+
[`port/PORT.md`](port/PORT.md) (C1 to C10) and the opt-in, disclosure and trust rules in the README.
|
|
22
|
+
Everything else follows the original's behavior, checked against the original under Pi.
|
package/LICENSE
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 y0usaf
|
|
4
|
+
Copyright (c) 2026 Michael Kinsy (the Go port)
|
|
5
|
+
|
|
6
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
7
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
8
|
+
in the Software without restriction, including without limitation the rights
|
|
9
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
10
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
11
|
+
furnished to do so, subject to the following conditions:
|
|
12
|
+
|
|
13
|
+
The above copyright notice and this permission notice shall be included in all
|
|
14
|
+
copies or substantial portions of the Software.
|
|
15
|
+
|
|
16
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
17
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
18
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
19
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
20
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
21
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
22
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
# Jev for PiG
|
|
2
|
+
|
|
3
|
+
A second opinion on what your agent is about to do, and on what it just printed. Jev asks a judge
|
|
4
|
+
typed questions ("is this destructive?", "does this send local data off the machine?", "does this
|
|
5
|
+
output contain a secret?") and gets back numbers, not prose, so the answer is something PiG can branch on.
|
|
6
|
+
|
|
7
|
+
```text
|
|
8
|
+
⚠ Jev flagged bash (shadow mode: reported, not blocked)
|
|
9
|
+
⚠ destructive ██████████ 0.99 ≥ 0.90
|
|
10
|
+
⚠ exfiltration ████████·· 0.79 ≥ 0.70
|
|
11
|
+
⚠ beyond scope ██████████ 0.98 ≥ 0.85
|
|
12
|
+
⚠ impact ██████████ 3.00/3 ≥ 2.50 (confidence 0.91)
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
**It is off until you turn it on.** Nothing is judged, nothing is sent, and it prints nothing at start-up
|
|
16
|
+
until you opt in. Judging sends parts of your session to a judge, so `/jev on` shows exactly what and where
|
|
17
|
+
before it does anything.
|
|
18
|
+
|
|
19
|
+
Three surfaces, from y0usaf's [pi-jev](https://github.com/y0usaf/pi-jev) (this is its Go port; see
|
|
20
|
+
[CREDITS.md](CREDITS.md)):
|
|
21
|
+
|
|
22
|
+
| Surface | What it does |
|
|
23
|
+
|---|---|
|
|
24
|
+
| **Gate** | Before `bash`, `write` or `edit` runs, judges the call: destructive? sends data off the machine? beyond what you asked? how much damage? In **shadow** mode (default) it only reports. In **enforce** mode a flagged call asks you first. |
|
|
25
|
+
| **Output judge** | After `bash` finishes, reads what it printed: does it carry a secret? What kind of failure is it? It adds one line to the result the model reads ("do not repeat the value", "fix the environment before retrying"). It never blocks. |
|
|
26
|
+
| **`jev_ask`** | A tool the model can call to ask typed questions (yes/no probability, pick one, rubric score) about a piece of text. Exists only while Jev is on. |
|
|
27
|
+
|
|
28
|
+
## Turn it on
|
|
29
|
+
|
|
30
|
+
```sh
|
|
31
|
+
pig install ./components/jev # or select the Package from a Piglet (pig-with-batteries does)
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
then, in a session:
|
|
35
|
+
|
|
36
|
+
```text
|
|
37
|
+
/jev on
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
```text
|
|
41
|
+
Turn on Jev for this session?
|
|
42
|
+
What leaves this machine for each judgment:
|
|
43
|
+
• the working directory and the tool name
|
|
44
|
+
• your last message (first 1200 characters)
|
|
45
|
+
• the tool arguments (bash/write/edit; long fields are cut at 400 characters, and write/edit arguments are file content)
|
|
46
|
+
• bash output (first 2000 characters)
|
|
47
|
+
Also sent: text the model passes to jev_ask (up to 8000 characters) and text you pass to /jev check.
|
|
48
|
+
Destination: the model acme/judge-1 (through PiG, the provider that already receives your conversation)
|
|
49
|
+
If the judge is unavailable, tool calls run unjudged (it fails open). A judgment is probabilistic advice, not a sandbox.
|
|
50
|
+
Turn Jev on?
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
To keep it on for every session, edit **your own** `<agent dir>/pi-jev.json` (`~/.pig/agent/pi-jev.json`;
|
|
54
|
+
`PIG_CODING_AGENT_DIR` moves it):
|
|
55
|
+
|
|
56
|
+
```json
|
|
57
|
+
{ "enabled": true }
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Jev then shows the notice above at every start-up until you add `"acknowledged": true`.
|
|
61
|
+
`/jev off` turns it off again for the session. Nothing turns it on for you: not a key in the environment,
|
|
62
|
+
not a file in a repository you opened (see [What a project can change](#what-a-project-can-change)).
|
|
63
|
+
|
|
64
|
+
## Who judges
|
|
65
|
+
|
|
66
|
+
| Backend | `"backend"` | Where the content goes | Needs |
|
|
67
|
+
|---|---|---|---|
|
|
68
|
+
| **The model PiG is configured with** (default) | `"model"` | The provider of your current model, through PiG's own model access and credentials. When you switch models, the judge follows (Jev says so) and earlier verdicts are dropped. Set `"model": "provider/id"` to use another model of your registry. | nothing else |
|
|
69
|
+
| **The TypeSafe API** | `"typesafe"` | The endpoint you configure. There is no built-in default and no default model name. | `endpoint`, `model`, and a key (`TYPESAFE_API_KEY`, `apiKey` or `apiKeyFile`) |
|
|
70
|
+
|
|
71
|
+
Both come from the shared Go client in [`components/typesafe`](../typesafe/README.md): the model backend asks the
|
|
72
|
+
model for typed answers in prompted JSON mode and validates them strictly; the TypeSafe backend speaks TypeSafe's
|
|
73
|
+
API. **Calibration caveat:** the thresholds below were measured by y0usaf against TypeSafe's Jev on a handful of
|
|
74
|
+
states ("a smoke calibration, not a labelled evaluation set", in the original's words). Nobody has measured them
|
|
75
|
+
against an arbitrary chat model. Read them as starting points; that is one reason shadow mode is the default.
|
|
76
|
+
|
|
77
|
+
```json
|
|
78
|
+
{
|
|
79
|
+
"enabled": true,
|
|
80
|
+
"backend": "typesafe",
|
|
81
|
+
"endpoint": "https://<your TypeSafe API root>",
|
|
82
|
+
"model": "<a TypeSafe model name>",
|
|
83
|
+
"apiKeyFile": "~/keys/typesafe.txt"
|
|
84
|
+
}
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
The key is sent only to that endpoint, which must be `https` (or `http` to `localhost`). `TYPESAFE_BASE_URL` and the
|
|
88
|
+
other `TYPESAFE_*` variables are ignored.
|
|
89
|
+
|
|
90
|
+
## The gate
|
|
91
|
+
|
|
92
|
+
| Question | Type | Reads | Threshold |
|
|
93
|
+
|---|---|---|---|
|
|
94
|
+
| Is this action destructive? | noul | `destructive` | 0.90 |
|
|
95
|
+
| Does this send local data or secrets off-machine? | noul | `exfiltration` | 0.70 |
|
|
96
|
+
| Does this affect anything beyond what the user asked for? | noul | `beyond_scope` | 0.85 |
|
|
97
|
+
| How much damage if the user did not want it? | score (4 levels) | `impact` | 2.50 |
|
|
98
|
+
|
|
99
|
+
All four go in one request. **Shadow mode is the default:** a flagged call produces a notification and a footer
|
|
100
|
+
status. In **enforce** mode (`/jev mode enforce`) a flagged call opens a dialog:
|
|
101
|
+
|
|
102
|
+
```text
|
|
103
|
+
Jev flagged this tool call
|
|
104
|
+
bash rm -rf src && git push --force origin main
|
|
105
|
+
|
|
106
|
+
⚠ destructive ██████████ 0.99 ≥ 0.90
|
|
107
|
+
⚠ exfiltration ████████·· 0.79 ≥ 0.70
|
|
108
|
+
⚠ beyond scope ██████████ 0.98 ≥ 0.85
|
|
109
|
+
⚠ impact ██████████ 3.00/3 ≥ 2.50 (confidence 0.91)
|
|
110
|
+
|
|
111
|
+
Judged by the model acme/judge-1 (through PiG, the provider that already receives your conversation)
|
|
112
|
+
Run it anyway?
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
Print and JSON runs have no UI (RPC and the terminal do), so enforcement there degrades to the same warning unless you set
|
|
116
|
+
`gate.blockWithoutUI`. If the dialog cannot be shown (the UI call fails), the flagged call is blocked, as in the original:
|
|
117
|
+
fail-open covers an unavailable judge, not a missing approval.
|
|
118
|
+
|
|
119
|
+
## Failure policy: it fails open
|
|
120
|
+
|
|
121
|
+
**If the judge is unavailable, tool calls run unjudged.** A missing key, a timeout, a 429, a malformed or incomplete
|
|
122
|
+
answer produces no verdict and the tool call proceeds; a dead endpoint never stops your agent. The footer says so
|
|
123
|
+
(`⚖ jev ✗ unavailable (failing open)`) instead of keeping an old "clear", and errors are reported at most once a minute.
|
|
124
|
+
This is y0usaf's policy, kept on purpose, and stated at start-up and in `/jev`. Do not treat a clear verdict as
|
|
125
|
+
authorization: a model's judgment is probabilistic advice, not a sandbox, and the judge can be wrong or steered by the
|
|
126
|
+
very content it reads.
|
|
127
|
+
|
|
128
|
+
## The output judge
|
|
129
|
+
|
|
130
|
+
`tool_result` asks two questions in one request and appends one line to the result when either fires:
|
|
131
|
+
|
|
132
|
+
| Question | Type | Reads | Threshold |
|
|
133
|
+
|---|---|---|---|
|
|
134
|
+
| Does this output contain a secret or credential? | noul | `leaks_secret` | 0.90 |
|
|
135
|
+
| What kind of failure is this? | choice (6 options) | `failure_class` | confidence 0.60 |
|
|
136
|
+
|
|
137
|
+
A leak appends "Do not repeat the value in a reply, a file, or a command; refer to it by name instead" and raises a
|
|
138
|
+
notification. A failure class appends what to do about it (retry a `transient` failure, fix the environment for
|
|
139
|
+
`environment`, fix the code for `code_bug`, do not retry `permission`, fix the invocation for `user_error`). `no_failure`
|
|
140
|
+
says nothing. Judged tools default to `["bash"]`.
|
|
141
|
+
|
|
142
|
+
## `jev_ask`
|
|
143
|
+
|
|
144
|
+
```json
|
|
145
|
+
{ "state": "the tool output, diff, or message to judge",
|
|
146
|
+
"questions": [
|
|
147
|
+
{ "id": "relevant", "type": "noul", "instructions": "Is this relevant to the user's question?" },
|
|
148
|
+
{ "id": "label", "type": "choice", "instructions": "Which bucket?",
|
|
149
|
+
"options": [{ "name": "bug", "description": "Defect in existing behaviour" }, { "name": "feature" }] },
|
|
150
|
+
{ "id": "quality", "type": "score", "instructions": "How thorough is this?",
|
|
151
|
+
"levels": ["Superficial", "Adequate", "Thorough"] } ] }
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
It is registered when Jev is on and refuses ("Jev is off") after `/jev off`. The text is cut to `maxStateChars`.
|
|
155
|
+
|
|
156
|
+
## Commands
|
|
157
|
+
|
|
158
|
+
- `/jev` shows the judge, what is sent, the failure policy and the last verdicts
|
|
159
|
+
- `/jev on` and `/jev off` toggle both judges for the session (`on` asks first, with the disclosure)
|
|
160
|
+
- `/jev mode shadow|enforce`
|
|
161
|
+
- `/jev last` prints the last gate verdict with all four answers
|
|
162
|
+
- `/jev output` prints the last judged output (also when it was clean)
|
|
163
|
+
- `/jev check <text>` runs the gate questions on text you supply (it sends that text, so it needs Jev on)
|
|
164
|
+
|
|
165
|
+
```text
|
|
166
|
+
Jev is ON — shadow mode, output judge on
|
|
167
|
+
Judge the model acme/judge-1 (through PiG, the provider that already receives your conversation)
|
|
168
|
+
Gate bash, write, edit (on)
|
|
169
|
+
Output bash (on)
|
|
170
|
+
Sends working directory, tool name, your last message, tool arguments, tool output, jev_ask text — cut to your limits
|
|
171
|
+
If down tool calls run unjudged (fails open)
|
|
172
|
+
Last bash: destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91
|
|
173
|
+
Commands /jev on|off · /jev mode shadow|enforce · /jev last · /jev output · /jev check <text>
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
## Configure
|
|
177
|
+
|
|
178
|
+
`<agent dir>/pi-jev.json` is yours. A project's `<cwd>/.pig/pi-jev.json` may tune the gate. Only keys you set override; the
|
|
179
|
+
rest are the defaults below.
|
|
180
|
+
|
|
181
|
+
```json
|
|
182
|
+
{
|
|
183
|
+
"enabled": false, "acknowledged": false, "backend": "model", "display": "rich",
|
|
184
|
+
"endpoint": "", "model": "", "apiKey": "", "apiKeyFile": "", "timeoutMs": 20000, "retries": 2,
|
|
185
|
+
"maxStateChars": 8000,
|
|
186
|
+
"gate": { "enabled": true, "mode": "shadow", "tools": ["bash", "write", "edit"], "argumentChars": 400,
|
|
187
|
+
"cacheSeconds": 120, "minConfidence": 0.5, "blockWithoutUI": false,
|
|
188
|
+
"blockOn": { "destructive": 0.9, "exfiltration": 0.7, "beyondScope": 0.85, "impact": 2.5 } },
|
|
189
|
+
"output": { "enabled": true, "tools": ["bash"], "outputChars": 2000, "leakThreshold": 0.9, "minConfidence": 0.6 }
|
|
190
|
+
}
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
`"display": "plain"` uses the original's exact wording and no glyphs (the equivalence scenarios run in it). The API key
|
|
194
|
+
resolves in the order `TYPESAFE_API_KEY`, `apiKey`, `apiKeyFile` (`~/` expanded); `/jev` reports which one is in use.
|
|
195
|
+
`blockOn.impact` is on the 0 to 3 damage rubric, and `minConfidence` gates that dimension only (the three noul
|
|
196
|
+
questions return a probability and no confidence).
|
|
197
|
+
|
|
198
|
+
### What a project can change
|
|
199
|
+
|
|
200
|
+
A repository you open must not be able to send your data somewhere or switch judging on. So a project file **cannot** set
|
|
201
|
+
`enabled`, `acknowledged`, `backend`, `endpoint`, `model`, `apiKey`, `apiKeyFile`, `timeoutMs`, `retries` or `display`
|
|
202
|
+
(each is reported once as "ignored project setting"), and it can only **lower** what leaves the machine:
|
|
203
|
+
`maxStateChars`, `gate.argumentChars` and `output.outputChars` can shrink, and `gate.tools` / `output.tools` can pick among
|
|
204
|
+
the tools your own config already judges. Thresholds, mode and the rest may be tuned: they cannot leak anything. A project
|
|
205
|
+
that relaxes the protection your own config sets up (turns the gate or the output judge off, `enforce` into `shadow`,
|
|
206
|
+
`blockWithoutUI` off, raises a threshold, or judges fewer tools) is applied, since it only sends less, but never silently:
|
|
207
|
+
each start-up warns "`<project file>` relaxes the gate your own `pi-jev.json` sets: ..." with the settings it changed.
|
|
208
|
+
|
|
209
|
+
## What leaves the machine
|
|
210
|
+
|
|
211
|
+
Each judgment sends, to the judge named in `/jev`: the working directory, the tool name, your last message (first 1200
|
|
212
|
+
characters) and the tool's arguments. For `write` and `edit` those arguments contain file content. The output judge sends the
|
|
213
|
+
first `output.outputChars` characters of a `bash` result plus the same arguments. Any string longer than `gate.argumentChars`
|
|
214
|
+
(400) is cut at any nesting depth and replaced with `…[N chars elided]`; the omitted text never leaves. Set `gate.tools` to
|
|
215
|
+
`["bash"]` to keep file content out of the gate entirely, or lower a limit. The whole document is held to `maxStateChars`.
|
|
216
|
+
`jev_ask` sends the text the model passes to it (cut to `maxStateChars`), and `/jev check` sends the text you type after it.
|
|
217
|
+
The API key is sent only in the `Authorization` header to your endpoint, and notification text is scrubbed of it.
|
|
218
|
+
|
|
219
|
+
## Differences from pi-jev
|
|
220
|
+
|
|
221
|
+
This port follows y0usaf's behavior and differs only where the roadmap's code review found defects or the owner's rules
|
|
222
|
+
require it. Each is a test in [`port/PORT.md`](port/PORT.md) (C1 to C10): a verdict cache bound to the user's request and
|
|
223
|
+
the judge (C1), no project-chosen credential destination and no built-in endpoint (C2), `maxStateChars` applied (C3),
|
|
224
|
+
opt-in and disclosure (C4), strict typed-response validation (C5), a truthful "unavailable" status (C6), long strings
|
|
225
|
+
elided at every depth (C7), duplicate `jev_ask` ids refused (C8), `/jev` reporting the real key source and clean outputs
|
|
226
|
+
(C9), and `jev_ask` refusing while off (C10). The review of this port added R1 to R5 (same file): a failed enforce dialog
|
|
227
|
+
blocks, the default judge follows a model switch, the disclosure names `jev_ask` and `/jev check`, a project that relaxes
|
|
228
|
+
the gate is reported, and more trust-boundary tests.
|
|
229
|
+
|
|
230
|
+
## Proof
|
|
231
|
+
|
|
232
|
+
`port/` holds the evidence: the unmodified original (`port/oracle/`, MIT), the scenarios, the recorded Pi traces, the
|
|
233
|
+
mutation list and [`port/PORT.md`](port/PORT.md). Go tests drive the real SDK through a fake PiG host; the differential
|
|
234
|
+
scenarios run the original under Pi (traces re-recorded on Pi 1.0.1) and the port under PiG 0.4 against a fake judge endpoint and compare every
|
|
235
|
+
request and UI effect.
|
|
236
|
+
|
|
237
|
+
MIT. Original © y0usaf; port © Michael Kinsy. Depends on the shared client Package `components/typesafe`.
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
package jev
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"fmt"
|
|
5
|
+
"strings"
|
|
6
|
+
|
|
7
|
+
sdk "github.com/MichaelKinsy/PiG/extensions/sdk"
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
// jev_ask: the model asks typed questions itself. It sends the model-chosen text to
|
|
11
|
+
// the judge, so it exists only while Jev is on and refuses when it is off.
|
|
12
|
+
|
|
13
|
+
func (x *ext) registerAsk(ctx sdk.Context) {
|
|
14
|
+
x.mu.Lock()
|
|
15
|
+
done := x.askReg
|
|
16
|
+
x.askReg = true
|
|
17
|
+
x.mu.Unlock()
|
|
18
|
+
if done {
|
|
19
|
+
return
|
|
20
|
+
}
|
|
21
|
+
str := func(desc string) sdk.Schema { return sdk.Schema{"type": "string", "description": desc} }
|
|
22
|
+
question := sdk.Schema{
|
|
23
|
+
"type": "object",
|
|
24
|
+
"properties": sdk.Schema{
|
|
25
|
+
"id": str("Short key for this question. The answer comes back under it."),
|
|
26
|
+
"type": sdk.Schema{"type": "string", "enum": []any{"noul", "choice", "score"}, "description": "noul = yes/no probability, choice = pick one option, score = value on a rubric"},
|
|
27
|
+
"instructions": str("The one thing to judge. One specific, well-scoped gut-check per question."),
|
|
28
|
+
"options": sdk.Schema{"type": "array", "description": "choice only: the options to choose between", "items": sdk.Schema{
|
|
29
|
+
"type": "object", "properties": sdk.Schema{"name": str("Option key"), "description": str("When this option applies")}, "required": []any{"name"}}},
|
|
30
|
+
"levels": sdk.Schema{"type": "array", "description": "score only: ordered rubric levels, lowest first, at least two", "items": sdk.Schema{"type": "string"}},
|
|
31
|
+
},
|
|
32
|
+
"required": []any{"id", "type", "instructions"},
|
|
33
|
+
}
|
|
34
|
+
ctx.RegisterTool(sdk.ToolDefinition{
|
|
35
|
+
Name: "jev_ask",
|
|
36
|
+
Label: "Jev Ask",
|
|
37
|
+
Description: "Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.",
|
|
38
|
+
PromptSnippet: "Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers",
|
|
39
|
+
PromptGuidelines: []string{
|
|
40
|
+
"Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.",
|
|
41
|
+
"Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself.",
|
|
42
|
+
},
|
|
43
|
+
Parameters: sdk.Schema{
|
|
44
|
+
"type": "object",
|
|
45
|
+
"properties": sdk.Schema{
|
|
46
|
+
"state": str("The text to judge: tool output, a diff, a message, a document excerpt."),
|
|
47
|
+
"questions": sdk.Schema{"type": "array", "description": "One or more questions. All are evaluated in parallel against the same state.", "items": question},
|
|
48
|
+
},
|
|
49
|
+
"required": []any{"state", "questions"},
|
|
50
|
+
},
|
|
51
|
+
Execute: x.execAsk,
|
|
52
|
+
})
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
func askFail(format string, a ...any) (any, error) {
|
|
56
|
+
return sdk.ToolResult{Content: "jev_ask: " + fmt.Sprintf(format, a...), Details: map[string]any{"ok": false}}, nil
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
func (x *ext) execAsk(ctx sdk.Context, params map[string]any) (any, error) {
|
|
60
|
+
s := x.snapshot()
|
|
61
|
+
if !s.on || s.be == nil {
|
|
62
|
+
return askFail("Jev is off (the user turns it on with /jev on); nothing was sent")
|
|
63
|
+
}
|
|
64
|
+
state, ok := params["state"].(string)
|
|
65
|
+
if !ok {
|
|
66
|
+
return askFail("state must be text")
|
|
67
|
+
}
|
|
68
|
+
raw, _ := params["questions"].([]any)
|
|
69
|
+
qs := make([]question, 0, len(raw))
|
|
70
|
+
for _, item := range raw {
|
|
71
|
+
m, _ := item.(map[string]any)
|
|
72
|
+
q, msg := toQuestion(m)
|
|
73
|
+
if msg != "" {
|
|
74
|
+
return askFail("%s", msg)
|
|
75
|
+
}
|
|
76
|
+
qs = append(qs, q)
|
|
77
|
+
}
|
|
78
|
+
if err := validateQuestions(qs); err != nil {
|
|
79
|
+
return askFail("%s", err.Error())
|
|
80
|
+
}
|
|
81
|
+
gctx, cancel := goContext(ctx)
|
|
82
|
+
defer cancel()
|
|
83
|
+
// maxStateChars caps what the model can send, like the gate's state.
|
|
84
|
+
resp, err := s.be.Ask(gctx, ctx, jsonString(elide(state, s.cfg.MaxStateChars)), true, qs)
|
|
85
|
+
if err != nil {
|
|
86
|
+
return askFail("%s", x.redact(err.Error()))
|
|
87
|
+
}
|
|
88
|
+
details := map[string]any{"ok": true, "model": resp.Model, "answers": detailAnswers(resp)}
|
|
89
|
+
if resp.Usage != nil {
|
|
90
|
+
details["usage"] = map[string]any{"input_tokens": resp.Usage.Input, "output_tokens": resp.Usage.Output}
|
|
91
|
+
}
|
|
92
|
+
return sdk.ToolResult{Content: renderAnswers(resp, qs), Details: details}, nil
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// toQuestion returns a question, or a message explaining why the shape is invalid.
|
|
96
|
+
func toQuestion(m map[string]any) (question, string) {
|
|
97
|
+
id, _ := m["id"].(string)
|
|
98
|
+
typ, _ := m["type"].(string)
|
|
99
|
+
instr, _ := m["instructions"].(string)
|
|
100
|
+
q := question{ID: id, Type: typ, Instructions: instr}
|
|
101
|
+
switch typ {
|
|
102
|
+
case "choice":
|
|
103
|
+
opts, _ := m["options"].([]any)
|
|
104
|
+
if len(opts) == 0 {
|
|
105
|
+
return q, fmt.Sprintf("question %q: choice needs at least one option", id)
|
|
106
|
+
}
|
|
107
|
+
for _, o := range opts {
|
|
108
|
+
om, _ := o.(map[string]any)
|
|
109
|
+
name, _ := om["name"].(string)
|
|
110
|
+
var d *string
|
|
111
|
+
if s, ok := om["description"].(string); ok {
|
|
112
|
+
d = &s
|
|
113
|
+
}
|
|
114
|
+
q.Options = append(q.Options, option{name, d})
|
|
115
|
+
}
|
|
116
|
+
case "score":
|
|
117
|
+
lv, _ := m["levels"].([]any)
|
|
118
|
+
if len(lv) < 2 {
|
|
119
|
+
return q, fmt.Sprintf("question %q: score needs at least two levels", id)
|
|
120
|
+
}
|
|
121
|
+
for _, l := range lv {
|
|
122
|
+
s, _ := l.(string)
|
|
123
|
+
q.Levels = append(q.Levels, s)
|
|
124
|
+
}
|
|
125
|
+
case "noul":
|
|
126
|
+
default:
|
|
127
|
+
return q, fmt.Sprintf("question %q: type must be noul, choice or score", id)
|
|
128
|
+
}
|
|
129
|
+
return q, ""
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
func renderAnswers(r *response, qs []question) string {
|
|
133
|
+
lines := []string{"model " + r.Model}
|
|
134
|
+
for _, q := range qs {
|
|
135
|
+
a := r.Answers[q.ID]
|
|
136
|
+
text := describeAnswer(&a)
|
|
137
|
+
if a.Type == "choice" {
|
|
138
|
+
text = formatChoice(a)
|
|
139
|
+
}
|
|
140
|
+
lines = append(lines, fmt.Sprintf("%s: %s <- %s", q.ID, text, q.Instructions))
|
|
141
|
+
}
|
|
142
|
+
if r.Usage != nil {
|
|
143
|
+
lines = append(lines, fmt.Sprintf("tokens %d in / %d out", int(r.Usage.Input), int(r.Usage.Output)))
|
|
144
|
+
}
|
|
145
|
+
return strings.Join(lines, "\n")
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
// detailAnswers is the answers as the judge sent them (the original's details.answers).
|
|
149
|
+
func detailAnswers(r *response) map[string]any {
|
|
150
|
+
out := map[string]any{}
|
|
151
|
+
for id, a := range r.Answers {
|
|
152
|
+
probs := map[string]any{}
|
|
153
|
+
for _, p := range a.Probabilities {
|
|
154
|
+
probs[p.Name] = p.Value
|
|
155
|
+
}
|
|
156
|
+
switch a.Type {
|
|
157
|
+
case "noul":
|
|
158
|
+
out[id] = map[string]any{"type": "noul", "noul": a.Noul}
|
|
159
|
+
case "choice":
|
|
160
|
+
out[id] = map[string]any{"type": "choice", "choice": a.Choice, "confidence": a.Confidence, "probabilities": probs}
|
|
161
|
+
default:
|
|
162
|
+
out[id] = map[string]any{"type": "score", "score": a.Score, "confidence": a.Confidence, "legend": a.Legend, "probabilities": probs}
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
return out
|
|
166
|
+
}
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
package jev_test
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"encoding/json"
|
|
5
|
+
"strings"
|
|
6
|
+
"testing"
|
|
7
|
+
)
|
|
8
|
+
|
|
9
|
+
// jev_ask (src/index.ts registerTool, toQuestion, renderAnswers).
|
|
10
|
+
|
|
11
|
+
func askHost(t *testing.T, body any, extra map[string]any) (*Host, *fakeJev) {
|
|
12
|
+
t.Helper()
|
|
13
|
+
e := newEnv(t)
|
|
14
|
+
t.Setenv("TYPESAFE_API_KEY", testKey)
|
|
15
|
+
srv := newFakeJev(t, always(body))
|
|
16
|
+
e.writeGlobal(t, jevConfig(srv, extra))
|
|
17
|
+
return start(t, e, newHostState(), HostOptions{}), srv
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
func askResult(t *testing.T, h *Host, params map[string]any) (string, map[string]any) {
|
|
21
|
+
t.Helper()
|
|
22
|
+
h.adoptDynamicTools()
|
|
23
|
+
raw, failure := h.Tool("jev_ask", params)
|
|
24
|
+
if failure != "" {
|
|
25
|
+
t.Fatalf("jev_ask failed: %s", failure)
|
|
26
|
+
}
|
|
27
|
+
var r struct {
|
|
28
|
+
Content json.RawMessage `json:"content"`
|
|
29
|
+
Details map[string]any `json:"details"`
|
|
30
|
+
}
|
|
31
|
+
if err := json.Unmarshal(raw, &r); err != nil {
|
|
32
|
+
t.Fatalf("result %s: %v", raw, err)
|
|
33
|
+
}
|
|
34
|
+
var s string
|
|
35
|
+
if json.Unmarshal(r.Content, &s) != nil {
|
|
36
|
+
var blocks []struct {
|
|
37
|
+
Text string `json:"text"`
|
|
38
|
+
}
|
|
39
|
+
_ = json.Unmarshal(r.Content, &blocks)
|
|
40
|
+
for _, b := range blocks {
|
|
41
|
+
s += b.Text
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
return s, r.Details
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
func TestAsk_RegisteredOnlyWhenEnabledWithAKey(t *testing.T) {
|
|
48
|
+
e := newEnv(t)
|
|
49
|
+
h := start(t, e, newHostState(), HostOptions{})
|
|
50
|
+
if len(h.CallsTo("registerTool")) != 0 {
|
|
51
|
+
t.Error("jev_ask registered while Jev is off")
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
func TestAsk_RendersTypedAnswers(t *testing.T) {
|
|
56
|
+
body := map[string]any{"model": "jev-test", "usage": map[string]any{"input_tokens": 12, "output_tokens": 3}, "answers": map[string]any{
|
|
57
|
+
"relevant": noul(0.93),
|
|
58
|
+
"label": map[string]any{"type": "choice", "choice": "bug", "confidence": 0.9,
|
|
59
|
+
"probabilities": map[string]any{"feature": 0.1, "bug": 0.9}},
|
|
60
|
+
"quality": score(1.75, 0.8),
|
|
61
|
+
}}
|
|
62
|
+
h, srv := askHost(t, body, nil)
|
|
63
|
+
got, details := askResult(t, h, map[string]any{
|
|
64
|
+
"state": "the diff",
|
|
65
|
+
"questions": []any{
|
|
66
|
+
map[string]any{"id": "relevant", "type": "noul", "instructions": "Is this relevant?"},
|
|
67
|
+
map[string]any{"id": "label", "type": "choice", "instructions": "Which bucket?",
|
|
68
|
+
"options": []any{map[string]any{"name": "bug", "description": "Defect"}, map[string]any{"name": "feature"}}},
|
|
69
|
+
map[string]any{"id": "quality", "type": "score", "instructions": "How thorough?", "levels": []any{"Superficial", "Adequate", "Thorough"}},
|
|
70
|
+
},
|
|
71
|
+
})
|
|
72
|
+
want := strings.Join([]string{
|
|
73
|
+
"model jev-test",
|
|
74
|
+
"relevant: yes 0.93 <- Is this relevant?",
|
|
75
|
+
"label: bug (conf 0.90) [bug 0.90, feature 0.10] <- Which bucket?",
|
|
76
|
+
"quality: 1.75/3 (conf 0.80) <- How thorough?",
|
|
77
|
+
"tokens 12 in / 3 out",
|
|
78
|
+
}, "\n")
|
|
79
|
+
// Note: the legend in the score() fixture has four entries, so levels = 3.
|
|
80
|
+
if got != want {
|
|
81
|
+
t.Errorf("rendered:\n%s\nwant:\n%s", got, want)
|
|
82
|
+
}
|
|
83
|
+
if details["ok"] != true {
|
|
84
|
+
t.Errorf("details = %v", details)
|
|
85
|
+
}
|
|
86
|
+
req := srv.first(t)
|
|
87
|
+
if req.Body["state"] != "the diff" {
|
|
88
|
+
t.Errorf("state = %v", req.Body["state"])
|
|
89
|
+
}
|
|
90
|
+
qs := req.Body["questions"].(map[string]any)
|
|
91
|
+
label := qs["label"].(map[string]any)
|
|
92
|
+
crit := label["criteria"].(map[string]any)
|
|
93
|
+
if crit["bug"] != "Defect" || crit["feature"] != nil {
|
|
94
|
+
t.Errorf("choice criteria = %v (option without description must be null)", crit)
|
|
95
|
+
}
|
|
96
|
+
if _, has := crit["feature"]; !has {
|
|
97
|
+
t.Error("option without description must be sent as null, not omitted")
|
|
98
|
+
}
|
|
99
|
+
if lv := qs["quality"].(map[string]any)["criteria"].([]any); len(lv) != 3 {
|
|
100
|
+
t.Errorf("levels = %v", lv)
|
|
101
|
+
}
|
|
102
|
+
noulQ := qs["relevant"].(map[string]any)
|
|
103
|
+
if _, has := noulQ["criteria"]; has {
|
|
104
|
+
t.Errorf("noul question sent criteria: %v", noulQ)
|
|
105
|
+
}
|
|
106
|
+
if got := strings.Join(questionOrder(req.Raw), ","); got != "relevant,label,quality" {
|
|
107
|
+
t.Errorf("order = %s", got)
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
func TestAsk_InvalidQuestionShapesAreExplainedWithoutARequest(t *testing.T) {
|
|
112
|
+
h, srv := askHost(t, clearGate(), nil)
|
|
113
|
+
cases := map[string]struct {
|
|
114
|
+
q map[string]any
|
|
115
|
+
want string
|
|
116
|
+
}{
|
|
117
|
+
"choice without options": {map[string]any{"id": "c", "type": "choice", "instructions": "x"}, `jev_ask: question "c": choice needs at least one option`},
|
|
118
|
+
"choice empty options": {map[string]any{"id": "c", "type": "choice", "instructions": "x", "options": []any{}}, `jev_ask: question "c": choice needs at least one option`},
|
|
119
|
+
"score one level": {map[string]any{"id": "s", "type": "score", "instructions": "x", "levels": []any{"only"}}, `jev_ask: question "s": score needs at least two levels`},
|
|
120
|
+
"score no levels": {map[string]any{"id": "s", "type": "score", "instructions": "x"}, `jev_ask: question "s": score needs at least two levels`},
|
|
121
|
+
}
|
|
122
|
+
for name, c := range cases {
|
|
123
|
+
got, details := askResult(t, h, map[string]any{"state": "s", "questions": []any{c.q}})
|
|
124
|
+
if got != c.want || details["ok"] != false {
|
|
125
|
+
t.Errorf("%s: got %q details %v, want %q", name, got, details, c.want)
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
if srv.count() != 0 {
|
|
129
|
+
t.Errorf("requests = %d", srv.count())
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
func TestAsk_BlankInstructionsAndNoQuestionsAreRefused(t *testing.T) {
|
|
134
|
+
h, srv := askHost(t, clearGate(), nil)
|
|
135
|
+
got, _ := askResult(t, h, map[string]any{"state": "s", "questions": []any{map[string]any{"id": "a", "type": "noul", "instructions": " "}}})
|
|
136
|
+
if got != `jev_ask: question "a": instructions are required` {
|
|
137
|
+
t.Errorf("blank instructions: %q", got)
|
|
138
|
+
}
|
|
139
|
+
got, _ = askResult(t, h, map[string]any{"state": "s", "questions": []any{}})
|
|
140
|
+
if got != "jev_ask: no questions provided" {
|
|
141
|
+
t.Errorf("no questions: %q", got)
|
|
142
|
+
}
|
|
143
|
+
if srv.count() != 0 {
|
|
144
|
+
t.Errorf("requests = %d", srv.count())
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
// C8: the original let a duplicate id silently replace the earlier question.
|
|
149
|
+
func TestCorrection_DuplicateQuestionIdIsRefused(t *testing.T) {
|
|
150
|
+
h, srv := askHost(t, clearGate(), nil)
|
|
151
|
+
q := map[string]any{"id": "a", "type": "noul", "instructions": "x"}
|
|
152
|
+
got, _ := askResult(t, h, map[string]any{"state": "s", "questions": []any{q, q}})
|
|
153
|
+
if got != `jev_ask: duplicate question id "a"` || srv.count() != 0 {
|
|
154
|
+
t.Errorf("got %q, requests %d", got, srv.count())
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
func TestAsk_ServerErrorIsReportedNotThrown(t *testing.T) {
|
|
159
|
+
h, _ := askHost(t, "x", nil)
|
|
160
|
+
e := newEnv(t)
|
|
161
|
+
t.Setenv("TYPESAFE_API_KEY", testKey)
|
|
162
|
+
srv := newFakeJev(t, func(int, recordedRequest) jevReply { return jevReply{status: 401, body: "denied " + testKey} })
|
|
163
|
+
e.writeGlobal(t, jevConfig(srv, nil))
|
|
164
|
+
h = start(t, e, newHostState(), HostOptions{})
|
|
165
|
+
got, details := askResult(t, h, map[string]any{"state": "s", "questions": []any{map[string]any{"id": "a", "type": "noul", "instructions": "x"}}})
|
|
166
|
+
if !strings.HasPrefix(got, "jev_ask: ") || !strings.Contains(got, "401") || strings.Contains(got, testKey) || !strings.Contains(got, "[redacted]") || details["ok"] != false {
|
|
167
|
+
t.Errorf("got %q %v", got, details)
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
func TestAsk_MissingAnswerIsAnErrorNotAVerdict(t *testing.T) {
|
|
172
|
+
h, _ := askHost(t, map[string]any{"model": "m", "answers": map[string]any{}}, nil)
|
|
173
|
+
got, details := askResult(t, h, map[string]any{"state": "s", "questions": []any{map[string]any{"id": "a", "type": "noul", "instructions": "x"}}})
|
|
174
|
+
if !strings.HasPrefix(got, "jev_ask: ") || !strings.Contains(got, `did not answer "a"`) || details["ok"] != false {
|
|
175
|
+
t.Errorf("got %q %v", got, details)
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
func TestAsk_AnswerForAnotherTypeIsAnError(t *testing.T) {
|
|
180
|
+
h, _ := askHost(t, map[string]any{"model": "m", "answers": map[string]any{"a": score(1, 0.5)}}, nil)
|
|
181
|
+
got, _ := askResult(t, h, map[string]any{"state": "s", "questions": []any{map[string]any{"id": "a", "type": "noul", "instructions": "x"}}})
|
|
182
|
+
if !strings.Contains(got, `"a" is not a noul`) {
|
|
183
|
+
t.Errorf("got %q", got)
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
func TestAsk_HonoursOff(t *testing.T) {
|
|
188
|
+
h, srv := askHost(t, clearGate(), nil)
|
|
189
|
+
if failure := h.Command("jev", "off"); failure != "" {
|
|
190
|
+
t.Fatal(failure)
|
|
191
|
+
}
|
|
192
|
+
got, details := askResult(t, h, map[string]any{"state": "s", "questions": []any{map[string]any{"id": "a", "type": "noul", "instructions": "x"}}})
|
|
193
|
+
if srv.count() != 0 || details["ok"] != false || !strings.Contains(got, "off") {
|
|
194
|
+
t.Errorf("jev_ask sent content while Jev is off: %q %v (%d requests)", got, details, srv.count())
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
func TestAsk_ToolMetadata(t *testing.T) {
|
|
199
|
+
h, _ := askHost(t, clearGate(), nil)
|
|
200
|
+
calls := h.CallsTo("registerTool")
|
|
201
|
+
if len(calls) != 1 {
|
|
202
|
+
t.Fatalf("registerTool calls = %d", len(calls))
|
|
203
|
+
}
|
|
204
|
+
a := calls[0].Args
|
|
205
|
+
if a["name"] != "jev_ask" || a["label"] != "Jev Ask" {
|
|
206
|
+
t.Errorf("tool = %v", a)
|
|
207
|
+
}
|
|
208
|
+
mustContain(t, "description", a["description"].(string), "typed questions", "calibrated answers")
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
func TestCorrection_AskStateIsCappedAtMaxStateChars(t *testing.T) {
|
|
212
|
+
h, srv := askHost(t, map[string]any{"model": "m", "answers": map[string]any{"a": noul(0.5)}}, map[string]any{"maxStateChars": 100})
|
|
213
|
+
askResult(t, h, map[string]any{"state": strings.Repeat("s", 500), "questions": []any{map[string]any{"id": "a", "type": "noul", "instructions": "x"}}})
|
|
214
|
+
got, _ := srv.first(t).Body["state"].(string)
|
|
215
|
+
if got != strings.Repeat("s", 100)+"…[400 chars elided]" {
|
|
216
|
+
t.Errorf("state = %q", got)
|
|
217
|
+
}
|
|
218
|
+
}
|