@pi-in-go/pigpen-jev 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CREDITS.md +22 -0
- package/LICENSE +22 -0
- package/README.md +237 -0
- package/extensions/jev/ask.go +166 -0
- package/extensions/jev/ask_test.go +218 -0
- package/extensions/jev/backend.go +128 -0
- package/extensions/jev/bench_test.go +64 -0
- package/extensions/jev/boundaries_test.go +159 -0
- package/extensions/jev/command.go +224 -0
- package/extensions/jev/commands_test.go +214 -0
- package/extensions/jev/config.go +450 -0
- package/extensions/jev/errors_test.go +191 -0
- package/extensions/jev/extension.go +391 -0
- package/extensions/jev/fakehost_test.go +548 -0
- package/extensions/jev/gate.go +125 -0
- package/extensions/jev/gate_test.go +610 -0
- package/extensions/jev/gatekey_test.go +24 -0
- package/extensions/jev/go.mod +9 -0
- package/extensions/jev/go.sum +2 -0
- package/extensions/jev/go.work +10 -0
- package/extensions/jev/helpers_test.go +404 -0
- package/extensions/jev/memo.go +88 -0
- package/extensions/jev/output.go +89 -0
- package/extensions/jev/output_test.go +187 -0
- package/extensions/jev/ownmodel_test.go +118 -0
- package/extensions/jev/render.go +136 -0
- package/extensions/jev/review_test.go +310 -0
- package/extensions/jev/source_test.go +57 -0
- package/extensions/jev/text.go +174 -0
- package/extensions/jev/trust_test.go +335 -0
- package/extensions/jev/types.go +227 -0
- package/libs/typesafe/CONTRACT.md +125 -0
- package/libs/typesafe/CREDITS.md +37 -0
- package/libs/typesafe/LICENSE +23 -0
- package/libs/typesafe/README.md +19 -0
- package/libs/typesafe/go.mod +3 -0
- package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
- package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
- package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
- package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
- package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
- package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
- package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
- package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
- package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
- package/libs/typesafe/libraries/ownmodel/run.go +288 -0
- package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
- package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
- package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
- package/libs/typesafe/libraries/typesafe/answers.go +268 -0
- package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
- package/libs/typesafe/libraries/typesafe/batch.go +80 -0
- package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
- package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
- package/libs/typesafe/libraries/typesafe/client.go +561 -0
- package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
- package/libs/typesafe/libraries/typesafe/doc.go +27 -0
- package/libs/typesafe/libraries/typesafe/entry.go +142 -0
- package/libs/typesafe/libraries/typesafe/env.go +11 -0
- package/libs/typesafe/libraries/typesafe/errors.go +310 -0
- package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
- package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
- package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
- package/libs/typesafe/libraries/typesafe/logging.go +160 -0
- package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
- package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
- package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
- package/libs/typesafe/libraries/typesafe/questions.go +490 -0
- package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
- package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
- package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
- package/libs/typesafe/libraries/typesafe/retry.go +350 -0
- package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
- package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
- package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
- package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
- package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
- package/libs/typesafe/libraries/typesafe/version.go +10 -0
- package/libs/typesafe/package.json +37 -0
- package/libs/typesafe/provenance.json +49 -0
- package/package.json +42 -0
- package/port/PORT.md +107 -0
- package/port/e2e/gate-and-output.py +35 -0
- package/port/e2e/jev-ask.py +36 -0
- package/port/e2e/model-switch.py +44 -0
- package/port/e2e/off-by-default.py +34 -0
- package/port/gen-scenarios.py +103 -0
- package/port/golden/cache-identical-calls.jsonl +30 -0
- package/port/golden/clear.jsonl +22 -0
- package/port/golden/commands.jsonl +43 -0
- package/port/golden/enforce-accept.jsonl +23 -0
- package/port/golden/enforce-decline.jsonl +22 -0
- package/port/golden/jev-ask.jsonl +20 -0
- package/port/golden/output-advice.jsonl +23 -0
- package/port/golden/output-leak.jsonl +24 -0
- package/port/golden/output-low-confidence.jsonl +22 -0
- package/port/golden/shadow-flagged.jsonl +23 -0
- package/port/golden/unjudged-tools.jsonl +19 -0
- package/port/golden/write-elision.jsonl +21 -0
- package/port/mutate-unit.py +63 -0
- package/port/mutations.json +578 -0
- package/port/oracle/LICENSE +21 -0
- package/port/oracle/README.md +181 -0
- package/port/oracle/SHA256SUMS +8 -0
- package/port/oracle/package.json +43 -0
- package/port/oracle/src/client.ts +409 -0
- package/port/oracle/src/config.ts +363 -0
- package/port/oracle/src/gate.ts +229 -0
- package/port/oracle/src/index.ts +649 -0
- package/port/oracle/src/output.ts +163 -0
- package/port/red-run.log +309 -0
- package/port/scenarios/cache-identical-calls.json +71 -0
- package/port/scenarios/clear.json +61 -0
- package/port/scenarios/commands.json +119 -0
- package/port/scenarios/enforce-accept.json +66 -0
- package/port/scenarios/enforce-decline.json +57 -0
- package/port/scenarios/jev-ask.json +83 -0
- package/port/scenarios/output-advice.json +61 -0
- package/port/scenarios/output-leak.json +61 -0
- package/port/scenarios/output-low-confidence.json +61 -0
- package/port/scenarios/shadow-flagged.json +61 -0
- package/port/scenarios/unjudged-tools.json +55 -0
- package/port/scenarios/write-elision.json +53 -0
- package/provenance.json +18 -0
package/port/PORT.md
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
# Port record: jev
|
|
2
|
+
|
|
3
|
+
| Input | Identity |
|
|
4
|
+
|---|---|
|
|
5
|
+
| Original | [y0usaf/pi-jev](https://github.com/y0usaf/pi-jev) 0.2.2, commit `88e5fb3888948e7065110d47cdf6ac57abb71ba4`, MIT (y0usaf). Vendored unmodified in `oracle/` (`src/*.ts`, `package.json`, `README.md`, `LICENSE`; sha256 of the sources in `oracle/SHA256SUMS`) |
|
|
6
|
+
| Oracle | Pi 0.87.1 (real, `--mode rpc`), Node 24.19.0; every scenario is also identical under PiG's Node runtime (host check) |
|
|
7
|
+
| Target | PiG 0.3.0+0.87.1 (a pre-release build, `63c6ba456`), go1.27.1; the 0.4.0 (Pi 0.99.1) notes are at the end |
|
|
8
|
+
| Client | Shared Go Package `components/typesafe` (lane `pigpen-typesafe-client`, CONTRACT `841b25f`): the TypeSafe backend and the own-model backend. This port has no HTTP client of its own |
|
|
9
|
+
| Original's tests | none (no test script, no test files upstream). The cases are derived from every branch of the source, so there are no upstream test titles to twin: **0 exact twins, 0 skipped**; the named skipped tests below record the gaps |
|
|
10
|
+
|
|
11
|
+
## Contract table
|
|
12
|
+
|
|
13
|
+
Upstream lines are in `oracle/src`. "Scenario" = `port/scenarios/<name>.json` (original under Pi == port under PiG); "Go" = a layer-1 test through the fake host.
|
|
14
|
+
|
|
15
|
+
| ID | Upstream | Behavior | Go | Scenario / test |
|
|
16
|
+
|---|---|---|---|---|
|
|
17
|
+
| M1 | `index.ts:130` `session_start` | reload config; warn about config problems; no key: one warning; status `jev: <mode>` | `sessionStart` | every scenario (startup status); `TestKey_MissingKeyWarnsOnceAndStaysInactive`, `TestConfig_*` |
|
|
18
|
+
| M2 | `index.ts:152` `tool_call` | judge `gate.tools`; shadow notifies, enforce asks, headless degrades unless `blockWithoutUI` | `toolCall` | shadow-flagged, clear, enforce-accept, enforce-decline; `TestGate_Enforce*` (incl. headless: **not reachable through RPC**, fake host, print/JSON mode) |
|
|
19
|
+
| M3 | `index.ts:195` `tool_result` | judge `output.tools`, append `[pi-jev] <notice>`, notify on a leak | `toolResult` | output-leak, output-advice, output-low-confidence; `TestOutput_*` |
|
|
20
|
+
| M4 | `index.ts:290-318` cache and in-flight sharing | identical input judged once per `cacheSeconds`; siblings share a request | `memo.go` | cache-identical-calls; `TestGate_IdenticalCallIsJudgedOncePerWindow`, `…SiblingCallsShareOneInFlightRequest`, `…CacheHoldsSixtyFourVerdicts` |
|
|
21
|
+
| M5 | `gate.ts:61` `GATE_QUESTIONS` | four questions, exact text, exact order | `gateQuestions` | every gate request in the goldens; `TestGate_RequestShape` |
|
|
22
|
+
| M6 | `gate.ts:101-142` `buildGateState`, `summarizeArguments` | state `{cwd, tool, arguments, platform, user_request}`; strings elided at `argumentChars` with `…[N chars elided]`; user request cut at 1200 with `…[truncated]` | `gateStateJSON`, `summarize` | write-elision; `TestGate_LongArgumentsAreElided`, `…ArgumentCharsCountUTF16Units`, `…UserRequestIsTruncatedTo1200Units` |
|
|
23
|
+
| M7 | `gate.ts:144-194` `evaluateGate` | thresholds inclusive; impact needs `minConfidence`; reasons text | `evaluateGate` | shadow-flagged; `TestGate_EachThresholdIsInclusive`, `…JustBelowEachThresholdIsClear`, `…ImpactBelowMinConfidenceDoesNotFlag` |
|
|
24
|
+
| M8 | `gate.ts:212` `judgmentKey` | stable key over tool and input (property order ignored) | `stableKey` | `TestGate_CacheKeyIgnoresPropertyOrder` |
|
|
25
|
+
| M9 | `output.ts:52-136` questions, `CLASS_ADVICE`, `evaluateOutput` | leak wins over class; advice table; class floor | `output.go` | output-leak, output-advice, output-low-confidence; `TestOutput_*` |
|
|
26
|
+
| M10 | `output.ts:89` `outputKey` | identical output judged once per 120 s | `extension.go` | cache-identical-calls; `TestOutput_IdenticalOutputIsJudgedOnce` |
|
|
27
|
+
| M11 | `index.ts:367-466` `/jev` | `on`, `off`, `mode`, `last`, `output`, `check`, status | `command.go` | commands; `TestCommand_*` |
|
|
28
|
+
| M12 | `index.ts:471-560` `jev_ask`, `toQuestion`, `renderAnswers` | typed questions, exact rendering, validation messages | `ask.go` | jev-ask; `TestAsk_*` |
|
|
29
|
+
| M13 | `index.ts:624` `lastUserRequest` | latest user text in the branch | `lastUserRequest` (`GetBranch`) | every gate request carries `user_request`; `TestGate_LatestUserMessageWins`, `…UserRequestIsTrimmed` |
|
|
30
|
+
| M14 | `client.ts:110-360` HTTP client, retries, timeouts, error texts, `redact` | request, retry on 429/529/5xx, per-attempt timeout | **shared client** `typesafe.Client` | the goldens' `http` events (request shape, headers); retries and error texts are the client Package's own tests |
|
|
31
|
+
| M15 | `config.ts` layering, key resolution, validation | defaults ← `<agent dir>/pi-jev.json` ← `<cwd>/.pig/pi-jev.json`; key env > `apiKey` > `apiKeyFile` | `config.go` | `TestKey_*`, `TestConfig_*`, `TestProject_*` |
|
|
32
|
+
|
|
33
|
+
SDK gap check (`pigeq gaps --ts port/oracle/src`): 0 blocking, 4 PARTIAL, all shown not to matter here: `ctx.model` (`GetModelInfo`, used only to name the judge), `ctx.signal` (`Context.Done`, mapped to a `context.Context`), `ctx.ui.confirm(opts.signal)` (line 188: the dialog closes when the request is cancelled), `tool.promptSnippet` (the scenarios' `system.inserted` part of the model request is identical under PiG 0.3.0: the snippet and the guidelines reach the prompt). `pigeq gaps --go extensions/jev`: clean.
|
|
34
|
+
|
|
35
|
+
## Corrections to the original (approved: roadmap "Jev identification", owner 2026-09-28, and the lane task)
|
|
36
|
+
|
|
37
|
+
Each has the upstream reproduction, the corrected expectation and its Go test (red first, in the RED commit). The equivalence scenarios avoid every one of them, so parity is claimed only where the port does what the original does.
|
|
38
|
+
|
|
39
|
+
| ID | Upstream defect (reproduction) | Correction | Test |
|
|
40
|
+
|---|---|---|---|
|
|
41
|
+
| C1 | `index.ts:290` cache key ignored the user's request: the same arguments after the request changed reused a verdict for the old intent (review probe 1: one HTTP request, second status `clear (enforce)`) | key binds tool, input, cwd and the user's request; the cache is cleared when the configuration (judge, model) is reloaded; errors are never cached | `TestCorrection_CacheIsBoundToTheUserRequest`, `…CacheIsBoundToTheModelAndEndpoint`, `TestGate_ErrorsAreNotCached` |
|
|
42
|
+
| C2 | `config.ts:166-197` a project file chose the endpoint while the environment key travelled to it (review probe 2) | backend, endpoint, model, key, timeout, retries, consent and display come **only** from the user's own config; a project can only lower what leaves the machine; no built-in endpoint or model name; the endpoint must be `https` (or `http` to localhost); `TYPESAFE_*` variables are ignored by the client | `TestCorrection_ProjectFileCannotRedirectTheEndpoint`, `…NoDefaultEndpointForTheHTTPBackend`, `…PlainHTTPToARemoteHostIsRefused`, `…ProjectCanOnlyNarrowWhatLeaves`, `…ProjectCannotRaiseTheStateCapOrOutputLimit`, `…TypeSafeEnvironmentCannotRedirectTheKey` |
|
|
43
|
+
| C3 | `config.ts:273` `maxStateChars` parsed, documented, never read (NEXT.md acknowledges it) | applied: long strings shrink first, then arguments, then the request are dropped; the same cap on `jev_ask`'s text | `TestCorrection_MaxStateCharsCapsTheState`, `…KeepsShortenedArguments`, `…AskStateIsCappedAtMaxStateChars` |
|
|
44
|
+
| C4 | a key alone switched judging on, in shadow mode, with no word about where content goes | opt-in: nothing is judged until `/jev on` (which shows the disclosure and asks) or `"enabled": true` in the user's own config; a start-up notice says what leaves and where until `"acknowledged": true`; `/jev` names the destination and the fail-open policy | `TestOptIn_*`, `TestDisclosure_*`, `TestCommand_Status*` |
|
|
45
|
+
| C5 | `client.ts:249-265` accepted a probability of 1.7, a score of 9, a choice outside its options, a confidence of 4 | strict typed validation; any violation is an unavailable verdict (fails open), never a number compared to a threshold | `TestCorrection_OutOfRangeProbabilityIsAnError`, `…ScoreOutsideTheRubricIsAnError`, `…ConfidenceOutsideZeroToOneIsAnError`, `…ChoiceConfidence…`, `TestErrors_ChoiceOutsideItsOptionsIsAnError`, `TestErrors_EmptyAnswersAreNotAClearVerdict` (twin of the 0.2.2 fix) |
|
|
46
|
+
| C6 | on an error `tool_call` returned silently and the footer kept the earlier `jev: clear` | the footer says `jev: unavailable (failing open)` | `TestCorrection_ErrorStatusIsUnavailableNotStaleClear`, `…OutputJudgeErrorAlsoShowsUnavailable` |
|
|
47
|
+
| C7 | `gate.ts:120`, `output.ts:143` elision stops below depth 4 / 3, so a deeper string left whole (found while porting; not in the review) | elided at every depth (hard stop at 64 levels) | `TestCorrection_DeepStringsAreElidedToo` |
|
|
48
|
+
| C8 | `index.ts:497` a duplicate `jev_ask` id silently replaced the earlier question | refused: `duplicate question id` | `TestCorrection_DuplicateQuestionIdIsRefused` |
|
|
49
|
+
| C9 | `/jev` said `apiKey (inline)` for a key from the environment; `lastOutput` was recorded only when a notice existed, so a clean output read as "no tool output judged yet" | the real key source; every judged output is recorded | `TestCommand_StatusIsTruthfulAboutTheKeySource`, `TestCorrection_CleanOutputIsRecordedToo` |
|
|
50
|
+
| C10 | `jev_ask` sent the model's text after `/jev off` | it exists only while Jev is on and refuses ("Jev is off") after `/jev off` | `TestAsk_HonoursOff`, `TestAsk_RegisteredOnlyWhenEnabledWithAKey` |
|
|
51
|
+
|
|
52
|
+
## Review corrections (rev-pigpen-jev)
|
|
53
|
+
|
|
54
|
+
Found by the adversarial review of this port; each has a test in `extensions/jev/review_test.go` that was red on `477a904`
|
|
55
|
+
(R1 to R4) or kills a mutant that survived the lane's tests (R5), and a `review-*` entry in `mutations.json`.
|
|
56
|
+
|
|
57
|
+
| ID | Defect in the port | Correction | Test |
|
|
58
|
+
|---|---|---|---|
|
|
59
|
+
| R1 | enforce mode: when the confirmation dialog failed (host UI error), `tool_call` failed open and ran the flagged call. The original's handler throws there and Pi blocks the call (`index.ts:188`; Pi `agent-session.ts` `_installAgentToolHooks`, "Extension failed, blocking execution") | blocked, reason `pi-jev: <verdict> (not confirmed: <error>)`. Fail-open covers an unavailable judge, not a missing approval | `TestReview_EnforceConfirmFailureBlocks` |
|
|
60
|
+
| R2 | the default judge was resolved once at `session_start`; after a model switch Jev kept sending content to the previous provider while the disclosure called it "the provider that already receives your conversation" | `model_select` rebuilds the default judge (unless the user's own config names `"model"`), drops cached verdicts, tells the user, and registers `jev_ask` if no judge was available before | `TestReview_DefaultJudgeFollowsTheSelectedModel`, `…ConfiguredJudgeIgnoresModelSwitch`, `…ModelSelectedLaterEnablesTheDefaultJudge`; end to end `e2e/model-switch.py` |
|
|
61
|
+
| R3 | the disclosure (`/jev on`, start-up notice) listed what the gate and output judge send, not `jev_ask`'s text (up to `maxStateChars`) or `/jev check`'s | listed; `/jev` "Sends" names `jev_ask` text | `TestReview_DisclosureNamesJevAskAndCheck` |
|
|
62
|
+
| R4 | a repository's `.pig/pi-jev.json` could silently turn the user's gate off, `enforce` into `shadow`, `blockWithoutUI` off, raise thresholds or drop judged tools | still applied (it only sends less), but reported at start-up with the settings it relaxed | `TestReview_ProjectThatRelaxesTheGateIsReported`, `…StricterProjectIsNotReported` |
|
|
63
|
+
| R5 | test gaps, not code defects: a project-applied `apiKeyFile`, `backend`, `model` or `acknowledged`, `http://` to a remote IP literal, an output cache key without the tool, and the C3 drop order all survived as mutants | guard tests | `TestReview_ProjectCannotChooseTheKeyFile`, `…CannotSwitchTheBackend`, `…CannotChooseTheModel`, `…CannotHideTheDisclosure`, `…PlainHTTPToARemoteIPIsRefused`, `…OutputCacheIsPerTool`, `…StateCapDropsArgumentsBeforeTheRequest` |
|
|
64
|
+
|
|
65
|
+
The Piglets selecting Jev (`piglets/jev`, `pig-with-batteries`) said `tools: []`, which per PiG's Piglet docs exposes none of
|
|
66
|
+
the extension's tools, yet the README promises `jev_ask`. It was exposed only because PiG 0.3.0 does not apply a Piglet's tool
|
|
67
|
+
scope to tools registered after start-up. Both now say `tools: [jev_ask]`.
|
|
68
|
+
|
|
69
|
+
## Other differences (mechanical, not corrections)
|
|
70
|
+
|
|
71
|
+
- **D1 endpoint.** The original's `endpoint` is the full request URL; the shared client wants the API root. A trailing `/v1/systemone` is removed, so the same setting works in both (the scenarios rely on it).
|
|
72
|
+
- **D2 argument key order.** The SDK decodes event data into `map[string]any`, so the arguments are sent with sorted keys. Named skipped test `TestGate_ArgumentsKeepInsertionOrder_GAP`.
|
|
73
|
+
- **D3 UTF-16.** Limits count UTF-16 units like JavaScript. A cut through a surrogate pair yields U+FFFD (a Go string cannot hold a lone surrogate); JSON stays valid. `TestGate_CutInsideSurrogatePairStaysValidUTF8`.
|
|
74
|
+
- **D4 `cacheSeconds` 0** disables the cache (upstream reused within the same millisecond).
|
|
75
|
+
- **D5 answer order.** `jev_ask` and `/jev last` list answers in question order; the original used the server's key order. The SDK result is a map, and the API answers in question order (the scenarios' fake server does).
|
|
76
|
+
- **D6 paths.** Agent directory `PIG_CODING_AGENT_DIR` (default `<PIG_HOME>/agent`), project directory `<cwd>/.pig`, file name kept (`pi-jev.json`).
|
|
77
|
+
- **D7 `/jev check`** needs Jev on (it sends the user's text): `TestCommand_CheckNeedsJevOn`.
|
|
78
|
+
- **D8 default judge.** The original always called the TypeSafe API (`api.typesafe.ai`, model `jev-latest`). Here the default judge is the model PiG is configured with (`"backend": "model"`), and the TypeSafe API is chosen with an explicit endpoint and model. `TestNoHardcodedProviderOrEndpoint` greps the sources.
|
|
79
|
+
- **D9 user message text blocks** are joined with no separator (SDK `BranchEntry`), the original joined with `\n`. `TestUserRequestJoinsTextBlocksWithNewline_GAP` (skipped, named).
|
|
80
|
+
- **D10 display.** `"display": "rich"` (default) adds glyphs, a verdict card and the destination; `"plain"` is the original's wording character for character (the scenarios run in it).
|
|
81
|
+
|
|
82
|
+
## Results (this revision)
|
|
83
|
+
|
|
84
|
+
- **Layer 1** (fake PiG host, real SDK over `net.Pipe`): 135 tests pass (121 from the lane, 14 from the review), 2 skipped with `-race -count=3` against the real shared client (`components/typesafe`). Two named skipped tests: `TestGate_ArgumentsKeepInsertionOrder_GAP` (D2), `TestUserRequestJoinsTextBlocksWithNewline_GAP` (D9). The own-model backend is tested through the template's `ModelStream` (`ownmodel_test.go`, 7 tests).
|
|
85
|
+
- **Red first**: `port/red-run.log`: on the registering-nothing stub 81 of 101 tests failed; the 18 that passed are the no-op cases (off by default, disabled, unjudged tool, no hardcoded provider), proved by the mutation check below.
|
|
86
|
+
- **`pigeq check`** (build 5, `agentFiles`): 12 of 12 scenarios identical to the traces the original recorded under Pi 0.87.1 (`pig-go == pi-ts`), 28 s. `pigeq record --ts` recorded them; each is also identical under PiG's Node runtime (`host check`). `port-gaps` and `exec-coverage` pass.
|
|
87
|
+
- **Mutations**: 81 from the lane, one per contract row and per no-op case, plus 15 `review-*` (`mutations.json`, `mutate-unit.py`). **96 of 96 killed, all by the unit layer** (review run); the lane's run was 81 of 81 (about 5 s each); none INVALID, none survived. The scenarios would also kill the wording and request-shape mutants, but the harness's own `pigeq mutate --unit` cannot run this port (its go.work makes a shared library Package unresolvable: FRICTION note, fix proposed), so `port/mutate-unit.py` builds each mutant with a go.work that works and runs the port's tests. `python3 port/mutate-unit.py` runs the list from any checkout (`PIG_SDK_DIR` or `pig reload --sdk-path`, `go` on PATH). A cache key without the working directory is also equivalent (the working directory is fixed for a session and every `session_start` clears the caches). Two mutants were removed as equivalent, with reasons: the judge id in the cache key (the caches are cleared on every configuration reload, so the judge cannot change under a cache) and the client's `Getenv` override (an explicit `BaseURL`, key, model and log level already take precedence over the environment; the line stays as defence in depth).
|
|
88
|
+
- **Binary**: `pig piglet build dist/staged/piglets/jev/piglet.yaml --format binary` builds with "Preparing fused Go members: jev" (pig 0.3.0+0.87.1, go1.27.1). `pig-with-batteries` cannot build a Binary because it selects the Node herdr reporter (`lock extension "extension": selected origin is unavailable`, RELEASE-BLOCKERS item 2, unrelated to Jev); the same manifest without herdr builds (58 MB).
|
|
89
|
+
- **End to end on the built Binaries** (`port/e2e/*.py`, real host in RPC mode, a scripted OpenAI-compatible model from `pigeq llm`, isolated dirs): (1) `gate-and-output.py`, the own-model backend, `/jev on` with its confirmation, a `bash` call: 4 model requests (agent, gate, output judge, agent), the shadow card and the leak notice appear, `/jev` shows the status; (2) `jev-ask.py`: `jev_ask` is registered at run time by `/jev on` and answered by the session model (3 requests); (3) `off-by-default.py` on the batteries Binary without herdr: no `/jev on` means 2 model requests (only the agent's) and no Jev UI output; (4) `model-switch.py` (review, R2): two scripted providers, `/jev on` on `eq-a`, RPC `set_model` to `eq-b`, one `bash` call: all 4 requests reach `eq-b` (the lane's Binary sent the gate and output judgments to `eq-a`). These use `--mode rpc`, not a TUI: no terminal screenshot was taken.
|
|
90
|
+
- **Cross-platform**: `GOOS=linux|darwin|windows go vet ./...` clean.
|
|
91
|
+
|
|
92
|
+
## Against PiG 0.4.0 (Pi 0.99.1)
|
|
93
|
+
|
|
94
|
+
Nothing here is Pi-version specific except the recorded goldens (Pi 0.87.1): re-record them under 0.99.1. Watch: the SDK may deliver event data with key order (D2 would go away); the `prompt snippet` host gap listed by `pigeq gaps` is already satisfied for 0.3.0 by the scenarios; built-in model access (`ModelRegistry`) is the own-model backend's only host dependency; the roadmap's built-in MCP does not touch this extension. `TYPESAFE_*` and the agent directory rules (D6) should be re-checked if 0.4.0 renames `PIG_CODING_AGENT_DIR`.
|
|
95
|
+
|
|
96
|
+
## Findings about the hosts
|
|
97
|
+
|
|
98
|
+
1. PiG hands `tool_call`/`tool_result` data over as `map[string]any` (D2).
|
|
99
|
+
2. `ModelRegistry.Complete` (used by the own-model backend) reaches the host through the `modelStream` call, which the SDK's fake-host template cannot answer; the own-model backend is proved end to end with a real binary instead (see the results).
|
|
100
|
+
3. Real Pi and PiG's Node runtime print a JSON-escaped `>` differently in the recorded model request (`>` vs `\u003e`): the scenarios avoid redirections in scripted commands.
|
|
101
|
+
|
|
102
|
+
## Re-verified on PiG 0.4.1 (porter-verify)
|
|
103
|
+
|
|
104
|
+
The golden traces were recorded again from the original under Pi 1.0.1 (host check: identical under PiG's Node runtime), with PiG 0.4.1 content (`5f948f86a`, `pig --version`
|
|
105
|
+
`0.3.1+1.0.1`) and normalizer v3, because Pi 1.0.x changed the trace format: a `prompt` response now carries
|
|
106
|
+
`data.disposition` (Pi 0.99.0, #9098). The bash tool's results now carry `structuredContent` (6 events), whose `wall_time_seconds` the harness normalizes (N6). An event-by-event diff against the previous traces shows no other
|
|
107
|
+
difference, and `pigeq check` passes on the Go port. Details: `docs/plan/progress/porter-verify.md`.
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import json, os, subprocess, sys, time, threading, queue, tempfile
|
|
2
|
+
BIN = sys.argv[1]; LLM = sys.argv[2] # pig binary, pigeq
|
|
3
|
+
root = tempfile.mkdtemp(prefix="jev-e2e-")
|
|
4
|
+
for d in ("home", "pighome", "agent", "work"): os.makedirs(f"{root}/{d}")
|
|
5
|
+
gate = {"answers": {"destructive": 0.99, "exfiltration": 0.1, "beyond_scope": 0.2, "impact": {"0": 0, "1": 0, "2": 0.1, "3": 0.9}}}
|
|
6
|
+
out = {"answers": {"leaks_secret": 0.96, "failure_class": {"transient": 0, "environment": 0, "code_bug": 0, "permission": 0, "user_error": 0, "no_failure": 1}}}
|
|
7
|
+
turns = [{"toolCalls": [{"name": "bash", "arguments": {"command": "echo API_KEY=abc123"}}]}, {"text": json.dumps(gate)}, {"text": json.dumps(out)}, {"text": "done"}]
|
|
8
|
+
json.dump(turns, open(f"{root}/turns.json", "w"))
|
|
9
|
+
llm = subprocess.Popen([LLM, "llm", "--script", f"{root}/turns.json", "--log", f"{root}/llm.log"], stdin=subprocess.PIPE, stdout=subprocess.PIPE, text=True)
|
|
10
|
+
base = llm.stdout.readline().strip().split()[-1]
|
|
11
|
+
json.dump({"providers": {"eq-llm": {"baseUrl": base, "api": "openai-completions", "apiKey": "eq-key", "models": [{"id": "eq-1", "name": "eq-1", "reasoning": False, "input": ["text"], "contextWindow": 100000, "maxTokens": 4096, "cost": {"input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0}}]}}}, open(f"{root}/agent/models.json", "w"))
|
|
12
|
+
env = {"PATH": os.environ["PATH"], "HOME": f"{root}/home", "PIG_HOME": f"{root}/pighome", "PIG_CODING_AGENT_DIR": f"{root}/agent", "PI_CODING_AGENT_DIR": f"{root}/agent", "TERM": "dumb", "LANG": "C.UTF-8", "PI_OFFLINE": "1", "PI_SKIP_VERSION_CHECK": "1"}
|
|
13
|
+
p = subprocess.Popen([BIN, "--mode", "rpc", "--no-session", "--offline", "--provider", "eq-llm", "--model", "eq-1"], cwd=f"{root}/work", env=env, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
|
|
14
|
+
q = queue.Queue()
|
|
15
|
+
threading.Thread(target=lambda: [q.put(json.loads(l)) for l in p.stdout if l.startswith("{")], daemon=True).start()
|
|
16
|
+
def send(o): p.stdin.write(json.dumps(o) + "\n"); p.stdin.flush()
|
|
17
|
+
ui = []
|
|
18
|
+
def pump(until, timeout=90):
|
|
19
|
+
end = time.time() + timeout
|
|
20
|
+
while time.time() < end:
|
|
21
|
+
try: ev = q.get(timeout=1)
|
|
22
|
+
except queue.Empty: continue
|
|
23
|
+
if ev.get("type") == "extension_ui_request":
|
|
24
|
+
ui.append({k: ev.get(k) for k in ("method", "title", "message", "statusKey", "statusText", "notifyType")})
|
|
25
|
+
if ev["method"] == "confirm": send({"type": "extension_ui_response", "id": ev["id"], "confirmed": True})
|
|
26
|
+
if until(ev): return ev
|
|
27
|
+
raise SystemExit("timeout waiting; ui=%s" % json.dumps(ui, indent=1))
|
|
28
|
+
send({"type": "prompt", "message": "/jev on", "id": "1"}); pump(lambda e: e.get("type") == "response" and e.get("id") == "1"); time.sleep(0.5)
|
|
29
|
+
send({"type": "prompt", "message": "please run echo API_KEY=abc123", "id": "2"}); pump(lambda e: e.get("type") == "agent_end")
|
|
30
|
+
time.sleep(0.5)
|
|
31
|
+
send({"type": "prompt", "message": "/jev", "id": "3"}); pump(lambda e: e.get("type") == "response" and e.get("id") == "3"); time.sleep(0.5)
|
|
32
|
+
for u in ui: print(json.dumps(u, ensure_ascii=False))
|
|
33
|
+
print("---- llm requests:", sum(1 for _ in open(f"{root}/llm.log")))
|
|
34
|
+
p.stdin.close(); llm.stdin.close(); p.wait(timeout=20)
|
|
35
|
+
print("root", root)
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import json, os, subprocess, sys, time, threading, queue, tempfile
|
|
2
|
+
BIN = sys.argv[1]; LLM = sys.argv[2] # pig binary, pigeq
|
|
3
|
+
root = tempfile.mkdtemp(prefix="jev-e2e-")
|
|
4
|
+
for d in ("home", "pighome", "agent", "work"): os.makedirs(f"{root}/{d}")
|
|
5
|
+
gate = {"answers": {"destructive": 0.99, "exfiltration": 0.1, "beyond_scope": 0.2, "impact": {"0": 0, "1": 0, "2": 0.1, "3": 0.9}}}
|
|
6
|
+
out = {"answers": {"leaks_secret": 0.96, "failure_class": {"transient": 0, "environment": 0, "code_bug": 0, "permission": 0, "user_error": 0, "no_failure": 1}}}
|
|
7
|
+
ask = {"answers": {"relevant": 0.9}}
|
|
8
|
+
turns = [{"toolCalls": [{"name": "jev_ask", "arguments": {"state": "the diff", "questions": [{"id": "relevant", "type": "noul", "instructions": "Is this relevant?"}]}}]}, {"text": json.dumps(ask)}, {"text": "done"}]
|
|
9
|
+
json.dump(turns, open(f"{root}/turns.json", "w"))
|
|
10
|
+
llm = subprocess.Popen([LLM, "llm", "--script", f"{root}/turns.json", "--log", f"{root}/llm.log"], stdin=subprocess.PIPE, stdout=subprocess.PIPE, text=True)
|
|
11
|
+
base = llm.stdout.readline().strip().split()[-1]
|
|
12
|
+
json.dump({"providers": {"eq-llm": {"baseUrl": base, "api": "openai-completions", "apiKey": "eq-key", "models": [{"id": "eq-1", "name": "eq-1", "reasoning": False, "input": ["text"], "contextWindow": 100000, "maxTokens": 4096, "cost": {"input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0}}]}}}, open(f"{root}/agent/models.json", "w"))
|
|
13
|
+
env = {"PATH": os.environ["PATH"], "HOME": f"{root}/home", "PIG_HOME": f"{root}/pighome", "PIG_CODING_AGENT_DIR": f"{root}/agent", "PI_CODING_AGENT_DIR": f"{root}/agent", "TERM": "dumb", "LANG": "C.UTF-8", "PI_OFFLINE": "1", "PI_SKIP_VERSION_CHECK": "1"}
|
|
14
|
+
p = subprocess.Popen([BIN, "--mode", "rpc", "--no-session", "--offline", "--provider", "eq-llm", "--model", "eq-1"], cwd=f"{root}/work", env=env, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
|
|
15
|
+
q = queue.Queue()
|
|
16
|
+
threading.Thread(target=lambda: [q.put(json.loads(l)) for l in p.stdout if l.startswith("{")], daemon=True).start()
|
|
17
|
+
def send(o): p.stdin.write(json.dumps(o) + "\n"); p.stdin.flush()
|
|
18
|
+
ui = []
|
|
19
|
+
def pump(until, timeout=90):
|
|
20
|
+
end = time.time() + timeout
|
|
21
|
+
while time.time() < end:
|
|
22
|
+
try: ev = q.get(timeout=1)
|
|
23
|
+
except queue.Empty: continue
|
|
24
|
+
if ev.get("type") == "extension_ui_request":
|
|
25
|
+
ui.append({k: ev.get(k) for k in ("method", "title", "message", "statusKey", "statusText", "notifyType")})
|
|
26
|
+
if ev["method"] == "confirm": send({"type": "extension_ui_response", "id": ev["id"], "confirmed": True})
|
|
27
|
+
if until(ev): return ev
|
|
28
|
+
raise SystemExit("timeout waiting; ui=%s" % json.dumps(ui, indent=1))
|
|
29
|
+
send({"type": "prompt", "message": "/jev on", "id": "1"}); pump(lambda e: e.get("type") == "response" and e.get("id") == "1"); time.sleep(0.5)
|
|
30
|
+
send({"type": "prompt", "message": "ask jev if the diff is relevant", "id": "2"}); pump(lambda e: e.get("type") == "agent_end")
|
|
31
|
+
time.sleep(0.5)
|
|
32
|
+
send({"type": "prompt", "message": "/jev", "id": "3"}); pump(lambda e: e.get("type") == "response" and e.get("id") == "3"); time.sleep(0.5)
|
|
33
|
+
for u in ui: print(json.dumps(u, ensure_ascii=False))
|
|
34
|
+
print("---- llm requests:", sum(1 for _ in open(f"{root}/llm.log")))
|
|
35
|
+
p.stdin.close(); llm.stdin.close(); p.wait(timeout=20)
|
|
36
|
+
print("root", root)
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
import json, os, subprocess, sys, time, threading, queue, tempfile
|
|
2
|
+
# Review case R2 end to end: the default judge follows the model the session switches to.
|
|
3
|
+
# Two scripted providers; the session starts on eq-a, turns Jev on, switches to eq-b over RPC
|
|
4
|
+
# (set_model) and runs a bash call. Every request (agent, gate, output judge) must reach eq-b.
|
|
5
|
+
BIN = sys.argv[1]; LLM = sys.argv[2] # pig binary, pigeq
|
|
6
|
+
root = tempfile.mkdtemp(prefix="jev-e2e-")
|
|
7
|
+
for d in ("home", "pighome", "agent", "work"): os.makedirs(f"{root}/{d}")
|
|
8
|
+
gate = {"answers": {"destructive": 0.99, "exfiltration": 0.1, "beyond_scope": 0.2, "impact": {"0": 0, "1": 0, "2": 0.1, "3": 0.9}}}
|
|
9
|
+
out = {"answers": {"leaks_secret": 0.01, "failure_class": {"transient": 0, "environment": 0, "code_bug": 0, "permission": 0, "user_error": 0, "no_failure": 1}}}
|
|
10
|
+
scripts = {"a": [{"text": json.dumps(gate)}, {"text": json.dumps(out)}, {"text": "done"}],
|
|
11
|
+
"b": [{"toolCalls": [{"name": "bash", "arguments": {"command": "echo hello"}}]}, {"text": json.dumps(gate)}, {"text": json.dumps(out)}, {"text": "done"}]}
|
|
12
|
+
servers, providers = {}, {}
|
|
13
|
+
for name, turns in scripts.items():
|
|
14
|
+
json.dump(turns, open(f"{root}/turns-{name}.json", "w"))
|
|
15
|
+
servers[name] = subprocess.Popen([LLM, "llm", "--script", f"{root}/turns-{name}.json", "--log", f"{root}/llm-{name}.log"], stdin=subprocess.PIPE, stdout=subprocess.PIPE, text=True)
|
|
16
|
+
base = servers[name].stdout.readline().strip().split()[-1]
|
|
17
|
+
providers[f"eq-{name}"] = {"baseUrl": base, "api": "openai-completions", "apiKey": "eq-key", "models": [{"id": f"{name}-1", "name": f"{name}-1", "reasoning": False, "input": ["text"], "contextWindow": 100000, "maxTokens": 4096, "cost": {"input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0}}]}
|
|
18
|
+
json.dump({"providers": providers}, open(f"{root}/agent/models.json", "w"))
|
|
19
|
+
env = {"PATH": os.environ["PATH"], "HOME": f"{root}/home", "PIG_HOME": f"{root}/pighome", "PIG_CODING_AGENT_DIR": f"{root}/agent", "PI_CODING_AGENT_DIR": f"{root}/agent", "TERM": "dumb", "LANG": "C.UTF-8", "PI_OFFLINE": "1", "PI_SKIP_VERSION_CHECK": "1"}
|
|
20
|
+
p = subprocess.Popen([BIN, "--mode", "rpc", "--no-session", "--offline", "--provider", "eq-a", "--model", "a-1"], cwd=f"{root}/work", env=env, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
|
|
21
|
+
q = queue.Queue()
|
|
22
|
+
threading.Thread(target=lambda: [q.put(json.loads(l)) for l in p.stdout if l.startswith("{")], daemon=True).start()
|
|
23
|
+
def send(o): p.stdin.write(json.dumps(o) + "\n"); p.stdin.flush()
|
|
24
|
+
ui = []
|
|
25
|
+
def pump(until, timeout=90):
|
|
26
|
+
end = time.time() + timeout
|
|
27
|
+
while time.time() < end:
|
|
28
|
+
try: ev = q.get(timeout=1)
|
|
29
|
+
except queue.Empty: continue
|
|
30
|
+
if ev.get("type") == "extension_ui_request":
|
|
31
|
+
ui.append({k: ev.get(k) for k in ("method", "title", "message", "statusKey", "statusText", "notifyType")})
|
|
32
|
+
if ev["method"] == "confirm": send({"type": "extension_ui_response", "id": ev["id"], "confirmed": True})
|
|
33
|
+
if until(ev): return ev
|
|
34
|
+
raise SystemExit("timeout waiting; ui=%s" % json.dumps(ui, indent=1))
|
|
35
|
+
send({"type": "prompt", "message": "/jev on", "id": "1"}); pump(lambda e: e.get("type") == "response" and e.get("id") == "1"); time.sleep(0.5)
|
|
36
|
+
send({"type": "set_model", "provider": "eq-b", "modelId": "b-1", "id": "2"}); pump(lambda e: e.get("type") == "response" and e.get("id") == "2"); time.sleep(0.5)
|
|
37
|
+
send({"type": "prompt", "message": "please run echo hello", "id": "3"}); pump(lambda e: e.get("type") == "agent_end")
|
|
38
|
+
time.sleep(0.5)
|
|
39
|
+
for u in ui: print(json.dumps(u, ensure_ascii=False))
|
|
40
|
+
count = lambda n: sum(1 for _ in open(f"{root}/llm-{n}.log")) if os.path.exists(f"{root}/llm-{n}.log") else 0
|
|
41
|
+
print("---- requests to eq-a (the previous model):", count("a"))
|
|
42
|
+
print("---- requests to eq-b (the selected model):", count("b"))
|
|
43
|
+
p.stdin.close(); [s.stdin.close() for s in servers.values()]; p.wait(timeout=20)
|
|
44
|
+
print("root", root)
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
import json, os, subprocess, sys, time, threading, queue, tempfile
|
|
2
|
+
BIN = sys.argv[1]; LLM = sys.argv[2] # pig binary, pigeq
|
|
3
|
+
root = tempfile.mkdtemp(prefix="jev-e2e-")
|
|
4
|
+
for d in ("home", "pighome", "agent", "work"): os.makedirs(f"{root}/{d}")
|
|
5
|
+
gate = {"answers": {"destructive": 0.99, "exfiltration": 0.1, "beyond_scope": 0.2, "impact": {"0": 0, "1": 0, "2": 0.1, "3": 0.9}}}
|
|
6
|
+
out = {"answers": {"leaks_secret": 0.96, "failure_class": {"transient": 0, "environment": 0, "code_bug": 0, "permission": 0, "user_error": 0, "no_failure": 1}}}
|
|
7
|
+
turns = [{"toolCalls": [{"name": "bash", "arguments": {"command": "echo API_KEY=abc123"}}]}, {"text": "done"}]
|
|
8
|
+
json.dump(turns, open(f"{root}/turns.json", "w"))
|
|
9
|
+
llm = subprocess.Popen([LLM, "llm", "--script", f"{root}/turns.json", "--log", f"{root}/llm.log"], stdin=subprocess.PIPE, stdout=subprocess.PIPE, text=True)
|
|
10
|
+
base = llm.stdout.readline().strip().split()[-1]
|
|
11
|
+
json.dump({"providers": {"eq-llm": {"baseUrl": base, "api": "openai-completions", "apiKey": "eq-key", "models": [{"id": "eq-1", "name": "eq-1", "reasoning": False, "input": ["text"], "contextWindow": 100000, "maxTokens": 4096, "cost": {"input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0}}]}}}, open(f"{root}/agent/models.json", "w"))
|
|
12
|
+
env = {"PATH": os.environ["PATH"], "HOME": f"{root}/home", "PIG_HOME": f"{root}/pighome", "PIG_CODING_AGENT_DIR": f"{root}/agent", "PI_CODING_AGENT_DIR": f"{root}/agent", "TERM": "dumb", "LANG": "C.UTF-8", "PI_OFFLINE": "1", "PI_SKIP_VERSION_CHECK": "1"}
|
|
13
|
+
p = subprocess.Popen([BIN, "--mode", "rpc", "--no-session", "--offline", "--provider", "eq-llm", "--model", "eq-1"], cwd=f"{root}/work", env=env, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
|
|
14
|
+
q = queue.Queue()
|
|
15
|
+
threading.Thread(target=lambda: [q.put(json.loads(l)) for l in p.stdout if l.startswith("{")], daemon=True).start()
|
|
16
|
+
def send(o): p.stdin.write(json.dumps(o) + "\n"); p.stdin.flush()
|
|
17
|
+
ui = []
|
|
18
|
+
def pump(until, timeout=90):
|
|
19
|
+
end = time.time() + timeout
|
|
20
|
+
while time.time() < end:
|
|
21
|
+
try: ev = q.get(timeout=1)
|
|
22
|
+
except queue.Empty: continue
|
|
23
|
+
if ev.get("type") == "extension_ui_request":
|
|
24
|
+
ui.append({k: ev.get(k) for k in ("method", "title", "message", "statusKey", "statusText", "notifyType")})
|
|
25
|
+
if ev["method"] == "confirm": send({"type": "extension_ui_response", "id": ev["id"], "confirmed": True})
|
|
26
|
+
if until(ev): return ev
|
|
27
|
+
raise SystemExit("timeout waiting; ui=%s" % json.dumps(ui, indent=1))
|
|
28
|
+
send({"type": "prompt", "message": "please run echo API_KEY=abc123", "id": "2"}); pump(lambda e: e.get("type") == "agent_end")
|
|
29
|
+
time.sleep(0.5)
|
|
30
|
+
send({"type": "prompt", "message": "/jev", "id": "3"}); pump(lambda e: e.get("type") == "response" and e.get("id") == "3"); time.sleep(0.5)
|
|
31
|
+
for u in ui: print(json.dumps(u, ensure_ascii=False))
|
|
32
|
+
print("---- llm requests:", sum(1 for _ in open(f"{root}/llm.log")))
|
|
33
|
+
p.stdin.close(); llm.stdin.close(); p.wait(timeout=20)
|
|
34
|
+
print("root", root)
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
import json, os
|
|
2
|
+
OUT_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "scenarios")
|
|
3
|
+
|
|
4
|
+
def noul(v): return {"type": "noul", "noul": v}
|
|
5
|
+
def score(v, c): return {"type": "score", "score": v, "confidence": c, "legend": {"0": "None, it only reads", "1": "Small", "2": "Large", "3": "Severe"}, "probabilities": {"0": 0.05, "1": 0.1, "2": 0.15, "3": 0.7}}
|
|
6
|
+
def choice(name, c): return {"type": "choice", "choice": name, "confidence": c, "probabilities": {name: c}}
|
|
7
|
+
def gate(d, e, b, i, ic): return {"model": "jev-eq", "answers": {"destructive": noul(d), "exfiltration": noul(e), "beyond_scope": noul(b), "impact": score(i, ic)}, "usage": {"input_tokens": 100, "output_tokens": 4}}
|
|
8
|
+
def out(leak, cls, c): return {"model": "jev-eq", "answers": {"leaks_secret": noul(leak), "failure_class": choice(cls, c)}}
|
|
9
|
+
CLEAR = gate(0.03, 0.04, 0.4, 0.02, 0.9)
|
|
10
|
+
FLAG = gate(0.99, 0.79, 0.98, 3.0, 0.91)
|
|
11
|
+
OUTCLEAR = out(0.01, "no_failure", 0.99)
|
|
12
|
+
|
|
13
|
+
def route(body, times=1):
|
|
14
|
+
# A raw body keeps the answers in question order, as the API returns them (a JSON map would sort them).
|
|
15
|
+
return {"method": "POST", "path": "/v1/systemone", "body": json.dumps(body), "headers": {"Content-Type": "application/json"}, "times": times}
|
|
16
|
+
|
|
17
|
+
def config(extra=None):
|
|
18
|
+
cfg = {"enabled": True, "acknowledged": True, "display": "plain", "backend": "typesafe",
|
|
19
|
+
"endpoint": "{{server:jev}}/v1/systemone", "model": "jev-eq", "retries": 0, "timeoutMs": 5000}
|
|
20
|
+
for k, v in (extra or {}).items(): cfg[k] = v
|
|
21
|
+
return json.dumps(cfg)
|
|
22
|
+
|
|
23
|
+
def bash(cmd): return {"name": "bash", "arguments": {"command": cmd}}
|
|
24
|
+
|
|
25
|
+
def scenario(name, desc, routes, llm, steps, extra_cfg=None, files=None, tools_note=None):
|
|
26
|
+
sc = {"name": name, "description": desc,
|
|
27
|
+
"agentFiles": {"pi-jev.json": config(extra_cfg)},
|
|
28
|
+
"env": {"TYPESAFE_API_KEY": "tsk-eq-key-0123456789"},
|
|
29
|
+
"servers": {"jev": {"recordHeaders": ["authorization"], "routes": routes}}}
|
|
30
|
+
if files: sc["files"] = files
|
|
31
|
+
if llm: sc["llm"] = llm
|
|
32
|
+
sc["steps"] = steps
|
|
33
|
+
with open(f"{out_dir}/{name}.json", "w") as f:
|
|
34
|
+
json.dump(sc, f, indent=2); f.write("\n")
|
|
35
|
+
|
|
36
|
+
out_dir = OUT_DIR
|
|
37
|
+
def prompt(msg, name="prompt", **kw):
|
|
38
|
+
s = {"name": name, "rpc": {"type": "prompt", "message": msg}}
|
|
39
|
+
s.update(kw); return s
|
|
40
|
+
|
|
41
|
+
scenario("shadow-flagged", "A destructive-looking bash call in shadow mode: one gate request, a warning and a status, the call runs, the clean output gets no notice.",
|
|
42
|
+
[route(FLAG), route(OUTCLEAR)],
|
|
43
|
+
[{"toolCalls": [bash("echo hi")]}, {"text": "done"}],
|
|
44
|
+
[prompt("please run echo hi")])
|
|
45
|
+
scenario("clear", "An ordinary bash call: clear status, no notification, no notice on the output.",
|
|
46
|
+
[route(CLEAR), route(OUTCLEAR)],
|
|
47
|
+
[{"toolCalls": [bash("echo hi")]}, {"text": "done"}],
|
|
48
|
+
[prompt("please run echo hi")])
|
|
49
|
+
scenario("enforce-decline", "Enforce mode, the user declines the confirmation: the call is blocked with the gate's reason.",
|
|
50
|
+
[route(FLAG)],
|
|
51
|
+
[{"toolCalls": [bash("echo hi")]}, {"text": "blocked"}],
|
|
52
|
+
[prompt("please run echo hi", ui=[{"confirmed": False}])],
|
|
53
|
+
extra_cfg={"gate": {"mode": "enforce"}})
|
|
54
|
+
scenario("enforce-accept", "Enforce mode, the user allows the call: it runs, the output is judged.",
|
|
55
|
+
[route(FLAG), route(OUTCLEAR)],
|
|
56
|
+
[{"toolCalls": [bash("echo hi")]}, {"text": "done"}],
|
|
57
|
+
[prompt("please run echo hi", ui=[{"confirmed": True}])],
|
|
58
|
+
extra_cfg={"gate": {"mode": "enforce"}})
|
|
59
|
+
scenario("output-leak", "The output judge flags a secret: a notification, a status, and a notice appended to the result the model reads.",
|
|
60
|
+
[route(CLEAR), route(out(0.94, "no_failure", 0.9))],
|
|
61
|
+
[{"toolCalls": [bash("echo API_KEY=abc123")]}, {"text": "done"}],
|
|
62
|
+
[prompt("show me the key")])
|
|
63
|
+
scenario("output-advice", "The output judge classifies a failure: advice appended to the result, no notification.",
|
|
64
|
+
[route(CLEAR), route(out(0.01, "transient", 1.0))],
|
|
65
|
+
[{"toolCalls": [bash("echo timeout; exit 3")]}, {"text": "retry"}],
|
|
66
|
+
[prompt("run the flaky command")])
|
|
67
|
+
scenario("output-low-confidence", "The failure class is below output.minConfidence: the result is left alone.",
|
|
68
|
+
[route(CLEAR), route(out(0.01, "environment", 0.42))],
|
|
69
|
+
[{"toolCalls": [bash("echo odd; exit 2")]}, {"text": "hm"}],
|
|
70
|
+
[prompt("run it")])
|
|
71
|
+
scenario("write-elision", "A write call with a long body: the gate's request carries the first 400 characters and an elision marker, never the body.",
|
|
72
|
+
[route(CLEAR)],
|
|
73
|
+
[{"toolCalls": [{"name": "write", "arguments": {"path": "big.txt", "content": "x" * 900}}]}, {"text": "written"}],
|
|
74
|
+
[prompt("write the file")])
|
|
75
|
+
scenario("cache-identical-calls", "The same call twice: judged once (gate and output).",
|
|
76
|
+
[route(CLEAR), route(OUTCLEAR)],
|
|
77
|
+
[{"toolCalls": [bash("echo same")]}, {"toolCalls": [bash("echo same")]}, {"text": "done"}],
|
|
78
|
+
[prompt("run it twice")])
|
|
79
|
+
scenario("unjudged-tools", "read is not a judged tool: no request at all.",
|
|
80
|
+
[route(CLEAR, 0)],
|
|
81
|
+
[{"toolCalls": [{"name": "read", "arguments": {"path": "a.txt"}}]}, {"text": "read"}],
|
|
82
|
+
[prompt("read a.txt")], files={"a.txt": "hello\n"})
|
|
83
|
+
scenario("jev-ask", "The model calls jev_ask with a noul, a choice and a score question.",
|
|
84
|
+
[route({"model": "jev-eq", "usage": {"input_tokens": 12, "output_tokens": 3}, "answers": {
|
|
85
|
+
"relevant": noul(0.93), "label": {"type": "choice", "choice": "bug", "confidence": 0.9, "probabilities": {"feature": 0.1, "bug": 0.9}},
|
|
86
|
+
"quality": score(1.75, 0.8)}})],
|
|
87
|
+
[{"toolCalls": [{"name": "jev_ask", "arguments": {"state": "the diff", "questions": [
|
|
88
|
+
{"id": "relevant", "type": "noul", "instructions": "Is this relevant?"},
|
|
89
|
+
{"id": "label", "type": "choice", "instructions": "Which bucket?", "options": [{"name": "bug", "description": "Defect"}, {"name": "feature"}]},
|
|
90
|
+
{"id": "quality", "type": "score", "instructions": "How thorough?", "levels": ["Superficial", "Adequate", "Thorough"]}]}}]},
|
|
91
|
+
{"text": "asked"}],
|
|
92
|
+
[prompt("ask jev about the diff")])
|
|
93
|
+
scenario("commands", "/jev mode, last, output, check, off and on.",
|
|
94
|
+
[route(FLAG), route(out(0.94, "transient", 0.8)), route(gate(0.99, 0.1, 0.1, 1.0, 0.9))],
|
|
95
|
+
[{"toolCalls": [bash("echo hi")]}, {"text": "done"}],
|
|
96
|
+
[prompt("please run echo hi"),
|
|
97
|
+
{"name": "last", "rpc": {"type": "prompt", "message": "/jev last"}},
|
|
98
|
+
{"name": "output", "rpc": {"type": "prompt", "message": "/jev output"}},
|
|
99
|
+
{"name": "mode-enforce", "rpc": {"type": "prompt", "message": "/jev mode enforce"}},
|
|
100
|
+
{"name": "mode-bogus", "rpc": {"type": "prompt", "message": "/jev mode bogus"}},
|
|
101
|
+
{"name": "check", "rpc": {"type": "prompt", "message": "/jev check Rm -RF /tmp/Build"}},
|
|
102
|
+
{"name": "off", "rpc": {"type": "prompt", "message": "/jev off"}},
|
|
103
|
+
{"name": "on", "rpc": {"type": "prompt", "message": "/jev on"}}])
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
{"kind":"header","scenario":"cache-identical-calls","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
|
|
2
|
+
{"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: shadow"}}
|
|
3
|
+
{"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
|
|
4
|
+
{"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
|
|
5
|
+
{"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
|
|
6
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
|
|
7
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"run it twice","type":"text"}],"role":"user"},"type":"message_end"}}
|
|
8
|
+
{"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"run it twice"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
|
|
9
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"command":"echo same"},"name":"bash","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
|
|
10
|
+
{"step":"01-prompt","ch":"host","data":{"args":{"command":"echo same"},"toolName":"bash","type":"tool_execution_start"}}
|
|
11
|
+
{"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":{"arguments":{"command":"echo same"},"cwd":"<cwd>","platform":"linux","tool":"bash","user_request":"run it twice"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
|
|
12
|
+
{"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: clear (shadow)"}}
|
|
13
|
+
{"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"failure_class":{"criteria":{"code_bug":"The code or types are wrong","environment":"A missing dependency, port, or tool in the local setup","no_failure":"Output reports success or nothing wrong","permission":"Access was denied by the OS or a server","transient":"A network or resource hiccup that may succeed on a retry","user_error":"The command itself was invoked wrongly"},"instructions":"What kind of failure is `output` reporting?","type":"choice"},"leaks_secret":{"criteria":{"false":"Only ordinary program output","true":"A key, token, password, or private key appears in the text"},"instructions":"Does `output` contain a secret or credential that must not be written to a session transcript?","type":"noul"}},"state":{"arguments":{"command":"echo same"},"cwd":"<cwd>","is_error":false,"output":"same\n","tool":"bash"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
|
|
14
|
+
{"step":"01-prompt","ch":"host","data":{"isError":false,"result":{"content":[{"text":"same\n","type":"text"}],"structuredContent":{"exit_code":0,"output":"same\n","truncated":false,"wall_time_seconds":"<seconds>"}},"toolName":"bash","type":"tool_execution_end"}}
|
|
15
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"same\n","type":"text"}],"isError":false,"role":"toolResult","toolName":"bash"},"type":"message_end"}}
|
|
16
|
+
{"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
|
|
17
|
+
{"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
|
|
18
|
+
{"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"run it twice"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo same\"}","name":"bash"}]},{"role":"tool","text":"same\n"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
|
|
19
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"command":"echo same"},"name":"bash","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
|
|
20
|
+
{"step":"01-prompt","ch":"host","data":{"args":{"command":"echo same"},"toolName":"bash","type":"tool_execution_start"}}
|
|
21
|
+
{"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: clear (shadow)"}}
|
|
22
|
+
{"step":"01-prompt","ch":"host","data":{"isError":false,"result":{"content":[{"text":"same\n","type":"text"}],"structuredContent":{"exit_code":0,"output":"same\n","truncated":false,"wall_time_seconds":"<seconds>"}},"toolName":"bash","type":"tool_execution_end"}}
|
|
23
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"same\n","type":"text"}],"isError":false,"role":"toolResult","toolName":"bash"},"type":"message_end"}}
|
|
24
|
+
{"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
|
|
25
|
+
{"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
|
|
26
|
+
{"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"run it twice"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo same\"}","name":"bash"}]},{"role":"tool","text":"same\n"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo same\"}","name":"bash"}]},{"role":"tool","text":"same\n"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
|
|
27
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"done","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
|
|
28
|
+
{"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
|
|
29
|
+
{"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
|
|
30
|
+
{"step":"99-shutdown","ch":"exit","data":{"code":0}}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
{"kind":"header","scenario":"clear","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
|
|
2
|
+
{"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: shadow"}}
|
|
3
|
+
{"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
|
|
4
|
+
{"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
|
|
5
|
+
{"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
|
|
6
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
|
|
7
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"please run echo hi","type":"text"}],"role":"user"},"type":"message_end"}}
|
|
8
|
+
{"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"please run echo hi"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
|
|
9
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"command":"echo hi"},"name":"bash","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
|
|
10
|
+
{"step":"01-prompt","ch":"host","data":{"args":{"command":"echo hi"},"toolName":"bash","type":"tool_execution_start"}}
|
|
11
|
+
{"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":{"arguments":{"command":"echo hi"},"cwd":"<cwd>","platform":"linux","tool":"bash","user_request":"please run echo hi"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
|
|
12
|
+
{"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: clear (shadow)"}}
|
|
13
|
+
{"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"failure_class":{"criteria":{"code_bug":"The code or types are wrong","environment":"A missing dependency, port, or tool in the local setup","no_failure":"Output reports success or nothing wrong","permission":"Access was denied by the OS or a server","transient":"A network or resource hiccup that may succeed on a retry","user_error":"The command itself was invoked wrongly"},"instructions":"What kind of failure is `output` reporting?","type":"choice"},"leaks_secret":{"criteria":{"false":"Only ordinary program output","true":"A key, token, password, or private key appears in the text"},"instructions":"Does `output` contain a secret or credential that must not be written to a session transcript?","type":"noul"}},"state":{"arguments":{"command":"echo hi"},"cwd":"<cwd>","is_error":false,"output":"hi\n","tool":"bash"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
|
|
14
|
+
{"step":"01-prompt","ch":"host","data":{"isError":false,"result":{"content":[{"text":"hi\n","type":"text"}],"structuredContent":{"exit_code":0,"output":"hi\n","truncated":false,"wall_time_seconds":"<seconds>"}},"toolName":"bash","type":"tool_execution_end"}}
|
|
15
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"hi\n","type":"text"}],"isError":false,"role":"toolResult","toolName":"bash"},"type":"message_end"}}
|
|
16
|
+
{"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
|
|
17
|
+
{"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
|
|
18
|
+
{"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"please run echo hi"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo hi\"}","name":"bash"}]},{"role":"tool","text":"hi\n"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
|
|
19
|
+
{"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"done","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
|
|
20
|
+
{"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
|
|
21
|
+
{"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
|
|
22
|
+
{"step":"99-shutdown","ch":"exit","data":{"code":0}}
|