@pi-in-go/pigpen-jev 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/CREDITS.md +22 -0
  2. package/LICENSE +22 -0
  3. package/README.md +237 -0
  4. package/extensions/jev/ask.go +166 -0
  5. package/extensions/jev/ask_test.go +218 -0
  6. package/extensions/jev/backend.go +128 -0
  7. package/extensions/jev/bench_test.go +64 -0
  8. package/extensions/jev/boundaries_test.go +159 -0
  9. package/extensions/jev/command.go +224 -0
  10. package/extensions/jev/commands_test.go +214 -0
  11. package/extensions/jev/config.go +450 -0
  12. package/extensions/jev/errors_test.go +191 -0
  13. package/extensions/jev/extension.go +391 -0
  14. package/extensions/jev/fakehost_test.go +548 -0
  15. package/extensions/jev/gate.go +125 -0
  16. package/extensions/jev/gate_test.go +610 -0
  17. package/extensions/jev/gatekey_test.go +24 -0
  18. package/extensions/jev/go.mod +9 -0
  19. package/extensions/jev/go.sum +2 -0
  20. package/extensions/jev/go.work +10 -0
  21. package/extensions/jev/helpers_test.go +404 -0
  22. package/extensions/jev/memo.go +88 -0
  23. package/extensions/jev/output.go +89 -0
  24. package/extensions/jev/output_test.go +187 -0
  25. package/extensions/jev/ownmodel_test.go +118 -0
  26. package/extensions/jev/render.go +136 -0
  27. package/extensions/jev/review_test.go +310 -0
  28. package/extensions/jev/source_test.go +57 -0
  29. package/extensions/jev/text.go +174 -0
  30. package/extensions/jev/trust_test.go +335 -0
  31. package/extensions/jev/types.go +227 -0
  32. package/libs/typesafe/CONTRACT.md +125 -0
  33. package/libs/typesafe/CREDITS.md +37 -0
  34. package/libs/typesafe/LICENSE +23 -0
  35. package/libs/typesafe/README.md +19 -0
  36. package/libs/typesafe/go.mod +3 -0
  37. package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
  38. package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
  39. package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
  40. package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
  41. package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
  42. package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
  43. package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
  44. package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
  45. package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
  46. package/libs/typesafe/libraries/ownmodel/run.go +288 -0
  47. package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
  48. package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
  49. package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
  50. package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
  51. package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
  52. package/libs/typesafe/libraries/typesafe/answers.go +268 -0
  53. package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
  54. package/libs/typesafe/libraries/typesafe/batch.go +80 -0
  55. package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
  56. package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
  57. package/libs/typesafe/libraries/typesafe/client.go +561 -0
  58. package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
  59. package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
  60. package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
  61. package/libs/typesafe/libraries/typesafe/doc.go +27 -0
  62. package/libs/typesafe/libraries/typesafe/entry.go +142 -0
  63. package/libs/typesafe/libraries/typesafe/env.go +11 -0
  64. package/libs/typesafe/libraries/typesafe/errors.go +310 -0
  65. package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
  66. package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
  67. package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
  68. package/libs/typesafe/libraries/typesafe/logging.go +160 -0
  69. package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
  70. package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
  71. package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
  72. package/libs/typesafe/libraries/typesafe/questions.go +490 -0
  73. package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
  74. package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
  75. package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
  76. package/libs/typesafe/libraries/typesafe/retry.go +350 -0
  77. package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
  78. package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
  79. package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
  80. package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
  81. package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
  82. package/libs/typesafe/libraries/typesafe/version.go +10 -0
  83. package/libs/typesafe/package.json +37 -0
  84. package/libs/typesafe/provenance.json +49 -0
  85. package/package.json +42 -0
  86. package/port/PORT.md +107 -0
  87. package/port/e2e/gate-and-output.py +35 -0
  88. package/port/e2e/jev-ask.py +36 -0
  89. package/port/e2e/model-switch.py +44 -0
  90. package/port/e2e/off-by-default.py +34 -0
  91. package/port/gen-scenarios.py +103 -0
  92. package/port/golden/cache-identical-calls.jsonl +30 -0
  93. package/port/golden/clear.jsonl +22 -0
  94. package/port/golden/commands.jsonl +43 -0
  95. package/port/golden/enforce-accept.jsonl +23 -0
  96. package/port/golden/enforce-decline.jsonl +22 -0
  97. package/port/golden/jev-ask.jsonl +20 -0
  98. package/port/golden/output-advice.jsonl +23 -0
  99. package/port/golden/output-leak.jsonl +24 -0
  100. package/port/golden/output-low-confidence.jsonl +22 -0
  101. package/port/golden/shadow-flagged.jsonl +23 -0
  102. package/port/golden/unjudged-tools.jsonl +19 -0
  103. package/port/golden/write-elision.jsonl +21 -0
  104. package/port/mutate-unit.py +63 -0
  105. package/port/mutations.json +578 -0
  106. package/port/oracle/LICENSE +21 -0
  107. package/port/oracle/README.md +181 -0
  108. package/port/oracle/SHA256SUMS +8 -0
  109. package/port/oracle/package.json +43 -0
  110. package/port/oracle/src/client.ts +409 -0
  111. package/port/oracle/src/config.ts +363 -0
  112. package/port/oracle/src/gate.ts +229 -0
  113. package/port/oracle/src/index.ts +649 -0
  114. package/port/oracle/src/output.ts +163 -0
  115. package/port/red-run.log +309 -0
  116. package/port/scenarios/cache-identical-calls.json +71 -0
  117. package/port/scenarios/clear.json +61 -0
  118. package/port/scenarios/commands.json +119 -0
  119. package/port/scenarios/enforce-accept.json +66 -0
  120. package/port/scenarios/enforce-decline.json +57 -0
  121. package/port/scenarios/jev-ask.json +83 -0
  122. package/port/scenarios/output-advice.json +61 -0
  123. package/port/scenarios/output-leak.json +61 -0
  124. package/port/scenarios/output-low-confidence.json +61 -0
  125. package/port/scenarios/shadow-flagged.json +61 -0
  126. package/port/scenarios/unjudged-tools.json +55 -0
  127. package/port/scenarios/write-elision.json +53 -0
  128. package/provenance.json +18 -0
@@ -0,0 +1,43 @@
1
+ {"kind":"header","scenario":"commands","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
2
+ {"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: shadow"}}
3
+ {"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
4
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
5
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
6
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
7
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"please run echo hi","type":"text"}],"role":"user"},"type":"message_end"}}
8
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"please run echo hi"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
9
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"command":"echo hi"},"name":"bash","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
10
+ {"step":"01-prompt","ch":"host","data":{"args":{"command":"echo hi"},"toolName":"bash","type":"tool_execution_start"}}
11
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":{"arguments":{"command":"echo hi"},"cwd":"<cwd>","platform":"linux","tool":"bash","user_request":"please run echo hi"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
12
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91"}}
13
+ {"step":"01-prompt","ch":"ui","data":{"message":"jev shadow: bash - destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91","method":"notify","notifyType":"warning"}}
14
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"failure_class":{"criteria":{"code_bug":"The code or types are wrong","environment":"A missing dependency, port, or tool in the local setup","no_failure":"Output reports success or nothing wrong","permission":"Access was denied by the OS or a server","transient":"A network or resource hiccup that may succeed on a retry","user_error":"The command itself was invoked wrongly"},"instructions":"What kind of failure is `output` reporting?","type":"choice"},"leaks_secret":{"criteria":{"false":"Only ordinary program output","true":"A key, token, password, or private key appears in the text"},"instructions":"Does `output` contain a secret or credential that must not be written to a session transcript?","type":"noul"}},"state":{"arguments":{"command":"echo hi"},"cwd":"<cwd>","is_error":false,"output":"hi\n","tool":"bash"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
15
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: leak (bash)"}}
16
+ {"step":"01-prompt","ch":"ui","data":{"message":"jev: bash output may carry a secret (0.94)","method":"notify","notifyType":"warning"}}
17
+ {"step":"01-prompt","ch":"host","data":{"isError":false,"result":{"content":[{"text":"hi\n","type":"text"},{"text":"[pi-jev] Jev flagged this output as containing a secret (0.94). Do not repeat the value in a reply, a file, or a command; refer to it by name instead.","type":"text"}]},"toolName":"bash","type":"tool_execution_end"}}
18
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"hi\n","type":"text"},{"text":"[pi-jev] Jev flagged this output as containing a secret (0.94). Do not repeat the value in a reply, a file, or a command; refer to it by name instead.","type":"text"}],"isError":false,"role":"toolResult","toolName":"bash"},"type":"message_end"}}
19
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
20
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
21
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"please run echo hi"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo hi\"}","name":"bash"}]},{"role":"tool","text":"hi\n\n[pi-jev] Jev flagged this output as containing a secret (0.94). Do not repeat the value in a reply, a file, or a command; refer to it by name instead."}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
22
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"done","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
23
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
24
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
25
+ {"step":"02-last","ch":"ui","data":{"message":"pi-jev: bash - destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91 | destructive=yes 0.99 exfiltration=yes 0.79 beyond_scope=yes 0.98 impact=3.00/3 (conf 0.91)","method":"notify","notifyType":"info"}}
26
+ {"step":"02-last","ch":"response","data":{"command":"prompt","data":{"disposition":"handled"},"success":true}}
27
+ {"step":"03-output","ch":"ui","data":{"message":"pi-jev output: bash - leak | leak 0.94 | class transient at 0.80","method":"notify","notifyType":"info"}}
28
+ {"step":"03-output","ch":"response","data":{"command":"prompt","data":{"disposition":"handled"},"success":true}}
29
+ {"step":"04-mode-enforce","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: enforce"}}
30
+ {"step":"04-mode-enforce","ch":"ui","data":{"message":"pi-jev: enforce (asks before running flagged calls)","method":"notify","notifyType":"info"}}
31
+ {"step":"04-mode-enforce","ch":"response","data":{"command":"prompt","data":{"disposition":"handled"},"success":true}}
32
+ {"step":"05-mode-bogus","ch":"ui","data":{"message":"pi-jev: mode is enforce (usage: /jev mode shadow|enforce)","method":"notify","notifyType":"warning"}}
33
+ {"step":"05-mode-bogus","ch":"response","data":{"command":"prompt","data":{"disposition":"handled"},"success":true}}
34
+ {"step":"06-check","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":"Rm -RF /tmp/Build"},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
35
+ {"step":"06-check","ch":"ui","data":{"message":"pi-jev check: destructive 0.99 | destructive=yes 0.99 exfiltration=yes 0.10 beyond_scope=yes 0.10 impact=1.00/3 (conf 0.90)","method":"notify","notifyType":"info"}}
36
+ {"step":"06-check","ch":"response","data":{"command":"prompt","data":{"disposition":"handled"},"success":true}}
37
+ {"step":"07-off","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: off"}}
38
+ {"step":"07-off","ch":"ui","data":{"message":"pi-jev: gate off","method":"notify","notifyType":"info"}}
39
+ {"step":"07-off","ch":"response","data":{"command":"prompt","data":{"disposition":"handled"},"success":true}}
40
+ {"step":"08-on","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: enforce"}}
41
+ {"step":"08-on","ch":"ui","data":{"message":"pi-jev: gate on","method":"notify","notifyType":"info"}}
42
+ {"step":"08-on","ch":"response","data":{"command":"prompt","data":{"disposition":"handled"},"success":true}}
43
+ {"step":"99-shutdown","ch":"exit","data":{"code":0}}
@@ -0,0 +1,23 @@
1
+ {"kind":"header","scenario":"enforce-accept","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
2
+ {"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: enforce"}}
3
+ {"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
4
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
5
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
6
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
7
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"please run echo hi","type":"text"}],"role":"user"},"type":"message_end"}}
8
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"please run echo hi"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
9
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"command":"echo hi"},"name":"bash","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
10
+ {"step":"01-prompt","ch":"host","data":{"args":{"command":"echo hi"},"toolName":"bash","type":"tool_execution_start"}}
11
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":{"arguments":{"command":"echo hi"},"cwd":"<cwd>","platform":"linux","tool":"bash","user_request":"please run echo hi"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
12
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91"}}
13
+ {"step":"01-prompt","ch":"ui","data":{"message":"bash\ndestructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91\n\nRun it anyway?","method":"confirm","title":"Jev flagged this tool call"}}
14
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"failure_class":{"criteria":{"code_bug":"The code or types are wrong","environment":"A missing dependency, port, or tool in the local setup","no_failure":"Output reports success or nothing wrong","permission":"Access was denied by the OS or a server","transient":"A network or resource hiccup that may succeed on a retry","user_error":"The command itself was invoked wrongly"},"instructions":"What kind of failure is `output` reporting?","type":"choice"},"leaks_secret":{"criteria":{"false":"Only ordinary program output","true":"A key, token, password, or private key appears in the text"},"instructions":"Does `output` contain a secret or credential that must not be written to a session transcript?","type":"noul"}},"state":{"arguments":{"command":"echo hi"},"cwd":"<cwd>","is_error":false,"output":"hi\n","tool":"bash"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
15
+ {"step":"01-prompt","ch":"host","data":{"isError":false,"result":{"content":[{"text":"hi\n","type":"text"}],"structuredContent":{"exit_code":0,"output":"hi\n","truncated":false,"wall_time_seconds":"<seconds>"}},"toolName":"bash","type":"tool_execution_end"}}
16
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"hi\n","type":"text"}],"isError":false,"role":"toolResult","toolName":"bash"},"type":"message_end"}}
17
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
18
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
19
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"please run echo hi"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo hi\"}","name":"bash"}]},{"role":"tool","text":"hi\n"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
20
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"done","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
21
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
22
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
23
+ {"step":"99-shutdown","ch":"exit","data":{"code":0}}
@@ -0,0 +1,22 @@
1
+ {"kind":"header","scenario":"enforce-decline","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
2
+ {"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: enforce"}}
3
+ {"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
4
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
5
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
6
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
7
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"please run echo hi","type":"text"}],"role":"user"},"type":"message_end"}}
8
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"please run echo hi"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
9
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"command":"echo hi"},"name":"bash","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
10
+ {"step":"01-prompt","ch":"host","data":{"args":{"command":"echo hi"},"toolName":"bash","type":"tool_execution_start"}}
11
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":{"arguments":{"command":"echo hi"},"cwd":"<cwd>","platform":"linux","tool":"bash","user_request":"please run echo hi"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
12
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91"}}
13
+ {"step":"01-prompt","ch":"ui","data":{"message":"bash\ndestructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91\n\nRun it anyway?","method":"confirm","title":"Jev flagged this tool call"}}
14
+ {"step":"01-prompt","ch":"host","data":{"isError":true,"result":{"content":[{"text":"pi-jev: destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91 (declined)","type":"text"}],"details":{}},"toolName":"bash","type":"tool_execution_end"}}
15
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"pi-jev: destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91 (declined)","type":"text"}],"details":{},"isError":true,"role":"toolResult","toolName":"bash"},"type":"message_end"}}
16
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
17
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
18
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"please run echo hi"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo hi\"}","name":"bash"}]},{"role":"tool","text":"pi-jev: destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91 (declined)"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
19
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"blocked","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
20
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
21
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
22
+ {"step":"99-shutdown","ch":"exit","data":{"code":0}}
@@ -0,0 +1,20 @@
1
+ {"kind":"header","scenario":"jev-ask","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
2
+ {"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: shadow"}}
3
+ {"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
4
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
5
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
6
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
7
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"ask jev about the diff","type":"text"}],"role":"user"},"type":"message_end"}}
8
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"ask jev about the diff"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
9
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"questions":[{"id":"relevant","instructions":"Is this relevant?","type":"noul"},{"id":"label","instructions":"Which bucket?","options":[{"description":"Defect","name":"bug"},{"name":"feature"}],"type":"choice"},{"id":"quality","instructions":"How thorough?","levels":["Superficial","Adequate","Thorough"],"type":"score"}],"state":"the diff"},"name":"jev_ask","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
10
+ {"step":"01-prompt","ch":"host","data":{"args":{"questions":[{"id":"relevant","instructions":"Is this relevant?","type":"noul"},{"id":"label","instructions":"Which bucket?","options":[{"description":"Defect","name":"bug"},{"name":"feature"}],"type":"choice"},{"id":"quality","instructions":"How thorough?","levels":["Superficial","Adequate","Thorough"],"type":"score"}],"state":"the diff"},"toolName":"jev_ask","type":"tool_execution_start"}}
11
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"label":{"criteria":{"bug":"Defect","feature":null},"instructions":"Which bucket?","type":"choice"},"quality":{"criteria":["Superficial","Adequate","Thorough"],"instructions":"How thorough?","type":"score"},"relevant":{"instructions":"Is this relevant?","type":"noul"}},"state":"the diff"},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
12
+ {"step":"01-prompt","ch":"host","data":{"isError":false,"result":{"content":[{"text":"model jev-eq\nrelevant: yes 0.93 <- Is this relevant?\nlabel: bug (conf 0.90) [bug 0.90, feature 0.10] <- Which bucket?\nquality: 1.75/3 (conf 0.80) <- How thorough?\ntokens 12 in / 3 out","type":"text"}],"details":{"answers":{"label":{"choice":"bug","confidence":0.9,"probabilities":{"bug":0.9,"feature":0.1},"type":"choice"},"quality":{"confidence":0.8,"legend":{"0":"None, it only reads","1":"Small","2":"Large","3":"Severe"},"probabilities":{"0":0.05,"1":0.1,"2":0.15,"3":0.7},"score":1.75,"type":"score"},"relevant":{"noul":0.93,"type":"noul"}},"model":"jev-eq","ok":true,"usage":{"input_tokens":12,"output_tokens":3}}},"toolName":"jev_ask","type":"tool_execution_end"}}
13
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"model jev-eq\nrelevant: yes 0.93 <- Is this relevant?\nlabel: bug (conf 0.90) [bug 0.90, feature 0.10] <- Which bucket?\nquality: 1.75/3 (conf 0.80) <- How thorough?\ntokens 12 in / 3 out","type":"text"}],"details":{"answers":{"label":{"choice":"bug","confidence":0.9,"probabilities":{"bug":0.9,"feature":0.1},"type":"choice"},"quality":{"confidence":0.8,"legend":{"0":"None, it only reads","1":"Small","2":"Large","3":"Severe"},"probabilities":{"0":0.05,"1":0.1,"2":0.15,"3":0.7},"score":1.75,"type":"score"},"relevant":{"noul":0.93,"type":"noul"}},"model":"jev-eq","ok":true,"usage":{"input_tokens":12,"output_tokens":3}},"isError":false,"role":"toolResult","toolName":"jev_ask"},"type":"message_end"}}
14
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
15
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
16
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"ask jev about the diff"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"questions\":[{\"id\":\"relevant\",\"instructions\":\"Is this relevant?\",\"type\":\"noul\"},{\"id\":\"label\",\"instructions\":\"Which bucket?\",\"options\":[{\"description\":\"Defect\",\"name\":\"bug\"},{\"name\":\"feature\"}],\"type\":\"choice\"},{\"id\":\"quality\",\"instructions\":\"How thorough?\",\"levels\":[\"Superficial\",\"Adequate\",\"Thorough\"],\"type\":\"score\"}],\"state\":\"the diff\"}","name":"jev_ask"}]},{"role":"tool","text":"model jev-eq\nrelevant: yes 0.93 <- Is this relevant?\nlabel: bug (conf 0.90) [bug 0.90, feature 0.10] <- Which bucket?\nquality: 1.75/3 (conf 0.80) <- How thorough?\ntokens 12 in / 3 out"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
17
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"asked","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
18
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
19
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
20
+ {"step":"99-shutdown","ch":"exit","data":{"code":0}}
@@ -0,0 +1,23 @@
1
+ {"kind":"header","scenario":"output-advice","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
2
+ {"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: shadow"}}
3
+ {"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
4
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
5
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
6
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
7
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"run the flaky command","type":"text"}],"role":"user"},"type":"message_end"}}
8
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"run the flaky command"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
9
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"command":"echo timeout; exit 3"},"name":"bash","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
10
+ {"step":"01-prompt","ch":"host","data":{"args":{"command":"echo timeout; exit 3"},"toolName":"bash","type":"tool_execution_start"}}
11
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":{"arguments":{"command":"echo timeout; exit 3"},"cwd":"<cwd>","platform":"linux","tool":"bash","user_request":"run the flaky command"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
12
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: clear (shadow)"}}
13
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"failure_class":{"criteria":{"code_bug":"The code or types are wrong","environment":"A missing dependency, port, or tool in the local setup","no_failure":"Output reports success or nothing wrong","permission":"Access was denied by the OS or a server","transient":"A network or resource hiccup that may succeed on a retry","user_error":"The command itself was invoked wrongly"},"instructions":"What kind of failure is `output` reporting?","type":"choice"},"leaks_secret":{"criteria":{"false":"Only ordinary program output","true":"A key, token, password, or private key appears in the text"},"instructions":"Does `output` contain a secret or credential that must not be written to a session transcript?","type":"noul"}},"state":{"arguments":{"command":"echo timeout; exit 3"},"cwd":"<cwd>","is_error":true,"output":"timeout\n\n\nCommand exited with code 3","tool":"bash"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
14
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: advice (bash)"}}
15
+ {"step":"01-prompt","ch":"host","data":{"isError":true,"result":{"content":[{"text":"timeout\n\n\nCommand exited with code 3","type":"text"},{"text":"[pi-jev] Jev read this as a transient failure (confidence 1.00): retrying the same command unchanged is reasonable.","type":"text"}],"isError":true},"toolName":"bash","type":"tool_execution_end"}}
16
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"timeout\n\n\nCommand exited with code 3","type":"text"},{"text":"[pi-jev] Jev read this as a transient failure (confidence 1.00): retrying the same command unchanged is reasonable.","type":"text"}],"isError":true,"role":"toolResult","toolName":"bash"},"type":"message_end"}}
17
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
18
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
19
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"run the flaky command"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo timeout; exit 3\"}","name":"bash"}]},{"role":"tool","text":"timeout\n\n\nCommand exited with code 3\n[pi-jev] Jev read this as a transient failure (confidence 1.00): retrying the same command unchanged is reasonable."}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
20
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"retry","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
21
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
22
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
23
+ {"step":"99-shutdown","ch":"exit","data":{"code":0}}
@@ -0,0 +1,24 @@
1
+ {"kind":"header","scenario":"output-leak","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
2
+ {"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: shadow"}}
3
+ {"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
4
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
5
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
6
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
7
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"show me the key","type":"text"}],"role":"user"},"type":"message_end"}}
8
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"show me the key"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
9
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"command":"echo API_KEY=abc123"},"name":"bash","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
10
+ {"step":"01-prompt","ch":"host","data":{"args":{"command":"echo API_KEY=abc123"},"toolName":"bash","type":"tool_execution_start"}}
11
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":{"arguments":{"command":"echo API_KEY=abc123"},"cwd":"<cwd>","platform":"linux","tool":"bash","user_request":"show me the key"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
12
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: clear (shadow)"}}
13
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"failure_class":{"criteria":{"code_bug":"The code or types are wrong","environment":"A missing dependency, port, or tool in the local setup","no_failure":"Output reports success or nothing wrong","permission":"Access was denied by the OS or a server","transient":"A network or resource hiccup that may succeed on a retry","user_error":"The command itself was invoked wrongly"},"instructions":"What kind of failure is `output` reporting?","type":"choice"},"leaks_secret":{"criteria":{"false":"Only ordinary program output","true":"A key, token, password, or private key appears in the text"},"instructions":"Does `output` contain a secret or credential that must not be written to a session transcript?","type":"noul"}},"state":{"arguments":{"command":"echo API_KEY=abc123"},"cwd":"<cwd>","is_error":false,"output":"API_KEY=abc123\n","tool":"bash"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
14
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: leak (bash)"}}
15
+ {"step":"01-prompt","ch":"ui","data":{"message":"jev: bash output may carry a secret (0.94)","method":"notify","notifyType":"warning"}}
16
+ {"step":"01-prompt","ch":"host","data":{"isError":false,"result":{"content":[{"text":"API_KEY=abc123\n","type":"text"},{"text":"[pi-jev] Jev flagged this output as containing a secret (0.94). Do not repeat the value in a reply, a file, or a command; refer to it by name instead.","type":"text"}]},"toolName":"bash","type":"tool_execution_end"}}
17
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"API_KEY=abc123\n","type":"text"},{"text":"[pi-jev] Jev flagged this output as containing a secret (0.94). Do not repeat the value in a reply, a file, or a command; refer to it by name instead.","type":"text"}],"isError":false,"role":"toolResult","toolName":"bash"},"type":"message_end"}}
18
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
19
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
20
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"show me the key"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo API_KEY=abc123\"}","name":"bash"}]},{"role":"tool","text":"API_KEY=abc123\n\n[pi-jev] Jev flagged this output as containing a secret (0.94). Do not repeat the value in a reply, a file, or a command; refer to it by name instead."}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
21
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"done","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
22
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
23
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
24
+ {"step":"99-shutdown","ch":"exit","data":{"code":0}}
@@ -0,0 +1,22 @@
1
+ {"kind":"header","scenario":"output-low-confidence","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
2
+ {"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: shadow"}}
3
+ {"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
4
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
5
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
6
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
7
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"run it","type":"text"}],"role":"user"},"type":"message_end"}}
8
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"run it"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
9
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"command":"echo odd; exit 2"},"name":"bash","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
10
+ {"step":"01-prompt","ch":"host","data":{"args":{"command":"echo odd; exit 2"},"toolName":"bash","type":"tool_execution_start"}}
11
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":{"arguments":{"command":"echo odd; exit 2"},"cwd":"<cwd>","platform":"linux","tool":"bash","user_request":"run it"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
12
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: clear (shadow)"}}
13
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"failure_class":{"criteria":{"code_bug":"The code or types are wrong","environment":"A missing dependency, port, or tool in the local setup","no_failure":"Output reports success or nothing wrong","permission":"Access was denied by the OS or a server","transient":"A network or resource hiccup that may succeed on a retry","user_error":"The command itself was invoked wrongly"},"instructions":"What kind of failure is `output` reporting?","type":"choice"},"leaks_secret":{"criteria":{"false":"Only ordinary program output","true":"A key, token, password, or private key appears in the text"},"instructions":"Does `output` contain a secret or credential that must not be written to a session transcript?","type":"noul"}},"state":{"arguments":{"command":"echo odd; exit 2"},"cwd":"<cwd>","is_error":true,"output":"odd\n\n\nCommand exited with code 2","tool":"bash"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
14
+ {"step":"01-prompt","ch":"host","data":{"isError":true,"result":{"content":[{"text":"odd\n\n\nCommand exited with code 2","type":"text"}],"isError":true,"structuredContent":{"exit_code":2,"output":"odd\n","truncated":false,"wall_time_seconds":"<seconds>"}},"toolName":"bash","type":"tool_execution_end"}}
15
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"odd\n\n\nCommand exited with code 2","type":"text"}],"isError":true,"role":"toolResult","toolName":"bash"},"type":"message_end"}}
16
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
17
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
18
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"run it"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo odd; exit 2\"}","name":"bash"}]},{"role":"tool","text":"odd\n\n\nCommand exited with code 2"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
19
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"hm","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
20
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
21
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
22
+ {"step":"99-shutdown","ch":"exit","data":{"code":0}}