@pi-in-go/pigpen-jev 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/CREDITS.md +22 -0
  2. package/LICENSE +22 -0
  3. package/README.md +237 -0
  4. package/extensions/jev/ask.go +166 -0
  5. package/extensions/jev/ask_test.go +218 -0
  6. package/extensions/jev/backend.go +128 -0
  7. package/extensions/jev/bench_test.go +64 -0
  8. package/extensions/jev/boundaries_test.go +159 -0
  9. package/extensions/jev/command.go +224 -0
  10. package/extensions/jev/commands_test.go +214 -0
  11. package/extensions/jev/config.go +450 -0
  12. package/extensions/jev/errors_test.go +191 -0
  13. package/extensions/jev/extension.go +391 -0
  14. package/extensions/jev/fakehost_test.go +548 -0
  15. package/extensions/jev/gate.go +125 -0
  16. package/extensions/jev/gate_test.go +610 -0
  17. package/extensions/jev/gatekey_test.go +24 -0
  18. package/extensions/jev/go.mod +9 -0
  19. package/extensions/jev/go.sum +2 -0
  20. package/extensions/jev/go.work +10 -0
  21. package/extensions/jev/helpers_test.go +404 -0
  22. package/extensions/jev/memo.go +88 -0
  23. package/extensions/jev/output.go +89 -0
  24. package/extensions/jev/output_test.go +187 -0
  25. package/extensions/jev/ownmodel_test.go +118 -0
  26. package/extensions/jev/render.go +136 -0
  27. package/extensions/jev/review_test.go +310 -0
  28. package/extensions/jev/source_test.go +57 -0
  29. package/extensions/jev/text.go +174 -0
  30. package/extensions/jev/trust_test.go +335 -0
  31. package/extensions/jev/types.go +227 -0
  32. package/libs/typesafe/CONTRACT.md +125 -0
  33. package/libs/typesafe/CREDITS.md +37 -0
  34. package/libs/typesafe/LICENSE +23 -0
  35. package/libs/typesafe/README.md +19 -0
  36. package/libs/typesafe/go.mod +3 -0
  37. package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
  38. package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
  39. package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
  40. package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
  41. package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
  42. package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
  43. package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
  44. package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
  45. package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
  46. package/libs/typesafe/libraries/ownmodel/run.go +288 -0
  47. package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
  48. package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
  49. package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
  50. package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
  51. package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
  52. package/libs/typesafe/libraries/typesafe/answers.go +268 -0
  53. package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
  54. package/libs/typesafe/libraries/typesafe/batch.go +80 -0
  55. package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
  56. package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
  57. package/libs/typesafe/libraries/typesafe/client.go +561 -0
  58. package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
  59. package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
  60. package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
  61. package/libs/typesafe/libraries/typesafe/doc.go +27 -0
  62. package/libs/typesafe/libraries/typesafe/entry.go +142 -0
  63. package/libs/typesafe/libraries/typesafe/env.go +11 -0
  64. package/libs/typesafe/libraries/typesafe/errors.go +310 -0
  65. package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
  66. package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
  67. package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
  68. package/libs/typesafe/libraries/typesafe/logging.go +160 -0
  69. package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
  70. package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
  71. package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
  72. package/libs/typesafe/libraries/typesafe/questions.go +490 -0
  73. package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
  74. package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
  75. package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
  76. package/libs/typesafe/libraries/typesafe/retry.go +350 -0
  77. package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
  78. package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
  79. package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
  80. package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
  81. package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
  82. package/libs/typesafe/libraries/typesafe/version.go +10 -0
  83. package/libs/typesafe/package.json +37 -0
  84. package/libs/typesafe/provenance.json +49 -0
  85. package/package.json +42 -0
  86. package/port/PORT.md +107 -0
  87. package/port/e2e/gate-and-output.py +35 -0
  88. package/port/e2e/jev-ask.py +36 -0
  89. package/port/e2e/model-switch.py +44 -0
  90. package/port/e2e/off-by-default.py +34 -0
  91. package/port/gen-scenarios.py +103 -0
  92. package/port/golden/cache-identical-calls.jsonl +30 -0
  93. package/port/golden/clear.jsonl +22 -0
  94. package/port/golden/commands.jsonl +43 -0
  95. package/port/golden/enforce-accept.jsonl +23 -0
  96. package/port/golden/enforce-decline.jsonl +22 -0
  97. package/port/golden/jev-ask.jsonl +20 -0
  98. package/port/golden/output-advice.jsonl +23 -0
  99. package/port/golden/output-leak.jsonl +24 -0
  100. package/port/golden/output-low-confidence.jsonl +22 -0
  101. package/port/golden/shadow-flagged.jsonl +23 -0
  102. package/port/golden/unjudged-tools.jsonl +19 -0
  103. package/port/golden/write-elision.jsonl +21 -0
  104. package/port/mutate-unit.py +63 -0
  105. package/port/mutations.json +578 -0
  106. package/port/oracle/LICENSE +21 -0
  107. package/port/oracle/README.md +181 -0
  108. package/port/oracle/SHA256SUMS +8 -0
  109. package/port/oracle/package.json +43 -0
  110. package/port/oracle/src/client.ts +409 -0
  111. package/port/oracle/src/config.ts +363 -0
  112. package/port/oracle/src/gate.ts +229 -0
  113. package/port/oracle/src/index.ts +649 -0
  114. package/port/oracle/src/output.ts +163 -0
  115. package/port/red-run.log +309 -0
  116. package/port/scenarios/cache-identical-calls.json +71 -0
  117. package/port/scenarios/clear.json +61 -0
  118. package/port/scenarios/commands.json +119 -0
  119. package/port/scenarios/enforce-accept.json +66 -0
  120. package/port/scenarios/enforce-decline.json +57 -0
  121. package/port/scenarios/jev-ask.json +83 -0
  122. package/port/scenarios/output-advice.json +61 -0
  123. package/port/scenarios/output-leak.json +61 -0
  124. package/port/scenarios/output-low-confidence.json +61 -0
  125. package/port/scenarios/shadow-flagged.json +61 -0
  126. package/port/scenarios/unjudged-tools.json +55 -0
  127. package/port/scenarios/write-elision.json +53 -0
  128. package/provenance.json +18 -0
@@ -0,0 +1,23 @@
1
+ {"kind":"header","scenario":"shadow-flagged","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
2
+ {"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: shadow"}}
3
+ {"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
4
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
5
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
6
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
7
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"please run echo hi","type":"text"}],"role":"user"},"type":"message_end"}}
8
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"please run echo hi"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
9
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"command":"echo hi"},"name":"bash","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
10
+ {"step":"01-prompt","ch":"host","data":{"args":{"command":"echo hi"},"toolName":"bash","type":"tool_execution_start"}}
11
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":{"arguments":{"command":"echo hi"},"cwd":"<cwd>","platform":"linux","tool":"bash","user_request":"please run echo hi"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
12
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91"}}
13
+ {"step":"01-prompt","ch":"ui","data":{"message":"jev shadow: bash - destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91","method":"notify","notifyType":"warning"}}
14
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"failure_class":{"criteria":{"code_bug":"The code or types are wrong","environment":"A missing dependency, port, or tool in the local setup","no_failure":"Output reports success or nothing wrong","permission":"Access was denied by the OS or a server","transient":"A network or resource hiccup that may succeed on a retry","user_error":"The command itself was invoked wrongly"},"instructions":"What kind of failure is `output` reporting?","type":"choice"},"leaks_secret":{"criteria":{"false":"Only ordinary program output","true":"A key, token, password, or private key appears in the text"},"instructions":"Does `output` contain a secret or credential that must not be written to a session transcript?","type":"noul"}},"state":{"arguments":{"command":"echo hi"},"cwd":"<cwd>","is_error":false,"output":"hi\n","tool":"bash"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
15
+ {"step":"01-prompt","ch":"host","data":{"isError":false,"result":{"content":[{"text":"hi\n","type":"text"}],"structuredContent":{"exit_code":0,"output":"hi\n","truncated":false,"wall_time_seconds":"<seconds>"}},"toolName":"bash","type":"tool_execution_end"}}
16
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"hi\n","type":"text"}],"isError":false,"role":"toolResult","toolName":"bash"},"type":"message_end"}}
17
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
18
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
19
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"please run echo hi"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"command\":\"echo hi\"}","name":"bash"}]},{"role":"tool","text":"hi\n"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
20
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"done","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
21
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
22
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
23
+ {"step":"99-shutdown","ch":"exit","data":{"code":0}}
@@ -0,0 +1,19 @@
1
+ {"kind":"header","scenario":"unjudged-tools","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
2
+ {"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: shadow"}}
3
+ {"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
4
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
5
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
6
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
7
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"read a.txt","type":"text"}],"role":"user"},"type":"message_end"}}
8
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"read a.txt"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
9
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"path":"a.txt"},"name":"read","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
10
+ {"step":"01-prompt","ch":"host","data":{"args":{"path":"a.txt"},"toolName":"read","type":"tool_execution_start"}}
11
+ {"step":"01-prompt","ch":"host","data":{"isError":false,"result":{"content":[{"text":"hello\n","type":"text"}]},"toolName":"read","type":"tool_execution_end"}}
12
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"hello\n","type":"text"}],"isError":false,"role":"toolResult","toolName":"read"},"type":"message_end"}}
13
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
14
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
15
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"read a.txt"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"path\":\"a.txt\"}","name":"read"}]},{"role":"tool","text":"hello\n"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
16
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"read","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
17
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
18
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
19
+ {"step":"99-shutdown","ch":"exit","data":{"code":0}}
@@ -0,0 +1,21 @@
1
+ {"kind":"header","scenario":"write-elision","lane":"pi-ts","host":"pi 1.0.1","extension":"sha256:7fecf89390f53ec308d08d182796b1340e4c9553ac5e6cbdc2df19d9ad1b63e0","normalizer":"v3"}
2
+ {"step":"00-startup","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: shadow"}}
3
+ {"step":"01-prompt","ch":"response","data":{"command":"prompt","data":{"disposition":"started"},"success":true}}
4
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_start"}}
5
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
6
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":"","role":"system"},"type":"message_end"}}
7
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"write the file","type":"text"}],"role":"user"},"type":"message_end"}}
8
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"write the file"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
9
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"arguments":{"content":"xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx","path":"big.txt"},"name":"write","type":"toolCall"}],"role":"assistant","stopReason":"toolUse"},"type":"message_end"}}
10
+ {"step":"01-prompt","ch":"host","data":{"args":{"content":"xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx","path":"big.txt"},"toolName":"write","type":"tool_execution_start"}}
11
+ {"step":"01-prompt","ch":"http","data":{"body":{"model":"jev-eq","questions":{"beyond_scope":{"criteria":{"false":"Confined to what the user asked for","true":"Touches things the user did not ask about"},"instructions":"Does this action affect anything beyond the change the user asked for?","type":"noul"},"destructive":{"criteria":{"false":"Nothing is deleted or overwritten","true":"Deletes or overwrites data, or rewrites shared history"},"instructions":"Is this action destructive?","type":"noul"},"exfiltration":{"criteria":{"false":"Nothing leaves the machine","true":"Local data or secrets leave the machine"},"instructions":"Does this action send local file contents, credentials, or environment variables to a network destination outside the project's own tooling?","type":"noul"},"impact":{"criteria":["None, it only reads","Small, one file or one reversible change","Large, many files or shared state","Severe, data loss or a forced overwrite of shared history"],"instructions":"How much damage would this action do if the user did not want it?","type":"score"}},"state":{"arguments":{"content":"xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx…[500 chars elided]","path":"big.txt"},"cwd":"<cwd>","platform":"linux","tool":"write","user_request":"write the file"}},"headers":{"authorization":"Bearer tsk-eq-key-0123456789","content-type":"application/json"},"matched":true,"method":"POST","path":"/v1/systemone","query":{},"server":"jev"}}
12
+ {"step":"01-prompt","ch":"ui","data":{"method":"setStatus","statusKey":"jev","statusText":"jev: clear (shadow)"}}
13
+ {"step":"01-prompt","ch":"host","data":{"isError":false,"result":{"content":[{"text":"Successfully wrote to big.txt","type":"text"}]},"toolName":"write","type":"tool_execution_end"}}
14
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"Successfully wrote to big.txt","type":"text"}],"isError":false,"role":"toolResult","toolName":"write"},"type":"message_end"}}
15
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
16
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_start"}}
17
+ {"step":"01-prompt","ch":"llm","data":{"messages":[{"role":"user","text":"write the file"},{"role":"assistant","text":"","toolCalls":[{"arguments":"{\"content\":\"xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx\",\"path\":\"big.txt\"}","name":"write"}]},{"role":"tool","text":"Successfully wrote to big.txt"}],"system":{"inserted":"- jev_ask: Ask Jev typed questions (yes/no, choice, rubric) about text and get calibrated answers\n\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites.\n- Use jev_ask when a judgement must be typed and calibrated rather than written: classification, relevance, yes/no checks, rubric scores.\n- Ask one specific question per entry in jev_ask; split multi-factor judgements into separate questions and combine the answers yourself","removed":"\nIn addition to the tools above, you may have access to other custom tools depending on the project.\n</tools>\n\n<rules>\n- Use bash for file operations like ls, rg, find\n- Use read to examine files instead of cat or sed.\n- You can inspect PI_* environment variables for current model and session details.\n- Use edit for precise changes (edits[].oldText must match exactly)\n- When changing multiple separate locations in one file, use one edit call with multiple entries in edits[] instead of multiple edit calls\n- Each edits[].oldText is matched against the original file, not after earlier edits are applied. Do not emit overlapping or nested edits. Merge nearby changes into one edit.\n- Keep edits[].oldText as small as possible while still being unique in the file. Do not pad with large unchanged regions.\n- Use write only for new files or complete rewrites"},"tools":[{"description":"Read the contents of a file. Supports text files and images (jpg, png, gif, webp, bmp). Images are sent as attachments. For text files, output is truncated to 2000 lines or 50KB (whichever is hit first). Use offset/limit for large files. When you need the full file, continue with offset until complete.","name":"read","parameters":{"properties":{"limit":{"description":"Maximum number of lines to read","type":"number"},"offset":{"description":"Line number to start reading from (1-indexed)","type":"number"},"path":{"description":"Path to the file to read (relative or absolute)","type":"string"}},"required":["path"],"type":"object"}},{"description":"Execute a bash command in the current working directory. Returns stdout and stderr. Output is truncated to last 2000 lines or 50KB (whichever is hit first). If truncated, full output is saved to a temp file. Optionally provide a timeout in seconds.","name":"bash","parameters":{"properties":{"command":{"description":"Shell command to execute","type":"string"},"timeout":{"description":"Timeout in seconds (optional, no default timeout)","type":"number"}},"required":["command"],"type":"object"}},{"description":"Edit a single file using exact text replacement. Every edits[].oldText must match a unique, non-overlapping region of the original file. If two changes affect the same block or nearby lines, merge them into one edit instead of emitting overlapping edits. Do not include large unchanged regions just to connect distant changes.","name":"edit","parameters":{"properties":{"edits":{"description":"One or more targeted replacements. Each edit is matched against the original file, not incrementally. Do not include overlapping or nested edits. If two changes touch the same block or nearby lines, merge them into one edit instead.","items":{"properties":{"newText":{"description":"Replacement text for this targeted edit.","type":"string"},"oldText":{"description":"Exact text for one targeted replacement. It must be unique in the original file and must not overlap with any other edits[].oldText in the same call.","type":"string"}},"required":["oldText","newText"],"type":"object"},"type":"array"},"path":{"description":"Path to the file to edit (relative or absolute)","type":"string"}},"required":["path","edits"],"type":"object"}},{"description":"Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Automatically creates parent directories.","name":"write","parameters":{"properties":{"content":{"description":"Content to write to the file","type":"string"},"path":{"description":"Path to the file to write (relative or absolute)","type":"string"}},"required":["path","content"],"type":"object"}},{"description":"Ask TypeSafe Jev typed questions about a piece of text and get calibrated answers (probabilities, a chosen option, a rubric score) instead of prose.","name":"jev_ask","parameters":{"properties":{"questions":{"description":"One or more questions. All are evaluated in parallel against the same state.","items":{"properties":{"id":{"description":"Short key for this question. The answer comes back under it.","type":"string"},"instructions":{"description":"The one thing to judge. One specific, well-scoped gut-check per question.","type":"string"},"levels":{"description":"score only: ordered rubric levels, lowest first, at least two","items":{"type":"string"},"type":"array"},"options":{"description":"choice only: the options to choose between","items":{"properties":{"description":{"description":"When this option applies","type":"string"},"name":{"description":"Option key","type":"string"}},"required":["name"],"type":"object"},"type":"array"},"type":{"description":"noul = yes/no probability, choice = pick one option, score = value on a rubric","enum":["noul","choice","score"],"type":"string"}},"required":["id","type","instructions"],"type":"object"},"type":"array"},"state":{"description":"The text to judge: tool output, a diff, a message, a document excerpt.","type":"string"}},"required":["state","questions"],"type":"object"}}]}}
18
+ {"step":"01-prompt","ch":"host","data":{"message":{"content":[{"text":"written","type":"text"}],"role":"assistant","stopReason":"stop"},"type":"message_end"}}
19
+ {"step":"01-prompt","ch":"host","data":{"type":"turn_end"}}
20
+ {"step":"01-prompt","ch":"host","data":{"type":"agent_end"}}
21
+ {"step":"99-shutdown","ch":"exit","data":{"code":0}}
@@ -0,0 +1,63 @@
1
+ """Unit-layer mutation run for the Jev port.
2
+
3
+ Each mutant in mutations.json is applied to a scratch copy of the extension and the shared
4
+ client, then the port's Go tests run against it. `pigeq mutate --unit` cannot run this port
5
+ (its go.work lists PiG's SDK as a replace, which leaves a shared library Package unresolvable),
6
+ so this script writes a go.work that uses the SDK, the extension and the library as modules.
7
+
8
+ python3 components/jev/port/mutate-unit.py [mutant names...]
9
+
10
+ Environment: PIG_SDK_DIR (PiG's staged Go SDK; default: `$PIG reload --sdk-path`, with PIG
11
+ defaulting to `pig`), GO (default: `go` on PATH), MUTATE_WORK (scratch directory; default: a
12
+ new temporary directory), MUTATIONS (another mutation list; default: mutations.json here). Run it with the same isolated HOME/PIG_HOME as any other pig run.
13
+ """
14
+ import json, os, shutil, subprocess, sys, tempfile, time
15
+
16
+ HERE = os.path.dirname(os.path.abspath(__file__))
17
+ JEV = os.path.dirname(HERE) # components/jev
18
+ TYPESAFE = os.path.join(os.path.dirname(JEV), "typesafe") # components/typesafe
19
+ GO = os.environ.get("GO", "go")
20
+ SDK = os.environ.get("PIG_SDK_DIR") or subprocess.run(
21
+ [os.environ.get("PIG", "pig"), "reload", "--sdk-path"], capture_output=True, text=True, check=True
22
+ ).stdout.strip().splitlines()[-1]
23
+ WORK = os.environ.get("MUTATE_WORK") or tempfile.mkdtemp(prefix="jev-mutate-")
24
+ muts = json.load(open(os.environ.get("MUTATIONS") or os.path.join(HERE, "mutations.json")))
25
+ env = dict(os.environ, GOFLAGS="", GOTOOLCHAIN="local")
26
+
27
+
28
+ def prep():
29
+ d = os.path.join(WORK, "u")
30
+ shutil.rmtree(d, ignore_errors=True)
31
+ os.makedirs(d)
32
+ shutil.copytree(JEV, os.path.join(d, "jev"), ignore=shutil.ignore_patterns("port"))
33
+ shutil.copytree(TYPESAFE, os.path.join(d, "typesafe"), ignore=shutil.ignore_patterns("port"))
34
+ return d
35
+
36
+
37
+ results = []
38
+ only = sys.argv[1:]
39
+ for m in muts:
40
+ if only and m["name"] not in only:
41
+ continue
42
+ d = prep()
43
+ p = os.path.join(d, "jev", "extensions", "jev", m["file"])
44
+ s = open(p).read()
45
+ assert s.count(m["find"]) == 1, m["name"]
46
+ open(p, "w").write(s.replace(m["find"], m["replace"], 1))
47
+ work = os.path.join(d, "go.work")
48
+ open(work, "w").write(f"go 1.26\n\nuse (\n\t{SDK}\n\t{d}/jev/extensions/jev\n\t{d}/typesafe\n)\n")
49
+ t = time.time()
50
+ r = subprocess.run([GO, "test", "-count=1", "-timeout", "120s", "./..."], cwd=os.path.join(d, "jev", "extensions", "jev"),
51
+ env=dict(env, GOWORK=work), capture_output=True, text=True)
52
+ out = r.stdout + r.stderr
53
+ if r.returncode == 0:
54
+ status, detail = "SURVIVED", ""
55
+ elif "build failed" in out or "setup failed" in out:
56
+ status, detail = "INVALID", out.strip().split("\n")[0]
57
+ else:
58
+ status = "KILLED"
59
+ detail = next((l.strip() for l in out.split("\n") if l.startswith("--- FAIL") or l.startswith("panic:")), "FAIL")
60
+ print(f"{status} {m['name']} [{time.time()-t:.0f}s] {detail}", flush=True)
61
+ results.append((m["name"], status))
62
+ print(sum(1 for _, s in results if s == "KILLED"), "killed of", len(results),
63
+ "survived:", [n for n, s in results if s == "SURVIVED"], "invalid:", [n for n, s in results if s == "INVALID"])