@pi-in-go/pigpen-jev 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CREDITS.md +22 -0
- package/LICENSE +22 -0
- package/README.md +237 -0
- package/extensions/jev/ask.go +166 -0
- package/extensions/jev/ask_test.go +218 -0
- package/extensions/jev/backend.go +128 -0
- package/extensions/jev/bench_test.go +64 -0
- package/extensions/jev/boundaries_test.go +159 -0
- package/extensions/jev/command.go +224 -0
- package/extensions/jev/commands_test.go +214 -0
- package/extensions/jev/config.go +450 -0
- package/extensions/jev/errors_test.go +191 -0
- package/extensions/jev/extension.go +391 -0
- package/extensions/jev/fakehost_test.go +548 -0
- package/extensions/jev/gate.go +125 -0
- package/extensions/jev/gate_test.go +610 -0
- package/extensions/jev/gatekey_test.go +24 -0
- package/extensions/jev/go.mod +9 -0
- package/extensions/jev/go.sum +2 -0
- package/extensions/jev/go.work +10 -0
- package/extensions/jev/helpers_test.go +404 -0
- package/extensions/jev/memo.go +88 -0
- package/extensions/jev/output.go +89 -0
- package/extensions/jev/output_test.go +187 -0
- package/extensions/jev/ownmodel_test.go +118 -0
- package/extensions/jev/render.go +136 -0
- package/extensions/jev/review_test.go +310 -0
- package/extensions/jev/source_test.go +57 -0
- package/extensions/jev/text.go +174 -0
- package/extensions/jev/trust_test.go +335 -0
- package/extensions/jev/types.go +227 -0
- package/libs/typesafe/CONTRACT.md +125 -0
- package/libs/typesafe/CREDITS.md +37 -0
- package/libs/typesafe/LICENSE +23 -0
- package/libs/typesafe/README.md +19 -0
- package/libs/typesafe/go.mod +3 -0
- package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
- package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
- package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
- package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
- package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
- package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
- package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
- package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
- package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
- package/libs/typesafe/libraries/ownmodel/run.go +288 -0
- package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
- package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
- package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
- package/libs/typesafe/libraries/typesafe/answers.go +268 -0
- package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
- package/libs/typesafe/libraries/typesafe/batch.go +80 -0
- package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
- package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
- package/libs/typesafe/libraries/typesafe/client.go +561 -0
- package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
- package/libs/typesafe/libraries/typesafe/doc.go +27 -0
- package/libs/typesafe/libraries/typesafe/entry.go +142 -0
- package/libs/typesafe/libraries/typesafe/env.go +11 -0
- package/libs/typesafe/libraries/typesafe/errors.go +310 -0
- package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
- package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
- package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
- package/libs/typesafe/libraries/typesafe/logging.go +160 -0
- package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
- package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
- package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
- package/libs/typesafe/libraries/typesafe/questions.go +490 -0
- package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
- package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
- package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
- package/libs/typesafe/libraries/typesafe/retry.go +350 -0
- package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
- package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
- package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
- package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
- package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
- package/libs/typesafe/libraries/typesafe/version.go +10 -0
- package/libs/typesafe/package.json +37 -0
- package/libs/typesafe/provenance.json +49 -0
- package/package.json +42 -0
- package/port/PORT.md +107 -0
- package/port/e2e/gate-and-output.py +35 -0
- package/port/e2e/jev-ask.py +36 -0
- package/port/e2e/model-switch.py +44 -0
- package/port/e2e/off-by-default.py +34 -0
- package/port/gen-scenarios.py +103 -0
- package/port/golden/cache-identical-calls.jsonl +30 -0
- package/port/golden/clear.jsonl +22 -0
- package/port/golden/commands.jsonl +43 -0
- package/port/golden/enforce-accept.jsonl +23 -0
- package/port/golden/enforce-decline.jsonl +22 -0
- package/port/golden/jev-ask.jsonl +20 -0
- package/port/golden/output-advice.jsonl +23 -0
- package/port/golden/output-leak.jsonl +24 -0
- package/port/golden/output-low-confidence.jsonl +22 -0
- package/port/golden/shadow-flagged.jsonl +23 -0
- package/port/golden/unjudged-tools.jsonl +19 -0
- package/port/golden/write-elision.jsonl +21 -0
- package/port/mutate-unit.py +63 -0
- package/port/mutations.json +578 -0
- package/port/oracle/LICENSE +21 -0
- package/port/oracle/README.md +181 -0
- package/port/oracle/SHA256SUMS +8 -0
- package/port/oracle/package.json +43 -0
- package/port/oracle/src/client.ts +409 -0
- package/port/oracle/src/config.ts +363 -0
- package/port/oracle/src/gate.ts +229 -0
- package/port/oracle/src/index.ts +649 -0
- package/port/oracle/src/output.ts +163 -0
- package/port/red-run.log +309 -0
- package/port/scenarios/cache-identical-calls.json +71 -0
- package/port/scenarios/clear.json +61 -0
- package/port/scenarios/commands.json +119 -0
- package/port/scenarios/enforce-accept.json +66 -0
- package/port/scenarios/enforce-decline.json +57 -0
- package/port/scenarios/jev-ask.json +83 -0
- package/port/scenarios/output-advice.json +61 -0
- package/port/scenarios/output-leak.json +61 -0
- package/port/scenarios/output-low-confidence.json +61 -0
- package/port/scenarios/shadow-flagged.json +61 -0
- package/port/scenarios/unjudged-tools.json +55 -0
- package/port/scenarios/write-elision.json +53 -0
- package/provenance.json +18 -0
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The output judge: what to ask Jev about a tool result that already ran, and
|
|
3
|
+
* how to turn the answers into something the model and the user act on.
|
|
4
|
+
*
|
|
5
|
+
* The tool_call gate sees intent. It cannot see what a command actually
|
|
6
|
+
* printed, so it cannot catch a credential echoed into the transcript or tell
|
|
7
|
+
* a transient failure from a permanent one. Both are judgements about text
|
|
8
|
+
* that exists only after the call.
|
|
9
|
+
*
|
|
10
|
+
* Question phrasing and thresholds come from the probe in README.md: the leak
|
|
11
|
+
* question separated 0.92-0.99 from 0.01-0.02 with no overlap, and the failure
|
|
12
|
+
* class answered at confidence 0.88-1.00 when it was right and 0.42 when it was
|
|
13
|
+
* unsure, which is why a confidence floor silences the advice.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import {
|
|
17
|
+
answerFor,
|
|
18
|
+
confidenceFor,
|
|
19
|
+
type JevQuestion,
|
|
20
|
+
type JevResponse,
|
|
21
|
+
type JevState,
|
|
22
|
+
} from "./client";
|
|
23
|
+
import type { JevConfig } from "./config";
|
|
24
|
+
|
|
25
|
+
export interface OutputInput {
|
|
26
|
+
cwd: string;
|
|
27
|
+
toolName: string;
|
|
28
|
+
input: unknown;
|
|
29
|
+
/** Text of the tool result, already rendered by the tool. */
|
|
30
|
+
output: string;
|
|
31
|
+
isError: boolean;
|
|
32
|
+
outputChars: number;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export interface OutputVerdict {
|
|
36
|
+
/** Probability that the output carries a secret. */
|
|
37
|
+
leaksSecret: number;
|
|
38
|
+
/** Option name, or undefined when the answer was unusable. */
|
|
39
|
+
failureClass: string | undefined;
|
|
40
|
+
classConfidence: number | undefined;
|
|
41
|
+
/** Leak reason when the leak threshold was crossed, else the advice line. */
|
|
42
|
+
notice: string | undefined;
|
|
43
|
+
kind: "leak" | "advice" | "none";
|
|
44
|
+
answers: Record<string, unknown>;
|
|
45
|
+
model: string;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Two questions, one request. The class question is asked even on success
|
|
50
|
+
* because a request costs the same either way and the answer is ignored.
|
|
51
|
+
*/
|
|
52
|
+
export const OUTPUT_QUESTIONS: Record<string, JevQuestion> = {
|
|
53
|
+
leaks_secret: {
|
|
54
|
+
type: "noul",
|
|
55
|
+
instructions:
|
|
56
|
+
"Does `output` contain a secret or credential that must not be written to a session transcript?",
|
|
57
|
+
criteria: {
|
|
58
|
+
true: "A key, token, password, or private key appears in the text",
|
|
59
|
+
false: "Only ordinary program output",
|
|
60
|
+
},
|
|
61
|
+
},
|
|
62
|
+
failure_class: {
|
|
63
|
+
type: "choice",
|
|
64
|
+
instructions: "What kind of failure is `output` reporting?",
|
|
65
|
+
criteria: {
|
|
66
|
+
transient: "A network or resource hiccup that may succeed on a retry",
|
|
67
|
+
environment: "A missing dependency, port, or tool in the local setup",
|
|
68
|
+
code_bug: "The code or types are wrong",
|
|
69
|
+
permission: "Access was denied by the OS or a server",
|
|
70
|
+
user_error: "The command itself was invoked wrongly",
|
|
71
|
+
no_failure: "Output reports success or nothing wrong",
|
|
72
|
+
},
|
|
73
|
+
},
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* What to tell the model per class. A table, not a branch: the class names are
|
|
78
|
+
* Jev's, and adding one is a row rather than a code path.
|
|
79
|
+
*/
|
|
80
|
+
export const CLASS_ADVICE: Record<string, string> = {
|
|
81
|
+
transient: "retrying the same command unchanged is reasonable",
|
|
82
|
+
environment: "fix the environment (missing tool, port, or service) before retrying",
|
|
83
|
+
code_bug: "fix the code or types; retrying unchanged will not help",
|
|
84
|
+
permission: "access was denied; change what is being accessed or ask the user",
|
|
85
|
+
user_error: "the invocation itself was wrong; fix the command",
|
|
86
|
+
};
|
|
87
|
+
|
|
88
|
+
/** Keyed by tool and output hash: identical output must cost one judgement. */
|
|
89
|
+
export function outputKey(toolName: string, output: string): string {
|
|
90
|
+
let hash = 0x811c9dc5;
|
|
91
|
+
for (let index = 0; index < output.length; index += 1) {
|
|
92
|
+
hash ^= output.charCodeAt(index);
|
|
93
|
+
hash = Math.imul(hash, 0x01000193) >>> 0;
|
|
94
|
+
}
|
|
95
|
+
return `${toolName}:${output.length}:${hash.toString(16)}`;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
export function buildOutputState(input: OutputInput): JevState {
|
|
99
|
+
return {
|
|
100
|
+
cwd: input.cwd,
|
|
101
|
+
tool: input.toolName,
|
|
102
|
+
is_error: input.isError,
|
|
103
|
+
arguments: summarize(input.input, 400),
|
|
104
|
+
output: truncate(input.output, input.outputChars),
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
export function evaluateOutput(
|
|
109
|
+
response: JevResponse,
|
|
110
|
+
config: JevConfig,
|
|
111
|
+
): OutputVerdict {
|
|
112
|
+
const leak = answerFor(response, "leaks_secret");
|
|
113
|
+
const leaksSecret = leak?.type === "noul" ? leak.noul : 0;
|
|
114
|
+
const classAnswer = answerFor(response, "failure_class");
|
|
115
|
+
const failureClass =
|
|
116
|
+
classAnswer?.type === "choice" ? classAnswer.choice : undefined;
|
|
117
|
+
const classConfidence = confidenceFor(response, "failure_class");
|
|
118
|
+
|
|
119
|
+
let notice: string | undefined;
|
|
120
|
+
let kind: OutputVerdict["kind"] = "none";
|
|
121
|
+
if (leaksSecret >= config.output.leakThreshold) {
|
|
122
|
+
kind = "leak";
|
|
123
|
+
notice = `Jev flagged this output as containing a secret (${leaksSecret.toFixed(2)}). Do not repeat the value in a reply, a file, or a command; refer to it by name instead.`;
|
|
124
|
+
} else if (
|
|
125
|
+
failureClass &&
|
|
126
|
+
CLASS_ADVICE[failureClass] &&
|
|
127
|
+
(classConfidence === undefined ||
|
|
128
|
+
classConfidence >= config.output.minConfidence)
|
|
129
|
+
) {
|
|
130
|
+
kind = "advice";
|
|
131
|
+
notice = `Jev read this as a ${failureClass} failure (confidence ${classConfidence?.toFixed(2) ?? "n/a"}): ${CLASS_ADVICE[failureClass]}.`;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
return {
|
|
135
|
+
leaksSecret,
|
|
136
|
+
failureClass,
|
|
137
|
+
classConfidence,
|
|
138
|
+
notice,
|
|
139
|
+
kind,
|
|
140
|
+
answers: response.answers,
|
|
141
|
+
model: response.model,
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** Same elision rule as the gate: long strings leave as a prefix plus a count. */
|
|
146
|
+
function summarize(value: unknown, maxChars: number, depth = 0): unknown {
|
|
147
|
+
if (typeof value === "string") return truncate(value, maxChars);
|
|
148
|
+
if (depth > 3 || value === null || typeof value !== "object") return value;
|
|
149
|
+
if (Array.isArray(value)) {
|
|
150
|
+
return value.map((item) => summarize(item, maxChars, depth + 1));
|
|
151
|
+
}
|
|
152
|
+
const out: Record<string, unknown> = {};
|
|
153
|
+
for (const [key, item] of Object.entries(value)) {
|
|
154
|
+
out[key] = summarize(item, maxChars, depth + 1);
|
|
155
|
+
}
|
|
156
|
+
return out;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
function truncate(text: string, maxChars: number): string {
|
|
160
|
+
return text.length > maxChars
|
|
161
|
+
? `${text.slice(0, maxChars)}\u2026[${text.length - maxChars} chars elided]`
|
|
162
|
+
: text;
|
|
163
|
+
}
|
package/port/red-run.log
ADDED
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
=== RUN TestAsk_RegisteredOnlyWhenEnabledWithAKey
|
|
2
|
+
--- PASS: TestAsk_RegisteredOnlyWhenEnabledWithAKey (0.01s)
|
|
3
|
+
=== RUN TestAsk_RendersTypedAnswers
|
|
4
|
+
ask_test.go:62: tool "jev_ask" is not registered
|
|
5
|
+
--- FAIL: TestAsk_RendersTypedAnswers (0.00s)
|
|
6
|
+
=== RUN TestAsk_InvalidQuestionShapesAreExplainedWithoutARequest
|
|
7
|
+
ask_test.go:122: tool "jev_ask" is not registered
|
|
8
|
+
--- FAIL: TestAsk_InvalidQuestionShapesAreExplainedWithoutARequest (0.00s)
|
|
9
|
+
=== RUN TestAsk_BlankInstructionsAndNoQuestionsAreRefused
|
|
10
|
+
ask_test.go:134: tool "jev_ask" is not registered
|
|
11
|
+
--- FAIL: TestAsk_BlankInstructionsAndNoQuestionsAreRefused (0.00s)
|
|
12
|
+
=== RUN TestCorrection_DuplicateQuestionIdIsRefused
|
|
13
|
+
ask_test.go:151: tool "jev_ask" is not registered
|
|
14
|
+
--- FAIL: TestCorrection_DuplicateQuestionIdIsRefused (0.00s)
|
|
15
|
+
=== RUN TestAsk_ServerErrorIsReportedNotThrown
|
|
16
|
+
ask_test.go:164: tool "jev_ask" is not registered
|
|
17
|
+
--- FAIL: TestAsk_ServerErrorIsReportedNotThrown (0.00s)
|
|
18
|
+
=== RUN TestAsk_MissingAnswerIsAnErrorNotAVerdict
|
|
19
|
+
ask_test.go:172: tool "jev_ask" is not registered
|
|
20
|
+
--- FAIL: TestAsk_MissingAnswerIsAnErrorNotAVerdict (0.00s)
|
|
21
|
+
=== RUN TestAsk_AnswerForAnotherTypeIsAnError
|
|
22
|
+
ask_test.go:180: tool "jev_ask" is not registered
|
|
23
|
+
--- FAIL: TestAsk_AnswerForAnotherTypeIsAnError (0.00s)
|
|
24
|
+
=== RUN TestAsk_HonoursOff
|
|
25
|
+
ask_test.go:188: command "jev" is not registered
|
|
26
|
+
--- FAIL: TestAsk_HonoursOff (0.00s)
|
|
27
|
+
=== RUN TestAsk_ToolMetadata
|
|
28
|
+
ask_test.go:201: registerTool calls = 0
|
|
29
|
+
--- FAIL: TestAsk_ToolMetadata (0.00s)
|
|
30
|
+
=== RUN TestCommand_OffAndOnToggleBothJudges
|
|
31
|
+
commands_test.go:32: command "jev" is not registered
|
|
32
|
+
--- FAIL: TestCommand_OffAndOnToggleBothJudges (0.00s)
|
|
33
|
+
=== RUN TestCommand_ModeSwitch
|
|
34
|
+
commands_test.go:53: command "jev" is not registered
|
|
35
|
+
--- FAIL: TestCommand_ModeSwitch (0.00s)
|
|
36
|
+
=== RUN TestCommand_LastVerdict
|
|
37
|
+
commands_test.go:72: command "jev" is not registered
|
|
38
|
+
--- FAIL: TestCommand_LastVerdict (0.00s)
|
|
39
|
+
=== RUN TestCommand_OutputVerdict
|
|
40
|
+
commands_test.go:84: command "jev" is not registered
|
|
41
|
+
--- FAIL: TestCommand_OutputVerdict (0.00s)
|
|
42
|
+
=== RUN TestCorrection_CleanOutputIsRecordedToo
|
|
43
|
+
commands_test.go:100: command "jev" is not registered
|
|
44
|
+
--- FAIL: TestCorrection_CleanOutputIsRecordedToo (0.00s)
|
|
45
|
+
=== RUN TestCommand_CheckRunsTheGateQuestionsOnSuppliedText
|
|
46
|
+
commands_test.go:108: command "jev" is not registered
|
|
47
|
+
--- FAIL: TestCommand_CheckRunsTheGateQuestionsOnSuppliedText (0.00s)
|
|
48
|
+
=== RUN TestCommand_CheckErrorIsReportedAsError
|
|
49
|
+
commands_test.go:128: command "jev" is not registered
|
|
50
|
+
--- FAIL: TestCommand_CheckErrorIsReportedAsError (0.00s)
|
|
51
|
+
=== RUN TestCommand_StatusShowsPolicyAndDestination
|
|
52
|
+
commands_test.go:134: command "jev" is not registered
|
|
53
|
+
--- FAIL: TestCommand_StatusShowsPolicyAndDestination (0.00s)
|
|
54
|
+
=== RUN TestCommand_StatusIsTruthfulAboutTheKeySource
|
|
55
|
+
commands_test.go:141: command "jev" is not registered
|
|
56
|
+
--- FAIL: TestCommand_StatusIsTruthfulAboutTheKeySource (0.00s)
|
|
57
|
+
=== RUN TestCommand_StatusWhenOffExplainsOptIn
|
|
58
|
+
commands_test.go:150: command "jev" is not registered
|
|
59
|
+
--- FAIL: TestCommand_StatusWhenOffExplainsOptIn (0.00s)
|
|
60
|
+
=== RUN TestOptIn_SlashOnAsksForConsentWithDisclosure
|
|
61
|
+
commands_test.go:164: command "jev" is not registered
|
|
62
|
+
--- FAIL: TestOptIn_SlashOnAsksForConsentWithDisclosure (0.00s)
|
|
63
|
+
=== RUN TestOptIn_ConsentIsPerSessionAndNeverWritesConfig
|
|
64
|
+
commands_test.go:194: command "jev" is not registered
|
|
65
|
+
--- FAIL: TestOptIn_ConsentIsPerSessionAndNeverWritesConfig (0.00s)
|
|
66
|
+
=== RUN TestFailOpen_ServerErrorNeverBlocks
|
|
67
|
+
errors_test.go:28: notifications = []
|
|
68
|
+
--- FAIL: TestFailOpen_ServerErrorNeverBlocks (0.00s)
|
|
69
|
+
=== RUN TestFailOpen_OutputJudgeLeavesResultAlone
|
|
70
|
+
--- PASS: TestFailOpen_OutputJudgeLeavesResultAlone (0.00s)
|
|
71
|
+
=== RUN TestErrors_TimeoutFailsOpen
|
|
72
|
+
errors_test.go:51: notifications = []
|
|
73
|
+
--- FAIL: TestErrors_TimeoutFailsOpen (0.00s)
|
|
74
|
+
=== RUN TestErrors_ReportedAtMostOncePerMinute
|
|
75
|
+
errors_test.go:67: error notifications = 0, want 1
|
|
76
|
+
--- FAIL: TestErrors_ReportedAtMostOncePerMinute (0.00s)
|
|
77
|
+
=== RUN TestErrors_EmptyAnswersAreNotAClearVerdict
|
|
78
|
+
errors_test.go:77: notifications = []
|
|
79
|
+
--- FAIL: TestErrors_EmptyAnswersAreNotAClearVerdict (0.00s)
|
|
80
|
+
=== RUN TestErrors_WrongAnswerTypeIsAnError
|
|
81
|
+
errors_test.go:90: notifications = []
|
|
82
|
+
--- FAIL: TestErrors_WrongAnswerTypeIsAnError (0.00s)
|
|
83
|
+
=== RUN TestErrors_MissingAnswersMapAndNonObject
|
|
84
|
+
errors_test.go:99: body {"model":"m"}: notifications = []
|
|
85
|
+
errors_test.go:99: body [1]: notifications = []
|
|
86
|
+
errors_test.go:99: body null: notifications = []
|
|
87
|
+
--- FAIL: TestErrors_MissingAnswersMapAndNonObject (0.00s)
|
|
88
|
+
=== RUN TestCorrection_OutOfRangeProbabilityIsAnError
|
|
89
|
+
errors_test.go:110: noul 1.7 accepted: []
|
|
90
|
+
errors_test.go:110: noul -0.2 accepted: []
|
|
91
|
+
--- FAIL: TestCorrection_OutOfRangeProbabilityIsAnError (0.00s)
|
|
92
|
+
=== RUN TestCorrection_ScoreOutsideTheRubricIsAnError
|
|
93
|
+
errors_test.go:119: notifications = []
|
|
94
|
+
--- FAIL: TestCorrection_ScoreOutsideTheRubricIsAnError (0.00s)
|
|
95
|
+
=== RUN TestCorrection_ConfidenceOutsideZeroToOneIsAnError
|
|
96
|
+
errors_test.go:127: notifications = []
|
|
97
|
+
--- FAIL: TestCorrection_ConfidenceOutsideZeroToOneIsAnError (0.00s)
|
|
98
|
+
=== RUN TestCorrection_ErrorStatusIsUnavailableNotStaleClear
|
|
99
|
+
errors_test.go:141: setup status "<none>"
|
|
100
|
+
--- FAIL: TestCorrection_ErrorStatusIsUnavailableNotStaleClear (0.00s)
|
|
101
|
+
=== RUN TestCorrection_OutputJudgeErrorAlsoShowsUnavailable
|
|
102
|
+
errors_test.go:153: status = "<none>"
|
|
103
|
+
--- FAIL: TestCorrection_OutputJudgeErrorAlsoShowsUnavailable (0.00s)
|
|
104
|
+
=== RUN TestErrors_ApiKeyIsRedactedFromNotifications
|
|
105
|
+
errors_test.go:166: notifications = []
|
|
106
|
+
--- FAIL: TestErrors_ApiKeyIsRedactedFromNotifications (0.00s)
|
|
107
|
+
=== RUN TestErrors_QuestionsAreValidatedBeforeAnyRequest
|
|
108
|
+
--- PASS: TestErrors_QuestionsAreValidatedBeforeAnyRequest (0.00s)
|
|
109
|
+
=== RUN TestGate_ShadowFlaggedNotifiesAndNeverBlocks
|
|
110
|
+
gate_test.go:29: notifications = [], want "warning: jev shadow: bash - destructive 0.99, exfiltration 0.79, beyond_scope 0.98, impact 3.00/3 at confidence 0.91"
|
|
111
|
+
gate_test.go:32: status = "<none>"
|
|
112
|
+
--- FAIL: TestGate_ShadowFlaggedNotifiesAndNeverBlocks (0.00s)
|
|
113
|
+
=== RUN TestGate_ClearVerdictSetsClearStatus
|
|
114
|
+
gate_test.go:46: status = "<none>"
|
|
115
|
+
--- FAIL: TestGate_ClearVerdictSetsClearStatus (0.00s)
|
|
116
|
+
=== RUN TestGate_SessionStartStatus
|
|
117
|
+
gate_test.go:60: status = "<none>", want "jev: shadow (out off)"
|
|
118
|
+
--- FAIL: TestGate_SessionStartStatus (0.00s)
|
|
119
|
+
=== RUN TestGate_EnforceAsksAndDeclineBlocks
|
|
120
|
+
gate_test.go:73: block=false reason=""
|
|
121
|
+
--- FAIL: TestGate_EnforceAsksAndDeclineBlocks (0.00s)
|
|
122
|
+
=== RUN TestGate_EnforceAcceptedRuns
|
|
123
|
+
--- PASS: TestGate_EnforceAcceptedRuns (0.00s)
|
|
124
|
+
=== RUN TestGate_EnforceHeadlessDegradesToWarning
|
|
125
|
+
gate_test.go:106: notifications = []
|
|
126
|
+
--- FAIL: TestGate_EnforceHeadlessDegradesToWarning (0.00s)
|
|
127
|
+
=== RUN TestGate_EnforceHeadlessBlocksWithBlockWithoutUI
|
|
128
|
+
gate_test.go:122: block=false reason=""
|
|
129
|
+
--- FAIL: TestGate_EnforceHeadlessBlocksWithBlockWithoutUI (0.00s)
|
|
130
|
+
=== RUN TestGate_OnlyJudgesConfiguredTools
|
|
131
|
+
gate_test.go:140: requests = 0, want 3 (bash, write, edit)
|
|
132
|
+
--- FAIL: TestGate_OnlyJudgesConfiguredTools (0.00s)
|
|
133
|
+
=== RUN TestGate_ConfiguredToolsReplaceDefaults
|
|
134
|
+
--- PASS: TestGate_ConfiguredToolsReplaceDefaults (0.00s)
|
|
135
|
+
=== RUN TestGate_RequestShape
|
|
136
|
+
gate_test.go:167: requests = 0
|
|
137
|
+
--- FAIL: TestGate_RequestShape (0.00s)
|
|
138
|
+
=== RUN TestGate_ArgumentsKeepInsertionOrder
|
|
139
|
+
gate_test.go:251: the extension made no request
|
|
140
|
+
--- FAIL: TestGate_ArgumentsKeepInsertionOrder (0.00s)
|
|
141
|
+
=== RUN TestGate_LongArgumentsAreElided
|
|
142
|
+
gate_test.go:265: the extension made no request
|
|
143
|
+
--- FAIL: TestGate_LongArgumentsAreElided (0.00s)
|
|
144
|
+
=== RUN TestGate_ArgumentCharsCountUTF16Units
|
|
145
|
+
gate_test.go:282: the extension made no request
|
|
146
|
+
--- FAIL: TestGate_ArgumentCharsCountUTF16Units (0.00s)
|
|
147
|
+
=== RUN TestGate_CutInsideSurrogatePairStaysValidUTF8
|
|
148
|
+
gate_test.go:295: the extension made no request
|
|
149
|
+
--- FAIL: TestGate_CutInsideSurrogatePairStaysValidUTF8 (0.00s)
|
|
150
|
+
=== RUN TestCorrection_DeepStringsAreElidedToo
|
|
151
|
+
gate_test.go:315: the extension made no request
|
|
152
|
+
--- FAIL: TestCorrection_DeepStringsAreElidedToo (0.00s)
|
|
153
|
+
=== RUN TestCorrection_MaxStateCharsCapsTheState
|
|
154
|
+
gate_test.go:334: the extension made no request
|
|
155
|
+
--- FAIL: TestCorrection_MaxStateCharsCapsTheState (0.00s)
|
|
156
|
+
=== RUN TestGate_UserRequestIsTruncatedTo1200Units
|
|
157
|
+
gate_test.go:353: the extension made no request
|
|
158
|
+
--- FAIL: TestGate_UserRequestIsTruncatedTo1200Units (0.00s)
|
|
159
|
+
=== RUN TestGate_NoUserRequestOmitsTheField
|
|
160
|
+
gate_test.go:365: the extension made no request
|
|
161
|
+
--- FAIL: TestGate_NoUserRequestOmitsTheField (0.00s)
|
|
162
|
+
=== RUN TestGate_ImpactBelowMinConfidenceDoesNotFlag
|
|
163
|
+
gate_test.go:378: status = "<none>"
|
|
164
|
+
--- FAIL: TestGate_ImpactBelowMinConfidenceDoesNotFlag (0.00s)
|
|
165
|
+
=== RUN TestGate_ThresholdIsInclusive
|
|
166
|
+
gate_test.go:390: status = "<none>"
|
|
167
|
+
--- FAIL: TestGate_ThresholdIsInclusive (0.00s)
|
|
168
|
+
=== RUN TestGate_ToFixedRoundsTiesUpLikeJavaScript
|
|
169
|
+
gate_test.go:403: status = "<none>", want the JavaScript rendering 0.13
|
|
170
|
+
--- FAIL: TestGate_ToFixedRoundsTiesUpLikeJavaScript (0.00s)
|
|
171
|
+
=== RUN TestGate_DisabledGateSkips
|
|
172
|
+
--- PASS: TestGate_DisabledGateSkips (0.00s)
|
|
173
|
+
=== RUN TestGate_IdenticalCallIsJudgedOncePerWindow
|
|
174
|
+
gate_test.go:432: requests = 0, want 1
|
|
175
|
+
--- FAIL: TestGate_IdenticalCallIsJudgedOncePerWindow (0.00s)
|
|
176
|
+
=== RUN TestGate_CacheKeyIgnoresPropertyOrder
|
|
177
|
+
gate_test.go:449: requests = 0, want 1
|
|
178
|
+
--- FAIL: TestGate_CacheKeyIgnoresPropertyOrder (0.00s)
|
|
179
|
+
=== RUN TestCorrection_CacheIsBoundToTheUserRequest
|
|
180
|
+
gate_test.go:472: requests = 0, want 2: the second judgment used a verdict from another request
|
|
181
|
+
--- FAIL: TestCorrection_CacheIsBoundToTheUserRequest (0.00s)
|
|
182
|
+
=== RUN TestCorrection_CacheIsBoundToTheModelAndEndpoint
|
|
183
|
+
gate_test.go:490: requests = 0: a verdict from another model was reused
|
|
184
|
+
--- FAIL: TestCorrection_CacheIsBoundToTheModelAndEndpoint (0.00s)
|
|
185
|
+
=== RUN TestGate_SiblingCallsShareOneInFlightRequest
|
|
186
|
+
gate_test.go:509: requests = 0, want 1 shared request
|
|
187
|
+
--- FAIL: TestGate_SiblingCallsShareOneInFlightRequest (0.00s)
|
|
188
|
+
=== RUN TestGate_ErrorsAreNotCached
|
|
189
|
+
gate_test.go:527: requests = 0, want 2
|
|
190
|
+
--- FAIL: TestGate_ErrorsAreNotCached (0.00s)
|
|
191
|
+
=== RUN TestOutput_LeakAppendsNoticeAndWarns
|
|
192
|
+
output_test.go:31: result not patched
|
|
193
|
+
--- FAIL: TestOutput_LeakAppendsNoticeAndWarns (0.00s)
|
|
194
|
+
=== RUN TestOutput_FailureClassAdviceTable
|
|
195
|
+
=== RUN TestOutput_FailureClassAdviceTable/transient
|
|
196
|
+
output_test.go:61: no advice
|
|
197
|
+
=== RUN TestOutput_FailureClassAdviceTable/environment
|
|
198
|
+
output_test.go:61: no advice
|
|
199
|
+
=== RUN TestOutput_FailureClassAdviceTable/code_bug
|
|
200
|
+
output_test.go:61: no advice
|
|
201
|
+
=== RUN TestOutput_FailureClassAdviceTable/permission
|
|
202
|
+
output_test.go:61: no advice
|
|
203
|
+
=== RUN TestOutput_FailureClassAdviceTable/user_error
|
|
204
|
+
output_test.go:61: no advice
|
|
205
|
+
--- FAIL: TestOutput_FailureClassAdviceTable (0.00s)
|
|
206
|
+
--- FAIL: TestOutput_FailureClassAdviceTable/transient (0.00s)
|
|
207
|
+
--- FAIL: TestOutput_FailureClassAdviceTable/environment (0.00s)
|
|
208
|
+
--- FAIL: TestOutput_FailureClassAdviceTable/code_bug (0.00s)
|
|
209
|
+
--- FAIL: TestOutput_FailureClassAdviceTable/permission (0.00s)
|
|
210
|
+
--- FAIL: TestOutput_FailureClassAdviceTable/user_error (0.00s)
|
|
211
|
+
=== RUN TestOutput_NoFailureAndUnknownClassSayNothing
|
|
212
|
+
--- PASS: TestOutput_NoFailureAndUnknownClassSayNothing (0.00s)
|
|
213
|
+
=== RUN TestOutput_LowClassConfidenceIsSilent
|
|
214
|
+
--- PASS: TestOutput_LowClassConfidenceIsSilent (0.00s)
|
|
215
|
+
=== RUN TestOutput_LeakThresholdIsInclusiveAndLeakWinsOverClass
|
|
216
|
+
output_test.go:96: result was not patched
|
|
217
|
+
--- FAIL: TestOutput_LeakThresholdIsInclusiveAndLeakWinsOverClass (0.00s)
|
|
218
|
+
=== RUN TestOutput_OnlyConfiguredToolsAreJudged
|
|
219
|
+
--- PASS: TestOutput_OnlyConfiguredToolsAreJudged (0.00s)
|
|
220
|
+
=== RUN TestOutput_EmptyOrNonTextResultIsNotJudged
|
|
221
|
+
--- PASS: TestOutput_EmptyOrNonTextResultIsNotJudged (0.00s)
|
|
222
|
+
=== RUN TestOutput_JoinsTextBlocksWithNewline
|
|
223
|
+
output_test.go:119: the extension made no request
|
|
224
|
+
--- FAIL: TestOutput_JoinsTextBlocksWithNewline (0.00s)
|
|
225
|
+
=== RUN TestOutput_RequestQuestions
|
|
226
|
+
output_test.go:128: the extension made no request
|
|
227
|
+
--- FAIL: TestOutput_RequestQuestions (0.00s)
|
|
228
|
+
=== RUN TestOutput_LongOutputIsElided
|
|
229
|
+
output_test.go:147: the extension made no request
|
|
230
|
+
--- FAIL: TestOutput_LongOutputIsElided (0.00s)
|
|
231
|
+
=== RUN TestOutput_ArgumentsAreElidedAt400
|
|
232
|
+
output_test.go:155: the extension made no request
|
|
233
|
+
--- FAIL: TestOutput_ArgumentsAreElidedAt400 (0.00s)
|
|
234
|
+
=== RUN TestOutput_IdenticalOutputIsJudgedOnce
|
|
235
|
+
output_test.go:166: requests = 0, want 1
|
|
236
|
+
--- FAIL: TestOutput_IdenticalOutputIsJudgedOnce (0.00s)
|
|
237
|
+
=== RUN TestOutput_DisabledSkips
|
|
238
|
+
--- PASS: TestOutput_DisabledSkips (0.00s)
|
|
239
|
+
=== RUN TestOutput_LeakThresholdConfigurable
|
|
240
|
+
output_test.go:185: 0.5 < configured 0.4 threshold not flagged
|
|
241
|
+
--- FAIL: TestOutput_LeakThresholdConfigurable (0.00s)
|
|
242
|
+
=== RUN TestNoHardcodedProviderOrEndpoint
|
|
243
|
+
--- PASS: TestNoHardcodedProviderOrEndpoint (0.00s)
|
|
244
|
+
=== RUN TestOnlyPublicSDKImports
|
|
245
|
+
--- PASS: TestOnlyPublicSDKImports (0.00s)
|
|
246
|
+
=== RUN TestOwnModelBackend_PENDING_CONTRACT
|
|
247
|
+
source_test.go:56: own-model backend: waiting for the components/typesafe CONTRACT commit (lane pigpen-typesafe-client); the gate must work through it
|
|
248
|
+
--- SKIP: TestOwnModelBackend_PENDING_CONTRACT (0.00s)
|
|
249
|
+
=== RUN TestUserRequestJoinsTextBlocksWithNewline_GAP
|
|
250
|
+
source_test.go:60: upstream joins a user message's text blocks with \n (index.ts messageText); the Go SDK's BranchEntry flattens them with no separator; a prompt sent over RPC is one block, so no scenario differs
|
|
251
|
+
--- SKIP: TestUserRequestJoinsTextBlocksWithNewline_GAP (0.00s)
|
|
252
|
+
=== RUN TestOptIn_NothingIsJudgedByDefault
|
|
253
|
+
--- PASS: TestOptIn_NothingIsJudgedByDefault (0.00s)
|
|
254
|
+
=== RUN TestOptIn_NoConfigAtAllIsInert
|
|
255
|
+
--- PASS: TestOptIn_NoConfigAtAllIsInert (0.00s)
|
|
256
|
+
=== RUN TestOptIn_ProjectFileCannotEnable
|
|
257
|
+
trust_test.go:59: no warning for the ignored key: []
|
|
258
|
+
--- FAIL: TestOptIn_ProjectFileCannotEnable (0.00s)
|
|
259
|
+
=== RUN TestOptIn_GlobalConfigEnables
|
|
260
|
+
trust_test.go:71: requests = 0
|
|
261
|
+
--- FAIL: TestOptIn_GlobalConfigEnables (0.00s)
|
|
262
|
+
=== RUN TestDisclosure_ShownAtStartUntilAcknowledged
|
|
263
|
+
trust_test.go:88: no disclosure: []
|
|
264
|
+
--- FAIL: TestDisclosure_ShownAtStartUntilAcknowledged (0.00s)
|
|
265
|
+
=== RUN TestDisclosure_AcknowledgedIsSilent
|
|
266
|
+
--- PASS: TestDisclosure_AcknowledgedIsSilent (0.00s)
|
|
267
|
+
=== RUN TestDisclosure_ModelBackendNamesTheSessionModel
|
|
268
|
+
trust_test.go:113: disclosure does not name the model that receives the content: []
|
|
269
|
+
--- FAIL: TestDisclosure_ModelBackendNamesTheSessionModel (0.00s)
|
|
270
|
+
=== RUN TestDisclosure_ModelBackendDoesNotMentionTypeSafe
|
|
271
|
+
--- PASS: TestDisclosure_ModelBackendDoesNotMentionTypeSafe (0.00s)
|
|
272
|
+
=== RUN TestCorrection_ProjectFileCannotRedirectTheEndpoint
|
|
273
|
+
trust_test.go:143: trusted endpoint requests = 0
|
|
274
|
+
--- FAIL: TestCorrection_ProjectFileCannotRedirectTheEndpoint (0.00s)
|
|
275
|
+
=== RUN TestCorrection_NoDefaultEndpointForTheHTTPBackend
|
|
276
|
+
trust_test.go:160: no explanation for the missing endpoint: []
|
|
277
|
+
--- FAIL: TestCorrection_NoDefaultEndpointForTheHTTPBackend (0.00s)
|
|
278
|
+
=== RUN TestCorrection_PlainHTTPToARemoteHostIsRefused
|
|
279
|
+
trust_test.go:172: an http:// endpoint on a remote host must be refused: []
|
|
280
|
+
--- FAIL: TestCorrection_PlainHTTPToARemoteHostIsRefused (0.00s)
|
|
281
|
+
=== RUN TestCorrection_ProjectCanOnlyNarrowWhatLeaves
|
|
282
|
+
trust_test.go:196: the extension made no request
|
|
283
|
+
--- FAIL: TestCorrection_ProjectCanOnlyNarrowWhatLeaves (0.00s)
|
|
284
|
+
=== RUN TestProject_CanNarrowAndTune
|
|
285
|
+
trust_test.go:214: the extension made no request
|
|
286
|
+
--- FAIL: TestProject_CanNarrowAndTune (0.00s)
|
|
287
|
+
=== RUN TestKey_EnvBeatsConfigBeatsFile
|
|
288
|
+
trust_test.go:249: env: "<no request>"
|
|
289
|
+
trust_test.go:252: inline: "<no request>"
|
|
290
|
+
trust_test.go:255: file: "<no request>"
|
|
291
|
+
--- FAIL: TestKey_EnvBeatsConfigBeatsFile (0.00s)
|
|
292
|
+
=== RUN TestKey_HomeIsExpandedInKeyFile
|
|
293
|
+
trust_test.go:270: requests = []
|
|
294
|
+
--- FAIL: TestKey_HomeIsExpandedInKeyFile (0.00s)
|
|
295
|
+
=== RUN TestKey_MissingKeyWarnsOnceAndStaysInactive
|
|
296
|
+
trust_test.go:287: missing-key warnings = 0, want 1: []
|
|
297
|
+
--- FAIL: TestKey_MissingKeyWarnsOnceAndStaysInactive (0.00s)
|
|
298
|
+
=== RUN TestConfig_UnreadableAndInvalidFilesWarn
|
|
299
|
+
trust_test.go:302: notifications = []
|
|
300
|
+
--- FAIL: TestConfig_UnreadableAndInvalidFilesWarn (0.00s)
|
|
301
|
+
=== RUN TestConfig_WarningsRedactTheKey
|
|
302
|
+
--- PASS: TestConfig_WarningsRedactTheKey (0.00s)
|
|
303
|
+
=== RUN TestConfig_InvalidValuesFallBackToDefaults
|
|
304
|
+
trust_test.go:330: status = "<none>"
|
|
305
|
+
trust_test.go:333: notifications = []
|
|
306
|
+
--- FAIL: TestConfig_InvalidValuesFallBackToDefaults (0.00s)
|
|
307
|
+
FAIL
|
|
308
|
+
FAIL github.com/MichaelKinsy/pigpen/jev 0.096s
|
|
309
|
+
FAIL
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "cache-identical-calls",
|
|
3
|
+
"description": "The same call twice: judged once (gate and output).",
|
|
4
|
+
"agentFiles": {
|
|
5
|
+
"pi-jev.json": "{\"enabled\": true, \"acknowledged\": true, \"display\": \"plain\", \"backend\": \"typesafe\", \"endpoint\": \"{{server:jev}}/v1/systemone\", \"model\": \"jev-eq\", \"retries\": 0, \"timeoutMs\": 5000}"
|
|
6
|
+
},
|
|
7
|
+
"env": {
|
|
8
|
+
"TYPESAFE_API_KEY": "tsk-eq-key-0123456789"
|
|
9
|
+
},
|
|
10
|
+
"servers": {
|
|
11
|
+
"jev": {
|
|
12
|
+
"recordHeaders": [
|
|
13
|
+
"authorization"
|
|
14
|
+
],
|
|
15
|
+
"routes": [
|
|
16
|
+
{
|
|
17
|
+
"method": "POST",
|
|
18
|
+
"path": "/v1/systemone",
|
|
19
|
+
"body": "{\"model\": \"jev-eq\", \"answers\": {\"destructive\": {\"type\": \"noul\", \"noul\": 0.03}, \"exfiltration\": {\"type\": \"noul\", \"noul\": 0.04}, \"beyond_scope\": {\"type\": \"noul\", \"noul\": 0.4}, \"impact\": {\"type\": \"score\", \"score\": 0.02, \"confidence\": 0.9, \"legend\": {\"0\": \"None, it only reads\", \"1\": \"Small\", \"2\": \"Large\", \"3\": \"Severe\"}, \"probabilities\": {\"0\": 0.05, \"1\": 0.1, \"2\": 0.15, \"3\": 0.7}}}, \"usage\": {\"input_tokens\": 100, \"output_tokens\": 4}}",
|
|
20
|
+
"headers": {
|
|
21
|
+
"Content-Type": "application/json"
|
|
22
|
+
},
|
|
23
|
+
"times": 1
|
|
24
|
+
},
|
|
25
|
+
{
|
|
26
|
+
"method": "POST",
|
|
27
|
+
"path": "/v1/systemone",
|
|
28
|
+
"body": "{\"model\": \"jev-eq\", \"answers\": {\"leaks_secret\": {\"type\": \"noul\", \"noul\": 0.01}, \"failure_class\": {\"type\": \"choice\", \"choice\": \"no_failure\", \"confidence\": 0.99, \"probabilities\": {\"no_failure\": 0.99}}}}",
|
|
29
|
+
"headers": {
|
|
30
|
+
"Content-Type": "application/json"
|
|
31
|
+
},
|
|
32
|
+
"times": 1
|
|
33
|
+
}
|
|
34
|
+
]
|
|
35
|
+
}
|
|
36
|
+
},
|
|
37
|
+
"llm": [
|
|
38
|
+
{
|
|
39
|
+
"toolCalls": [
|
|
40
|
+
{
|
|
41
|
+
"name": "bash",
|
|
42
|
+
"arguments": {
|
|
43
|
+
"command": "echo same"
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
]
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
"toolCalls": [
|
|
50
|
+
{
|
|
51
|
+
"name": "bash",
|
|
52
|
+
"arguments": {
|
|
53
|
+
"command": "echo same"
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
]
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
"text": "done"
|
|
60
|
+
}
|
|
61
|
+
],
|
|
62
|
+
"steps": [
|
|
63
|
+
{
|
|
64
|
+
"name": "prompt",
|
|
65
|
+
"rpc": {
|
|
66
|
+
"type": "prompt",
|
|
67
|
+
"message": "run it twice"
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
]
|
|
71
|
+
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "clear",
|
|
3
|
+
"description": "An ordinary bash call: clear status, no notification, no notice on the output.",
|
|
4
|
+
"agentFiles": {
|
|
5
|
+
"pi-jev.json": "{\"enabled\": true, \"acknowledged\": true, \"display\": \"plain\", \"backend\": \"typesafe\", \"endpoint\": \"{{server:jev}}/v1/systemone\", \"model\": \"jev-eq\", \"retries\": 0, \"timeoutMs\": 5000}"
|
|
6
|
+
},
|
|
7
|
+
"env": {
|
|
8
|
+
"TYPESAFE_API_KEY": "tsk-eq-key-0123456789"
|
|
9
|
+
},
|
|
10
|
+
"servers": {
|
|
11
|
+
"jev": {
|
|
12
|
+
"recordHeaders": [
|
|
13
|
+
"authorization"
|
|
14
|
+
],
|
|
15
|
+
"routes": [
|
|
16
|
+
{
|
|
17
|
+
"method": "POST",
|
|
18
|
+
"path": "/v1/systemone",
|
|
19
|
+
"body": "{\"model\": \"jev-eq\", \"answers\": {\"destructive\": {\"type\": \"noul\", \"noul\": 0.03}, \"exfiltration\": {\"type\": \"noul\", \"noul\": 0.04}, \"beyond_scope\": {\"type\": \"noul\", \"noul\": 0.4}, \"impact\": {\"type\": \"score\", \"score\": 0.02, \"confidence\": 0.9, \"legend\": {\"0\": \"None, it only reads\", \"1\": \"Small\", \"2\": \"Large\", \"3\": \"Severe\"}, \"probabilities\": {\"0\": 0.05, \"1\": 0.1, \"2\": 0.15, \"3\": 0.7}}}, \"usage\": {\"input_tokens\": 100, \"output_tokens\": 4}}",
|
|
20
|
+
"headers": {
|
|
21
|
+
"Content-Type": "application/json"
|
|
22
|
+
},
|
|
23
|
+
"times": 1
|
|
24
|
+
},
|
|
25
|
+
{
|
|
26
|
+
"method": "POST",
|
|
27
|
+
"path": "/v1/systemone",
|
|
28
|
+
"body": "{\"model\": \"jev-eq\", \"answers\": {\"leaks_secret\": {\"type\": \"noul\", \"noul\": 0.01}, \"failure_class\": {\"type\": \"choice\", \"choice\": \"no_failure\", \"confidence\": 0.99, \"probabilities\": {\"no_failure\": 0.99}}}}",
|
|
29
|
+
"headers": {
|
|
30
|
+
"Content-Type": "application/json"
|
|
31
|
+
},
|
|
32
|
+
"times": 1
|
|
33
|
+
}
|
|
34
|
+
]
|
|
35
|
+
}
|
|
36
|
+
},
|
|
37
|
+
"llm": [
|
|
38
|
+
{
|
|
39
|
+
"toolCalls": [
|
|
40
|
+
{
|
|
41
|
+
"name": "bash",
|
|
42
|
+
"arguments": {
|
|
43
|
+
"command": "echo hi"
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
]
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
"text": "done"
|
|
50
|
+
}
|
|
51
|
+
],
|
|
52
|
+
"steps": [
|
|
53
|
+
{
|
|
54
|
+
"name": "prompt",
|
|
55
|
+
"rpc": {
|
|
56
|
+
"type": "prompt",
|
|
57
|
+
"message": "please run echo hi"
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
]
|
|
61
|
+
}
|