jev-agent-tools 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +37 -1
- package/CONTRIBUTING.md +3 -0
- package/README.md +22 -14
- package/SECURITY.md +17 -1
- package/dist/adapters/ask-files.js +11 -2
- package/dist/adapters/ask-proof.js +63 -7
- package/dist/adapters/command.js +82 -29
- package/dist/adapters/docs.js +30 -10
- package/dist/adapters/evidence-context.js +119 -0
- package/dist/adapters/files.js +141 -16
- package/dist/adapters/find.js +34 -6
- package/dist/adapters/git-base.js +7 -1
- package/dist/adapters/git.js +51 -7
- package/dist/adapters/locate-file.js +47 -9
- package/dist/adapters/private-storage.js +14 -6
- package/dist/adapters/risk-callers.js +3 -0
- package/dist/adapters/shell.js +23 -7
- package/dist/adapters/test-inventory.js +10 -2
- package/dist/configuration.js +17 -7
- package/dist/constants.js +26 -5
- package/dist/core/ask-references.js +193 -109
- package/dist/core/asks.js +78 -7
- package/dist/core/locate.js +8 -8
- package/dist/core/output.js +17 -0
- package/dist/core/result-report.js +302 -0
- package/dist/core/secret-path.js +34 -0
- package/dist/core/state.js +8 -1
- package/dist/core/units.js +1 -1
- package/dist/jev/client.js +34 -12
- package/dist/mcp/protocol.js +50 -27
- package/dist/mcp/tools.js +20 -7
- package/dist/render.js +72 -0
- package/dist/report-schema.js +1356 -0
- package/dist/result-types.js +1 -0
- package/dist/texts/ask-files.js +3 -1
- package/dist/texts/ask.js +3 -1
- package/dist/texts/check-diff.js +7 -4
- package/dist/texts/find.js +7 -2
- package/dist/texts/guide.js +3 -16
- package/dist/texts/instructions.js +72 -0
- package/dist/texts/locate.js +7 -2
- package/dist/texts/select-tests.js +3 -1
- package/dist/tools/ask-files.js +248 -15
- package/dist/tools/ask.js +523 -62
- package/dist/tools/check-diff.js +222 -30
- package/dist/tools/docs-check.js +122 -13
- package/dist/tools/find.js +320 -27
- package/dist/tools/locate.js +317 -18
- package/dist/tools/review-report.js +230 -0
- package/dist/tools/select-tests.js +273 -19
- package/dist/tools/spec-check.js +119 -22
- package/docs/adr/0001-strict-typescript-pure-core-offline-tests.md +3 -3
- package/docs/agent-instructions.md +59 -30
- package/docs/design.md +13 -1
- package/docs/mcp.md +8 -6
- package/docs/tools/jev_ask.md +8 -5
- package/docs/tools/jev_ask_files.md +2 -1
- package/docs/tools/jev_check_diff.md +4 -1
- package/docs/tools/jev_find_files.md +2 -1
- package/docs/tools/jev_locate_in_file.md +5 -0
- package/docs/tools/jev_select_tests.md +4 -1
- package/package.json +1 -1
- package/rules/jev-ask.md +22 -1
- package/server.json +2 -2
- package/src/adapters/ask-files.ts +11 -3
- package/src/adapters/ask-proof.ts +69 -11
- package/src/adapters/command.ts +96 -33
- package/src/adapters/docs.ts +33 -14
- package/src/adapters/evidence-context.ts +169 -0
- package/src/adapters/files.ts +146 -16
- package/src/adapters/find.ts +37 -7
- package/src/adapters/git-base.ts +7 -1
- package/src/adapters/git.ts +61 -8
- package/src/adapters/locate-file.ts +51 -9
- package/src/adapters/private-storage.ts +17 -5
- package/src/adapters/risk-callers.ts +3 -0
- package/src/adapters/shell.ts +23 -7
- package/src/adapters/test-inventory.ts +12 -4
- package/src/configuration.ts +16 -2
- package/src/constants.ts +26 -5
- package/src/core/ask-references.ts +262 -146
- package/src/core/asks.ts +79 -7
- package/src/core/import-boundaries.ts +8 -3
- package/src/core/locate.ts +8 -5
- package/src/core/output.ts +34 -0
- package/src/core/result-report.ts +410 -0
- package/src/core/secret-path.ts +37 -0
- package/src/core/state.ts +8 -1
- package/src/core/units.ts +3 -2
- package/src/index.ts +3 -0
- package/src/jev/client.ts +54 -16
- package/src/jev/types.ts +18 -3
- package/src/mcp/protocol.ts +91 -41
- package/src/mcp/tools.ts +26 -13
- package/src/render.ts +109 -0
- package/src/report-schema.ts +1380 -0
- package/src/result-types.ts +234 -0
- package/src/result.ts +4 -1
- package/src/runtime.ts +6 -0
- package/src/texts/ask-files.ts +4 -1
- package/src/texts/ask.ts +8 -1
- package/src/texts/check-diff.ts +7 -4
- package/src/texts/find.ts +8 -2
- package/src/texts/guide.ts +8 -16
- package/src/texts/instructions.ts +98 -0
- package/src/texts/locate.ts +8 -2
- package/src/texts/run-end.ts +2 -2
- package/src/texts/select-tests.ts +4 -1
- package/src/tools/ask-files.ts +309 -14
- package/src/tools/ask.ts +700 -77
- package/src/tools/check-diff.ts +331 -28
- package/src/tools/docs-check.ts +241 -39
- package/src/tools/find.ts +386 -29
- package/src/tools/locate.ts +384 -19
- package/src/tools/review-report.ts +308 -0
- package/src/tools/select-tests.ts +479 -21
- package/src/tools/spec-check.ts +193 -19
package/dist/tools/spec-check.js
CHANGED
|
@@ -1,12 +1,27 @@
|
|
|
1
|
+
import { resolveEvidenceContext, withEvidenceContext, } from "../adapters/evidence-context.js";
|
|
1
2
|
import { collectFiles } from "../adapters/files.js";
|
|
2
3
|
import { collectUnits } from "../adapters/git.js";
|
|
3
4
|
import { resolveBase } from "../adapters/git-base.js";
|
|
4
5
|
import { CHOICE_MAX_OPTIONS, STATE_MAX_CHARS, TIMEOUT_MS, } from "../constants.js";
|
|
5
6
|
import { buildEnvelope, } from "../core/output.js";
|
|
7
|
+
import { known, } from "../core/result-report.js";
|
|
6
8
|
import { prepareSpecCheck, readSpecJudgment, } from "../presets/spec.js";
|
|
7
9
|
import { NOT_CONFIGURED } from "../texts/configuration.js";
|
|
10
|
+
import { ReviewReport, reportMetrics } from "./review-report.js";
|
|
8
11
|
export async function runSpecCheck(deps, input) {
|
|
9
12
|
const started = performance.now();
|
|
13
|
+
const report = new ReviewReport();
|
|
14
|
+
const admission = input.evidenceContext
|
|
15
|
+
? undefined
|
|
16
|
+
: await resolveEvidenceContext(input.cwd, undefined, {
|
|
17
|
+
exec: deps.exec,
|
|
18
|
+
signal: input.signal,
|
|
19
|
+
origin: deps.evidenceOrigin,
|
|
20
|
+
});
|
|
21
|
+
const evidenceContext = input.evidenceContext ?? admission?.context;
|
|
22
|
+
if (!evidenceContext)
|
|
23
|
+
throw new Error("Evidence context was not established");
|
|
24
|
+
evidenceContext.requestedBase = input.base ?? "HEAD";
|
|
10
25
|
const judgments = [];
|
|
11
26
|
const findings = [];
|
|
12
27
|
const limitations = [];
|
|
@@ -15,7 +30,7 @@ export async function runSpecCheck(deps, input) {
|
|
|
15
30
|
let sent = 0;
|
|
16
31
|
let incomplete = false;
|
|
17
32
|
let emptyBase;
|
|
18
|
-
const finish = (refusal) => {
|
|
33
|
+
const finish = (refusal, cause = "internal_error") => {
|
|
19
34
|
const answers = findings.map((finding) => ({
|
|
20
35
|
label: finding.kind === "requirement"
|
|
21
36
|
? `${input.specPath}:${finding.requirement?.start}-${finding.requirement?.end} ${finding.label} — requirement violated`
|
|
@@ -47,7 +62,9 @@ export async function runSpecCheck(deps, input) {
|
|
|
47
62
|
!budget &&
|
|
48
63
|
!unchecked.length &&
|
|
49
64
|
!findings.length &&
|
|
50
|
-
emptyBase === undefined
|
|
65
|
+
emptyBase === undefined &&
|
|
66
|
+
report.items.size > 0 &&
|
|
67
|
+
[...report.items.values()].every((item) => item.treatment === "judged")
|
|
51
68
|
? {
|
|
52
69
|
lines: [
|
|
53
70
|
{
|
|
@@ -61,55 +78,97 @@ export async function runSpecCheck(deps, input) {
|
|
|
61
78
|
yield: {
|
|
62
79
|
calls: judgments.reduce((n, j) => n + (j.calls ?? 0), 0),
|
|
63
80
|
questions: judgments.reduce((n, j) => n + (j.questions ?? 0), 0),
|
|
64
|
-
costUsd: judgments.
|
|
81
|
+
costUsd: judgments.length && judgments.every((j) => j.usage !== undefined)
|
|
82
|
+
? judgments.reduce((n, j) => n + (j.usage?.costUsd ?? 0), 0)
|
|
83
|
+
: undefined,
|
|
65
84
|
cacheHits: judgments.reduce((n, j) => n + (j.cacheHits ?? 0), 0),
|
|
66
85
|
cacheRequests: judgments.reduce((n, j) => n + (j.cacheRequests ?? 0), 0),
|
|
67
86
|
elapsedMs: performance.now() - started,
|
|
68
87
|
},
|
|
69
88
|
});
|
|
70
|
-
|
|
89
|
+
if (emptyBase !== undefined)
|
|
90
|
+
report.diagnose("no_changed_units", `No changed units against ${emptyBase}`, emptyBase, [], false);
|
|
91
|
+
if (refusal) {
|
|
92
|
+
report.refusal = cause !== "not_configured";
|
|
93
|
+
report.diagnose(cause, refusal, input.specPath);
|
|
94
|
+
}
|
|
95
|
+
const result = report.build("jev_check_diff", evidenceContext, reportMetrics(envelope));
|
|
96
|
+
return { ok: !refusal, envelope, judgments, findings, result };
|
|
71
97
|
};
|
|
72
98
|
if (!input.specPath)
|
|
73
|
-
return finish("spec_path required: no specification to check against.");
|
|
99
|
+
return finish("spec_path required: no specification to check against.", "missing_required");
|
|
74
100
|
if (!deps.client)
|
|
75
|
-
return finish(NOT_CONFIGURED);
|
|
101
|
+
return finish(NOT_CONFIGURED, "not_configured");
|
|
76
102
|
const comparison = await resolveBase(deps.exec, input.cwd, input.base, input.signal);
|
|
77
103
|
if (!comparison.ok)
|
|
78
|
-
return finish(comparison.error);
|
|
104
|
+
return finish(comparison.error, comparison.cause ?? "invalid_base");
|
|
79
105
|
const base = comparison.base;
|
|
106
|
+
evidenceContext.resolvedBase = base;
|
|
80
107
|
const root = await deps.exec("git", ["rev-parse", "--show-toplevel"], {
|
|
81
108
|
cwd: input.cwd,
|
|
82
109
|
timeout: TIMEOUT_MS,
|
|
83
110
|
signal: input.signal,
|
|
84
111
|
});
|
|
85
112
|
if (root.code || root.killed)
|
|
86
|
-
return finish("Repository root not found.");
|
|
113
|
+
return finish("Repository root not found.", root.killed ? "cancelled" : "git_failure");
|
|
87
114
|
const cwd = root.stdout.trim();
|
|
115
|
+
if (evidenceContext.effectiveRoot)
|
|
116
|
+
evidenceContext.effectiveRoot.path = cwd;
|
|
88
117
|
const [collected, specification] = await Promise.all([
|
|
89
118
|
collectUnits(deps.exec, { cwd, base, signal: input.signal }),
|
|
90
119
|
collectFiles(cwd, [input.specPath], input.signal, { exec: deps.exec }),
|
|
91
120
|
]);
|
|
92
121
|
if (!collected.ok)
|
|
93
|
-
return finish(collected.error);
|
|
122
|
+
return finish(collected.error, collected.cause ?? "git_failure");
|
|
94
123
|
if (!specification.ok)
|
|
95
|
-
return finish(specification.error);
|
|
124
|
+
return finish(specification.error, specification.cause ?? "file_unavailable");
|
|
96
125
|
const text = Object.values(specification.files)[0];
|
|
97
126
|
if (!text)
|
|
98
|
-
return finish("The specification is empty: nothing to check.");
|
|
127
|
+
return finish("The specification is empty: nothing to check.", "empty_required");
|
|
99
128
|
for (const limit of collected.limits)
|
|
100
129
|
limitations.push({
|
|
101
130
|
fact: `${limit.file} : ${limit.kind}`,
|
|
102
131
|
next: "Read the complete change before concluding.",
|
|
103
132
|
});
|
|
104
|
-
const
|
|
105
|
-
|
|
106
|
-
|
|
133
|
+
const unavailableUnits = collected.units.filter((unit) => unit.before === null && unit.after === null);
|
|
134
|
+
const units = collected.units.filter((unit) => unit.before !== null || unit.after !== null);
|
|
135
|
+
for (const unit of unavailableUnits)
|
|
107
136
|
unchecked.push(`${unit.file} ${unit.name} (changed source unavailable)`);
|
|
108
|
-
return false;
|
|
109
|
-
});
|
|
110
137
|
const prepared = prepareSpecCheck(text, units);
|
|
138
|
+
for (const requirement of prepared.requirements)
|
|
139
|
+
report.expect(`spec:${requirement.id}`, requirement.label, "requirement");
|
|
140
|
+
report.expect("spec:drift", "Specification drift pointer", "pointer");
|
|
141
|
+
report.inventories.push({
|
|
142
|
+
id: "spec-requirements",
|
|
143
|
+
kind: "sections",
|
|
144
|
+
rules: ["### REQ- headings in the supplied specification"],
|
|
145
|
+
restrictions: [input.specPath],
|
|
146
|
+
discovered: known(prepared.requirements.length),
|
|
147
|
+
considered: known(prepared.requirements.length),
|
|
148
|
+
scopeRestricted: true,
|
|
149
|
+
criteria: [],
|
|
150
|
+
});
|
|
151
|
+
report.inventories.push({
|
|
152
|
+
id: "changed-units",
|
|
153
|
+
kind: "units",
|
|
154
|
+
rules: ["Changed source units against resolved base"],
|
|
155
|
+
restrictions: [],
|
|
156
|
+
discovered: known(collected.units.length),
|
|
157
|
+
considered: known(units.length),
|
|
158
|
+
scopeRestricted: false,
|
|
159
|
+
criteria: [],
|
|
160
|
+
});
|
|
161
|
+
if (!collected.units.length && prepared.requirements.length)
|
|
162
|
+
report.items.clear();
|
|
163
|
+
const reportIds = [...report.items.keys()];
|
|
164
|
+
for (const unit of unavailableUnits)
|
|
165
|
+
report.diagnose("binary_or_non_utf8", "Changed source unavailable", unit.file, reportIds, units.length === 0, [`${unit.file} ${unit.name}`]);
|
|
166
|
+
for (const limit of collected.limits)
|
|
167
|
+
report.diagnose(limit.kind === "secret_pattern" ? "secret_pattern" : "collection_omitted", limit.kind, limit.file, reportIds, units.length === 0, ["unit" in limit ? `${limit.file} ${limit.unit}` : limit.file]);
|
|
168
|
+
report.missingWork =
|
|
169
|
+
unavailableUnits.length > 0 || collected.limits.length > 0;
|
|
111
170
|
if (!prepared.requirements.length)
|
|
112
|
-
return finish("No ### REQ-… requirements in the specification: nothing to check.");
|
|
171
|
+
return finish("No ### REQ-… requirements in the specification: nothing to check.", "missing_required");
|
|
113
172
|
if (!collected.units.length) {
|
|
114
173
|
emptyBase = base;
|
|
115
174
|
return finish();
|
|
@@ -117,20 +176,24 @@ export async function runSpecCheck(deps, input) {
|
|
|
117
176
|
if (unchecked.length) {
|
|
118
177
|
unchecked.push(...prepared.requirements.map((requirement) => `${requirement.label} (changed source unavailable)`), "drift (changed source unavailable)");
|
|
119
178
|
}
|
|
120
|
-
if (prepared.tableWarning)
|
|
179
|
+
if (prepared.tableWarning) {
|
|
180
|
+
report.diagnose("unsupported_syntax", "Specification Markdown table interpretation is uncalibrated", input.specPath, [], false);
|
|
121
181
|
limitations.push({
|
|
122
182
|
fact: "The specification contains a Markdown table.",
|
|
123
183
|
next: "Read the table requirements: their interpretation is uncalibrated.",
|
|
124
184
|
});
|
|
125
|
-
|
|
185
|
+
}
|
|
186
|
+
incomplete = report.missingWork;
|
|
187
|
+
const state = withEvidenceContext(prepared.state, evidenceContext);
|
|
126
188
|
if (units.length + 1 > CHOICE_MAX_OPTIONS ||
|
|
127
|
-
JSON.stringify(
|
|
128
|
-
return finish(`State or spec pointer exceeds the limits; compare against a closer base (STATE_MAX_CHARS=${STATE_MAX_CHARS})
|
|
189
|
+
JSON.stringify(state).length > STATE_MAX_CHARS)
|
|
190
|
+
return finish(`State or spec pointer exceeds the limits; compare against a closer base (STATE_MAX_CHARS=${STATE_MAX_CHARS}).`, "evidence_too_large");
|
|
129
191
|
if (!units.length)
|
|
130
192
|
return finish();
|
|
131
|
-
const judgment = await deps.client.judge(
|
|
193
|
+
const judgment = await deps.client.judge(state, prepared.questions, {
|
|
132
194
|
signal: input.signal,
|
|
133
195
|
...deps.runtime.session.requestGate(),
|
|
196
|
+
admissionCause: () => budget?.kind === "session" ? "session_budget" : "call_budget",
|
|
134
197
|
beforeRequest(questionCount) {
|
|
135
198
|
if (input.maxCalls !== undefined && sent >= input.maxCalls) {
|
|
136
199
|
budget = {
|
|
@@ -149,6 +212,40 @@ export async function runSpecCheck(deps, input) {
|
|
|
149
212
|
onUsage: (usage) => deps.runtime.session.recordUsage(usage),
|
|
150
213
|
});
|
|
151
214
|
judgments.push(judgment);
|
|
215
|
+
if (!judgment.ok)
|
|
216
|
+
report.failure(judgment, reportIds, budget
|
|
217
|
+
? budget.kind === "session"
|
|
218
|
+
? "session_budget"
|
|
219
|
+
: "call_budget"
|
|
220
|
+
: undefined);
|
|
221
|
+
else {
|
|
222
|
+
for (const requirement of prepared.requirements)
|
|
223
|
+
report.answer(`spec:${requirement.id}`, judgment.answers[requirement.id], { band: incomplete ? "unsure" : "verdict" });
|
|
224
|
+
const driftAnswer = judgment.answers.drift;
|
|
225
|
+
const selectedUnit = driftAnswer?.type === "choice"
|
|
226
|
+
? units.find((unit) => unit.id === driftAnswer.choice)
|
|
227
|
+
: undefined;
|
|
228
|
+
const pointer = report.items.get("spec:drift");
|
|
229
|
+
if (pointer && selectedUnit)
|
|
230
|
+
report.items.set("spec:drift", {
|
|
231
|
+
...pointer,
|
|
232
|
+
label: `${selectedUnit.file} ${selectedUnit.name} specification drift`,
|
|
233
|
+
});
|
|
234
|
+
const selectedProbability = driftAnswer?.type === "choice"
|
|
235
|
+
? driftAnswer.probabilities[driftAnswer.choice]
|
|
236
|
+
: undefined;
|
|
237
|
+
report.answer("spec:drift", driftAnswer, {
|
|
238
|
+
band: incomplete ? "unsure" : "verdict",
|
|
239
|
+
...(driftAnswer?.type === "choice" && selectedProbability !== undefined
|
|
240
|
+
? {
|
|
241
|
+
value: {
|
|
242
|
+
head: driftAnswer.choice,
|
|
243
|
+
p: selectedProbability,
|
|
244
|
+
},
|
|
245
|
+
}
|
|
246
|
+
: {}),
|
|
247
|
+
});
|
|
248
|
+
}
|
|
152
249
|
if (!judgment.ok) {
|
|
153
250
|
unchecked.push(...prepared.requirements.map((req) => `${req.label} (${judgment.error})`), `drift (${judgment.error})`);
|
|
154
251
|
return finish();
|
|
@@ -14,11 +14,11 @@ Keep limits and thresholds in src/constants.ts. Batch questions for the same evi
|
|
|
14
14
|
|
|
15
15
|
Enforce downward import layers, including erased type dependencies:
|
|
16
16
|
|
|
17
|
-
- core/ imports core/, constants.ts, result.ts and type-only contracts from jev/types.ts.
|
|
17
|
+
- core/ imports core/, constants.ts, result.ts, result-types.ts and type-only contracts from jev/types.ts.
|
|
18
18
|
- presets/ and adapters/ each import their own layer, core/ and those neutral modules; adapters/ does not import presets/ or tools/.
|
|
19
|
-
- jev/ imports its own layer, core/, constants.ts and result.ts.
|
|
19
|
+
- jev/ imports its own layer, core/, constants.ts, result.ts and result-types.ts.
|
|
20
20
|
- texts/ imports its own layer, constants.ts and presets/; it has no external dependencies.
|
|
21
|
-
- constants.ts and result.ts import nothing. Root integration modules and tools/ compose the layers.
|
|
21
|
+
- constants.ts and result-types.ts import nothing. result.ts may import the dependency-free result-types.ts contract. Root integration modules and tools/ compose the layers.
|
|
22
22
|
|
|
23
23
|
Reject all cycles, including type-only cycles, and nonliteral module loading. Shared contracts belong below their consumers. Pure layers do not import external modules, with explicit core exceptions for pure node:path functions and erased import type contracts from @ast-grep/napi. Inline type specifiers that retain runtime loading do not qualify. Filesystem, process and network modules, direct fetch calls and Node builtin-module loaders are forbidden in the core; import checks are not an exhaustive proof against indirect global effects.
|
|
24
24
|
|
|
@@ -1,54 +1,77 @@
|
|
|
1
1
|
# Agent instructions for MCP clients
|
|
2
2
|
|
|
3
|
-
pi and
|
|
3
|
+
pi and OMP deliver the reading guide and discretionary Jev policy automatically. The MCP server sends the same semantic policy as its `instructions`, with MCP's explicit absence of an automatic docs hook, but clients may ignore that field. Add the generated block below to the project's instruction file so the agent can choose useful evidence tools and report their results accurately.
|
|
4
4
|
|
|
5
|
-
Copy
|
|
5
|
+
Copy the block unchanged, or trim the optional tool table to the tools you enable. It contains no secrets and is safe to commit. Copies are not updated automatically: replace them from this page when upgrading a release.
|
|
6
6
|
|
|
7
7
|
## The block
|
|
8
8
|
|
|
9
|
+
<!-- BEGIN GENERATED JEV INSTRUCTIONS -->
|
|
10
|
+
|
|
9
11
|
```markdown
|
|
10
12
|
## Jev evidence tools (jev_* via MCP)
|
|
11
13
|
|
|
12
|
-
|
|
14
|
+
<!-- Generated policy 2026-10-03.1 from src/texts/instructions.ts. Update this copied block from docs/agent-instructions.md on each release; pasted copies are not updated automatically. -->
|
|
13
15
|
|
|
14
|
-
|
|
16
|
+
jev_* tools: choosing evidence and reading results.
|
|
15
17
|
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
18
|
+
Use Jev for a bounded semantic judgment when its answer could change an open decision or focus the next inspection, and the relevant evidence is available. Use decisive reading, search or authorized execution directly when it settles the question. Exact lookup, counting and runtime causality belong to native tools. A Jev call is not a prerequisite for a conclusion, a review or task completion; no explanation is needed for choosing native tools.
|
|
19
|
+
|
|
20
|
+
Before a chosen call, identify the open decision and supply only the evidence needed for a bounded question; a complete prior analysis is not required. Ask about one positive, self-contained observable fact per claim, with its sources, and show both sides of a comparison.
|
|
21
|
+
- Failure: relevant output, the failing test and the implementation it exercises; output alone is a lead, not a bug-versus-wrong-test diagnosis. Include configuration or runtime evidence when the hypothesis depends on it. A static judgment does not establish runtime causality.
|
|
22
|
+
- Change: current evidence and the earlier reference via base. Use the checkout/root corresponding to the work and admitted by the tool; another clone is not the same context.
|
|
23
|
+
- Plan/documentation: precise observable commitments and the passages that constrain them. Local confirmation does not establish that omitted obligations were searched.
|
|
24
|
+
- Selection/review: compatible inventory, references and configuration. Suggested commands are not executed commands; existing tests need not cover a new scenario. Read the diff natively for its contents, not as proof of safety or global coverage.
|
|
25
|
+
|
|
26
|
+
The report begins with execution state: complete means the requested admitted work was processed; partial means some requested work remains unjudged or limited; not_judged means no admissible judgment; refused means a call-wide refusal. These states describe execution, not correctness, safety, documentation completeness or global test coverage. Static selection or conservative fallback can be useful without a Jev judgment; inspect the selection reason and remaining limits.
|
|
27
|
+
|
|
28
|
+
Context identifies canonical host/server authority, requested and effective root, base and resolved reference, inventory restrictions and command execution/cwd when collected. Compare that context to your actual work before using a result. Unknown, not_collected and not_applicable are distinct, not guessed values or a substitute checkout.
|
|
29
|
+
|
|
30
|
+
Each item distinguishes a judged result sourced from fresh or cache, static treatment without a Jev judgment, or unjudged work. A cached judgment is not a fresh verification. Only judged items carry a judgment measure; unjudged and static items have no invented probability or confidence. Source and treatment are independent of the probability band.
|
|
31
|
+
|
|
32
|
+
Jev reads what you pass at a glance. A verdict is a lead to check before an irreversible action, not a proof. About 1 in 100 clear verdicts is wrong, and far more when the evidence is cut, from the wrong file, or depends on a file that was not passed; no mark can see that, so check the evidence yourself before you edit, delete or report done.
|
|
24
33
|
|
|
25
|
-
|
|
34
|
+
Marks:
|
|
35
|
+
- unsure: Jev does not see it clearly (yes/no between 0.20 and 0.80; a category or level under 0.85 confidence; findings between their thresholds), or a control of the call failed and the line says which. Inspect the passage or obtain the decisive file; unchanged rewording does not settle it.
|
|
36
|
+
- abstain: a necessary piece is missing from what you passed; the line names it. Obtain that piece or leave the conclusion open. A further Jev call is optional when the changed evidence makes it useful.
|
|
37
|
+
- "no (not shown)" or "not addressed": the file or state does not show it. That is not "false".
|
|
38
|
+
- uncalibrated: no error rate has been measured for this kind of ask; read it as a hint.
|
|
39
|
+
- Diagnostics: typed cause, origin, affected scope and materiality explain limits or blocking work, with referenced next actions. An uncertain judged item and work never judged are different. Inspect the named evidence or action without treating a diagnostic as a probability or a clean bill of health.
|
|
26
40
|
|
|
27
|
-
|
|
41
|
+
Accounting separates tool invocation, HTTP attempts, questions actually sent, requested results (fresh/cache/not judged/static), cache probes (hits/requests), auxiliary controls and passage work, current reported cost and elapsed time. These counts are not interchangeable: an HTTP attempt is not a result, a cache probe is not a requested judgment, and a cached result does not imply a request. Unknown current cost means unreported, not zero or reconstructed historical cached cost. Use max_calls to bound a chosen wide ask.
|
|
28
42
|
|
|
29
|
-
|
|
30
|
-
- Ask about facts the files show, in positive sentences, with the evidence attached.
|
|
31
|
-
- Before reporting a change as done, call jev_check_diff with check "risk", then with check "docs" (MCP has no automatic documentation check). Update or justify every flagged sentence.
|
|
32
|
-
- Bound wide asks with max_calls.
|
|
43
|
+
After an unavailable, refused or out-of-scope result, continue natively within the existing permissions; do not widen sharing or confinement to obtain a judgment. For uncertainty or missing evidence, inspect or obtain the decisive piece, or leave the conclusion open. Revisit Jev only when new evidence, a material context change or a new useful question makes the judgment useful; rewording unchanged evidence is not a reason to retry. There is no retry quota for genuinely changed evidence, and no certification is needed once decisive evidence settles the question. A reached cap or an unjudged result is not evidence of safety or zero affected tests.
|
|
33
44
|
|
|
34
|
-
|
|
45
|
+
Report current conclusions first, then decisive evidence, origin and scope, then material reservations. Established means supported by relevant decisive evidence; a reported check remains explicitly reported, not observed execution. Distinguish native reading/execution, Jev's static judgment and testimony. A call count or global status is not proof.
|
|
46
|
+
If native evidence settles the same scope and context after a Jev uncertainty, attribute the current conclusion to that evidence; old unsure or abstain results need not be recited as current reservations. A later Jev confirmation is attributed to Jev and retains its static limits. Independent reservations survive either resolution.
|
|
47
|
+
Where an uncertainty actually returned by Jev remains material, explicitly say “Jev did not confirm X”, identify the missing evidence or guarantee, its known or indeterminate impact and the evidence to obtain. Preserve that reservation in the main text, summary and recommendation. Name native reservations and unjudged work as such, not as Jev uncertainty. Unresolved contradiction, stale evidence or a different checkout/base/scenario remains a reservation; the latest favorable answer does not win by default.
|
|
48
|
+
Keep material limits in the main text, not only behind a link. Do not generalize a checked scenario into a universal guarantee. Separate relevant unestablished hypotheses from observations; omit irrelevant hypotheses. Explain unavailable historical causes only when material to action or requested, without inventing causality.
|
|
49
|
+
Use existing history references when useful or requested; no exhaustive historical relay, persistent register or History block is required. A requested audit details available steps separately from the current state and does not reopen settled conclusions. Preserve existing traces; identify unavailable traces when they limit evidence or audit, without reconstructing evidence, executions or call counts. pi/OMP may reference an existing trace; MCP references must be client-accessible or explicitly unavailable.
|
|
35
50
|
|
|
36
|
-
|
|
37
|
-
- unsure: the answer is ambiguous or a control failed. Read the passage or file the line points to, or add the file that settles it. Rewording the question does not help.
|
|
38
|
-
- abstain: a necessary piece is missing; the line names it. Add that file or command and ask once.
|
|
39
|
-
- "no (not shown)" or "not addressed": the evidence does not show it. That is not "false".
|
|
40
|
-
- uncalibrated: no measured error rate; treat it as a hint.
|
|
41
|
-
- Lines in brackets: what limited the call and the next action or parameter to use.
|
|
51
|
+
Ask about facts the files show, in positive sentences, with the evidence attached; do counting and searching with your text search tool or code.
|
|
42
52
|
|
|
43
|
-
|
|
53
|
+
Evidence passed to jev_* tools leaves the machine for the configured endpoint. Review its data handling; share only authorized evidence, never secrets or credentials. Host/client approvals still apply. jev_ask commands run with normal shell permissions and no sandbox; prefer read-only commands. Tool choice does not relax quality, proof, approval or confidentiality obligations.
|
|
44
54
|
|
|
45
|
-
|
|
55
|
+
jev_check_diff: Review a stable diff for semantic risks, stale documentation or specification drift when that review can inform an open decision. Read the diff natively when you need its contents. Choose risk, docs or spec for the question at hand; neither risk followed by docs nor a Jev review before done is required. Findings are leads within the inspected scope, not proof of global safety or completeness.
|
|
46
56
|
|
|
47
|
-
|
|
57
|
+
MCP has no automatic run-end documentation hook. Its absence does not require manual replacement calls, including risk followed by docs. pi and OMP retain an opt-out host hook; that is an explicit automatic exception, not an agent call requirement.
|
|
48
58
|
|
|
49
|
-
|
|
59
|
+
### Optional tool choices
|
|
60
|
+
|
|
61
|
+
| Tool | Useful open decision |
|
|
62
|
+
|---|---|
|
|
63
|
+
| jev_ask | One bounded judgment combines a note, files, before/after evidence or command output. |
|
|
64
|
+
| jev_ask_files | The same questions apply independently to many candidate files. |
|
|
65
|
+
| jev_find_files | Find an entry point by behavior when the filename is unknown. |
|
|
66
|
+
| jev_locate_in_file | Find a useful range in one file over 19 KB. |
|
|
67
|
+
| jev_check_diff | Review a stable diff for the chosen risk, docs or spec question. |
|
|
68
|
+
| jev_select_tests | Suggest commands for existing affected tests when selection can change the execution plan; it never runs them. |
|
|
69
|
+
|
|
70
|
+
Use your text search tool for exact strings and known symbols, your file-name search tool for known filenames, and native reading or authorized execution for decisive evidence. Tool descriptions retain their full evidence recipes and limitations.
|
|
50
71
|
```
|
|
51
72
|
|
|
73
|
+
<!-- END GENERATED JEV INSTRUCTIONS -->
|
|
74
|
+
|
|
52
75
|
## Where to put it
|
|
53
76
|
|
|
54
77
|
| Client | File | Notes |
|
|
@@ -88,4 +111,10 @@ inclusion: always
|
|
|
88
111
|
|
|
89
112
|
## Keeping it current
|
|
90
113
|
|
|
91
|
-
The
|
|
114
|
+
The versioned canonical fragments and host renderers live in [`src/texts/instructions.ts`](../src/texts/instructions.ts). Runtime guides, descriptions/guidelines, the OMP rule and this MCP block reuse them; the MCP block includes policy version, source and copy-update instructions. Thresholds continue to come from [policy constants](design.md#policy-thresholds).
|
|
115
|
+
|
|
116
|
+
After changing the canonical policy, run `node scripts/generate-instructions.ts` to regenerate this block and `rules/jev-ask.md`. Run `node scripts/generate-instructions.ts --check` to reject stale generated outputs. During release verification, inspect delivery across pi/OMP/MCP for semantic parity, including the automatic pi/OMP hook versus no MCP hook; shared wording is not evidence of agent behavior. A release updates runtime/server instructions, not instruction blocks already pasted into external projects. Update those copies from this page; no synchronization with external copies is claimed.
|
|
117
|
+
|
|
118
|
+
## Automatic hook and evaluation costs
|
|
119
|
+
|
|
120
|
+
pi and OMP preserve the existing automatic run-end docs hook and its conditions/budget; `JEV_TOOLS_AUTO_DOCS=0` disables it. This is an explicit exception to agent-discretionary calls, not a requirement to invoke Jev, and does not certify complete documentation. MCP has no hook and no required manual replacement. In usage evaluation, record hook cost separately from discretionary calls and include both in full-task cost; no product price, budget setting or new runtime counter is introduced by this instruction policy.
|
package/docs/design.md
CHANGED
|
@@ -6,10 +6,18 @@ The tools construct bounded evidence before asking for judgment. They preserve u
|
|
|
6
6
|
|
|
7
7
|
Supply the discriminating evidence, not an argument about it. For comparisons, show both sides: a failing test and its implementation, or a file before and after. A bounded static import closure adds declarations before judgment where supported; it cannot establish completeness or discover relationships without imports. Commands are evidence sources, not a sandbox or a substitute for reading the output directly.
|
|
8
8
|
|
|
9
|
+
Every call captures its host/server authority independently. Optional `root` admits only an exact repository top-level or live registered worktree with the same Git common directory, before content, commands, cache or HTTP. No override preserves existing cwd behavior; an invalid override never falls back. The effective context, requested/resolved base and inventory limits travel with evidence and results. See [root admission](../README.md#evidence-root).
|
|
10
|
+
|
|
11
|
+
Explicit selectors define required evidence; lexical hints do not become vetoes merely because they resemble filenames. Exact directory paths cannot fall back to suffix matches. Canonical paths serialize each current/base version once, with aliases as metadata, and file admission counts identities rather than versions. Missing requirements gate their full question/control group; global-note requirements remain global. Captured output and repository files keep separate provenance.
|
|
12
|
+
|
|
13
|
+
Historical-only files pass the same protected-path, ignore, regular-file and text admission checks before their base content is disclosed. Current and base versions share the distinct-file ceiling; unresolved or over-budget selectors are not retried through raw historical reads.
|
|
14
|
+
|
|
9
15
|
## Pure core, thin adapters
|
|
10
16
|
|
|
11
17
|
Pure core transformations construct states, units, questions and display envelopes. Adapters own filesystem, Git, parsing and host effects; the HTTP client owns transport. Presets define fixed review questions and texts define host-facing guidance. Hosts sit on top: `src/index.ts` registers the tools with pi and omp, and `src/mcp/` serves the same tool factories to any MCP client over stdio ([ADR 0010](adr/0010-mcp-server-thin-host.md), [setup](mcp.md)). Dependency checks keep shared contracts below consumers, reject cycles and account for erased type imports. Pure path operations and erased parser types are explicit architectural exceptions. Both hosts use the same HTTP judgment protocol.
|
|
12
18
|
|
|
19
|
+
The dependency-free `src/result-types.ts` defines `ResultReportV1`; `core/result-report.ts` builds reports and validates semantic invariants, while `report-schema.ts` supplies the closed transport schema and structural validator. Producers record actual observations rather than reconstructing them from rendered prose. `renderResultReport` projects the same report to self-contained text, retaining native read/runner details. pi/omp publish `details.result`; supported MCP versions publish `structuredContent.result` without introducing a second judgment policy.
|
|
20
|
+
|
|
13
21
|
## Typed intents and fixed checks
|
|
14
22
|
|
|
15
23
|
Caller intents compile to typed questions with canonical options and exact statement text. Verification of one combined situation preserves the distinction between holds, contradicted, not addressed and cannot tell, with an exact-statement cross-check that cannot override missing evidence. Custom questions remain uncalibrated. Diff checks use reviewed questions over before/after evidence units; tests are evidence, not changed units to judge. A matrix asks boolean cells across units; a pointer chooses one candidate or none. Neither executes the code.
|
|
@@ -18,11 +26,15 @@ Caller intents compile to typed questions with canonical options and exact state
|
|
|
18
26
|
|
|
19
27
|
Leading-option probability (p_max) is the probability of the most likely choice, not the response's separate confidence field. Gray bands and applicable option-order checks expose ambiguity. Missing evidence remains abstention. Preset witnesses use a decoy and a known positive reference to detect a biased setup; failed controls do not promote findings. No mark establishes that unseen evidence is complete. See [result reading](../README.md#read-the-results).
|
|
20
28
|
|
|
29
|
+
Execution state and judgment band are independent: complete/partial/refused/not judged describe processing, while verdict/unsure/abstain describe admitted judgments. Every requested item is fresh, cache, static or unjudged; static/fallback decisions have no invented probability. Auxiliary controls and passage decisions do not inflate requested-result counts. Missing required controls leave the requested group unjudged. Accounting distinguishes HTTP attempts, questions sent, cache probes, current cost and elapsed time; unknown cost is never reconstructed as zero or historical cached cost. Scoped diagnostics preserve material omissions and useful conditional actions.
|
|
30
|
+
|
|
21
31
|
## Bounded work, visible limits
|
|
22
32
|
|
|
23
33
|
Admission limits, request planning, rate limiting and retry bounds are distinct. Oversized evidence is refused or visibly omitted, never silently converted into an ordinary verdict. Per-invocation max_calls and session call/cost limits are separate; control and severity requests count too. Under a session USD limit, requests are admitted one at a time so each admission sees all cost reported so far. Unjudged tests stay selected. Automatic documentation review is the only run-end automation and can request at most one extra turn.
|
|
24
34
|
|
|
25
|
-
Successful non-command judgments are cached only in session memory by canonical evidence, question and requested model string. Errors are not cached; command judgments bypass caching. The default openjev alias can move, and an echoed model name does not prove served-model identity. Fractional budgets bound admission, not an exact prediction of the final request's cost.
|
|
35
|
+
Successful non-command judgments are cached only in session memory by canonical evidence, question and requested model string. Every judgment input, including control and passage decisions, carries admitted authority/root provenance and the resolved base when applicable; that metadata participates in both cache identity and serialized evidence admission. Identical content from different roots or base revisions cannot reuse a judgment. Errors are not cached; command judgments bypass caching. The default openjev alias can move, and an echoed model name does not prove served-model identity. Fractional budgets bound admission, not an exact prediction of the final request's cost.
|
|
36
|
+
|
|
37
|
+
`src/texts/instructions.ts` is the versioned source for host policy, evidence guidance and current-state presentation. `node scripts/generate-instructions.ts` updates checked-in omp/MCP fragments; `--check` detects drift. Jev is discretionary when it can inform an open decision; native decisive evidence requires no certification call. Unchanged retries and mandatory risk/docs sequences are not recovery policy. pi/omp retain the opt-out documentation hook; MCP has no replacement obligation. Final presentation distinguishes native execution, Jev's static judgment and reported evidence, preserving independent material reservations without requiring an exhaustive history registry.
|
|
26
38
|
|
|
27
39
|
## Policy thresholds
|
|
28
40
|
|
package/docs/mcp.md
CHANGED
|
@@ -18,8 +18,8 @@ The server reads, per field, the environment first and then the configuration sa
|
|
|
18
18
|
|
|
19
19
|
| Variable | Required | Meaning |
|
|
20
20
|
|---|---|---|
|
|
21
|
-
| `JEV_TOOLS_URL` | yes | Complete endpoint URL compatible with the Jev API format. |
|
|
22
|
-
| `JEV_TOOLS_API_KEY` | yes | Bearer credential. Never printed in tool output. |
|
|
21
|
+
| `JEV_TOOLS_URL` | yes | Complete endpoint URL compatible with the Jev API format. Must be `https:`; plain `http:` is accepted only for `127.0.0.1`, `::1` or `localhost`. |
|
|
22
|
+
| `JEV_TOOLS_API_KEY` | yes | Bearer credential. Never printed in tool output, never passed to `jev_ask` commands, and replaced with `[redacted]` in their output. |
|
|
23
23
|
| `JEV_TOOLS_MODEL` | no | Requested model, default `openjev`. |
|
|
24
24
|
| `JEV_TOOLS_ROOT` | no | Repository directory when `--root` is not given. |
|
|
25
25
|
| `JEV_TOOLS_MAX_CALLS`, `JEV_TOOLS_MAX_USD` | no | Session call and cost limits for this server process. |
|
|
@@ -194,7 +194,7 @@ npx -y -p jev-agent-tools jev-agent-tools-mcp --root . < /dev/null
|
|
|
194
194
|
|
|
195
195
|
In PowerShell, run the second as `$null | npx -y -p jev-agent-tools jev-agent-tools-mcp --root .`.
|
|
196
196
|
|
|
197
|
-
The second command prints `jev-agent-tools MCP server ready (root ...; endpoint configured)` on stderr and exits when stdin closes. In the client, the six tools should be listed: `jev_ask`, `jev_ask_files`, `jev_find_files`, `jev_locate_in_file`, `jev_check_diff`, `jev_select_tests`.
|
|
197
|
+
The second command prints `jev-agent-tools MCP server ready (root ...; endpoint configured)` on stderr and exits when stdin closes. In the client, the six tools should be listed: `jev_ask`, `jev_ask_files`, `jev_find_files`, `jev_locate_in_file`, `jev_check_diff`, `jev_select_tests`. A smoke call with a one-line note and a yes/no question reports execution, evidence context, fresh/cache/static/unjudged items, diagnostics and separate request/result accounting. This verifies integration, not model accuracy.
|
|
198
198
|
|
|
199
199
|
## From a clone
|
|
200
200
|
|
|
@@ -207,12 +207,14 @@ Then use `"command": "node"` with `"args": ["/absolute/path/to/jev-tools/dist/mc
|
|
|
207
207
|
|
|
208
208
|
## Differences from pi and omp
|
|
209
209
|
|
|
210
|
-
- **No run-end documentation check.**
|
|
210
|
+
- **No run-end documentation check.** pi/OMP retain an automatic opt-out host hook (`JEV_TOOLS_AUTO_DOCS=0`). MCP has no hook and its absence does not require manual replacement calls, including risk followed by docs; choose a review only when it can inform an open decision.
|
|
211
211
|
- **Approval is the client's.** `jev_ask` is annotated as not read-only and potentially destructive while it accepts `command`; the other five are read-only. All six are open-world because evidence goes to your endpoint. `JEV_TOOLS_ALLOW_COMMAND=0` removes `command` from the schema and makes `jev_ask` read-only.
|
|
212
|
-
- **Instructions.** The reading guide and
|
|
212
|
+
- **Instructions.** The full reading guide and discretionary policy are sent as the server's `instructions`, using the same canonical fragments as pi/OMP with MCP's explicit hook difference. Clients may ignore them, so also copy the versioned [agent instructions](agent-instructions.md) into the project and update the copy on release upgrades.
|
|
213
213
|
- **Tool names in descriptions** refer to "your text search tool" and "your file-name search tool" instead of pi or omp tool names.
|
|
214
|
-
- **Protocol.** Versions 2024-11-05 through 2025-11-25
|
|
214
|
+
- **Protocol.** Versions 2024-11-05 through 2025-11-25 use `initialize`; 2026-07-28 uses `server/discover` and per-request `_meta`. Versions 2025-06-18 and later advertise `outputSchema` and return `structuredContent.result`; 2024-11-05 and 2025-03-26 receive self-contained text only. Each request retains its negotiated version even if another request changes the session version. Only 2026-07-28 adds `resultType: "complete"` and `ttlMs: 0` / `cacheScope: "private"`; this transport completion is independent of the report's execution state. Modern discovery supplies identity in `_meta["io.modelcontextprotocol/serverInfo"]`. Expected refusals use typed reports and appropriate `isError`; unexpected server exceptions remain JSON-RPC internal errors without invented report accounting. Tools only; no resources, prompts or sampling. Cancelling a call aborts its Jev requests and command.
|
|
215
|
+
- **Malformed arguments.** Input-schema violations return JSON-RPC `INVALID_PARAMS` without a result report or judgment; well-formed calls rejected by root/evidence admission retain their typed refusal report.
|
|
215
216
|
- **One process, one session.** Limits, cache and counters last as long as the connection; restart the server to reset them.
|
|
217
|
+
- **Per-call evidence root.** All six tools accept `root` for a registered worktree of the same repository, with the [shared admission restrictions](../README.md#evidence-root). The startup directory remains the authority; `root` is not permission to access unrelated repositories.
|
|
216
218
|
- **Command shutdown.** Cancellation, command timeout, stdin closure, SIGTERM and SIGINT wait for bounded command-tree termination. On POSIX the managed group receives SIGTERM, then SIGKILL after two seconds if it remains. On Windows, the server maps ordinary MSYS descendants through the same Bash installation's process table before running the system `taskkill /T /F`; each subprocess is bounded to five seconds. Cancelled calls receive no response. This is not a sandbox: descendants that deliberately detach from the managed group or escape the tracked tree are not contained.
|
|
217
219
|
|
|
218
220
|
## Troubleshooting
|
package/docs/tools/jev_ask.md
CHANGED
|
@@ -14,7 +14,9 @@ Use [jev_ask_files](jev_ask_files.md) for independent answers per file. Use nati
|
|
|
14
14
|
|
|
15
15
|
The tool assembles one JSON situation. A note appears as `state`; repository files as `files["path"]`; historical versions as `files_before["path"]`; and command evidence as `output` with `command`, `exit_code`, `timed_out`, `stdout` and `stderr`. You receive answers, not the assembled file contents or captured output. Jev reads the evidence at a glance; it does not count, calculate or infer missing dependencies. Evidence outweighs an explanation of its role.
|
|
16
16
|
|
|
17
|
-
Before judgment, bounded depth-one static import closure adds supported used declarations and data files, including before/after versions when applicable. A
|
|
17
|
+
Before judgment, bounded depth-one static import closure adds supported used declarations and data files, including before/after versions when applicable. A closure diagnostic records additions and gaps; it cannot establish completeness or discover relationships without imports. Explicit `files[...]` and `files_before[...]` selectors name required evidence. Ordinary lexical mentions such as `.only` or `process.env` do not create missing-file vetoes, and names printed in command output are not disk evidence. An exact path with directory segments never falls back to another file with the same basename. For abbreviated references, a uniquely matching supplied file takes precedence over inventory discovery; ambiguity is reported, not guessed.
|
|
18
|
+
|
|
19
|
+
One canonical path stores each file version; lightweight aliases do not duplicate content. Current and base versions remain distinct, and the 20-file limit counts distinct file identities, not their versions. A before-only selector can load historical content without a current file. `null` denotes a proven absent base version, not unavailable evidence. Missing/empty requirements exclude their question group and its controls; a requirement in the global note applies to all groups. Numeric zero and boolean false are not empty evidence.
|
|
18
20
|
|
|
19
21
|
For recognizable assertion failures, the tool attempts to attach the failing test named by the log. An unidentifiable target produces a warning; any bug-versus-wrong-test conclusion under that warning remains unproven. Pass the failing test and implementation yourself whenever possible.
|
|
20
22
|
|
|
@@ -24,10 +26,11 @@ At least one of `state`, `paths` or `command` must provide evidence.
|
|
|
24
26
|
|
|
25
27
|
| Field | Type / default | Meaning |
|
|
26
28
|
|---|---|---|
|
|
27
|
-
| `
|
|
29
|
+
| `root` | Optional nonempty string | Exact initial Git root or registered worktree of the same repository; see [evidence-root admission](../../README.md#evidence-root). |
|
|
30
|
+
| `state` | Optional string, at most 8,000 characters | Short request, plan or observation not shown by the files; not a place to paste files or logs. JSON object text is accepted, but reserved evidence keys such as `files`, `files_before`, `evidence` and `output` cannot be supplied by the note. |
|
|
28
31
|
| `paths` | Optional array of nonempty strings, at most 20 | Repository-relative files to read; not recursive directory or glob triage. |
|
|
29
|
-
| `base` | Optional nonempty Git ref | Add
|
|
30
|
-
| `command` | Optional nonempty string | Execute `bash -c` at the repository root with `CI=1`, normal shell permissions and no additional sandbox. |
|
|
32
|
+
| `base` | Optional nonempty Git ref | Add historical versions for the resolved revision; only established absence is `null`. Use `HEAD` for uncommitted before/after questions. Both versions count toward serialized evidence admission. |
|
|
33
|
+
| `command` | Optional nonempty string | Execute `bash -c` at the repository root with `CI=1` and without `JEV_TOOLS_API_KEY`, normal shell permissions and no additional sandbox. The configured key is replaced with `[redacted]` in captured output, with the count reported. |
|
|
31
34
|
| `timeout_s` | Optional number, default 60, range 1–300 | Command timeout in seconds. |
|
|
32
35
|
| `asks` | Required intent, nonempty intent array, or JSON string encoding them | Typed questions; one intent per judgment. |
|
|
33
36
|
| `max_calls` | Optional non-negative integer | Limit requests including controls and output passage finding; unjudged work is reported. |
|
|
@@ -81,7 +84,7 @@ Verification distinguishes `holds`, `contradicted`, `not addressed by the state`
|
|
|
81
84
|
|
|
82
85
|
Serialized evidence, including keys and escapes, is bounded to 80,000 characters. Oversized files/state are refused with a split suggestion, not silently shortened. Repository file confinement rejects escaping paths and internal URLs.
|
|
83
86
|
|
|
84
|
-
Commands are captured in private temporary files and cleaned up. Captured streams over 64 MiB are refused after execution; this neither caps disk usage nor interrupts execution. Repeated line shapes are compressed; lines over 8,192 characters and more than 2,048 distinct shapes produce visible notices. If compression still cannot fit, bounded passage judgments retain failure details and beginning/end context with omission markers. Excessive passage work is refused with guidance to narrow the command. A timeout gives `exit_code: null` and `timed_out: true`, not success; cancellation stops without judgment. Command calls bypass caching. `JEV_TOOLS_ALLOW_COMMAND=0` disables command evidence. Missing endpoint/key refuses judgment without a chat-model fallback.
|
|
87
|
+
Commands are captured in private temporary files and cleaned up. Captured streams over 64 MiB are refused after execution; this neither caps disk usage nor interrupts execution. Repeated line shapes are compressed; lines over 8,192 characters and more than 2,048 distinct shapes produce visible notices. If compression still cannot fit, bounded passage judgments retain failure details and beginning/end context with omission markers. Command output reaches this stage before immutable file/note budget refusal, including calls with base snapshots and import closure. Excessive passage work is refused with guidance to narrow the command. Reports distinguish not started, finished and unknown execution, preserve observed cwd/exit/timeout even if later processing fails, and do not equate a judgment refusal with an unexecuted command. A timeout gives `exit_code: null` and `timed_out: true`, not success; cancellation stops without judgment. Command calls bypass caching. `JEV_TOOLS_ALLOW_COMMAND=0` disables command evidence. Missing endpoint/key refuses judgment without a chat-model fallback.
|
|
85
88
|
|
|
86
89
|
## Host differences
|
|
87
90
|
|
|
@@ -18,6 +18,7 @@ Each admitted file is judged alone as `path` and `content`. Every ask applies to
|
|
|
18
18
|
|
|
19
19
|
| Field | Type / default | Meaning |
|
|
20
20
|
|---|---|---|
|
|
21
|
+
| `root` | Optional nonempty string | Exact initial Git root or registered worktree of the same repository; see [evidence-root admission](../../README.md#evidence-root). |
|
|
21
22
|
| `paths` | Required nonempty array of nonempty strings | Repository-relative files, recursive directories or globs; at most 255 admitted files. |
|
|
22
23
|
| `asks` | Required typed intent, nonempty array of intents, or JSON string encoding them | All asks apply to every admitted file. An array is the clearest form. |
|
|
23
24
|
| `max_calls` | Optional non-negative integer | Request limit for this invocation; remaining files are listed unchecked. Separate from session budgets. |
|
|
@@ -65,7 +66,7 @@ Illustrative call with fictional repository paths, not a recorded execution:
|
|
|
65
66
|
|
|
66
67
|
## Results and next action
|
|
67
68
|
|
|
68
|
-
Results preserve the exact statement next to each file's answer. Boolean `yes` or `no (not shown)` includes the probability of yes; intermediate probabilities between 0.20 and 0.80 are unsure. Categories and levels need a leading-option probability of at least 0.85 after applicable controls. `classify`, `rate`, `decide` and `free` are uncalibrated; sharing a display band does not establish an error rate. Failed integrity or option-order controls make clear-looking results unsure. Read the candidate file before acting.
|
|
69
|
+
Results preserve the exact statement next to each file's answer and distinguish fresh/cache judgments from unjudged files. Boolean `yes` or `no (not shown)` includes the probability of yes; intermediate probabilities between 0.20 and 0.80 are unsure. Categories and levels need a leading-option probability of at least 0.85 after applicable controls. `classify`, `rate`, `decide` and `free` are uncalibrated; sharing a display band does not establish an error rate. Failed integrity or option-order controls make clear-looking results unsure; missing required judgments never become fabricated probabilities. Read the candidate file before acting. Typed diagnostics and actions explain skipped/unchecked files, causes and scope. See [shared result reading](../../README.md#read-the-results).
|
|
69
70
|
|
|
70
71
|
## Limits and failure behavior
|
|
71
72
|
|
|
@@ -18,6 +18,7 @@ Changed declarations, slices, files or hunks become before/after evidence units.
|
|
|
18
18
|
|
|
19
19
|
| Field | Type / default | Meaning |
|
|
20
20
|
|---|---|---|
|
|
21
|
+
| `root` | Optional nonempty string | Exact initial Git root or registered worktree of the same repository; see [evidence-root admission](../../README.md#evidence-root). |
|
|
21
22
|
| `check` | Required `"risk"`, `"docs"` or `"spec"` | Select the preset below. |
|
|
22
23
|
| `base` | Optional nonempty Git ref, default `HEAD` | Compare the current tree, including untracked files, with this revision. |
|
|
23
24
|
| `spec_path` | Optional nonempty repository-relative string | Required for `spec`: Markdown specification with `### REQ-…` headings. |
|
|
@@ -36,7 +37,7 @@ For replaced member accesses, separate local-caller checks inspect statically re
|
|
|
36
37
|
|
|
37
38
|
### docs
|
|
38
39
|
|
|
39
|
-
Judge up to 40 existing tracked Markdown sections mentioning changed code or importing source, and identify
|
|
40
|
+
Judge up to 40 existing tracked Markdown sections mentioning changed code or importing source, and identify an existing sentence that may no longer match the change. Collection and traversal limits remain material reservations in the report. This is not an exhaustive detector of missing documentation, arbitrary companion edits or every stale sentence; inspect the sentence against the code before changing it.
|
|
40
41
|
|
|
41
42
|
### spec
|
|
42
43
|
|
|
@@ -64,6 +65,8 @@ For another preset, use `{"check":"docs","base":"HEAD"}` or `{"check":"spec","ba
|
|
|
64
65
|
|
|
65
66
|
Findings name the evidence unit, stale sentence or requirement and its probability. Fixed findings require at least 0.7. Docs probabilities from 0.2 up to 0.7 are unsure without an additional judgment; read the indicated wording. Local-caller probabilities between 0.2 and 0.7 are unsure; `cannot_tell` at least 0.3 is abstain naming the missing provider or binding. Failed witness batches remain unsure. Read flagged source, callers or documentation before editing and preserve unresolved uncertainty in reports. See [shared result reading](../../README.md#read-the-results).
|
|
66
67
|
|
|
68
|
+
Structured items retain negative as well as positive review results, actual control values and fresh/cache provenance. Budget-exhausted or incomplete groups are unjudged, not negative findings. Documentation completeness limits, collection omissions and witness failures remain scoped diagnostics even when another part of the review yields usable results.
|
|
69
|
+
|
|
67
70
|
## Limits and failure behavior
|
|
68
71
|
|
|
69
72
|
No findings is not proof of safety, unseen-caller coverage or complete documentation. Unjudged units, callers, sections, requirements and severity are named when request or session budgets stop work. Parser gaps and unresolved dynamic providers remain limits, not invented answers. Missing configuration refuses judgment without a chat-model fallback. Static collection is bounded and does not evaluate third-party code.
|
|
@@ -18,6 +18,7 @@ Deterministic lexical pre-ranking and name judgments produce candidates; the too
|
|
|
18
18
|
|
|
19
19
|
| Field | Type / default | Meaning |
|
|
20
20
|
|---|---|---|
|
|
21
|
+
| `root` | Optional nonempty string | Exact initial Git root or registered worktree of the same repository; see [evidence-root admission](../../README.md#evidence-root). |
|
|
21
22
|
| `goal` | Required nonempty string | Behavioral sentence with at least five content words for an unqualified entry verdict. Shorter goals are accepted but capped at unsure. |
|
|
22
23
|
| `keywords` | Optional array of nonempty strings, default empty | Known identifiers or terms steering lexical ranking and excerpt selection. |
|
|
23
24
|
| `scope` | Optional directory string or array of directory strings, default repository | Restrict search; an incorrect scope may yield none. |
|
|
@@ -44,7 +45,7 @@ Illustrative call with fictional repository paths, not a recorded execution:
|
|
|
44
45
|
|
|
45
46
|
## Results and next action
|
|
46
47
|
|
|
47
|
-
|
|
48
|
+
The report distinguishes the primary entry decision from auxiliary ranking and controls, retains actual fresh/cache provenance and supplies useful read actions. Unsure can mean two plausible entries, order sensitivity, inconclusive excerpts or a short goal: inspect the leading candidates. A judged `none` means no candidate fit the supplied evidence: clarify the behavioral goal or widen the relevant scope. An empty collection, exhausted budget or missing response is not a `none` judgment and carries no invented probability. Paths and probabilities are not source text; read the files before editing. See [shared result reading](../../README.md#read-the-results).
|
|
48
49
|
|
|
49
50
|
## Limits and failure behavior
|
|
50
51
|
|