jev-agent-tools 0.1.4 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +106 -1
- package/CONTRIBUTING.md +43 -0
- package/README.md +58 -17
- package/SECURITY.md +43 -0
- package/dist/adapters/analysis-context.js +75 -0
- package/dist/adapters/ask-files.js +198 -0
- package/dist/adapters/ask-proof.js +200 -0
- package/dist/adapters/ask-syntax.js +385 -0
- package/dist/adapters/canonical-path.js +17 -0
- package/dist/adapters/command.js +234 -0
- package/dist/adapters/docs.js +192 -0
- package/dist/adapters/evidence-context.js +119 -0
- package/dist/adapters/exec.js +207 -0
- package/dist/adapters/files.js +418 -0
- package/dist/adapters/find.js +150 -0
- package/dist/adapters/git-base.js +32 -0
- package/dist/adapters/git-inventory.js +71 -0
- package/dist/adapters/git.js +483 -0
- package/dist/adapters/locate-file.js +197 -0
- package/dist/adapters/output-lines.js +46 -0
- package/dist/adapters/private-storage.js +106 -0
- package/dist/adapters/risk-callers.js +429 -0
- package/dist/adapters/runner-version.js +78 -0
- package/dist/adapters/shell.js +92 -0
- package/dist/adapters/syntax.js +187 -0
- package/dist/adapters/test-inventory.js +139 -0
- package/dist/adapters/usage.js +20 -0
- package/dist/adapters/utf8.js +47 -0
- package/dist/configuration.js +267 -0
- package/dist/constants.js +140 -0
- package/dist/core/ask-closure.js +282 -0
- package/dist/core/ask-proof.js +1 -0
- package/dist/core/ask-references.js +278 -0
- package/dist/core/asks.js +507 -0
- package/dist/core/batches.js +65 -0
- package/dist/core/command-output.js +224 -0
- package/dist/core/diff.js +178 -0
- package/dist/core/docs.js +302 -0
- package/dist/core/find.js +108 -0
- package/dist/core/git.js +1 -0
- package/dist/core/imports.js +550 -0
- package/dist/core/integrity.js +45 -0
- package/dist/core/lexical.js +132 -0
- package/dist/core/locate.js +169 -0
- package/dist/core/output.js +137 -0
- package/dist/core/pointer.js +29 -0
- package/dist/core/result-report.js +302 -0
- package/dist/core/risk-callers.js +851 -0
- package/dist/core/runner-version.js +45 -0
- package/dist/core/secret-path.js +34 -0
- package/dist/core/sections.js +230 -0
- package/dist/core/state.js +51 -0
- package/dist/core/syntax.js +1 -0
- package/dist/core/test-commands.js +334 -0
- package/dist/core/test-coverage.js +74 -0
- package/dist/core/test-discovery.js +1382 -0
- package/dist/core/test-evidence.js +527 -0
- package/dist/core/test-state.js +81 -0
- package/dist/core/truncate.js +12 -0
- package/dist/core/units.js +349 -0
- package/dist/describe.js +23 -0
- package/dist/guide.js +33 -0
- package/dist/host.js +24 -0
- package/dist/jev/client.js +456 -0
- package/dist/jev/pool.js +54 -0
- package/dist/jev/types.js +1 -0
- package/dist/mcp/main.js +124 -0
- package/dist/mcp/protocol.js +210 -0
- package/dist/mcp/tools.js +129 -0
- package/dist/presets/docs.js +62 -0
- package/dist/presets/risk.js +179 -0
- package/dist/presets/spec.js +81 -0
- package/dist/presets/witnesses.js +249 -0
- package/dist/render.js +114 -0
- package/dist/report-schema.js +1356 -0
- package/dist/result-types.js +1 -0
- package/dist/result.js +3 -0
- package/dist/runtime.js +1 -0
- package/dist/session.js +147 -0
- package/dist/texts/ask-files.js +3 -0
- package/dist/texts/ask.js +4 -0
- package/dist/texts/check-diff.js +20 -0
- package/dist/texts/configuration.js +1 -0
- package/dist/texts/find.js +19 -0
- package/dist/texts/guide.js +3 -0
- package/dist/texts/instructions.js +72 -0
- package/dist/texts/locate.js +15 -0
- package/dist/texts/select-tests.js +4 -0
- package/dist/tools/ask-files.js +450 -0
- package/dist/tools/ask-schema.js +70 -0
- package/dist/tools/ask.js +1147 -0
- package/dist/tools/check-diff.js +594 -0
- package/dist/tools/docs-check.js +408 -0
- package/dist/tools/find.js +682 -0
- package/dist/tools/locate.js +602 -0
- package/dist/tools/review-report.js +230 -0
- package/dist/tools/select-tests.js +821 -0
- package/dist/tools/spec-check.js +263 -0
- package/docs/adr/0001-strict-typescript-pure-core-offline-tests.md +31 -0
- package/docs/adr/0002-one-http-protocol-across-hosts.md +17 -0
- package/docs/adr/0003-explicit-scope-conservative-automation.md +19 -0
- package/docs/adr/0004-compiled-typed-intents.md +19 -0
- package/docs/adr/0005-evidence-construction-before-judgment.md +19 -0
- package/docs/adr/0006-visible-uncertainty-constrained-controls.md +21 -0
- package/docs/adr/0007-bounded-evidence-visible-limits.md +21 -0
- package/docs/adr/0008-static-test-discovery-conservative-plans.md +19 -0
- package/docs/adr/0009-session-cache-requested-model-identity.md +17 -0
- package/docs/adr/0010-mcp-server-thin-host.md +23 -0
- package/docs/agent-instructions.md +120 -0
- package/docs/design.md +16 -4
- package/docs/mcp.md +233 -0
- package/docs/tools/jev_ask.md +8 -5
- package/docs/tools/jev_ask_files.md +2 -1
- package/docs/tools/jev_check_diff.md +4 -1
- package/docs/tools/jev_find_files.md +2 -1
- package/docs/tools/jev_locate_in_file.md +5 -0
- package/docs/tools/jev_select_tests.md +4 -1
- package/package.json +19 -4
- package/rules/jev-ask.md +22 -1
- package/server.json +57 -0
- package/src/adapters/ask-files.ts +11 -3
- package/src/adapters/ask-proof.ts +69 -11
- package/src/adapters/canonical-path.ts +18 -0
- package/src/adapters/command.ts +102 -36
- package/src/adapters/docs.ts +33 -14
- package/src/adapters/evidence-context.ts +169 -0
- package/src/adapters/exec.ts +226 -0
- package/src/adapters/files.ts +146 -16
- package/src/adapters/find.ts +37 -7
- package/src/adapters/git-base.ts +7 -1
- package/src/adapters/git.ts +61 -8
- package/src/adapters/locate-file.ts +51 -9
- package/src/adapters/private-storage.ts +155 -0
- package/src/adapters/risk-callers.ts +7 -2
- package/src/adapters/shell.ts +113 -0
- package/src/adapters/test-inventory.ts +12 -4
- package/src/configuration.ts +55 -14
- package/src/constants.ts +37 -5
- package/src/core/ask-references.ts +262 -146
- package/src/core/asks.ts +79 -7
- package/src/core/command-output.ts +17 -1
- package/src/core/import-boundaries.ts +8 -3
- package/src/core/locate.ts +8 -5
- package/src/core/output.ts +34 -0
- package/src/core/result-report.ts +410 -0
- package/src/core/secret-path.ts +37 -0
- package/src/core/state.ts +8 -1
- package/src/core/units.ts +3 -2
- package/src/host.ts +11 -0
- package/src/index.ts +3 -0
- package/src/jev/client.ts +66 -16
- package/src/jev/types.ts +24 -3
- package/src/mcp/main.ts +135 -0
- package/src/mcp/protocol.ts +332 -0
- package/src/mcp/tools.ts +179 -0
- package/src/render.ts +109 -0
- package/src/report-schema.ts +1380 -0
- package/src/result-types.ts +234 -0
- package/src/result.ts +4 -1
- package/src/runtime.ts +6 -0
- package/src/session.ts +59 -0
- package/src/setup.ts +13 -5
- package/src/texts/ask-files.ts +4 -1
- package/src/texts/ask.ts +8 -1
- package/src/texts/check-diff.ts +7 -4
- package/src/texts/find.ts +8 -2
- package/src/texts/guide.ts +8 -16
- package/src/texts/instructions.ts +98 -0
- package/src/texts/locate.ts +8 -2
- package/src/texts/run-end.ts +2 -2
- package/src/texts/select-tests.ts +4 -1
- package/src/tools/ask-files.ts +311 -18
- package/src/tools/ask.ts +722 -95
- package/src/tools/check-diff.ts +337 -31
- package/src/tools/docs-check.ts +241 -38
- package/src/tools/find.ts +389 -29
- package/src/tools/locate.ts +387 -25
- package/src/tools/review-report.ts +308 -0
- package/src/tools/select-tests.ts +484 -23
- package/src/tools/spec-check.ts +194 -19
package/docs/design.md
CHANGED
|
@@ -6,9 +6,17 @@ The tools construct bounded evidence before asking for judgment. They preserve u
|
|
|
6
6
|
|
|
7
7
|
Supply the discriminating evidence, not an argument about it. For comparisons, show both sides: a failing test and its implementation, or a file before and after. A bounded static import closure adds declarations before judgment where supported; it cannot establish completeness or discover relationships without imports. Commands are evidence sources, not a sandbox or a substitute for reading the output directly.
|
|
8
8
|
|
|
9
|
+
Every call captures its host/server authority independently. Optional `root` admits only an exact repository top-level or live registered worktree with the same Git common directory, before content, commands, cache or HTTP. No override preserves existing cwd behavior; an invalid override never falls back. The effective context, requested/resolved base and inventory limits travel with evidence and results. See [root admission](../README.md#evidence-root).
|
|
10
|
+
|
|
11
|
+
Explicit selectors define required evidence; lexical hints do not become vetoes merely because they resemble filenames. Exact directory paths cannot fall back to suffix matches. Canonical paths serialize each current/base version once, with aliases as metadata, and file admission counts identities rather than versions. Missing requirements gate their full question/control group; global-note requirements remain global. Captured output and repository files keep separate provenance.
|
|
12
|
+
|
|
13
|
+
Historical-only files pass the same protected-path, ignore, regular-file and text admission checks before their base content is disclosed. Current and base versions share the distinct-file ceiling; unresolved or over-budget selectors are not retried through raw historical reads.
|
|
14
|
+
|
|
9
15
|
## Pure core, thin adapters
|
|
10
16
|
|
|
11
|
-
Pure core transformations construct states, units, questions and display envelopes. Adapters own filesystem, Git, parsing and host effects; the HTTP client owns transport. Presets define fixed review questions and texts define host-facing guidance. Dependency checks keep shared contracts below consumers, reject cycles and account for erased type imports. Pure path operations and erased parser types are explicit architectural exceptions. Both hosts use the same HTTP judgment protocol.
|
|
17
|
+
Pure core transformations construct states, units, questions and display envelopes. Adapters own filesystem, Git, parsing and host effects; the HTTP client owns transport. Presets define fixed review questions and texts define host-facing guidance. Hosts sit on top: `src/index.ts` registers the tools with pi and omp, and `src/mcp/` serves the same tool factories to any MCP client over stdio ([ADR 0010](adr/0010-mcp-server-thin-host.md), [setup](mcp.md)). Dependency checks keep shared contracts below consumers, reject cycles and account for erased type imports. Pure path operations and erased parser types are explicit architectural exceptions. Both hosts use the same HTTP judgment protocol.
|
|
18
|
+
|
|
19
|
+
The dependency-free `src/result-types.ts` defines `ResultReportV1`; `core/result-report.ts` builds reports and validates semantic invariants, while `report-schema.ts` supplies the closed transport schema and structural validator. Producers record actual observations rather than reconstructing them from rendered prose. `renderResultReport` projects the same report to self-contained text, retaining native read/runner details. pi/omp publish `details.result`; supported MCP versions publish `structuredContent.result` without introducing a second judgment policy.
|
|
12
20
|
|
|
13
21
|
## Typed intents and fixed checks
|
|
14
22
|
|
|
@@ -18,11 +26,15 @@ Caller intents compile to typed questions with canonical options and exact state
|
|
|
18
26
|
|
|
19
27
|
Leading-option probability (p_max) is the probability of the most likely choice, not the response's separate confidence field. Gray bands and applicable option-order checks expose ambiguity. Missing evidence remains abstention. Preset witnesses use a decoy and a known positive reference to detect a biased setup; failed controls do not promote findings. No mark establishes that unseen evidence is complete. See [result reading](../README.md#read-the-results).
|
|
20
28
|
|
|
29
|
+
Execution state and judgment band are independent: complete/partial/refused/not judged describe processing, while verdict/unsure/abstain describe admitted judgments. Every requested item is fresh, cache, static or unjudged; static/fallback decisions have no invented probability. Auxiliary controls and passage decisions do not inflate requested-result counts. Missing required controls leave the requested group unjudged. Accounting distinguishes HTTP attempts, questions sent, cache probes, current cost and elapsed time; unknown cost is never reconstructed as zero or historical cached cost. Scoped diagnostics preserve material omissions and useful conditional actions.
|
|
30
|
+
|
|
21
31
|
## Bounded work, visible limits
|
|
22
32
|
|
|
23
|
-
Admission limits, request planning, rate limiting and retry bounds are distinct. Oversized evidence is refused or visibly omitted, never silently converted into an ordinary verdict. Per-invocation max_calls and session call/cost limits are separate; control and severity requests count too. Unjudged tests stay selected. Automatic documentation review is the only run-end automation and can request at most one extra turn.
|
|
33
|
+
Admission limits, request planning, rate limiting and retry bounds are distinct. Oversized evidence is refused or visibly omitted, never silently converted into an ordinary verdict. Per-invocation max_calls and session call/cost limits are separate; control and severity requests count too. Under a session USD limit, requests are admitted one at a time so each admission sees all cost reported so far. Unjudged tests stay selected. Automatic documentation review is the only run-end automation and can request at most one extra turn.
|
|
34
|
+
|
|
35
|
+
Successful non-command judgments are cached only in session memory by canonical evidence, question and requested model string. Every judgment input, including control and passage decisions, carries admitted authority/root provenance and the resolved base when applicable; that metadata participates in both cache identity and serialized evidence admission. Identical content from different roots or base revisions cannot reuse a judgment. Errors are not cached; command judgments bypass caching. The default openjev alias can move, and an echoed model name does not prove served-model identity. Fractional budgets bound admission, not an exact prediction of the final request's cost.
|
|
24
36
|
|
|
25
|
-
|
|
37
|
+
`src/texts/instructions.ts` is the versioned source for host policy, evidence guidance and current-state presentation. `node scripts/generate-instructions.ts` updates checked-in omp/MCP fragments; `--check` detects drift. Jev is discretionary when it can inform an open decision; native decisive evidence requires no certification call. Unchanged retries and mandatory risk/docs sequences are not recovery policy. pi/omp retain the opt-out documentation hook; MCP has no replacement obligation. Final presentation distinguishes native execution, Jev's static judgment and reported evidence, preserving independent material reservations without requiring an exhaustive history registry.
|
|
26
38
|
|
|
27
39
|
## Policy thresholds
|
|
28
40
|
|
|
@@ -184,4 +196,4 @@ Optional native syntax parsing and file-search acceleration can be absent. Tools
|
|
|
184
196
|
|
|
185
197
|
## Architecture decisions
|
|
186
198
|
|
|
187
|
-
See the [architecture decision records](adr/) for durable trade-offs. Start with the [README](../README.md) for installation and follow its six tool references for complete parameter contracts.
|
|
199
|
+
See the [architecture decision records](adr/) for durable trade-offs. Start with the [README](../README.md) for installation and follow its six tool references for complete parameter contracts. MCP clients: [setup guide](mcp.md) and [agent instructions](agent-instructions.md).
|
package/docs/mcp.md
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
# MCP setup guide
|
|
2
|
+
|
|
3
|
+
`jev-agent-tools-mcp` is a stdio [MCP](https://modelcontextprotocol.io) server that exposes the same six `jev_*` tools as the pi and omp extension to any MCP client. It ships in the `jev-agent-tools` npm package, has no runtime dependencies beyond that package, and runs one session per server process.
|
|
4
|
+
|
|
5
|
+
Setup takes three steps: provide the endpoint and key, register the server with your client, and add the [agent instructions](agent-instructions.md) to your project.
|
|
6
|
+
|
|
7
|
+
## 1. Requirements
|
|
8
|
+
|
|
9
|
+
- Node.js 24 or later, and Git, on the machine that runs the client.
|
|
10
|
+
- Bash for optional `jev_ask` command evidence. On Windows this is Git for Windows bash, found automatically next to `git` or under Program Files; set `JEV_TOOLS_BASH` to use another. The WSL `bash.exe` launchers are never used.
|
|
11
|
+
- A Jev endpoint URL and API key.
|
|
12
|
+
|
|
13
|
+
The server ships in `jev-agent-tools` version 0.2.0 and later. For a development checkout, see [From a clone](#from-a-clone).
|
|
14
|
+
|
|
15
|
+
## 2. Provide the endpoint and key
|
|
16
|
+
|
|
17
|
+
The server reads, per field, the environment first and then the configuration saved by `/jev-setup` in pi or omp. There is no interactive setup inside MCP.
|
|
18
|
+
|
|
19
|
+
| Variable | Required | Meaning |
|
|
20
|
+
|---|---|---|
|
|
21
|
+
| `JEV_TOOLS_URL` | yes | Complete endpoint URL compatible with the Jev API format. Must be `https:`; plain `http:` is accepted only for `127.0.0.1`, `::1` or `localhost`. |
|
|
22
|
+
| `JEV_TOOLS_API_KEY` | yes | Bearer credential. Never printed in tool output, never passed to `jev_ask` commands, and replaced with `[redacted]` in their output. |
|
|
23
|
+
| `JEV_TOOLS_MODEL` | no | Requested model, default `openjev`. |
|
|
24
|
+
| `JEV_TOOLS_ROOT` | no | Repository directory when `--root` is not given. |
|
|
25
|
+
| `JEV_TOOLS_MAX_CALLS`, `JEV_TOOLS_MAX_USD` | no | Session call and cost limits for this server process. |
|
|
26
|
+
| `JEV_TOOLS_ALLOW_COMMAND` | no | `0` removes `command` from `jev_ask`. |
|
|
27
|
+
| `JEV_TOOLS_BASH` | no | Full path of the bash used for commands on Windows. |
|
|
28
|
+
|
|
29
|
+
Prefer passing secrets through your client's environment-variable interpolation (shown per client below) rather than writing the key into a committed file. Saved configuration from `/jev-setup` lives in `~/.config/jev-agent-tools/config.json` and is refused unless the directory and file are private: owner-only mode bits on Linux and macOS, an access list limited to you, SYSTEM and Administrators on Windows. Unusable saved storage is reported on the server's stderr and never stops the server.
|
|
30
|
+
|
|
31
|
+
Without an endpoint and key the server still starts, lists the tools and answers each call with what is missing.
|
|
32
|
+
|
|
33
|
+
## 3. Register the server
|
|
34
|
+
|
|
35
|
+
Every example registers a server named `jev`. Replace the repository path. The repository the tools work in is `--root`, else `JEV_TOOLS_ROOT`, else the directory the client starts the server in.
|
|
36
|
+
|
|
37
|
+
**Windows:** most clients start the command without a shell, and `npx` is a `.cmd` script there, so `"command": "npx"` fails to start. Use `"command": "cmd"` with `"args": ["/c", "npx", ...]`, as in the Windows examples. This was checked on Windows 11 with Node.js 24: a shell-less spawn of `npx` failed with `ENOENT`, `cmd /c npx` started the server.
|
|
38
|
+
|
|
39
|
+
The shared arguments are:
|
|
40
|
+
|
|
41
|
+
```text
|
|
42
|
+
npx -y -p jev-agent-tools jev-agent-tools-mcp --root <repository>
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
`-p jev-agent-tools` is required because the binary name differs from the package name. Pin a version (`jev-agent-tools@X.Y.Z`) for reproducible setups.
|
|
46
|
+
|
|
47
|
+
### From the MCP Registry
|
|
48
|
+
|
|
49
|
+
Releases are also published to the official [MCP Registry](https://registry.modelcontextprotocol.io) as `io.github.NomenAK/jev-agent-tools`, from [`server.json`](../server.json). Clients and catalogs that read the registry can install the server from there; they start it as `npx jev-agent-tools` (the package's only binary) and ask for `JEV_TOOLS_URL` and the secret `JEV_TOOLS_API_KEY`. The manual entries below give the same result.
|
|
50
|
+
|
|
51
|
+
### Claude Code
|
|
52
|
+
|
|
53
|
+
For a private registration, add it from the project directory with the CLI. Your shell expands the variables, so the values are stored in your private `~/.claude.json`, not in the project:
|
|
54
|
+
|
|
55
|
+
```sh
|
|
56
|
+
claude mcp add --scope local --env JEV_TOOLS_URL="$JEV_TOOLS_URL" --env JEV_TOOLS_API_KEY="$JEV_TOOLS_API_KEY" --transport stdio jev -- npx -y -p jev-agent-tools jev-agent-tools-mcp --root .
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
To share the server with the team, write `.mcp.json` at the project root instead. Claude Code expands `${VAR}` and `${VAR:-default}` from each user's environment when it loads the file, so no key is committed:
|
|
60
|
+
|
|
61
|
+
```json
|
|
62
|
+
{
|
|
63
|
+
"mcpServers": {
|
|
64
|
+
"jev": {
|
|
65
|
+
"command": "npx",
|
|
66
|
+
"args": ["-y", "-p", "jev-agent-tools", "jev-agent-tools-mcp", "--root", "${CLAUDE_PROJECT_DIR:-.}"],
|
|
67
|
+
"env": {
|
|
68
|
+
"JEV_TOOLS_URL": "${JEV_TOOLS_URL}",
|
|
69
|
+
"JEV_TOOLS_API_KEY": "${JEV_TOOLS_API_KEY}"
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
On Windows use `"command": "cmd"` and prepend `"/c", "npx"` to `args`. Check with `claude mcp list`. Add the instructions to `CLAUDE.md` ([template](agent-instructions.md#claudemd)).
|
|
77
|
+
|
|
78
|
+
### Claude Desktop
|
|
79
|
+
|
|
80
|
+
Edit `claude_desktop_config.json` (Settings, Developer, Edit Config): `~/Library/Application Support/Claude/claude_desktop_config.json` on macOS, `%APPDATA%\Claude\claude_desktop_config.json` on Windows. Claude Desktop has no project directory, so `--root` is required. Its documentation does not describe variable interpolation, so values here are literal; keep this file private.
|
|
81
|
+
|
|
82
|
+
```json
|
|
83
|
+
{
|
|
84
|
+
"mcpServers": {
|
|
85
|
+
"jev": {
|
|
86
|
+
"command": "cmd",
|
|
87
|
+
"args": ["/c", "npx", "-y", "-p", "jev-agent-tools", "jev-agent-tools-mcp", "--root", "C:\\path\\to\\repository"],
|
|
88
|
+
"env": {
|
|
89
|
+
"JEV_TOOLS_URL": "https://your-jev-endpoint.example/judge",
|
|
90
|
+
"JEV_TOOLS_API_KEY": "your-key"
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
On macOS use `"command": "npx"` without `"/c", "npx"`. Restart Claude Desktop after saving. One server serves one repository; add a second entry with another name for another repository.
|
|
98
|
+
|
|
99
|
+
### Kiro (IDE and CLI)
|
|
100
|
+
|
|
101
|
+
Workspace: `.kiro/settings/mcp.json`. User: `~/.kiro/settings/mcp.json`. The workspace file wins for a server of the same name. Kiro expands `${VAR}`.
|
|
102
|
+
|
|
103
|
+
```json
|
|
104
|
+
{
|
|
105
|
+
"mcpServers": {
|
|
106
|
+
"jev": {
|
|
107
|
+
"command": "npx",
|
|
108
|
+
"args": ["-y", "-p", "jev-agent-tools", "jev-agent-tools-mcp", "--root", "."],
|
|
109
|
+
"env": {
|
|
110
|
+
"JEV_TOOLS_URL": "${JEV_TOOLS_URL}",
|
|
111
|
+
"JEV_TOOLS_API_KEY": "${JEV_TOOLS_API_KEY}"
|
|
112
|
+
},
|
|
113
|
+
"disabled": false,
|
|
114
|
+
"autoApprove": ["jev_ask_files", "jev_find_files", "jev_locate_in_file", "jev_check_diff", "jev_select_tests"]
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
`autoApprove` above leaves `jev_ask` on manual approval because it can run shell commands. On Windows use `"command": "cmd"` with `"/c", "npx"` first in `args`. Add the instructions as a steering file ([template](agent-instructions.md#kiro-steering)).
|
|
121
|
+
|
|
122
|
+
### Cursor
|
|
123
|
+
|
|
124
|
+
Project: `.cursor/mcp.json`. Global: `~/.cursor/mcp.json`. Cursor expands `${env:NAME}` and `${workspaceFolder}`.
|
|
125
|
+
|
|
126
|
+
```json
|
|
127
|
+
{
|
|
128
|
+
"mcpServers": {
|
|
129
|
+
"jev": {
|
|
130
|
+
"command": "npx",
|
|
131
|
+
"args": ["-y", "-p", "jev-agent-tools", "jev-agent-tools-mcp", "--root", "${workspaceFolder}"],
|
|
132
|
+
"env": {
|
|
133
|
+
"JEV_TOOLS_URL": "${env:JEV_TOOLS_URL}",
|
|
134
|
+
"JEV_TOOLS_API_KEY": "${env:JEV_TOOLS_API_KEY}"
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
### VS Code (GitHub Copilot)
|
|
142
|
+
|
|
143
|
+
Workspace: `.vscode/mcp.json`. The top-level key is `servers`, not `mcpServers`.
|
|
144
|
+
|
|
145
|
+
```json
|
|
146
|
+
{
|
|
147
|
+
"servers": {
|
|
148
|
+
"jev": {
|
|
149
|
+
"command": "npx",
|
|
150
|
+
"args": ["-y", "-p", "jev-agent-tools", "jev-agent-tools-mcp", "--root", "${workspaceFolder}"],
|
|
151
|
+
"env": {
|
|
152
|
+
"JEV_TOOLS_URL": "https://your-jev-endpoint.example/judge",
|
|
153
|
+
"JEV_TOOLS_API_KEY": "${input:jev-api-key}"
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
},
|
|
157
|
+
"inputs": [
|
|
158
|
+
{ "type": "promptString", "id": "jev-api-key", "description": "Jev API key", "password": true }
|
|
159
|
+
]
|
|
160
|
+
}
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
The `inputs` prompt follows VS Code's MCP configuration reference; check it for your VS Code version.
|
|
164
|
+
|
|
165
|
+
### OpenAI Codex CLI
|
|
166
|
+
|
|
167
|
+
```sh
|
|
168
|
+
codex mcp add jev --env JEV_TOOLS_URL=https://your-jev-endpoint.example/judge -- npx -y -p jev-agent-tools jev-agent-tools-mcp --root .
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
Or in `~/.codex/config.toml` (or a trusted project's `.codex/config.toml`). `env_vars` forwards variables from your environment without writing them into the file:
|
|
172
|
+
|
|
173
|
+
```toml
|
|
174
|
+
[mcp_servers.jev]
|
|
175
|
+
command = "npx"
|
|
176
|
+
args = ["-y", "-p", "jev-agent-tools", "jev-agent-tools-mcp", "--root", "."]
|
|
177
|
+
env_vars = ["JEV_TOOLS_URL", "JEV_TOOLS_API_KEY"]
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
Add the instructions to `AGENTS.md` ([template](agent-instructions.md#agentsmd)).
|
|
181
|
+
|
|
182
|
+
### Windsurf and other clients
|
|
183
|
+
|
|
184
|
+
Clients that use an `mcpServers` object accept the Claude Code JSON entry above. Windsurf reads `mcp_config.json` (see its MCP documentation for the current path) and expands `${env:NAME}`.
|
|
185
|
+
|
|
186
|
+
## 4. Verify
|
|
187
|
+
|
|
188
|
+
Run the server once by hand; it reads protocol messages from stdin and writes diagnostics to stderr:
|
|
189
|
+
|
|
190
|
+
```sh
|
|
191
|
+
npx -y -p jev-agent-tools jev-agent-tools-mcp --version
|
|
192
|
+
npx -y -p jev-agent-tools jev-agent-tools-mcp --root . < /dev/null
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
In PowerShell, run the second as `$null | npx -y -p jev-agent-tools jev-agent-tools-mcp --root .`.
|
|
196
|
+
|
|
197
|
+
The second command prints `jev-agent-tools MCP server ready (root ...; endpoint configured)` on stderr and exits when stdin closes. In the client, the six tools should be listed: `jev_ask`, `jev_ask_files`, `jev_find_files`, `jev_locate_in_file`, `jev_check_diff`, `jev_select_tests`. A smoke call with a one-line note and a yes/no question reports execution, evidence context, fresh/cache/static/unjudged items, diagnostics and separate request/result accounting. This verifies integration, not model accuracy.
|
|
198
|
+
|
|
199
|
+
## From a clone
|
|
200
|
+
|
|
201
|
+
```sh
|
|
202
|
+
npm ci --include=optional
|
|
203
|
+
npm run build
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
Then use `"command": "node"` with `"args": ["/absolute/path/to/jev-tools/dist/mcp/main.js", "--root", "<repository>"]`. This form also avoids the Windows `npx` issue. Node does not strip TypeScript types under `node_modules`, which is why the server ships as compiled `dist/` while pi and omp load `src/`.
|
|
207
|
+
|
|
208
|
+
## Differences from pi and omp
|
|
209
|
+
|
|
210
|
+
- **No run-end documentation check.** pi/OMP retain an automatic opt-out host hook (`JEV_TOOLS_AUTO_DOCS=0`). MCP has no hook and its absence does not require manual replacement calls, including risk followed by docs; choose a review only when it can inform an open decision.
|
|
211
|
+
- **Approval is the client's.** `jev_ask` is annotated as not read-only and potentially destructive while it accepts `command`; the other five are read-only. All six are open-world because evidence goes to your endpoint. `JEV_TOOLS_ALLOW_COMMAND=0` removes `command` from the schema and makes `jev_ask` read-only.
|
|
212
|
+
- **Instructions.** The full reading guide and discretionary policy are sent as the server's `instructions`, using the same canonical fragments as pi/OMP with MCP's explicit hook difference. Clients may ignore them, so also copy the versioned [agent instructions](agent-instructions.md) into the project and update the copy on release upgrades.
|
|
213
|
+
- **Tool names in descriptions** refer to "your text search tool" and "your file-name search tool" instead of pi or omp tool names.
|
|
214
|
+
- **Protocol.** Versions 2024-11-05 through 2025-11-25 use `initialize`; 2026-07-28 uses `server/discover` and per-request `_meta`. Versions 2025-06-18 and later advertise `outputSchema` and return `structuredContent.result`; 2024-11-05 and 2025-03-26 receive self-contained text only. Each request retains its negotiated version even if another request changes the session version. Only 2026-07-28 adds `resultType: "complete"` and `ttlMs: 0` / `cacheScope: "private"`; this transport completion is independent of the report's execution state. Modern discovery supplies identity in `_meta["io.modelcontextprotocol/serverInfo"]`. Expected refusals use typed reports and appropriate `isError`; unexpected server exceptions remain JSON-RPC internal errors without invented report accounting. Tools only; no resources, prompts or sampling. Cancelling a call aborts its Jev requests and command.
|
|
215
|
+
- **Malformed arguments.** Input-schema violations return JSON-RPC `INVALID_PARAMS` without a result report or judgment; well-formed calls rejected by root/evidence admission retain their typed refusal report.
|
|
216
|
+
- **One process, one session.** Limits, cache and counters last as long as the connection; restart the server to reset them.
|
|
217
|
+
- **Per-call evidence root.** All six tools accept `root` for a registered worktree of the same repository, with the [shared admission restrictions](../README.md#evidence-root). The startup directory remains the authority; `root` is not permission to access unrelated repositories.
|
|
218
|
+
- **Command shutdown.** Cancellation, command timeout, stdin closure, SIGTERM and SIGINT wait for bounded command-tree termination. On POSIX the managed group receives SIGTERM, then SIGKILL after two seconds if it remains. On Windows, the server maps ordinary MSYS descendants through the same Bash installation's process table before running the system `taskkill /T /F`; each subprocess is bounded to five seconds. Cancelled calls receive no response. This is not a sandbox: descendants that deliberately detach from the managed group or escape the tracked tree are not contained.
|
|
219
|
+
|
|
220
|
+
## Troubleshooting
|
|
221
|
+
|
|
222
|
+
| Symptom | Cause and fix |
|
|
223
|
+
|---|---|
|
|
224
|
+
| Client shows the server failed to start on Windows | `npx` cannot start without a shell. Use `cmd /c npx` or `node <path>/dist/mcp/main.js`. |
|
|
225
|
+
| `jev-tools is not configured` in every result | The server did not receive `JEV_TOOLS_URL` and `JEV_TOOLS_API_KEY`. Check the client's `env` block and that interpolated variables exist where the client was started. |
|
|
226
|
+
| stderr: `Cannot read or save Jev configuration` | Saved `/jev-setup` storage is not private or is malformed. Environment variables still apply. Fix permissions (on Windows, remove other accounts from the folder's Security tab) or delete the file and save again. |
|
|
227
|
+
| `Repository directory not found` | `--root` or `JEV_TOOLS_ROOT` points to a missing directory. |
|
|
228
|
+
| Command evidence reports `Command executable unavailable` | No usable bash. Install Git for Windows or set `JEV_TOOLS_BASH`. |
|
|
229
|
+
| Results from the wrong repository | The client started the server elsewhere. Pass an absolute `--root`. |
|
|
230
|
+
|
|
231
|
+
## Data and safety
|
|
232
|
+
|
|
233
|
+
Repository evidence, notes and command output are sent to your configured endpoint; review its data-handling policy first. File collection is confined to the repository, but `jev_ask` commands run with your shell permissions and no sandbox. See [SECURITY.md](../SECURITY.md).
|
package/docs/tools/jev_ask.md
CHANGED
|
@@ -14,7 +14,9 @@ Use [jev_ask_files](jev_ask_files.md) for independent answers per file. Use nati
|
|
|
14
14
|
|
|
15
15
|
The tool assembles one JSON situation. A note appears as `state`; repository files as `files["path"]`; historical versions as `files_before["path"]`; and command evidence as `output` with `command`, `exit_code`, `timed_out`, `stdout` and `stderr`. You receive answers, not the assembled file contents or captured output. Jev reads the evidence at a glance; it does not count, calculate or infer missing dependencies. Evidence outweighs an explanation of its role.
|
|
16
16
|
|
|
17
|
-
Before judgment, bounded depth-one static import closure adds supported used declarations and data files, including before/after versions when applicable. A
|
|
17
|
+
Before judgment, bounded depth-one static import closure adds supported used declarations and data files, including before/after versions when applicable. A closure diagnostic records additions and gaps; it cannot establish completeness or discover relationships without imports. Explicit `files[...]` and `files_before[...]` selectors name required evidence. Ordinary lexical mentions such as `.only` or `process.env` do not create missing-file vetoes, and names printed in command output are not disk evidence. An exact path with directory segments never falls back to another file with the same basename. For abbreviated references, a uniquely matching supplied file takes precedence over inventory discovery; ambiguity is reported, not guessed.
|
|
18
|
+
|
|
19
|
+
One canonical path stores each file version; lightweight aliases do not duplicate content. Current and base versions remain distinct, and the 20-file limit counts distinct file identities, not their versions. A before-only selector can load historical content without a current file. `null` denotes a proven absent base version, not unavailable evidence. Missing/empty requirements exclude their question group and its controls; a requirement in the global note applies to all groups. Numeric zero and boolean false are not empty evidence.
|
|
18
20
|
|
|
19
21
|
For recognizable assertion failures, the tool attempts to attach the failing test named by the log. An unidentifiable target produces a warning; any bug-versus-wrong-test conclusion under that warning remains unproven. Pass the failing test and implementation yourself whenever possible.
|
|
20
22
|
|
|
@@ -24,10 +26,11 @@ At least one of `state`, `paths` or `command` must provide evidence.
|
|
|
24
26
|
|
|
25
27
|
| Field | Type / default | Meaning |
|
|
26
28
|
|---|---|---|
|
|
27
|
-
| `
|
|
29
|
+
| `root` | Optional nonempty string | Exact initial Git root or registered worktree of the same repository; see [evidence-root admission](../../README.md#evidence-root). |
|
|
30
|
+
| `state` | Optional string, at most 8,000 characters | Short request, plan or observation not shown by the files; not a place to paste files or logs. JSON object text is accepted, but reserved evidence keys such as `files`, `files_before`, `evidence` and `output` cannot be supplied by the note. |
|
|
28
31
|
| `paths` | Optional array of nonempty strings, at most 20 | Repository-relative files to read; not recursive directory or glob triage. |
|
|
29
|
-
| `base` | Optional nonempty Git ref | Add
|
|
30
|
-
| `command` | Optional nonempty string | Execute `bash -c` at the repository root with `CI=1`, normal shell permissions and no additional sandbox. |
|
|
32
|
+
| `base` | Optional nonempty Git ref | Add historical versions for the resolved revision; only established absence is `null`. Use `HEAD` for uncommitted before/after questions. Both versions count toward serialized evidence admission. |
|
|
33
|
+
| `command` | Optional nonempty string | Execute `bash -c` at the repository root with `CI=1` and without `JEV_TOOLS_API_KEY`, normal shell permissions and no additional sandbox. The configured key is replaced with `[redacted]` in captured output, with the count reported. |
|
|
31
34
|
| `timeout_s` | Optional number, default 60, range 1–300 | Command timeout in seconds. |
|
|
32
35
|
| `asks` | Required intent, nonempty intent array, or JSON string encoding them | Typed questions; one intent per judgment. |
|
|
33
36
|
| `max_calls` | Optional non-negative integer | Limit requests including controls and output passage finding; unjudged work is reported. |
|
|
@@ -81,7 +84,7 @@ Verification distinguishes `holds`, `contradicted`, `not addressed by the state`
|
|
|
81
84
|
|
|
82
85
|
Serialized evidence, including keys and escapes, is bounded to 80,000 characters. Oversized files/state are refused with a split suggestion, not silently shortened. Repository file confinement rejects escaping paths and internal URLs.
|
|
83
86
|
|
|
84
|
-
Commands are captured in private temporary files and cleaned up. Captured streams over 64 MiB are refused after execution; this neither caps disk usage nor interrupts execution. Repeated line shapes are compressed; lines over 8,192 characters and more than 2,048 distinct shapes produce visible notices. If compression still cannot fit, bounded passage judgments retain failure details and beginning/end context with omission markers. Excessive passage work is refused with guidance to narrow the command. A timeout gives `exit_code: null` and `timed_out: true`, not success; cancellation stops without judgment. Command calls bypass caching. `JEV_TOOLS_ALLOW_COMMAND=0` disables command evidence. Missing endpoint/key refuses judgment without a chat-model fallback.
|
|
87
|
+
Commands are captured in private temporary files and cleaned up. Captured streams over 64 MiB are refused after execution; this neither caps disk usage nor interrupts execution. Repeated line shapes are compressed; lines over 8,192 characters and more than 2,048 distinct shapes produce visible notices. If compression still cannot fit, bounded passage judgments retain failure details and beginning/end context with omission markers. Command output reaches this stage before immutable file/note budget refusal, including calls with base snapshots and import closure. Excessive passage work is refused with guidance to narrow the command. Reports distinguish not started, finished and unknown execution, preserve observed cwd/exit/timeout even if later processing fails, and do not equate a judgment refusal with an unexecuted command. A timeout gives `exit_code: null` and `timed_out: true`, not success; cancellation stops without judgment. Command calls bypass caching. `JEV_TOOLS_ALLOW_COMMAND=0` disables command evidence. Missing endpoint/key refuses judgment without a chat-model fallback.
|
|
85
88
|
|
|
86
89
|
## Host differences
|
|
87
90
|
|
|
@@ -18,6 +18,7 @@ Each admitted file is judged alone as `path` and `content`. Every ask applies to
|
|
|
18
18
|
|
|
19
19
|
| Field | Type / default | Meaning |
|
|
20
20
|
|---|---|---|
|
|
21
|
+
| `root` | Optional nonempty string | Exact initial Git root or registered worktree of the same repository; see [evidence-root admission](../../README.md#evidence-root). |
|
|
21
22
|
| `paths` | Required nonempty array of nonempty strings | Repository-relative files, recursive directories or globs; at most 255 admitted files. |
|
|
22
23
|
| `asks` | Required typed intent, nonempty array of intents, or JSON string encoding them | All asks apply to every admitted file. An array is the clearest form. |
|
|
23
24
|
| `max_calls` | Optional non-negative integer | Request limit for this invocation; remaining files are listed unchecked. Separate from session budgets. |
|
|
@@ -65,7 +66,7 @@ Illustrative call with fictional repository paths, not a recorded execution:
|
|
|
65
66
|
|
|
66
67
|
## Results and next action
|
|
67
68
|
|
|
68
|
-
Results preserve the exact statement next to each file's answer. Boolean `yes` or `no (not shown)` includes the probability of yes; intermediate probabilities between 0.20 and 0.80 are unsure. Categories and levels need a leading-option probability of at least 0.85 after applicable controls. `classify`, `rate`, `decide` and `free` are uncalibrated; sharing a display band does not establish an error rate. Failed integrity or option-order controls make clear-looking results unsure. Read the candidate file before acting.
|
|
69
|
+
Results preserve the exact statement next to each file's answer and distinguish fresh/cache judgments from unjudged files. Boolean `yes` or `no (not shown)` includes the probability of yes; intermediate probabilities between 0.20 and 0.80 are unsure. Categories and levels need a leading-option probability of at least 0.85 after applicable controls. `classify`, `rate`, `decide` and `free` are uncalibrated; sharing a display band does not establish an error rate. Failed integrity or option-order controls make clear-looking results unsure; missing required judgments never become fabricated probabilities. Read the candidate file before acting. Typed diagnostics and actions explain skipped/unchecked files, causes and scope. See [shared result reading](../../README.md#read-the-results).
|
|
69
70
|
|
|
70
71
|
## Limits and failure behavior
|
|
71
72
|
|
|
@@ -18,6 +18,7 @@ Changed declarations, slices, files or hunks become before/after evidence units.
|
|
|
18
18
|
|
|
19
19
|
| Field | Type / default | Meaning |
|
|
20
20
|
|---|---|---|
|
|
21
|
+
| `root` | Optional nonempty string | Exact initial Git root or registered worktree of the same repository; see [evidence-root admission](../../README.md#evidence-root). |
|
|
21
22
|
| `check` | Required `"risk"`, `"docs"` or `"spec"` | Select the preset below. |
|
|
22
23
|
| `base` | Optional nonempty Git ref, default `HEAD` | Compare the current tree, including untracked files, with this revision. |
|
|
23
24
|
| `spec_path` | Optional nonempty repository-relative string | Required for `spec`: Markdown specification with `### REQ-…` headings. |
|
|
@@ -36,7 +37,7 @@ For replaced member accesses, separate local-caller checks inspect statically re
|
|
|
36
37
|
|
|
37
38
|
### docs
|
|
38
39
|
|
|
39
|
-
Judge up to 40 existing tracked Markdown sections mentioning changed code or importing source, and identify
|
|
40
|
+
Judge up to 40 existing tracked Markdown sections mentioning changed code or importing source, and identify an existing sentence that may no longer match the change. Collection and traversal limits remain material reservations in the report. This is not an exhaustive detector of missing documentation, arbitrary companion edits or every stale sentence; inspect the sentence against the code before changing it.
|
|
40
41
|
|
|
41
42
|
### spec
|
|
42
43
|
|
|
@@ -64,6 +65,8 @@ For another preset, use `{"check":"docs","base":"HEAD"}` or `{"check":"spec","ba
|
|
|
64
65
|
|
|
65
66
|
Findings name the evidence unit, stale sentence or requirement and its probability. Fixed findings require at least 0.7. Docs probabilities from 0.2 up to 0.7 are unsure without an additional judgment; read the indicated wording. Local-caller probabilities between 0.2 and 0.7 are unsure; `cannot_tell` at least 0.3 is abstain naming the missing provider or binding. Failed witness batches remain unsure. Read flagged source, callers or documentation before editing and preserve unresolved uncertainty in reports. See [shared result reading](../../README.md#read-the-results).
|
|
66
67
|
|
|
68
|
+
Structured items retain negative as well as positive review results, actual control values and fresh/cache provenance. Budget-exhausted or incomplete groups are unjudged, not negative findings. Documentation completeness limits, collection omissions and witness failures remain scoped diagnostics even when another part of the review yields usable results.
|
|
69
|
+
|
|
67
70
|
## Limits and failure behavior
|
|
68
71
|
|
|
69
72
|
No findings is not proof of safety, unseen-caller coverage or complete documentation. Unjudged units, callers, sections, requirements and severity are named when request or session budgets stop work. Parser gaps and unresolved dynamic providers remain limits, not invented answers. Missing configuration refuses judgment without a chat-model fallback. Static collection is bounded and does not evaluate third-party code.
|
|
@@ -18,6 +18,7 @@ Deterministic lexical pre-ranking and name judgments produce candidates; the too
|
|
|
18
18
|
|
|
19
19
|
| Field | Type / default | Meaning |
|
|
20
20
|
|---|---|---|
|
|
21
|
+
| `root` | Optional nonempty string | Exact initial Git root or registered worktree of the same repository; see [evidence-root admission](../../README.md#evidence-root). |
|
|
21
22
|
| `goal` | Required nonempty string | Behavioral sentence with at least five content words for an unqualified entry verdict. Shorter goals are accepted but capped at unsure. |
|
|
22
23
|
| `keywords` | Optional array of nonempty strings, default empty | Known identifiers or terms steering lexical ranking and excerpt selection. |
|
|
23
24
|
| `scope` | Optional directory string or array of directory strings, default repository | Restrict search; an incorrect scope may yield none. |
|
|
@@ -44,7 +45,7 @@ Illustrative call with fictional repository paths, not a recorded execution:
|
|
|
44
45
|
|
|
45
46
|
## Results and next action
|
|
46
47
|
|
|
47
|
-
|
|
48
|
+
The report distinguishes the primary entry decision from auxiliary ranking and controls, retains actual fresh/cache provenance and supplies useful read actions. Unsure can mean two plausible entries, order sensitivity, inconclusive excerpts or a short goal: inspect the leading candidates. A judged `none` means no candidate fit the supplied evidence: clarify the behavioral goal or widen the relevant scope. An empty collection, exhausted budget or missing response is not a `none` judgment and carries no invented probability. Paths and probabilities are not source text; read the files before editing. See [shared result reading](../../README.md#read-the-results).
|
|
48
49
|
|
|
49
50
|
## Limits and failure behavior
|
|
50
51
|
|
|
@@ -18,6 +18,7 @@ Code splits the file into declarations or sections and constructs a choice over
|
|
|
18
18
|
|
|
19
19
|
| Field | Type / default | Meaning |
|
|
20
20
|
|---|---|---|
|
|
21
|
+
| `root` | Optional nonempty string | Exact initial Git root or registered worktree of the same repository; see [evidence-root admission](../../README.md#evidence-root). |
|
|
21
22
|
| `path` | Required nonempty string | One repository-relative file, at least 19,000 bytes. |
|
|
22
23
|
| `goal` | Required nonempty string | Sentence describing the behavior you need to find. |
|
|
23
24
|
|
|
@@ -38,10 +39,14 @@ Illustrative call with a fictional repository path, not a recorded execution:
|
|
|
38
39
|
|
|
39
40
|
A result names `path:start-end`, a label, probability and the next read. Above 0.7 an unmarked range is a verdict: read it. From 0.4 through 0.7, unsure lists the two best ranges by probability: read both, not an arbitrary next section. Below 0.4 the tool narrows to three leading candidates plus none and judges again, retaining the original unsure band. Option-order sensitivity can also leave the result unsure. `none` means no supplied section fits: search elsewhere or clarify the goal. See [shared result reading](../../README.md#read-the-results).
|
|
40
41
|
|
|
42
|
+
The versioned report preserves actual primary/control values and their fresh/cache source. Missing required control answers produce unjudged work rather than a synthesized range or probability. Typed diagnostics retain parsing/window omissions and native next actions, including searching elsewhere for a judged `none`.
|
|
43
|
+
|
|
41
44
|
## Limits and failure behavior
|
|
42
45
|
|
|
43
46
|
Small files are refused with direct-read guidance. Whole-file admission is bounded by 320,000 characters and 1,280,000 bytes; larger files take the streamed-window path, not silent whole-file truncation. Window serialization, fallback section sizes, parsing support and choice cardinality are separately bounded. Unreadable or invalid evidence, missing configuration and exhausted session budgets are reported rather than producing a range verdict. Missing optional Python grammar is named with installation guidance. A selected window does not prove every relevant declaration was inspected.
|
|
44
47
|
|
|
48
|
+
The serialized judgment budget includes per-call evidence provenance. Block planning, excerpt allocation and full-text refinement reserve that metadata capacity before selecting evidence; provenance never silently pushes an otherwise planned request past admission. Selected blocks still require at least two sections and obey the choice limit.
|
|
49
|
+
|
|
45
50
|
## Host differences
|
|
46
51
|
|
|
47
52
|
omp `read` already outlines code files: use this tool when that outline does not tell you which range serves your goal. Both hosts use the same judgment protocol; omp marks the tool read-only.
|
|
@@ -20,6 +20,7 @@ Residual exported-unit checks concern changed units not exercised by any discove
|
|
|
20
20
|
|
|
21
21
|
| Field | Type / default | Meaning |
|
|
22
22
|
|---|---|---|
|
|
23
|
+
| `root` | Optional nonempty string | Exact initial Git root or registered worktree of the same repository; see [evidence-root admission](../../README.md#evidence-root). |
|
|
23
24
|
| `base` | Optional nonempty Git ref, default `HEAD` | Compare the current tree with this revision. |
|
|
24
25
|
| `paths` | Optional array of nonempty strings, default discovered inventory | Candidate test files or globs, not changed source paths. |
|
|
25
26
|
| `witnesses` | Optional `"off"`, `"auto"` (default), `"on"` | Controls residual coverage checks; test pointers have no witnesses. |
|
|
@@ -27,6 +28,8 @@ Residual exported-unit checks concern changed units not exercised by any discove
|
|
|
27
28
|
|
|
28
29
|
Unknown fields and unresolved refs are rejected with guidance. Repository-relative evidence paths are confined to the repository; internal URLs are not file inputs. Narrowing candidate paths also narrows the inventory to which coverage observations apply.
|
|
29
30
|
|
|
31
|
+
Each requested criterion reports its matches and any excluded or unmatched inventory. Partial matches keep the matched selection and identify the unmatched criteria; zero matches are not proof of no affected tests. Changed files outside the supported dependency graph retain conservative widening, with the triggering paths named. Runner commands remain static plans and use the admitted root.
|
|
32
|
+
|
|
30
33
|
## Example
|
|
31
34
|
|
|
32
35
|
Illustrative call with fictional repository paths, not a recorded execution:
|
|
@@ -44,7 +47,7 @@ Illustrative call with fictional repository paths, not a recorded execution:
|
|
|
44
47
|
|
|
45
48
|
Inspect selected scenarios and the returned runner commands, then execute them yourself. A scenario is selected when `1 - p(none) >= 0.5`; missing answers are selected too. If at least 80% of a file's scenarios are selected, names are uncertain or counts unknown, the command runs the whole file instead of a fragile name filter. Commands remain separate when runner context differs.
|
|
46
49
|
|
|
47
|
-
|
|
50
|
+
Static selection identifies its deterministic reason separately from fresh/cache Jev judgments. Conservative fallback preserves tests when judgments are unavailable; it is not confirmation that every scenario is affected. Residual absence is a derived reservation linked to retained coverage items or the considered inventory, not an extra requested judgment or fabricated probability. It does not establish repository-wide uncovered behavior. Read material limits and follow the named native action. See [shared result reading](../../README.md#read-the-results).
|
|
48
51
|
|
|
49
52
|
## Limits and failure behavior
|
|
50
53
|
|
package/package.json
CHANGED
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jev-agent-tools",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"license": "MIT",
|
|
6
|
-
"
|
|
6
|
+
"mcpName": "io.github.NomenAK/jev-agent-tools",
|
|
7
|
+
"description": "Six evidence-oriented tools for pi and omp coding agents and any MCP client, compatible with the Jev API format.",
|
|
7
8
|
"repository": {
|
|
8
9
|
"type": "git",
|
|
9
10
|
"url": "git+https://github.com/NomenAK/jev-tools.git"
|
|
@@ -19,7 +20,9 @@
|
|
|
19
20
|
"test-selection",
|
|
20
21
|
"pi",
|
|
21
22
|
"omp",
|
|
22
|
-
"jev"
|
|
23
|
+
"jev",
|
|
24
|
+
"mcp",
|
|
25
|
+
"model-context-protocol"
|
|
23
26
|
],
|
|
24
27
|
"files": [
|
|
25
28
|
"src/",
|
|
@@ -33,8 +36,18 @@
|
|
|
33
36
|
"docs/tools/jev_find_files.md",
|
|
34
37
|
"docs/tools/jev_locate_in_file.md",
|
|
35
38
|
"docs/tools/jev_check_diff.md",
|
|
36
|
-
"docs/tools/jev_select_tests.md"
|
|
39
|
+
"docs/tools/jev_select_tests.md",
|
|
40
|
+
"docs/mcp.md",
|
|
41
|
+
"docs/agent-instructions.md",
|
|
42
|
+
"dist/",
|
|
43
|
+
"server.json",
|
|
44
|
+
"SECURITY.md",
|
|
45
|
+
"CONTRIBUTING.md",
|
|
46
|
+
"docs/adr/"
|
|
37
47
|
],
|
|
48
|
+
"bin": {
|
|
49
|
+
"jev-agent-tools-mcp": "dist/mcp/main.js"
|
|
50
|
+
},
|
|
38
51
|
"pi": {
|
|
39
52
|
"extensions": [
|
|
40
53
|
"./src/index.ts"
|
|
@@ -49,6 +62,8 @@
|
|
|
49
62
|
"node": ">=24"
|
|
50
63
|
},
|
|
51
64
|
"scripts": {
|
|
65
|
+
"build": "tsc -p tsconfig.build.json",
|
|
66
|
+
"prepack": "npm run build",
|
|
52
67
|
"typecheck": "tsc --noEmit",
|
|
53
68
|
"check:imports": "node scripts/check-imports.ts",
|
|
54
69
|
"lint": "biome check src test scripts",
|
package/rules/jev-ask.md
CHANGED
|
@@ -1,4 +1,25 @@
|
|
|
1
1
|
---
|
|
2
2
|
alwaysApply: true
|
|
3
3
|
---
|
|
4
|
-
|
|
4
|
+
<!-- Generated policy 2026-10-03.1 from src/texts/instructions.ts; run node scripts/generate-instructions.ts after editing the source. -->
|
|
5
|
+
Use Jev for a bounded semantic judgment when its answer could change an open decision or focus the next inspection, and the relevant evidence is available. Use decisive reading, search or authorized execution directly when it settles the question. Exact lookup, counting and runtime causality belong to native tools. A Jev call is not a prerequisite for a conclusion, a review or task completion; no explanation is needed for choosing native tools.
|
|
6
|
+
|
|
7
|
+
Before a chosen call, identify the open decision and supply only the evidence needed for a bounded question; a complete prior analysis is not required. Ask about one positive, self-contained observable fact per claim, with its sources, and show both sides of a comparison.
|
|
8
|
+
- Failure: relevant output, the failing test and the implementation it exercises; output alone is a lead, not a bug-versus-wrong-test diagnosis. Include configuration or runtime evidence when the hypothesis depends on it. A static judgment does not establish runtime causality.
|
|
9
|
+
- Change: current evidence and the earlier reference via base. Use the checkout/root corresponding to the work and admitted by the tool; another clone is not the same context.
|
|
10
|
+
- Plan/documentation: precise observable commitments and the passages that constrain them. Local confirmation does not establish that omitted obligations were searched.
|
|
11
|
+
- Selection/review: compatible inventory, references and configuration. Suggested commands are not executed commands; existing tests need not cover a new scenario. Read the diff natively for its contents, not as proof of safety or global coverage.
|
|
12
|
+
|
|
13
|
+
After an unavailable, refused or out-of-scope result, continue natively within the existing permissions; do not widen sharing or confinement to obtain a judgment. For uncertainty or missing evidence, inspect or obtain the decisive piece, or leave the conclusion open. Revisit Jev only when new evidence, a material context change or a new useful question makes the judgment useful; rewording unchanged evidence is not a reason to retry. There is no retry quota for genuinely changed evidence, and no certification is needed once decisive evidence settles the question. A reached cap or an unjudged result is not evidence of safety or zero affected tests.
|
|
14
|
+
|
|
15
|
+
jev_check_diff: Review a stable diff for semantic risks, stale documentation or specification drift when that review can inform an open decision. Read the diff natively when you need its contents. Choose risk, docs or spec for the question at hand; neither risk followed by docs nor a Jev review before done is required. Findings are leads within the inspected scope, not proof of global safety or completeness.
|
|
16
|
+
|
|
17
|
+
Report current conclusions first, then decisive evidence, origin and scope, then material reservations. Established means supported by relevant decisive evidence; a reported check remains explicitly reported, not observed execution. Distinguish native reading/execution, Jev's static judgment and testimony. A call count or global status is not proof.
|
|
18
|
+
If native evidence settles the same scope and context after a Jev uncertainty, attribute the current conclusion to that evidence; old unsure or abstain results need not be recited as current reservations. A later Jev confirmation is attributed to Jev and retains its static limits. Independent reservations survive either resolution.
|
|
19
|
+
Where an uncertainty actually returned by Jev remains material, explicitly say “Jev did not confirm X”, identify the missing evidence or guarantee, its known or indeterminate impact and the evidence to obtain. Preserve that reservation in the main text, summary and recommendation. Name native reservations and unjudged work as such, not as Jev uncertainty. Unresolved contradiction, stale evidence or a different checkout/base/scenario remains a reservation; the latest favorable answer does not win by default.
|
|
20
|
+
Keep material limits in the main text, not only behind a link. Do not generalize a checked scenario into a universal guarantee. Separate relevant unestablished hypotheses from observations; omit irrelevant hypotheses. Explain unavailable historical causes only when material to action or requested, without inventing causality.
|
|
21
|
+
Use existing history references when useful or requested; no exhaustive historical relay, persistent register or History block is required. A requested audit details available steps separately from the current state and does not reopen settled conclusions. Preserve existing traces; identify unavailable traces when they limit evidence or audit, without reconstructing evidence, executions or call counts. pi/OMP may reference an existing trace; MCP references must be client-accessible or explicitly unavailable.
|
|
22
|
+
|
|
23
|
+
pi and OMP retain an automatic, opt-out documentation check at task completion (JEV_TOOLS_AUTO_DOCS=0 disables it). It is a host feature, not an agent requirement to call Jev. Inspect any flagged sentence against the change and update it or explain with evidence why it remains correct. The check is limited and does not certify documentation completeness or demonstrated usefulness. A flagged sentence does not require a second call or recursive reviews.
|
|
24
|
+
|
|
25
|
+
Evidence passed to jev_* tools leaves the machine for the configured endpoint. Review its data handling; share only authorized evidence, never secrets or credentials. Host/client approvals still apply. jev_ask commands run with normal shell permissions and no sandbox; prefer read-only commands. Tool choice does not relax quality, proof, approval or confidentiality obligations.
|
package/server.json
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json",
|
|
3
|
+
"name": "io.github.NomenAK/jev-agent-tools",
|
|
4
|
+
"title": "Jev agent tools",
|
|
5
|
+
"description": "Evidence-oriented code questions, diff review and test selection via a Jev endpoint.",
|
|
6
|
+
"version": "0.3.0",
|
|
7
|
+
"repository": {
|
|
8
|
+
"url": "https://github.com/NomenAK/jev-tools",
|
|
9
|
+
"source": "github"
|
|
10
|
+
},
|
|
11
|
+
"websiteUrl": "https://github.com/NomenAK/jev-tools/blob/main/docs/mcp.md",
|
|
12
|
+
"packages": [
|
|
13
|
+
{
|
|
14
|
+
"registryType": "npm",
|
|
15
|
+
"registryBaseUrl": "https://registry.npmjs.org",
|
|
16
|
+
"identifier": "jev-agent-tools",
|
|
17
|
+
"version": "0.3.0",
|
|
18
|
+
"runtimeHint": "npx",
|
|
19
|
+
"transport": {
|
|
20
|
+
"type": "stdio"
|
|
21
|
+
},
|
|
22
|
+
"packageArguments": [
|
|
23
|
+
{
|
|
24
|
+
"type": "named",
|
|
25
|
+
"name": "--root",
|
|
26
|
+
"description": "Repository directory the tools work in. Defaults to the directory the client starts the server in.",
|
|
27
|
+
"format": "filepath",
|
|
28
|
+
"isRequired": false
|
|
29
|
+
}
|
|
30
|
+
],
|
|
31
|
+
"environmentVariables": [
|
|
32
|
+
{
|
|
33
|
+
"name": "JEV_TOOLS_URL",
|
|
34
|
+
"description": "Complete endpoint URL compatible with the Jev API format.",
|
|
35
|
+
"isRequired": true
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"name": "JEV_TOOLS_API_KEY",
|
|
39
|
+
"description": "Bearer credential for the endpoint.",
|
|
40
|
+
"isRequired": true,
|
|
41
|
+
"isSecret": true
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
"name": "JEV_TOOLS_MODEL",
|
|
45
|
+
"description": "Requested Jev model.",
|
|
46
|
+
"default": "openjev",
|
|
47
|
+
"isRequired": false
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
"name": "JEV_TOOLS_ALLOW_COMMAND",
|
|
51
|
+
"description": "Set to 0 to remove shell commands from jev_ask.",
|
|
52
|
+
"isRequired": false
|
|
53
|
+
}
|
|
54
|
+
]
|
|
55
|
+
}
|
|
56
|
+
]
|
|
57
|
+
}
|