jev-agent-tools 0.1.4 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +106 -1
- package/CONTRIBUTING.md +43 -0
- package/README.md +58 -17
- package/SECURITY.md +43 -0
- package/dist/adapters/analysis-context.js +75 -0
- package/dist/adapters/ask-files.js +198 -0
- package/dist/adapters/ask-proof.js +200 -0
- package/dist/adapters/ask-syntax.js +385 -0
- package/dist/adapters/canonical-path.js +17 -0
- package/dist/adapters/command.js +234 -0
- package/dist/adapters/docs.js +192 -0
- package/dist/adapters/evidence-context.js +119 -0
- package/dist/adapters/exec.js +207 -0
- package/dist/adapters/files.js +418 -0
- package/dist/adapters/find.js +150 -0
- package/dist/adapters/git-base.js +32 -0
- package/dist/adapters/git-inventory.js +71 -0
- package/dist/adapters/git.js +483 -0
- package/dist/adapters/locate-file.js +197 -0
- package/dist/adapters/output-lines.js +46 -0
- package/dist/adapters/private-storage.js +106 -0
- package/dist/adapters/risk-callers.js +429 -0
- package/dist/adapters/runner-version.js +78 -0
- package/dist/adapters/shell.js +92 -0
- package/dist/adapters/syntax.js +187 -0
- package/dist/adapters/test-inventory.js +139 -0
- package/dist/adapters/usage.js +20 -0
- package/dist/adapters/utf8.js +47 -0
- package/dist/configuration.js +267 -0
- package/dist/constants.js +140 -0
- package/dist/core/ask-closure.js +282 -0
- package/dist/core/ask-proof.js +1 -0
- package/dist/core/ask-references.js +278 -0
- package/dist/core/asks.js +507 -0
- package/dist/core/batches.js +65 -0
- package/dist/core/command-output.js +224 -0
- package/dist/core/diff.js +178 -0
- package/dist/core/docs.js +302 -0
- package/dist/core/find.js +108 -0
- package/dist/core/git.js +1 -0
- package/dist/core/imports.js +550 -0
- package/dist/core/integrity.js +45 -0
- package/dist/core/lexical.js +132 -0
- package/dist/core/locate.js +169 -0
- package/dist/core/output.js +137 -0
- package/dist/core/pointer.js +29 -0
- package/dist/core/result-report.js +302 -0
- package/dist/core/risk-callers.js +851 -0
- package/dist/core/runner-version.js +45 -0
- package/dist/core/secret-path.js +34 -0
- package/dist/core/sections.js +230 -0
- package/dist/core/state.js +51 -0
- package/dist/core/syntax.js +1 -0
- package/dist/core/test-commands.js +334 -0
- package/dist/core/test-coverage.js +74 -0
- package/dist/core/test-discovery.js +1382 -0
- package/dist/core/test-evidence.js +527 -0
- package/dist/core/test-state.js +81 -0
- package/dist/core/truncate.js +12 -0
- package/dist/core/units.js +349 -0
- package/dist/describe.js +23 -0
- package/dist/guide.js +33 -0
- package/dist/host.js +24 -0
- package/dist/jev/client.js +456 -0
- package/dist/jev/pool.js +54 -0
- package/dist/jev/types.js +1 -0
- package/dist/mcp/main.js +124 -0
- package/dist/mcp/protocol.js +210 -0
- package/dist/mcp/tools.js +129 -0
- package/dist/presets/docs.js +62 -0
- package/dist/presets/risk.js +179 -0
- package/dist/presets/spec.js +81 -0
- package/dist/presets/witnesses.js +249 -0
- package/dist/render.js +114 -0
- package/dist/report-schema.js +1356 -0
- package/dist/result-types.js +1 -0
- package/dist/result.js +3 -0
- package/dist/runtime.js +1 -0
- package/dist/session.js +147 -0
- package/dist/texts/ask-files.js +3 -0
- package/dist/texts/ask.js +4 -0
- package/dist/texts/check-diff.js +20 -0
- package/dist/texts/configuration.js +1 -0
- package/dist/texts/find.js +19 -0
- package/dist/texts/guide.js +3 -0
- package/dist/texts/instructions.js +72 -0
- package/dist/texts/locate.js +15 -0
- package/dist/texts/select-tests.js +4 -0
- package/dist/tools/ask-files.js +450 -0
- package/dist/tools/ask-schema.js +70 -0
- package/dist/tools/ask.js +1147 -0
- package/dist/tools/check-diff.js +594 -0
- package/dist/tools/docs-check.js +408 -0
- package/dist/tools/find.js +682 -0
- package/dist/tools/locate.js +602 -0
- package/dist/tools/review-report.js +230 -0
- package/dist/tools/select-tests.js +821 -0
- package/dist/tools/spec-check.js +263 -0
- package/docs/adr/0001-strict-typescript-pure-core-offline-tests.md +31 -0
- package/docs/adr/0002-one-http-protocol-across-hosts.md +17 -0
- package/docs/adr/0003-explicit-scope-conservative-automation.md +19 -0
- package/docs/adr/0004-compiled-typed-intents.md +19 -0
- package/docs/adr/0005-evidence-construction-before-judgment.md +19 -0
- package/docs/adr/0006-visible-uncertainty-constrained-controls.md +21 -0
- package/docs/adr/0007-bounded-evidence-visible-limits.md +21 -0
- package/docs/adr/0008-static-test-discovery-conservative-plans.md +19 -0
- package/docs/adr/0009-session-cache-requested-model-identity.md +17 -0
- package/docs/adr/0010-mcp-server-thin-host.md +23 -0
- package/docs/agent-instructions.md +120 -0
- package/docs/design.md +16 -4
- package/docs/mcp.md +233 -0
- package/docs/tools/jev_ask.md +8 -5
- package/docs/tools/jev_ask_files.md +2 -1
- package/docs/tools/jev_check_diff.md +4 -1
- package/docs/tools/jev_find_files.md +2 -1
- package/docs/tools/jev_locate_in_file.md +5 -0
- package/docs/tools/jev_select_tests.md +4 -1
- package/package.json +19 -4
- package/rules/jev-ask.md +22 -1
- package/server.json +57 -0
- package/src/adapters/ask-files.ts +11 -3
- package/src/adapters/ask-proof.ts +69 -11
- package/src/adapters/canonical-path.ts +18 -0
- package/src/adapters/command.ts +102 -36
- package/src/adapters/docs.ts +33 -14
- package/src/adapters/evidence-context.ts +169 -0
- package/src/adapters/exec.ts +226 -0
- package/src/adapters/files.ts +146 -16
- package/src/adapters/find.ts +37 -7
- package/src/adapters/git-base.ts +7 -1
- package/src/adapters/git.ts +61 -8
- package/src/adapters/locate-file.ts +51 -9
- package/src/adapters/private-storage.ts +155 -0
- package/src/adapters/risk-callers.ts +7 -2
- package/src/adapters/shell.ts +113 -0
- package/src/adapters/test-inventory.ts +12 -4
- package/src/configuration.ts +55 -14
- package/src/constants.ts +37 -5
- package/src/core/ask-references.ts +262 -146
- package/src/core/asks.ts +79 -7
- package/src/core/command-output.ts +17 -1
- package/src/core/import-boundaries.ts +8 -3
- package/src/core/locate.ts +8 -5
- package/src/core/output.ts +34 -0
- package/src/core/result-report.ts +410 -0
- package/src/core/secret-path.ts +37 -0
- package/src/core/state.ts +8 -1
- package/src/core/units.ts +3 -2
- package/src/host.ts +11 -0
- package/src/index.ts +3 -0
- package/src/jev/client.ts +66 -16
- package/src/jev/types.ts +24 -3
- package/src/mcp/main.ts +135 -0
- package/src/mcp/protocol.ts +332 -0
- package/src/mcp/tools.ts +179 -0
- package/src/render.ts +109 -0
- package/src/report-schema.ts +1380 -0
- package/src/result-types.ts +234 -0
- package/src/result.ts +4 -1
- package/src/runtime.ts +6 -0
- package/src/session.ts +59 -0
- package/src/setup.ts +13 -5
- package/src/texts/ask-files.ts +4 -1
- package/src/texts/ask.ts +8 -1
- package/src/texts/check-diff.ts +7 -4
- package/src/texts/find.ts +8 -2
- package/src/texts/guide.ts +8 -16
- package/src/texts/instructions.ts +98 -0
- package/src/texts/locate.ts +8 -2
- package/src/texts/run-end.ts +2 -2
- package/src/texts/select-tests.ts +4 -1
- package/src/tools/ask-files.ts +311 -18
- package/src/tools/ask.ts +722 -95
- package/src/tools/check-diff.ts +337 -31
- package/src/tools/docs-check.ts +241 -38
- package/src/tools/find.ts +389 -29
- package/src/tools/locate.ts +387 -25
- package/src/tools/review-report.ts +308 -0
- package/src/tools/select-tests.ts +484 -23
- package/src/tools/spec-check.ts +194 -19
|
@@ -0,0 +1,594 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import { Type } from "@sinclair/typebox";
|
|
3
|
+
import { createAnalysisContext } from "../adapters/analysis-context.js";
|
|
4
|
+
import { resolveEvidenceContext, withEvidenceContext, } from "../adapters/evidence-context.js";
|
|
5
|
+
import { collectUnits } from "../adapters/git.js";
|
|
6
|
+
import { resolveBase } from "../adapters/git-base.js";
|
|
7
|
+
import { shareGitInventory } from "../adapters/git-inventory.js";
|
|
8
|
+
import { collectRiskCallers } from "../adapters/risk-callers.js";
|
|
9
|
+
import { hostUsage } from "../adapters/usage.js";
|
|
10
|
+
import { BAND_BOOL_GRAY_A, CALLER_DISPLAY_MAX_SPANS, CANNOT_TELL_MIN, FLAG_MIN, STATE_MAX_CHARS, TIMEOUT_MS, } from "../constants.js";
|
|
11
|
+
import { prepareBatches } from "../core/batches.js";
|
|
12
|
+
import { isTestFile } from "../core/diff.js";
|
|
13
|
+
import { buildEnvelope, } from "../core/output.js";
|
|
14
|
+
import { known, } from "../core/result-report.js";
|
|
15
|
+
import { evaluateWitnessHealth, prepareRiskMatrix, prepareRiskSeverity, } from "../presets/risk.js";
|
|
16
|
+
import { renderResultReport } from "../render.js";
|
|
17
|
+
import { CHECK_DIFF_DESCRIPTION } from "../texts/check-diff.js";
|
|
18
|
+
import { NOT_CONFIGURED } from "../texts/configuration.js";
|
|
19
|
+
import { CHECK_DIFF_GUIDELINE } from "../texts/instructions.js";
|
|
20
|
+
import { runDocsCheck } from "./docs-check.js";
|
|
21
|
+
import { controlsFor, ReviewReport, reportMetrics } from "./review-report.js";
|
|
22
|
+
import { runSpecCheck } from "./spec-check.js";
|
|
23
|
+
export const checkDiffParameters = Type.Object({
|
|
24
|
+
root: Type.Optional(Type.String()),
|
|
25
|
+
check: Type.Union([
|
|
26
|
+
Type.Literal("risk"),
|
|
27
|
+
Type.Literal("docs"),
|
|
28
|
+
Type.Literal("spec"),
|
|
29
|
+
]),
|
|
30
|
+
base: Type.Optional(Type.String({ minLength: 1 })),
|
|
31
|
+
spec_path: Type.Optional(Type.String({ minLength: 1 })),
|
|
32
|
+
witnesses: Type.Optional(Type.Union([
|
|
33
|
+
Type.Literal("off"),
|
|
34
|
+
Type.Literal("auto"),
|
|
35
|
+
Type.Literal("on"),
|
|
36
|
+
])),
|
|
37
|
+
dimensions: Type.Optional(Type.Record(Type.String(), Type.String({ minLength: 1 }))),
|
|
38
|
+
only: Type.Optional(Type.Array(Type.String({ minLength: 1 }))),
|
|
39
|
+
max_calls: Type.Optional(Type.Integer({ minimum: 0 })),
|
|
40
|
+
}, { additionalProperties: false });
|
|
41
|
+
function unitLabel(unit) {
|
|
42
|
+
const range = unit.afterRange ?? unit.beforeRange;
|
|
43
|
+
const fingerprint = createHash("sha256")
|
|
44
|
+
.update(JSON.stringify([unit.file, unit.name, unit.before, unit.after]))
|
|
45
|
+
.digest("hex")
|
|
46
|
+
.slice(0, 8);
|
|
47
|
+
return `${unit.file}${range ? `:${range.start}-${range.end}` : ""} ${unit.name} [${fingerprint}]`;
|
|
48
|
+
}
|
|
49
|
+
export function createCheckDiffTool(dependencies) {
|
|
50
|
+
const { host, runtime, exec: execute } = dependencies;
|
|
51
|
+
return {
|
|
52
|
+
name: "jev_check_diff",
|
|
53
|
+
label: "Jev check diff",
|
|
54
|
+
description: CHECK_DIFF_DESCRIPTION,
|
|
55
|
+
parameters: checkDiffParameters,
|
|
56
|
+
...(host.isOmp
|
|
57
|
+
? { approval: "read", loadMode: "essential" }
|
|
58
|
+
: {
|
|
59
|
+
promptSnippet: "Check the uncommitted diff for risky changes, stale docs or spec drift",
|
|
60
|
+
promptGuidelines: [CHECK_DIFF_GUIDELINE],
|
|
61
|
+
}),
|
|
62
|
+
async execute(_id, args, signal, _update, ctx) {
|
|
63
|
+
const client = dependencies.client;
|
|
64
|
+
const evidence = await resolveEvidenceContext(ctx.cwd, args.root, {
|
|
65
|
+
exec: execute,
|
|
66
|
+
signal,
|
|
67
|
+
origin: dependencies.evidenceOrigin,
|
|
68
|
+
});
|
|
69
|
+
const evidenceContext = evidence.context;
|
|
70
|
+
const report = new ReviewReport();
|
|
71
|
+
const cwd = evidence.ok ? evidence.cwd : evidenceContext.authority.path;
|
|
72
|
+
const exec = shareGitInventory(execute);
|
|
73
|
+
if (!evidence.ok) {
|
|
74
|
+
const envelope = buildEnvelope({
|
|
75
|
+
refusal: evidence.error,
|
|
76
|
+
yield: {
|
|
77
|
+
calls: 0,
|
|
78
|
+
questions: 0,
|
|
79
|
+
cacheHits: 0,
|
|
80
|
+
cacheRequests: 0,
|
|
81
|
+
elapsedMs: 0,
|
|
82
|
+
},
|
|
83
|
+
});
|
|
84
|
+
runtime.session.record(envelope);
|
|
85
|
+
runtime.guide.deliver(ctx);
|
|
86
|
+
report.refusal = true;
|
|
87
|
+
report.diagnose(evidence.cause, evidence.error, args.root);
|
|
88
|
+
const result = report.build("jev_check_diff", evidenceContext, reportMetrics(envelope));
|
|
89
|
+
return {
|
|
90
|
+
content: [
|
|
91
|
+
{
|
|
92
|
+
type: "text",
|
|
93
|
+
text: renderResultReport(result, { details: envelope }),
|
|
94
|
+
},
|
|
95
|
+
],
|
|
96
|
+
details: {
|
|
97
|
+
ok: false,
|
|
98
|
+
envelope,
|
|
99
|
+
judgments: [],
|
|
100
|
+
callerProofs: [],
|
|
101
|
+
evidenceContext,
|
|
102
|
+
result,
|
|
103
|
+
},
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
evidenceContext.requestedBase = args.base ?? "HEAD";
|
|
107
|
+
if (args.check === "docs" || args.check === "spec") {
|
|
108
|
+
const deps = { client, host, runtime, exec };
|
|
109
|
+
const result = args.check === "docs"
|
|
110
|
+
? await runDocsCheck(deps, {
|
|
111
|
+
cwd: cwd,
|
|
112
|
+
base: args.base,
|
|
113
|
+
signal,
|
|
114
|
+
maxCalls: args.max_calls,
|
|
115
|
+
evidenceContext,
|
|
116
|
+
})
|
|
117
|
+
: await runSpecCheck(deps, {
|
|
118
|
+
cwd: cwd,
|
|
119
|
+
base: args.base,
|
|
120
|
+
specPath: args.spec_path,
|
|
121
|
+
signal,
|
|
122
|
+
maxCalls: args.max_calls,
|
|
123
|
+
evidenceContext,
|
|
124
|
+
});
|
|
125
|
+
runtime.session.record(result.envelope);
|
|
126
|
+
runtime.guide.deliver(ctx);
|
|
127
|
+
const usage = result.judgments.reduce((sum, judgment) => ({
|
|
128
|
+
inputTokens: sum.inputTokens + (judgment.usage?.inputTokens ?? 0),
|
|
129
|
+
costUsd: sum.costUsd + (judgment.usage?.costUsd ?? 0),
|
|
130
|
+
}), { inputTokens: 0, costUsd: 0 });
|
|
131
|
+
return {
|
|
132
|
+
content: [
|
|
133
|
+
{
|
|
134
|
+
type: "text",
|
|
135
|
+
text: renderResultReport(result.result, {
|
|
136
|
+
details: result.envelope,
|
|
137
|
+
}),
|
|
138
|
+
},
|
|
139
|
+
],
|
|
140
|
+
details: { ...result, callerProofs: [], evidenceContext },
|
|
141
|
+
...hostUsage(host.isOmp, usage),
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
const started = performance.now();
|
|
145
|
+
const judgments = [];
|
|
146
|
+
const callerProofs = [];
|
|
147
|
+
const answers = [];
|
|
148
|
+
const unitOrder = new Map();
|
|
149
|
+
const limitations = [];
|
|
150
|
+
const unchecked = [];
|
|
151
|
+
let budget;
|
|
152
|
+
let sent = 0;
|
|
153
|
+
let matrixHealthy = true;
|
|
154
|
+
let emptyBase;
|
|
155
|
+
const finish = (refusal, cause) => {
|
|
156
|
+
answers.sort((a, b) => (unitOrder.get(a) ?? "").localeCompare(unitOrder.get(b) ?? ""));
|
|
157
|
+
const uniqueLimitations = [
|
|
158
|
+
...new Map(limitations.map((limit) => [`${limit.fact}\0${limit.next}`, limit])).values(),
|
|
159
|
+
];
|
|
160
|
+
const usage = judgments.reduce((sum, value) => ({
|
|
161
|
+
inputTokens: sum.inputTokens + (value.usage?.inputTokens ?? 0),
|
|
162
|
+
costUsd: sum.costUsd + (value.usage?.costUsd ?? 0),
|
|
163
|
+
}), { inputTokens: 0, costUsd: 0 });
|
|
164
|
+
const noFindings = !refusal &&
|
|
165
|
+
matrixHealthy &&
|
|
166
|
+
!answers.length &&
|
|
167
|
+
!unchecked.length &&
|
|
168
|
+
report.items.size > 0 &&
|
|
169
|
+
[...report.items.values()].every((item) => item.treatment === "judged");
|
|
170
|
+
const envelope = buildEnvelope({
|
|
171
|
+
answers,
|
|
172
|
+
limitations: emptyBase !== undefined && !refusal
|
|
173
|
+
? [
|
|
174
|
+
...uniqueLimitations,
|
|
175
|
+
{
|
|
176
|
+
fact: `no changed units against ${emptyBase}`,
|
|
177
|
+
next: "nothing judged; pass base= or check the working directory",
|
|
178
|
+
},
|
|
179
|
+
]
|
|
180
|
+
: uniqueLimitations,
|
|
181
|
+
unchecked: [...new Set(unchecked)],
|
|
182
|
+
budget,
|
|
183
|
+
...(refusal ? { refusal } : {}),
|
|
184
|
+
...(noFindings && emptyBase === undefined
|
|
185
|
+
? {
|
|
186
|
+
lines: [
|
|
187
|
+
{
|
|
188
|
+
type: "list",
|
|
189
|
+
title: "risk: no findings",
|
|
190
|
+
items: [],
|
|
191
|
+
},
|
|
192
|
+
],
|
|
193
|
+
}
|
|
194
|
+
: {}),
|
|
195
|
+
yield: {
|
|
196
|
+
calls: judgments.reduce((n, j) => n + (j.calls ?? 0), 0),
|
|
197
|
+
questions: judgments.reduce((n, j) => n + (j.questions ?? 0), 0),
|
|
198
|
+
costUsd: judgments.every((j) => j.usage !== undefined) &&
|
|
199
|
+
judgments.length > 0
|
|
200
|
+
? usage.costUsd
|
|
201
|
+
: undefined,
|
|
202
|
+
cacheHits: judgments.reduce((n, j) => n + (j.cacheHits ?? 0), 0),
|
|
203
|
+
cacheRequests: judgments.reduce((n, j) => n + (j.cacheRequests ?? 0), 0),
|
|
204
|
+
elapsedMs: performance.now() - started,
|
|
205
|
+
},
|
|
206
|
+
});
|
|
207
|
+
if (emptyBase !== undefined)
|
|
208
|
+
report.diagnose("no_changed_units", `No changed units against ${emptyBase}; no risk judgment requested.`, emptyBase, [], false);
|
|
209
|
+
if (refusal &&
|
|
210
|
+
!report.diagnostics.some((d) => d.effect === "blocking")) {
|
|
211
|
+
report.refusal = refusal !== NOT_CONFIGURED;
|
|
212
|
+
report.diagnose(cause ?? "internal_error", refusal);
|
|
213
|
+
}
|
|
214
|
+
const result = report.build("jev_check_diff", evidenceContext, reportMetrics(envelope));
|
|
215
|
+
runtime.session.record(envelope);
|
|
216
|
+
runtime.guide.deliver(ctx);
|
|
217
|
+
return {
|
|
218
|
+
content: [
|
|
219
|
+
{
|
|
220
|
+
type: "text",
|
|
221
|
+
text: renderResultReport(result, { details: envelope }),
|
|
222
|
+
},
|
|
223
|
+
],
|
|
224
|
+
details: {
|
|
225
|
+
evidenceContext,
|
|
226
|
+
result,
|
|
227
|
+
ok: !refusal,
|
|
228
|
+
envelope,
|
|
229
|
+
judgments,
|
|
230
|
+
callerProofs,
|
|
231
|
+
},
|
|
232
|
+
...hostUsage(host.isOmp, usage),
|
|
233
|
+
};
|
|
234
|
+
};
|
|
235
|
+
if (!client)
|
|
236
|
+
return finish(NOT_CONFIGURED, "not_configured");
|
|
237
|
+
const comparison = await resolveBase(exec, cwd, args.base, signal);
|
|
238
|
+
if (!comparison.ok)
|
|
239
|
+
return finish(comparison.error, comparison.cause ?? "invalid_base");
|
|
240
|
+
evidenceContext.resolvedBase = comparison.base;
|
|
241
|
+
const base = comparison.base;
|
|
242
|
+
const analysis = await createAnalysisContext();
|
|
243
|
+
const collected = await collectUnits(exec, {
|
|
244
|
+
cwd: cwd,
|
|
245
|
+
base,
|
|
246
|
+
signal,
|
|
247
|
+
}, analysis.parser);
|
|
248
|
+
if (!collected.ok)
|
|
249
|
+
return finish(collected.error, collected.cause ?? "git_failure");
|
|
250
|
+
const repositoryRoot = await exec("git", ["rev-parse", "--show-toplevel"], { cwd, timeout: TIMEOUT_MS, signal });
|
|
251
|
+
if (!repositoryRoot.code &&
|
|
252
|
+
!repositoryRoot.killed &&
|
|
253
|
+
evidenceContext.effectiveRoot)
|
|
254
|
+
evidenceContext.effectiveRoot.path = repositoryRoot.stdout.trim();
|
|
255
|
+
report.inventories.push({
|
|
256
|
+
id: "changed-units",
|
|
257
|
+
kind: "units",
|
|
258
|
+
rules: [
|
|
259
|
+
"Changed source units against resolved base; tests are evidence, not risk units",
|
|
260
|
+
],
|
|
261
|
+
restrictions: args.only ?? [],
|
|
262
|
+
discovered: known(collected.units.length),
|
|
263
|
+
considered: known(collected.units.length),
|
|
264
|
+
scopeRestricted: Boolean(args.only?.length),
|
|
265
|
+
criteria: [],
|
|
266
|
+
});
|
|
267
|
+
for (const unit of collected.units)
|
|
268
|
+
report.expect(`unit:${unit.id}`, unitLabel(unit), "unit");
|
|
269
|
+
for (const limit of collected.limits)
|
|
270
|
+
report.diagnose(limit.kind === "secret_pattern"
|
|
271
|
+
? "secret_pattern"
|
|
272
|
+
: "collection_omitted", `${limit.file}: ${limit.kind}`, limit.file);
|
|
273
|
+
for (const limit of collected.limits)
|
|
274
|
+
limitations.push({
|
|
275
|
+
fact: `${limit.file}: ${limit.kind}`,
|
|
276
|
+
next: "read the complete before/after source or rerun with base= a nearer ref",
|
|
277
|
+
});
|
|
278
|
+
const readableUnits = collected.units.filter((unit) => {
|
|
279
|
+
if (unit.before !== null || unit.after !== null)
|
|
280
|
+
return true;
|
|
281
|
+
unchecked.push(`${unitLabel(unit)} source unavailable`);
|
|
282
|
+
report.diagnose("binary_or_non_utf8", "Changed source unavailable", unit.file, [`unit:${unit.id}`]);
|
|
283
|
+
return false;
|
|
284
|
+
});
|
|
285
|
+
if (!collected.units.length) {
|
|
286
|
+
emptyBase = base;
|
|
287
|
+
return finish();
|
|
288
|
+
}
|
|
289
|
+
if (!readableUnits.length)
|
|
290
|
+
return finish();
|
|
291
|
+
const matrix = prepareRiskMatrix(readableUnits, {
|
|
292
|
+
...args,
|
|
293
|
+
testFilesPresent: collected.files.some((file) => isTestFile(file.path)),
|
|
294
|
+
});
|
|
295
|
+
if (!matrix.ok)
|
|
296
|
+
return finish(matrix.error, "invalid_arguments");
|
|
297
|
+
const prepared = matrix;
|
|
298
|
+
prepared.state = withEvidenceContext(prepared.state, evidenceContext);
|
|
299
|
+
for (const unit of readableUnits)
|
|
300
|
+
report.items.delete(`unit:${unit.id}`);
|
|
301
|
+
for (const cell of prepared.cells)
|
|
302
|
+
report.expect(`risk:${cell.id}`, `${cell.unitId} ${cell.dimension}`, "unit", `risk:${cell.unitId}`);
|
|
303
|
+
for (const warning of prepared.warnings) {
|
|
304
|
+
limitations.push(warning);
|
|
305
|
+
report.diagnose("unsupported_syntax", warning.fact, undefined, [], false);
|
|
306
|
+
}
|
|
307
|
+
if (JSON.stringify(prepared.state).length > STATE_MAX_CHARS)
|
|
308
|
+
return finish(`risk state exceeds STATE_MAX_CHARS=${STATE_MAX_CHARS}; rerun with base= a nearer ref`, "evidence_too_large");
|
|
309
|
+
const batches = prepareBatches(prepared.state, prepared.questions, {
|
|
310
|
+
groups: prepared.groups,
|
|
311
|
+
witnesses: prepared.witnessQuestionIds,
|
|
312
|
+
});
|
|
313
|
+
if (!batches.ok)
|
|
314
|
+
return finish(batches.error, "group_too_large");
|
|
315
|
+
let reservedMatrixCalls = batches.batches.length;
|
|
316
|
+
const optionsFor = (matrixRequest = false) => ({
|
|
317
|
+
signal,
|
|
318
|
+
...runtime.session.requestGate(),
|
|
319
|
+
admissionCause: () => budget?.kind === "session" ? "session_budget" : "call_budget",
|
|
320
|
+
beforeRequest(questionCount) {
|
|
321
|
+
if (matrixRequest && reservedMatrixCalls > 0)
|
|
322
|
+
reservedMatrixCalls--;
|
|
323
|
+
const limit = args.max_calls === undefined
|
|
324
|
+
? undefined
|
|
325
|
+
: args.max_calls - (matrixRequest ? 0 : reservedMatrixCalls);
|
|
326
|
+
if (limit !== undefined && sent >= limit) {
|
|
327
|
+
budget = {
|
|
328
|
+
kind: "max_calls",
|
|
329
|
+
message: `max_calls=${args.max_calls} reached`,
|
|
330
|
+
};
|
|
331
|
+
return { ok: false, error: budget.message };
|
|
332
|
+
}
|
|
333
|
+
const admitted = runtime.session.admit(questionCount);
|
|
334
|
+
if (!admitted.ok) {
|
|
335
|
+
budget = { kind: "session", message: admitted.error };
|
|
336
|
+
return admitted;
|
|
337
|
+
}
|
|
338
|
+
sent++;
|
|
339
|
+
return admitted;
|
|
340
|
+
},
|
|
341
|
+
onUsage: (usage) => runtime.session.recordUsage(usage),
|
|
342
|
+
});
|
|
343
|
+
const judge = async (state, questions, extra, matrixRequest = false) => {
|
|
344
|
+
state = withEvidenceContext(state, evidenceContext);
|
|
345
|
+
if (JSON.stringify(state).length > STATE_MAX_CHARS)
|
|
346
|
+
return {
|
|
347
|
+
ok: false,
|
|
348
|
+
error: `required evidence exceeds STATE_MAX_CHARS=${STATE_MAX_CHARS}`,
|
|
349
|
+
};
|
|
350
|
+
const result = await client.judge(state, questions, {
|
|
351
|
+
...optionsFor(matrixRequest),
|
|
352
|
+
...extra,
|
|
353
|
+
});
|
|
354
|
+
judgments.push(result);
|
|
355
|
+
return result;
|
|
356
|
+
};
|
|
357
|
+
const local = prepared.dimensions.some((dimension) => dimension.name === "reliability")
|
|
358
|
+
? await collectRiskCallers(exec, { cwd: cwd, base, signal }, collected.units, collected.files, analysis.parser)
|
|
359
|
+
: { proofs: [], limits: [] };
|
|
360
|
+
const uncheckedCallers = new Set();
|
|
361
|
+
for (const limit of local.limits) {
|
|
362
|
+
if (limit.kind !== "dynamic_access_uncovered") {
|
|
363
|
+
const id = `caller:${limit.unitId}`;
|
|
364
|
+
report.expect(id, `${limit.unitId} local caller`, "unit");
|
|
365
|
+
report.diagnose("collection_omitted", limit.reason, limit.paths.join(" + "), [id]);
|
|
366
|
+
}
|
|
367
|
+
else
|
|
368
|
+
report.diagnose("dynamic_dependency", limit.reason, limit.paths.join(" + "), [], false);
|
|
369
|
+
limitations.push({
|
|
370
|
+
fact: `${limit.unitId}: ${limit.reason} — ${limit.paths.join(" + ")}`,
|
|
371
|
+
next: "read the named caller/provider pieces or supply an observation via jev_ask",
|
|
372
|
+
});
|
|
373
|
+
if (limit.kind !== "dynamic_access_uncovered" &&
|
|
374
|
+
!uncheckedCallers.has(limit.unitId)) {
|
|
375
|
+
uncheckedCallers.add(limit.unitId);
|
|
376
|
+
unchecked.push(`${limit.unitId} local caller ${limit.paths.join(" + ")}`);
|
|
377
|
+
}
|
|
378
|
+
}
|
|
379
|
+
const matrixPromise = judge(prepared.state, prepared.questions, { groups: prepared.groups, witnesses: prepared.witnessQuestionIds }, true).finally(() => {
|
|
380
|
+
reservedMatrixCalls = 0;
|
|
381
|
+
});
|
|
382
|
+
const localPromises = local.proofs.map(async (proof) => {
|
|
383
|
+
if (args.max_calls !== undefined &&
|
|
384
|
+
sent + reservedMatrixCalls >= args.max_calls)
|
|
385
|
+
await matrixPromise;
|
|
386
|
+
return {
|
|
387
|
+
proof,
|
|
388
|
+
result: await judge(proof.state, { caller: proof.question }),
|
|
389
|
+
};
|
|
390
|
+
});
|
|
391
|
+
const [result, locals] = await Promise.all([
|
|
392
|
+
matrixPromise,
|
|
393
|
+
Promise.all(localPromises),
|
|
394
|
+
]);
|
|
395
|
+
const health = result.ok
|
|
396
|
+
? evaluateWitnessHealth(prepared, result)
|
|
397
|
+
: {
|
|
398
|
+
healthy: false,
|
|
399
|
+
controls: [],
|
|
400
|
+
unhealthyQuestionIds: new Map(),
|
|
401
|
+
};
|
|
402
|
+
matrixHealthy = health.healthy;
|
|
403
|
+
for (const failure of health.controls)
|
|
404
|
+
report.diagnose(health.healthy ? "conservative_widening" : "control_failure", failure.fact, undefined, prepared.cells
|
|
405
|
+
.filter((cell) => health.unhealthyQuestionIds.has(cell.id))
|
|
406
|
+
.map((cell) => `risk:${cell.id}`), false);
|
|
407
|
+
report.countControls(result, prepared.witnessQuestionIds);
|
|
408
|
+
for (const cell of prepared.cells) {
|
|
409
|
+
const rawAnswer = result.ok ? result.answers[cell.id] : undefined;
|
|
410
|
+
const answer = rawAnswer?.type === "unjudged" && !rawAnswer.cause && budget
|
|
411
|
+
? {
|
|
412
|
+
...rawAnswer,
|
|
413
|
+
cause: budget.kind === "session"
|
|
414
|
+
? "session_budget"
|
|
415
|
+
: "call_budget",
|
|
416
|
+
}
|
|
417
|
+
: rawAnswer;
|
|
418
|
+
const invalid = health.unhealthyQuestionIds.get(cell.id);
|
|
419
|
+
if (!result.ok)
|
|
420
|
+
report.failure(result, [`risk:${cell.id}`], budget
|
|
421
|
+
? budget.kind === "session"
|
|
422
|
+
? "session_budget"
|
|
423
|
+
: "call_budget"
|
|
424
|
+
: undefined);
|
|
425
|
+
else
|
|
426
|
+
report.answer(`risk:${cell.id}`, answer, {
|
|
427
|
+
band: invalid ? "unsure" : "verdict",
|
|
428
|
+
reason: invalid,
|
|
429
|
+
uncalibrated: cell.uncalibrated,
|
|
430
|
+
}, controlsFor(cell.id, result, prepared.witnessQuestionIds), false);
|
|
431
|
+
}
|
|
432
|
+
const decoys = new Set(prepared.witnesses
|
|
433
|
+
.filter((witness) => witness.expected === "no")
|
|
434
|
+
.map((witness) => witness.unitId)).size;
|
|
435
|
+
const references = new Set(prepared.witnesses
|
|
436
|
+
.filter((witness) => witness.expected === "yes")
|
|
437
|
+
.map((witness) => witness.unitId)).size;
|
|
438
|
+
const witnessStatus = health.healthy
|
|
439
|
+
? "healthy"
|
|
440
|
+
: !result.ok ||
|
|
441
|
+
health.controls.some((control) => control.fact.includes("unavailable"))
|
|
442
|
+
? "unavailable"
|
|
443
|
+
: "unhealthy";
|
|
444
|
+
limitations.push({
|
|
445
|
+
fact: prepared.witnesses.length
|
|
446
|
+
? `risk matrix: ${decoys} decoys and ${references} references ${witnessStatus}`
|
|
447
|
+
: "risk matrix: witnesses disabled",
|
|
448
|
+
next: "read the findings and stated limits",
|
|
449
|
+
});
|
|
450
|
+
limitations.push(...health.controls);
|
|
451
|
+
if (!result.ok)
|
|
452
|
+
limitations.push({
|
|
453
|
+
fact: `risk matrix: ${result.error}`,
|
|
454
|
+
next: "check Jev configuration and availability, then retry",
|
|
455
|
+
});
|
|
456
|
+
const severity = [];
|
|
457
|
+
for (const cell of prepared.cells) {
|
|
458
|
+
const unit = collected.units.find((u) => u.id === cell.unitId);
|
|
459
|
+
if (!unit)
|
|
460
|
+
continue;
|
|
461
|
+
const answer = result.ok ? result.answers[cell.id] : undefined;
|
|
462
|
+
if (answer?.type !== "bool") {
|
|
463
|
+
unchecked.push(`${unitLabel(unit)} ${cell.dimension}`);
|
|
464
|
+
continue;
|
|
465
|
+
}
|
|
466
|
+
if (answer.p < FLAG_MIN)
|
|
467
|
+
continue;
|
|
468
|
+
const reason = health.unhealthyQuestionIds.get(cell.id);
|
|
469
|
+
const line = {
|
|
470
|
+
label: `${unitLabel(unit)} ${cell.dimension}${reason ? ` — ${reason}` : ""}`,
|
|
471
|
+
value: { head: "risk", p: answer.p },
|
|
472
|
+
band: reason ? "unsure" : "verdict",
|
|
473
|
+
uncalibrated: cell.uncalibrated,
|
|
474
|
+
};
|
|
475
|
+
answers.push(line);
|
|
476
|
+
unitOrder.set(line, unit.id);
|
|
477
|
+
if (!cell.uncalibrated)
|
|
478
|
+
severity.push({ unit, dimension: cell.dimension, lines: [line] });
|
|
479
|
+
}
|
|
480
|
+
for (const { proof, result: localResult } of locals) {
|
|
481
|
+
callerProofs.push(proof);
|
|
482
|
+
const unit = collected.units.find((u) => u.id === proof.unitId);
|
|
483
|
+
const reportId = `caller:${proof.unitId}`;
|
|
484
|
+
report.expect(reportId, `${unit ? unitLabel(unit) : proof.unitId} local caller`, "unit");
|
|
485
|
+
if (!localResult.ok) {
|
|
486
|
+
report.failure(localResult, [reportId], budget
|
|
487
|
+
? budget.kind === "session"
|
|
488
|
+
? "session_budget"
|
|
489
|
+
: "call_budget"
|
|
490
|
+
: undefined);
|
|
491
|
+
continue;
|
|
492
|
+
}
|
|
493
|
+
const answer = localResult.answers.caller;
|
|
494
|
+
if (answer?.type !== "choice") {
|
|
495
|
+
report.answer(reportId, answer);
|
|
496
|
+
unchecked.push(`${unit ? unitLabel(unit) : proof.unitId} local caller`);
|
|
497
|
+
continue;
|
|
498
|
+
}
|
|
499
|
+
const p = answer.probabilities.new_failure ?? 0;
|
|
500
|
+
const missing = (answer.probabilities.cannot_tell ?? 0) >= CANNOT_TELL_MIN;
|
|
501
|
+
const negative = !missing && p < BAND_BOOL_GRAY_A;
|
|
502
|
+
const merged = [];
|
|
503
|
+
for (const span of [...proof.paths].sort((a, b) => a.path.localeCompare(b.path) || a.start - b.start || a.end - b.end)) {
|
|
504
|
+
const previous = merged.at(-1);
|
|
505
|
+
if (previous &&
|
|
506
|
+
previous.path === span.path &&
|
|
507
|
+
span.start <= previous.end + 1)
|
|
508
|
+
previous.end = Math.max(previous.end, span.end);
|
|
509
|
+
else
|
|
510
|
+
merged.push({ ...span });
|
|
511
|
+
}
|
|
512
|
+
const omitted = Math.max(0, merged.length - CALLER_DISPLAY_MAX_SPANS);
|
|
513
|
+
const passages = merged
|
|
514
|
+
.slice(0, CALLER_DISPLAY_MAX_SPANS)
|
|
515
|
+
.map((span) => `${span.path}:${span.start}-${span.end}`)
|
|
516
|
+
.join(" + ");
|
|
517
|
+
const line = {
|
|
518
|
+
label: `${unit ? unitLabel(unit) : proof.unitId} reliability — static code only; ${passages}${omitted ? ` (+${omitted} passages)` : ""}`,
|
|
519
|
+
value: {
|
|
520
|
+
head: missing
|
|
521
|
+
? "cannot_tell"
|
|
522
|
+
: negative
|
|
523
|
+
? "no_new_failure"
|
|
524
|
+
: "new_failure",
|
|
525
|
+
p: missing
|
|
526
|
+
? (answer.probabilities.cannot_tell ?? 0)
|
|
527
|
+
: negative
|
|
528
|
+
? (answer.probabilities.no_new_failure ?? 0)
|
|
529
|
+
: p,
|
|
530
|
+
},
|
|
531
|
+
band: missing
|
|
532
|
+
? "abstain"
|
|
533
|
+
: negative || p >= FLAG_MIN
|
|
534
|
+
? "verdict"
|
|
535
|
+
: "unsure",
|
|
536
|
+
reason: missing
|
|
537
|
+
? "actual provider or binding missing; supply its declaration or an observation via jev_ask"
|
|
538
|
+
: negative
|
|
539
|
+
? "Local caller check does not flag a new failure; static evidence does not prove every caller safe."
|
|
540
|
+
: p < FLAG_MIN
|
|
541
|
+
? "Local caller new-failure probability is in the gray band; inspect the stated proof passages natively."
|
|
542
|
+
: "Local caller new-failure probability meets the finding threshold; static code evidence only.",
|
|
543
|
+
};
|
|
544
|
+
const callerItem = report.items.get(reportId);
|
|
545
|
+
if (callerItem)
|
|
546
|
+
report.items.set(reportId, { ...callerItem, label: line.label });
|
|
547
|
+
report.answer(reportId, answer, line);
|
|
548
|
+
if (missing && line.reason)
|
|
549
|
+
report.diagnose("missing_required", line.reason, line.label, [reportId], false);
|
|
550
|
+
if (!unit || negative)
|
|
551
|
+
continue;
|
|
552
|
+
answers.push(line);
|
|
553
|
+
unitOrder.set(line, unit.id);
|
|
554
|
+
if (!missing && p >= FLAG_MIN) {
|
|
555
|
+
const existing = severity.find((item) => item.unit.id === unit.id && item.dimension === "reliability");
|
|
556
|
+
if (existing) {
|
|
557
|
+
existing.evidence = proof.state;
|
|
558
|
+
existing.lines.push(line);
|
|
559
|
+
}
|
|
560
|
+
else
|
|
561
|
+
severity.push({
|
|
562
|
+
unit,
|
|
563
|
+
dimension: "reliability",
|
|
564
|
+
lines: [line],
|
|
565
|
+
evidence: proof.state,
|
|
566
|
+
});
|
|
567
|
+
}
|
|
568
|
+
}
|
|
569
|
+
await Promise.all(severity.map(async (item) => {
|
|
570
|
+
const request = prepareRiskSeverity(item.unit, item.dimension, item.evidence?.callerEvidence);
|
|
571
|
+
const value = await judge(request.state, request.questions);
|
|
572
|
+
const reportId = `severity:${item.unit.id}:${item.dimension}`;
|
|
573
|
+
report.expect(reportId, `${item.unit.id} ${item.dimension} severity`, "unit");
|
|
574
|
+
if (value.ok)
|
|
575
|
+
report.answer(reportId, value.answers.severity);
|
|
576
|
+
else
|
|
577
|
+
report.failure(value, [reportId], budget
|
|
578
|
+
? budget.kind === "session"
|
|
579
|
+
? "session_budget"
|
|
580
|
+
: "call_budget"
|
|
581
|
+
: undefined);
|
|
582
|
+
const answer = value.ok ? value.answers.severity : undefined;
|
|
583
|
+
if (answer?.type === "score") {
|
|
584
|
+
const suffix = ` · severity ${answer.score.toFixed(1)}/3`;
|
|
585
|
+
for (const line of item.lines)
|
|
586
|
+
line.label += suffix;
|
|
587
|
+
}
|
|
588
|
+
else
|
|
589
|
+
unchecked.push(`${unitLabel(item.unit)} ${item.dimension} severity`);
|
|
590
|
+
}));
|
|
591
|
+
return finish();
|
|
592
|
+
},
|
|
593
|
+
};
|
|
594
|
+
}
|