jev-agent-tools 0.1.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (180) hide show
  1. package/CHANGELOG.md +106 -1
  2. package/CONTRIBUTING.md +43 -0
  3. package/README.md +58 -17
  4. package/SECURITY.md +43 -0
  5. package/dist/adapters/analysis-context.js +75 -0
  6. package/dist/adapters/ask-files.js +198 -0
  7. package/dist/adapters/ask-proof.js +200 -0
  8. package/dist/adapters/ask-syntax.js +385 -0
  9. package/dist/adapters/canonical-path.js +17 -0
  10. package/dist/adapters/command.js +234 -0
  11. package/dist/adapters/docs.js +192 -0
  12. package/dist/adapters/evidence-context.js +119 -0
  13. package/dist/adapters/exec.js +207 -0
  14. package/dist/adapters/files.js +418 -0
  15. package/dist/adapters/find.js +150 -0
  16. package/dist/adapters/git-base.js +32 -0
  17. package/dist/adapters/git-inventory.js +71 -0
  18. package/dist/adapters/git.js +483 -0
  19. package/dist/adapters/locate-file.js +197 -0
  20. package/dist/adapters/output-lines.js +46 -0
  21. package/dist/adapters/private-storage.js +106 -0
  22. package/dist/adapters/risk-callers.js +429 -0
  23. package/dist/adapters/runner-version.js +78 -0
  24. package/dist/adapters/shell.js +92 -0
  25. package/dist/adapters/syntax.js +187 -0
  26. package/dist/adapters/test-inventory.js +139 -0
  27. package/dist/adapters/usage.js +20 -0
  28. package/dist/adapters/utf8.js +47 -0
  29. package/dist/configuration.js +267 -0
  30. package/dist/constants.js +140 -0
  31. package/dist/core/ask-closure.js +282 -0
  32. package/dist/core/ask-proof.js +1 -0
  33. package/dist/core/ask-references.js +278 -0
  34. package/dist/core/asks.js +507 -0
  35. package/dist/core/batches.js +65 -0
  36. package/dist/core/command-output.js +224 -0
  37. package/dist/core/diff.js +178 -0
  38. package/dist/core/docs.js +302 -0
  39. package/dist/core/find.js +108 -0
  40. package/dist/core/git.js +1 -0
  41. package/dist/core/imports.js +550 -0
  42. package/dist/core/integrity.js +45 -0
  43. package/dist/core/lexical.js +132 -0
  44. package/dist/core/locate.js +169 -0
  45. package/dist/core/output.js +137 -0
  46. package/dist/core/pointer.js +29 -0
  47. package/dist/core/result-report.js +302 -0
  48. package/dist/core/risk-callers.js +851 -0
  49. package/dist/core/runner-version.js +45 -0
  50. package/dist/core/secret-path.js +34 -0
  51. package/dist/core/sections.js +230 -0
  52. package/dist/core/state.js +51 -0
  53. package/dist/core/syntax.js +1 -0
  54. package/dist/core/test-commands.js +334 -0
  55. package/dist/core/test-coverage.js +74 -0
  56. package/dist/core/test-discovery.js +1382 -0
  57. package/dist/core/test-evidence.js +527 -0
  58. package/dist/core/test-state.js +81 -0
  59. package/dist/core/truncate.js +12 -0
  60. package/dist/core/units.js +349 -0
  61. package/dist/describe.js +23 -0
  62. package/dist/guide.js +33 -0
  63. package/dist/host.js +24 -0
  64. package/dist/jev/client.js +456 -0
  65. package/dist/jev/pool.js +54 -0
  66. package/dist/jev/types.js +1 -0
  67. package/dist/mcp/main.js +124 -0
  68. package/dist/mcp/protocol.js +210 -0
  69. package/dist/mcp/tools.js +129 -0
  70. package/dist/presets/docs.js +62 -0
  71. package/dist/presets/risk.js +179 -0
  72. package/dist/presets/spec.js +81 -0
  73. package/dist/presets/witnesses.js +249 -0
  74. package/dist/render.js +114 -0
  75. package/dist/report-schema.js +1356 -0
  76. package/dist/result-types.js +1 -0
  77. package/dist/result.js +3 -0
  78. package/dist/runtime.js +1 -0
  79. package/dist/session.js +147 -0
  80. package/dist/texts/ask-files.js +3 -0
  81. package/dist/texts/ask.js +4 -0
  82. package/dist/texts/check-diff.js +20 -0
  83. package/dist/texts/configuration.js +1 -0
  84. package/dist/texts/find.js +19 -0
  85. package/dist/texts/guide.js +3 -0
  86. package/dist/texts/instructions.js +72 -0
  87. package/dist/texts/locate.js +15 -0
  88. package/dist/texts/select-tests.js +4 -0
  89. package/dist/tools/ask-files.js +450 -0
  90. package/dist/tools/ask-schema.js +70 -0
  91. package/dist/tools/ask.js +1147 -0
  92. package/dist/tools/check-diff.js +594 -0
  93. package/dist/tools/docs-check.js +408 -0
  94. package/dist/tools/find.js +682 -0
  95. package/dist/tools/locate.js +602 -0
  96. package/dist/tools/review-report.js +230 -0
  97. package/dist/tools/select-tests.js +821 -0
  98. package/dist/tools/spec-check.js +263 -0
  99. package/docs/adr/0001-strict-typescript-pure-core-offline-tests.md +31 -0
  100. package/docs/adr/0002-one-http-protocol-across-hosts.md +17 -0
  101. package/docs/adr/0003-explicit-scope-conservative-automation.md +19 -0
  102. package/docs/adr/0004-compiled-typed-intents.md +19 -0
  103. package/docs/adr/0005-evidence-construction-before-judgment.md +19 -0
  104. package/docs/adr/0006-visible-uncertainty-constrained-controls.md +21 -0
  105. package/docs/adr/0007-bounded-evidence-visible-limits.md +21 -0
  106. package/docs/adr/0008-static-test-discovery-conservative-plans.md +19 -0
  107. package/docs/adr/0009-session-cache-requested-model-identity.md +17 -0
  108. package/docs/adr/0010-mcp-server-thin-host.md +23 -0
  109. package/docs/agent-instructions.md +120 -0
  110. package/docs/design.md +16 -4
  111. package/docs/mcp.md +233 -0
  112. package/docs/tools/jev_ask.md +8 -5
  113. package/docs/tools/jev_ask_files.md +2 -1
  114. package/docs/tools/jev_check_diff.md +4 -1
  115. package/docs/tools/jev_find_files.md +2 -1
  116. package/docs/tools/jev_locate_in_file.md +5 -0
  117. package/docs/tools/jev_select_tests.md +4 -1
  118. package/package.json +19 -4
  119. package/rules/jev-ask.md +22 -1
  120. package/server.json +57 -0
  121. package/src/adapters/ask-files.ts +11 -3
  122. package/src/adapters/ask-proof.ts +69 -11
  123. package/src/adapters/canonical-path.ts +18 -0
  124. package/src/adapters/command.ts +102 -36
  125. package/src/adapters/docs.ts +33 -14
  126. package/src/adapters/evidence-context.ts +169 -0
  127. package/src/adapters/exec.ts +226 -0
  128. package/src/adapters/files.ts +146 -16
  129. package/src/adapters/find.ts +37 -7
  130. package/src/adapters/git-base.ts +7 -1
  131. package/src/adapters/git.ts +61 -8
  132. package/src/adapters/locate-file.ts +51 -9
  133. package/src/adapters/private-storage.ts +155 -0
  134. package/src/adapters/risk-callers.ts +7 -2
  135. package/src/adapters/shell.ts +113 -0
  136. package/src/adapters/test-inventory.ts +12 -4
  137. package/src/configuration.ts +55 -14
  138. package/src/constants.ts +37 -5
  139. package/src/core/ask-references.ts +262 -146
  140. package/src/core/asks.ts +79 -7
  141. package/src/core/command-output.ts +17 -1
  142. package/src/core/import-boundaries.ts +8 -3
  143. package/src/core/locate.ts +8 -5
  144. package/src/core/output.ts +34 -0
  145. package/src/core/result-report.ts +410 -0
  146. package/src/core/secret-path.ts +37 -0
  147. package/src/core/state.ts +8 -1
  148. package/src/core/units.ts +3 -2
  149. package/src/host.ts +11 -0
  150. package/src/index.ts +3 -0
  151. package/src/jev/client.ts +66 -16
  152. package/src/jev/types.ts +24 -3
  153. package/src/mcp/main.ts +135 -0
  154. package/src/mcp/protocol.ts +332 -0
  155. package/src/mcp/tools.ts +179 -0
  156. package/src/render.ts +109 -0
  157. package/src/report-schema.ts +1380 -0
  158. package/src/result-types.ts +234 -0
  159. package/src/result.ts +4 -1
  160. package/src/runtime.ts +6 -0
  161. package/src/session.ts +59 -0
  162. package/src/setup.ts +13 -5
  163. package/src/texts/ask-files.ts +4 -1
  164. package/src/texts/ask.ts +8 -1
  165. package/src/texts/check-diff.ts +7 -4
  166. package/src/texts/find.ts +8 -2
  167. package/src/texts/guide.ts +8 -16
  168. package/src/texts/instructions.ts +98 -0
  169. package/src/texts/locate.ts +8 -2
  170. package/src/texts/run-end.ts +2 -2
  171. package/src/texts/select-tests.ts +4 -1
  172. package/src/tools/ask-files.ts +311 -18
  173. package/src/tools/ask.ts +722 -95
  174. package/src/tools/check-diff.ts +337 -31
  175. package/src/tools/docs-check.ts +241 -38
  176. package/src/tools/find.ts +389 -29
  177. package/src/tools/locate.ts +387 -25
  178. package/src/tools/review-report.ts +308 -0
  179. package/src/tools/select-tests.ts +484 -23
  180. package/src/tools/spec-check.ts +194 -19
@@ -0,0 +1,594 @@
1
+ import { createHash } from "node:crypto";
2
+ import { Type } from "@sinclair/typebox";
3
+ import { createAnalysisContext } from "../adapters/analysis-context.js";
4
+ import { resolveEvidenceContext, withEvidenceContext, } from "../adapters/evidence-context.js";
5
+ import { collectUnits } from "../adapters/git.js";
6
+ import { resolveBase } from "../adapters/git-base.js";
7
+ import { shareGitInventory } from "../adapters/git-inventory.js";
8
+ import { collectRiskCallers } from "../adapters/risk-callers.js";
9
+ import { hostUsage } from "../adapters/usage.js";
10
+ import { BAND_BOOL_GRAY_A, CALLER_DISPLAY_MAX_SPANS, CANNOT_TELL_MIN, FLAG_MIN, STATE_MAX_CHARS, TIMEOUT_MS, } from "../constants.js";
11
+ import { prepareBatches } from "../core/batches.js";
12
+ import { isTestFile } from "../core/diff.js";
13
+ import { buildEnvelope, } from "../core/output.js";
14
+ import { known, } from "../core/result-report.js";
15
+ import { evaluateWitnessHealth, prepareRiskMatrix, prepareRiskSeverity, } from "../presets/risk.js";
16
+ import { renderResultReport } from "../render.js";
17
+ import { CHECK_DIFF_DESCRIPTION } from "../texts/check-diff.js";
18
+ import { NOT_CONFIGURED } from "../texts/configuration.js";
19
+ import { CHECK_DIFF_GUIDELINE } from "../texts/instructions.js";
20
+ import { runDocsCheck } from "./docs-check.js";
21
+ import { controlsFor, ReviewReport, reportMetrics } from "./review-report.js";
22
+ import { runSpecCheck } from "./spec-check.js";
23
+ export const checkDiffParameters = Type.Object({
24
+ root: Type.Optional(Type.String()),
25
+ check: Type.Union([
26
+ Type.Literal("risk"),
27
+ Type.Literal("docs"),
28
+ Type.Literal("spec"),
29
+ ]),
30
+ base: Type.Optional(Type.String({ minLength: 1 })),
31
+ spec_path: Type.Optional(Type.String({ minLength: 1 })),
32
+ witnesses: Type.Optional(Type.Union([
33
+ Type.Literal("off"),
34
+ Type.Literal("auto"),
35
+ Type.Literal("on"),
36
+ ])),
37
+ dimensions: Type.Optional(Type.Record(Type.String(), Type.String({ minLength: 1 }))),
38
+ only: Type.Optional(Type.Array(Type.String({ minLength: 1 }))),
39
+ max_calls: Type.Optional(Type.Integer({ minimum: 0 })),
40
+ }, { additionalProperties: false });
41
+ function unitLabel(unit) {
42
+ const range = unit.afterRange ?? unit.beforeRange;
43
+ const fingerprint = createHash("sha256")
44
+ .update(JSON.stringify([unit.file, unit.name, unit.before, unit.after]))
45
+ .digest("hex")
46
+ .slice(0, 8);
47
+ return `${unit.file}${range ? `:${range.start}-${range.end}` : ""} ${unit.name} [${fingerprint}]`;
48
+ }
49
+ export function createCheckDiffTool(dependencies) {
50
+ const { host, runtime, exec: execute } = dependencies;
51
+ return {
52
+ name: "jev_check_diff",
53
+ label: "Jev check diff",
54
+ description: CHECK_DIFF_DESCRIPTION,
55
+ parameters: checkDiffParameters,
56
+ ...(host.isOmp
57
+ ? { approval: "read", loadMode: "essential" }
58
+ : {
59
+ promptSnippet: "Check the uncommitted diff for risky changes, stale docs or spec drift",
60
+ promptGuidelines: [CHECK_DIFF_GUIDELINE],
61
+ }),
62
+ async execute(_id, args, signal, _update, ctx) {
63
+ const client = dependencies.client;
64
+ const evidence = await resolveEvidenceContext(ctx.cwd, args.root, {
65
+ exec: execute,
66
+ signal,
67
+ origin: dependencies.evidenceOrigin,
68
+ });
69
+ const evidenceContext = evidence.context;
70
+ const report = new ReviewReport();
71
+ const cwd = evidence.ok ? evidence.cwd : evidenceContext.authority.path;
72
+ const exec = shareGitInventory(execute);
73
+ if (!evidence.ok) {
74
+ const envelope = buildEnvelope({
75
+ refusal: evidence.error,
76
+ yield: {
77
+ calls: 0,
78
+ questions: 0,
79
+ cacheHits: 0,
80
+ cacheRequests: 0,
81
+ elapsedMs: 0,
82
+ },
83
+ });
84
+ runtime.session.record(envelope);
85
+ runtime.guide.deliver(ctx);
86
+ report.refusal = true;
87
+ report.diagnose(evidence.cause, evidence.error, args.root);
88
+ const result = report.build("jev_check_diff", evidenceContext, reportMetrics(envelope));
89
+ return {
90
+ content: [
91
+ {
92
+ type: "text",
93
+ text: renderResultReport(result, { details: envelope }),
94
+ },
95
+ ],
96
+ details: {
97
+ ok: false,
98
+ envelope,
99
+ judgments: [],
100
+ callerProofs: [],
101
+ evidenceContext,
102
+ result,
103
+ },
104
+ };
105
+ }
106
+ evidenceContext.requestedBase = args.base ?? "HEAD";
107
+ if (args.check === "docs" || args.check === "spec") {
108
+ const deps = { client, host, runtime, exec };
109
+ const result = args.check === "docs"
110
+ ? await runDocsCheck(deps, {
111
+ cwd: cwd,
112
+ base: args.base,
113
+ signal,
114
+ maxCalls: args.max_calls,
115
+ evidenceContext,
116
+ })
117
+ : await runSpecCheck(deps, {
118
+ cwd: cwd,
119
+ base: args.base,
120
+ specPath: args.spec_path,
121
+ signal,
122
+ maxCalls: args.max_calls,
123
+ evidenceContext,
124
+ });
125
+ runtime.session.record(result.envelope);
126
+ runtime.guide.deliver(ctx);
127
+ const usage = result.judgments.reduce((sum, judgment) => ({
128
+ inputTokens: sum.inputTokens + (judgment.usage?.inputTokens ?? 0),
129
+ costUsd: sum.costUsd + (judgment.usage?.costUsd ?? 0),
130
+ }), { inputTokens: 0, costUsd: 0 });
131
+ return {
132
+ content: [
133
+ {
134
+ type: "text",
135
+ text: renderResultReport(result.result, {
136
+ details: result.envelope,
137
+ }),
138
+ },
139
+ ],
140
+ details: { ...result, callerProofs: [], evidenceContext },
141
+ ...hostUsage(host.isOmp, usage),
142
+ };
143
+ }
144
+ const started = performance.now();
145
+ const judgments = [];
146
+ const callerProofs = [];
147
+ const answers = [];
148
+ const unitOrder = new Map();
149
+ const limitations = [];
150
+ const unchecked = [];
151
+ let budget;
152
+ let sent = 0;
153
+ let matrixHealthy = true;
154
+ let emptyBase;
155
+ const finish = (refusal, cause) => {
156
+ answers.sort((a, b) => (unitOrder.get(a) ?? "").localeCompare(unitOrder.get(b) ?? ""));
157
+ const uniqueLimitations = [
158
+ ...new Map(limitations.map((limit) => [`${limit.fact}\0${limit.next}`, limit])).values(),
159
+ ];
160
+ const usage = judgments.reduce((sum, value) => ({
161
+ inputTokens: sum.inputTokens + (value.usage?.inputTokens ?? 0),
162
+ costUsd: sum.costUsd + (value.usage?.costUsd ?? 0),
163
+ }), { inputTokens: 0, costUsd: 0 });
164
+ const noFindings = !refusal &&
165
+ matrixHealthy &&
166
+ !answers.length &&
167
+ !unchecked.length &&
168
+ report.items.size > 0 &&
169
+ [...report.items.values()].every((item) => item.treatment === "judged");
170
+ const envelope = buildEnvelope({
171
+ answers,
172
+ limitations: emptyBase !== undefined && !refusal
173
+ ? [
174
+ ...uniqueLimitations,
175
+ {
176
+ fact: `no changed units against ${emptyBase}`,
177
+ next: "nothing judged; pass base= or check the working directory",
178
+ },
179
+ ]
180
+ : uniqueLimitations,
181
+ unchecked: [...new Set(unchecked)],
182
+ budget,
183
+ ...(refusal ? { refusal } : {}),
184
+ ...(noFindings && emptyBase === undefined
185
+ ? {
186
+ lines: [
187
+ {
188
+ type: "list",
189
+ title: "risk: no findings",
190
+ items: [],
191
+ },
192
+ ],
193
+ }
194
+ : {}),
195
+ yield: {
196
+ calls: judgments.reduce((n, j) => n + (j.calls ?? 0), 0),
197
+ questions: judgments.reduce((n, j) => n + (j.questions ?? 0), 0),
198
+ costUsd: judgments.every((j) => j.usage !== undefined) &&
199
+ judgments.length > 0
200
+ ? usage.costUsd
201
+ : undefined,
202
+ cacheHits: judgments.reduce((n, j) => n + (j.cacheHits ?? 0), 0),
203
+ cacheRequests: judgments.reduce((n, j) => n + (j.cacheRequests ?? 0), 0),
204
+ elapsedMs: performance.now() - started,
205
+ },
206
+ });
207
+ if (emptyBase !== undefined)
208
+ report.diagnose("no_changed_units", `No changed units against ${emptyBase}; no risk judgment requested.`, emptyBase, [], false);
209
+ if (refusal &&
210
+ !report.diagnostics.some((d) => d.effect === "blocking")) {
211
+ report.refusal = refusal !== NOT_CONFIGURED;
212
+ report.diagnose(cause ?? "internal_error", refusal);
213
+ }
214
+ const result = report.build("jev_check_diff", evidenceContext, reportMetrics(envelope));
215
+ runtime.session.record(envelope);
216
+ runtime.guide.deliver(ctx);
217
+ return {
218
+ content: [
219
+ {
220
+ type: "text",
221
+ text: renderResultReport(result, { details: envelope }),
222
+ },
223
+ ],
224
+ details: {
225
+ evidenceContext,
226
+ result,
227
+ ok: !refusal,
228
+ envelope,
229
+ judgments,
230
+ callerProofs,
231
+ },
232
+ ...hostUsage(host.isOmp, usage),
233
+ };
234
+ };
235
+ if (!client)
236
+ return finish(NOT_CONFIGURED, "not_configured");
237
+ const comparison = await resolveBase(exec, cwd, args.base, signal);
238
+ if (!comparison.ok)
239
+ return finish(comparison.error, comparison.cause ?? "invalid_base");
240
+ evidenceContext.resolvedBase = comparison.base;
241
+ const base = comparison.base;
242
+ const analysis = await createAnalysisContext();
243
+ const collected = await collectUnits(exec, {
244
+ cwd: cwd,
245
+ base,
246
+ signal,
247
+ }, analysis.parser);
248
+ if (!collected.ok)
249
+ return finish(collected.error, collected.cause ?? "git_failure");
250
+ const repositoryRoot = await exec("git", ["rev-parse", "--show-toplevel"], { cwd, timeout: TIMEOUT_MS, signal });
251
+ if (!repositoryRoot.code &&
252
+ !repositoryRoot.killed &&
253
+ evidenceContext.effectiveRoot)
254
+ evidenceContext.effectiveRoot.path = repositoryRoot.stdout.trim();
255
+ report.inventories.push({
256
+ id: "changed-units",
257
+ kind: "units",
258
+ rules: [
259
+ "Changed source units against resolved base; tests are evidence, not risk units",
260
+ ],
261
+ restrictions: args.only ?? [],
262
+ discovered: known(collected.units.length),
263
+ considered: known(collected.units.length),
264
+ scopeRestricted: Boolean(args.only?.length),
265
+ criteria: [],
266
+ });
267
+ for (const unit of collected.units)
268
+ report.expect(`unit:${unit.id}`, unitLabel(unit), "unit");
269
+ for (const limit of collected.limits)
270
+ report.diagnose(limit.kind === "secret_pattern"
271
+ ? "secret_pattern"
272
+ : "collection_omitted", `${limit.file}: ${limit.kind}`, limit.file);
273
+ for (const limit of collected.limits)
274
+ limitations.push({
275
+ fact: `${limit.file}: ${limit.kind}`,
276
+ next: "read the complete before/after source or rerun with base= a nearer ref",
277
+ });
278
+ const readableUnits = collected.units.filter((unit) => {
279
+ if (unit.before !== null || unit.after !== null)
280
+ return true;
281
+ unchecked.push(`${unitLabel(unit)} source unavailable`);
282
+ report.diagnose("binary_or_non_utf8", "Changed source unavailable", unit.file, [`unit:${unit.id}`]);
283
+ return false;
284
+ });
285
+ if (!collected.units.length) {
286
+ emptyBase = base;
287
+ return finish();
288
+ }
289
+ if (!readableUnits.length)
290
+ return finish();
291
+ const matrix = prepareRiskMatrix(readableUnits, {
292
+ ...args,
293
+ testFilesPresent: collected.files.some((file) => isTestFile(file.path)),
294
+ });
295
+ if (!matrix.ok)
296
+ return finish(matrix.error, "invalid_arguments");
297
+ const prepared = matrix;
298
+ prepared.state = withEvidenceContext(prepared.state, evidenceContext);
299
+ for (const unit of readableUnits)
300
+ report.items.delete(`unit:${unit.id}`);
301
+ for (const cell of prepared.cells)
302
+ report.expect(`risk:${cell.id}`, `${cell.unitId} ${cell.dimension}`, "unit", `risk:${cell.unitId}`);
303
+ for (const warning of prepared.warnings) {
304
+ limitations.push(warning);
305
+ report.diagnose("unsupported_syntax", warning.fact, undefined, [], false);
306
+ }
307
+ if (JSON.stringify(prepared.state).length > STATE_MAX_CHARS)
308
+ return finish(`risk state exceeds STATE_MAX_CHARS=${STATE_MAX_CHARS}; rerun with base= a nearer ref`, "evidence_too_large");
309
+ const batches = prepareBatches(prepared.state, prepared.questions, {
310
+ groups: prepared.groups,
311
+ witnesses: prepared.witnessQuestionIds,
312
+ });
313
+ if (!batches.ok)
314
+ return finish(batches.error, "group_too_large");
315
+ let reservedMatrixCalls = batches.batches.length;
316
+ const optionsFor = (matrixRequest = false) => ({
317
+ signal,
318
+ ...runtime.session.requestGate(),
319
+ admissionCause: () => budget?.kind === "session" ? "session_budget" : "call_budget",
320
+ beforeRequest(questionCount) {
321
+ if (matrixRequest && reservedMatrixCalls > 0)
322
+ reservedMatrixCalls--;
323
+ const limit = args.max_calls === undefined
324
+ ? undefined
325
+ : args.max_calls - (matrixRequest ? 0 : reservedMatrixCalls);
326
+ if (limit !== undefined && sent >= limit) {
327
+ budget = {
328
+ kind: "max_calls",
329
+ message: `max_calls=${args.max_calls} reached`,
330
+ };
331
+ return { ok: false, error: budget.message };
332
+ }
333
+ const admitted = runtime.session.admit(questionCount);
334
+ if (!admitted.ok) {
335
+ budget = { kind: "session", message: admitted.error };
336
+ return admitted;
337
+ }
338
+ sent++;
339
+ return admitted;
340
+ },
341
+ onUsage: (usage) => runtime.session.recordUsage(usage),
342
+ });
343
+ const judge = async (state, questions, extra, matrixRequest = false) => {
344
+ state = withEvidenceContext(state, evidenceContext);
345
+ if (JSON.stringify(state).length > STATE_MAX_CHARS)
346
+ return {
347
+ ok: false,
348
+ error: `required evidence exceeds STATE_MAX_CHARS=${STATE_MAX_CHARS}`,
349
+ };
350
+ const result = await client.judge(state, questions, {
351
+ ...optionsFor(matrixRequest),
352
+ ...extra,
353
+ });
354
+ judgments.push(result);
355
+ return result;
356
+ };
357
+ const local = prepared.dimensions.some((dimension) => dimension.name === "reliability")
358
+ ? await collectRiskCallers(exec, { cwd: cwd, base, signal }, collected.units, collected.files, analysis.parser)
359
+ : { proofs: [], limits: [] };
360
+ const uncheckedCallers = new Set();
361
+ for (const limit of local.limits) {
362
+ if (limit.kind !== "dynamic_access_uncovered") {
363
+ const id = `caller:${limit.unitId}`;
364
+ report.expect(id, `${limit.unitId} local caller`, "unit");
365
+ report.diagnose("collection_omitted", limit.reason, limit.paths.join(" + "), [id]);
366
+ }
367
+ else
368
+ report.diagnose("dynamic_dependency", limit.reason, limit.paths.join(" + "), [], false);
369
+ limitations.push({
370
+ fact: `${limit.unitId}: ${limit.reason} — ${limit.paths.join(" + ")}`,
371
+ next: "read the named caller/provider pieces or supply an observation via jev_ask",
372
+ });
373
+ if (limit.kind !== "dynamic_access_uncovered" &&
374
+ !uncheckedCallers.has(limit.unitId)) {
375
+ uncheckedCallers.add(limit.unitId);
376
+ unchecked.push(`${limit.unitId} local caller ${limit.paths.join(" + ")}`);
377
+ }
378
+ }
379
+ const matrixPromise = judge(prepared.state, prepared.questions, { groups: prepared.groups, witnesses: prepared.witnessQuestionIds }, true).finally(() => {
380
+ reservedMatrixCalls = 0;
381
+ });
382
+ const localPromises = local.proofs.map(async (proof) => {
383
+ if (args.max_calls !== undefined &&
384
+ sent + reservedMatrixCalls >= args.max_calls)
385
+ await matrixPromise;
386
+ return {
387
+ proof,
388
+ result: await judge(proof.state, { caller: proof.question }),
389
+ };
390
+ });
391
+ const [result, locals] = await Promise.all([
392
+ matrixPromise,
393
+ Promise.all(localPromises),
394
+ ]);
395
+ const health = result.ok
396
+ ? evaluateWitnessHealth(prepared, result)
397
+ : {
398
+ healthy: false,
399
+ controls: [],
400
+ unhealthyQuestionIds: new Map(),
401
+ };
402
+ matrixHealthy = health.healthy;
403
+ for (const failure of health.controls)
404
+ report.diagnose(health.healthy ? "conservative_widening" : "control_failure", failure.fact, undefined, prepared.cells
405
+ .filter((cell) => health.unhealthyQuestionIds.has(cell.id))
406
+ .map((cell) => `risk:${cell.id}`), false);
407
+ report.countControls(result, prepared.witnessQuestionIds);
408
+ for (const cell of prepared.cells) {
409
+ const rawAnswer = result.ok ? result.answers[cell.id] : undefined;
410
+ const answer = rawAnswer?.type === "unjudged" && !rawAnswer.cause && budget
411
+ ? {
412
+ ...rawAnswer,
413
+ cause: budget.kind === "session"
414
+ ? "session_budget"
415
+ : "call_budget",
416
+ }
417
+ : rawAnswer;
418
+ const invalid = health.unhealthyQuestionIds.get(cell.id);
419
+ if (!result.ok)
420
+ report.failure(result, [`risk:${cell.id}`], budget
421
+ ? budget.kind === "session"
422
+ ? "session_budget"
423
+ : "call_budget"
424
+ : undefined);
425
+ else
426
+ report.answer(`risk:${cell.id}`, answer, {
427
+ band: invalid ? "unsure" : "verdict",
428
+ reason: invalid,
429
+ uncalibrated: cell.uncalibrated,
430
+ }, controlsFor(cell.id, result, prepared.witnessQuestionIds), false);
431
+ }
432
+ const decoys = new Set(prepared.witnesses
433
+ .filter((witness) => witness.expected === "no")
434
+ .map((witness) => witness.unitId)).size;
435
+ const references = new Set(prepared.witnesses
436
+ .filter((witness) => witness.expected === "yes")
437
+ .map((witness) => witness.unitId)).size;
438
+ const witnessStatus = health.healthy
439
+ ? "healthy"
440
+ : !result.ok ||
441
+ health.controls.some((control) => control.fact.includes("unavailable"))
442
+ ? "unavailable"
443
+ : "unhealthy";
444
+ limitations.push({
445
+ fact: prepared.witnesses.length
446
+ ? `risk matrix: ${decoys} decoys and ${references} references ${witnessStatus}`
447
+ : "risk matrix: witnesses disabled",
448
+ next: "read the findings and stated limits",
449
+ });
450
+ limitations.push(...health.controls);
451
+ if (!result.ok)
452
+ limitations.push({
453
+ fact: `risk matrix: ${result.error}`,
454
+ next: "check Jev configuration and availability, then retry",
455
+ });
456
+ const severity = [];
457
+ for (const cell of prepared.cells) {
458
+ const unit = collected.units.find((u) => u.id === cell.unitId);
459
+ if (!unit)
460
+ continue;
461
+ const answer = result.ok ? result.answers[cell.id] : undefined;
462
+ if (answer?.type !== "bool") {
463
+ unchecked.push(`${unitLabel(unit)} ${cell.dimension}`);
464
+ continue;
465
+ }
466
+ if (answer.p < FLAG_MIN)
467
+ continue;
468
+ const reason = health.unhealthyQuestionIds.get(cell.id);
469
+ const line = {
470
+ label: `${unitLabel(unit)} ${cell.dimension}${reason ? ` — ${reason}` : ""}`,
471
+ value: { head: "risk", p: answer.p },
472
+ band: reason ? "unsure" : "verdict",
473
+ uncalibrated: cell.uncalibrated,
474
+ };
475
+ answers.push(line);
476
+ unitOrder.set(line, unit.id);
477
+ if (!cell.uncalibrated)
478
+ severity.push({ unit, dimension: cell.dimension, lines: [line] });
479
+ }
480
+ for (const { proof, result: localResult } of locals) {
481
+ callerProofs.push(proof);
482
+ const unit = collected.units.find((u) => u.id === proof.unitId);
483
+ const reportId = `caller:${proof.unitId}`;
484
+ report.expect(reportId, `${unit ? unitLabel(unit) : proof.unitId} local caller`, "unit");
485
+ if (!localResult.ok) {
486
+ report.failure(localResult, [reportId], budget
487
+ ? budget.kind === "session"
488
+ ? "session_budget"
489
+ : "call_budget"
490
+ : undefined);
491
+ continue;
492
+ }
493
+ const answer = localResult.answers.caller;
494
+ if (answer?.type !== "choice") {
495
+ report.answer(reportId, answer);
496
+ unchecked.push(`${unit ? unitLabel(unit) : proof.unitId} local caller`);
497
+ continue;
498
+ }
499
+ const p = answer.probabilities.new_failure ?? 0;
500
+ const missing = (answer.probabilities.cannot_tell ?? 0) >= CANNOT_TELL_MIN;
501
+ const negative = !missing && p < BAND_BOOL_GRAY_A;
502
+ const merged = [];
503
+ for (const span of [...proof.paths].sort((a, b) => a.path.localeCompare(b.path) || a.start - b.start || a.end - b.end)) {
504
+ const previous = merged.at(-1);
505
+ if (previous &&
506
+ previous.path === span.path &&
507
+ span.start <= previous.end + 1)
508
+ previous.end = Math.max(previous.end, span.end);
509
+ else
510
+ merged.push({ ...span });
511
+ }
512
+ const omitted = Math.max(0, merged.length - CALLER_DISPLAY_MAX_SPANS);
513
+ const passages = merged
514
+ .slice(0, CALLER_DISPLAY_MAX_SPANS)
515
+ .map((span) => `${span.path}:${span.start}-${span.end}`)
516
+ .join(" + ");
517
+ const line = {
518
+ label: `${unit ? unitLabel(unit) : proof.unitId} reliability — static code only; ${passages}${omitted ? ` (+${omitted} passages)` : ""}`,
519
+ value: {
520
+ head: missing
521
+ ? "cannot_tell"
522
+ : negative
523
+ ? "no_new_failure"
524
+ : "new_failure",
525
+ p: missing
526
+ ? (answer.probabilities.cannot_tell ?? 0)
527
+ : negative
528
+ ? (answer.probabilities.no_new_failure ?? 0)
529
+ : p,
530
+ },
531
+ band: missing
532
+ ? "abstain"
533
+ : negative || p >= FLAG_MIN
534
+ ? "verdict"
535
+ : "unsure",
536
+ reason: missing
537
+ ? "actual provider or binding missing; supply its declaration or an observation via jev_ask"
538
+ : negative
539
+ ? "Local caller check does not flag a new failure; static evidence does not prove every caller safe."
540
+ : p < FLAG_MIN
541
+ ? "Local caller new-failure probability is in the gray band; inspect the stated proof passages natively."
542
+ : "Local caller new-failure probability meets the finding threshold; static code evidence only.",
543
+ };
544
+ const callerItem = report.items.get(reportId);
545
+ if (callerItem)
546
+ report.items.set(reportId, { ...callerItem, label: line.label });
547
+ report.answer(reportId, answer, line);
548
+ if (missing && line.reason)
549
+ report.diagnose("missing_required", line.reason, line.label, [reportId], false);
550
+ if (!unit || negative)
551
+ continue;
552
+ answers.push(line);
553
+ unitOrder.set(line, unit.id);
554
+ if (!missing && p >= FLAG_MIN) {
555
+ const existing = severity.find((item) => item.unit.id === unit.id && item.dimension === "reliability");
556
+ if (existing) {
557
+ existing.evidence = proof.state;
558
+ existing.lines.push(line);
559
+ }
560
+ else
561
+ severity.push({
562
+ unit,
563
+ dimension: "reliability",
564
+ lines: [line],
565
+ evidence: proof.state,
566
+ });
567
+ }
568
+ }
569
+ await Promise.all(severity.map(async (item) => {
570
+ const request = prepareRiskSeverity(item.unit, item.dimension, item.evidence?.callerEvidence);
571
+ const value = await judge(request.state, request.questions);
572
+ const reportId = `severity:${item.unit.id}:${item.dimension}`;
573
+ report.expect(reportId, `${item.unit.id} ${item.dimension} severity`, "unit");
574
+ if (value.ok)
575
+ report.answer(reportId, value.answers.severity);
576
+ else
577
+ report.failure(value, [reportId], budget
578
+ ? budget.kind === "session"
579
+ ? "session_budget"
580
+ : "call_budget"
581
+ : undefined);
582
+ const answer = value.ok ? value.answers.severity : undefined;
583
+ if (answer?.type === "score") {
584
+ const suffix = ` · severity ${answer.score.toFixed(1)}/3`;
585
+ for (const line of item.lines)
586
+ line.label += suffix;
587
+ }
588
+ else
589
+ unchecked.push(`${unitLabel(item.unit)} ${item.dimension} severity`);
590
+ }));
591
+ return finish();
592
+ },
593
+ };
594
+ }