jev-agent-tools 0.1.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (180) hide show
  1. package/CHANGELOG.md +106 -1
  2. package/CONTRIBUTING.md +43 -0
  3. package/README.md +58 -17
  4. package/SECURITY.md +43 -0
  5. package/dist/adapters/analysis-context.js +75 -0
  6. package/dist/adapters/ask-files.js +198 -0
  7. package/dist/adapters/ask-proof.js +200 -0
  8. package/dist/adapters/ask-syntax.js +385 -0
  9. package/dist/adapters/canonical-path.js +17 -0
  10. package/dist/adapters/command.js +234 -0
  11. package/dist/adapters/docs.js +192 -0
  12. package/dist/adapters/evidence-context.js +119 -0
  13. package/dist/adapters/exec.js +207 -0
  14. package/dist/adapters/files.js +418 -0
  15. package/dist/adapters/find.js +150 -0
  16. package/dist/adapters/git-base.js +32 -0
  17. package/dist/adapters/git-inventory.js +71 -0
  18. package/dist/adapters/git.js +483 -0
  19. package/dist/adapters/locate-file.js +197 -0
  20. package/dist/adapters/output-lines.js +46 -0
  21. package/dist/adapters/private-storage.js +106 -0
  22. package/dist/adapters/risk-callers.js +429 -0
  23. package/dist/adapters/runner-version.js +78 -0
  24. package/dist/adapters/shell.js +92 -0
  25. package/dist/adapters/syntax.js +187 -0
  26. package/dist/adapters/test-inventory.js +139 -0
  27. package/dist/adapters/usage.js +20 -0
  28. package/dist/adapters/utf8.js +47 -0
  29. package/dist/configuration.js +267 -0
  30. package/dist/constants.js +140 -0
  31. package/dist/core/ask-closure.js +282 -0
  32. package/dist/core/ask-proof.js +1 -0
  33. package/dist/core/ask-references.js +278 -0
  34. package/dist/core/asks.js +507 -0
  35. package/dist/core/batches.js +65 -0
  36. package/dist/core/command-output.js +224 -0
  37. package/dist/core/diff.js +178 -0
  38. package/dist/core/docs.js +302 -0
  39. package/dist/core/find.js +108 -0
  40. package/dist/core/git.js +1 -0
  41. package/dist/core/imports.js +550 -0
  42. package/dist/core/integrity.js +45 -0
  43. package/dist/core/lexical.js +132 -0
  44. package/dist/core/locate.js +169 -0
  45. package/dist/core/output.js +137 -0
  46. package/dist/core/pointer.js +29 -0
  47. package/dist/core/result-report.js +302 -0
  48. package/dist/core/risk-callers.js +851 -0
  49. package/dist/core/runner-version.js +45 -0
  50. package/dist/core/secret-path.js +34 -0
  51. package/dist/core/sections.js +230 -0
  52. package/dist/core/state.js +51 -0
  53. package/dist/core/syntax.js +1 -0
  54. package/dist/core/test-commands.js +334 -0
  55. package/dist/core/test-coverage.js +74 -0
  56. package/dist/core/test-discovery.js +1382 -0
  57. package/dist/core/test-evidence.js +527 -0
  58. package/dist/core/test-state.js +81 -0
  59. package/dist/core/truncate.js +12 -0
  60. package/dist/core/units.js +349 -0
  61. package/dist/describe.js +23 -0
  62. package/dist/guide.js +33 -0
  63. package/dist/host.js +24 -0
  64. package/dist/jev/client.js +456 -0
  65. package/dist/jev/pool.js +54 -0
  66. package/dist/jev/types.js +1 -0
  67. package/dist/mcp/main.js +124 -0
  68. package/dist/mcp/protocol.js +210 -0
  69. package/dist/mcp/tools.js +129 -0
  70. package/dist/presets/docs.js +62 -0
  71. package/dist/presets/risk.js +179 -0
  72. package/dist/presets/spec.js +81 -0
  73. package/dist/presets/witnesses.js +249 -0
  74. package/dist/render.js +114 -0
  75. package/dist/report-schema.js +1356 -0
  76. package/dist/result-types.js +1 -0
  77. package/dist/result.js +3 -0
  78. package/dist/runtime.js +1 -0
  79. package/dist/session.js +147 -0
  80. package/dist/texts/ask-files.js +3 -0
  81. package/dist/texts/ask.js +4 -0
  82. package/dist/texts/check-diff.js +20 -0
  83. package/dist/texts/configuration.js +1 -0
  84. package/dist/texts/find.js +19 -0
  85. package/dist/texts/guide.js +3 -0
  86. package/dist/texts/instructions.js +72 -0
  87. package/dist/texts/locate.js +15 -0
  88. package/dist/texts/select-tests.js +4 -0
  89. package/dist/tools/ask-files.js +450 -0
  90. package/dist/tools/ask-schema.js +70 -0
  91. package/dist/tools/ask.js +1147 -0
  92. package/dist/tools/check-diff.js +594 -0
  93. package/dist/tools/docs-check.js +408 -0
  94. package/dist/tools/find.js +682 -0
  95. package/dist/tools/locate.js +602 -0
  96. package/dist/tools/review-report.js +230 -0
  97. package/dist/tools/select-tests.js +821 -0
  98. package/dist/tools/spec-check.js +263 -0
  99. package/docs/adr/0001-strict-typescript-pure-core-offline-tests.md +31 -0
  100. package/docs/adr/0002-one-http-protocol-across-hosts.md +17 -0
  101. package/docs/adr/0003-explicit-scope-conservative-automation.md +19 -0
  102. package/docs/adr/0004-compiled-typed-intents.md +19 -0
  103. package/docs/adr/0005-evidence-construction-before-judgment.md +19 -0
  104. package/docs/adr/0006-visible-uncertainty-constrained-controls.md +21 -0
  105. package/docs/adr/0007-bounded-evidence-visible-limits.md +21 -0
  106. package/docs/adr/0008-static-test-discovery-conservative-plans.md +19 -0
  107. package/docs/adr/0009-session-cache-requested-model-identity.md +17 -0
  108. package/docs/adr/0010-mcp-server-thin-host.md +23 -0
  109. package/docs/agent-instructions.md +120 -0
  110. package/docs/design.md +16 -4
  111. package/docs/mcp.md +233 -0
  112. package/docs/tools/jev_ask.md +8 -5
  113. package/docs/tools/jev_ask_files.md +2 -1
  114. package/docs/tools/jev_check_diff.md +4 -1
  115. package/docs/tools/jev_find_files.md +2 -1
  116. package/docs/tools/jev_locate_in_file.md +5 -0
  117. package/docs/tools/jev_select_tests.md +4 -1
  118. package/package.json +19 -4
  119. package/rules/jev-ask.md +22 -1
  120. package/server.json +57 -0
  121. package/src/adapters/ask-files.ts +11 -3
  122. package/src/adapters/ask-proof.ts +69 -11
  123. package/src/adapters/canonical-path.ts +18 -0
  124. package/src/adapters/command.ts +102 -36
  125. package/src/adapters/docs.ts +33 -14
  126. package/src/adapters/evidence-context.ts +169 -0
  127. package/src/adapters/exec.ts +226 -0
  128. package/src/adapters/files.ts +146 -16
  129. package/src/adapters/find.ts +37 -7
  130. package/src/adapters/git-base.ts +7 -1
  131. package/src/adapters/git.ts +61 -8
  132. package/src/adapters/locate-file.ts +51 -9
  133. package/src/adapters/private-storage.ts +155 -0
  134. package/src/adapters/risk-callers.ts +7 -2
  135. package/src/adapters/shell.ts +113 -0
  136. package/src/adapters/test-inventory.ts +12 -4
  137. package/src/configuration.ts +55 -14
  138. package/src/constants.ts +37 -5
  139. package/src/core/ask-references.ts +262 -146
  140. package/src/core/asks.ts +79 -7
  141. package/src/core/command-output.ts +17 -1
  142. package/src/core/import-boundaries.ts +8 -3
  143. package/src/core/locate.ts +8 -5
  144. package/src/core/output.ts +34 -0
  145. package/src/core/result-report.ts +410 -0
  146. package/src/core/secret-path.ts +37 -0
  147. package/src/core/state.ts +8 -1
  148. package/src/core/units.ts +3 -2
  149. package/src/host.ts +11 -0
  150. package/src/index.ts +3 -0
  151. package/src/jev/client.ts +66 -16
  152. package/src/jev/types.ts +24 -3
  153. package/src/mcp/main.ts +135 -0
  154. package/src/mcp/protocol.ts +332 -0
  155. package/src/mcp/tools.ts +179 -0
  156. package/src/render.ts +109 -0
  157. package/src/report-schema.ts +1380 -0
  158. package/src/result-types.ts +234 -0
  159. package/src/result.ts +4 -1
  160. package/src/runtime.ts +6 -0
  161. package/src/session.ts +59 -0
  162. package/src/setup.ts +13 -5
  163. package/src/texts/ask-files.ts +4 -1
  164. package/src/texts/ask.ts +8 -1
  165. package/src/texts/check-diff.ts +7 -4
  166. package/src/texts/find.ts +8 -2
  167. package/src/texts/guide.ts +8 -16
  168. package/src/texts/instructions.ts +98 -0
  169. package/src/texts/locate.ts +8 -2
  170. package/src/texts/run-end.ts +2 -2
  171. package/src/texts/select-tests.ts +4 -1
  172. package/src/tools/ask-files.ts +311 -18
  173. package/src/tools/ask.ts +722 -95
  174. package/src/tools/check-diff.ts +337 -31
  175. package/src/tools/docs-check.ts +241 -38
  176. package/src/tools/find.ts +389 -29
  177. package/src/tools/locate.ts +387 -25
  178. package/src/tools/review-report.ts +308 -0
  179. package/src/tools/select-tests.ts +484 -23
  180. package/src/tools/spec-check.ts +194 -19
@@ -1,7 +1,12 @@
1
- import { matchesGlob, relative, resolve } from "node:path";
1
+ import { isAbsolute, matchesGlob, relative, resolve } from "node:path";
2
2
  import type { ToolDefinition } from "@earendil-works/pi-coding-agent";
3
3
  import { type Static, Type } from "@sinclair/typebox";
4
4
  import { createAnalysisContext } from "../adapters/analysis-context.ts";
5
+ import {
6
+ type EvidenceContext,
7
+ resolveEvidenceContext,
8
+ withEvidenceContext,
9
+ } from "../adapters/evidence-context.ts";
5
10
  import { collectUnits } from "../adapters/git.ts";
6
11
  import { resolveBase } from "../adapters/git-base.ts";
7
12
  import { shareGitInventory } from "../adapters/git-inventory.ts";
@@ -25,6 +30,11 @@ import type {
25
30
  Limitation,
26
31
  } from "../core/output.ts";
27
32
  import { buildEnvelope } from "../core/output.ts";
33
+ import {
34
+ type Cause,
35
+ type ResultReportV1,
36
+ known as reportKnown,
37
+ } from "../core/result-report.ts";
28
38
  import { buildRunnerCommands } from "../core/test-commands.ts";
29
39
  import { prepareCoverageWitnesses } from "../core/test-coverage.ts";
30
40
  import type { TestEntry } from "../core/test-discovery.ts";
@@ -41,12 +51,14 @@ import {
41
51
  buildCoverageWitnessUnits,
42
52
  evaluateBatchWitnessHealth,
43
53
  } from "../presets/witnesses.ts";
44
- import { renderEnvelope } from "../render.ts";
54
+ import { renderResultReport } from "../render.ts";
45
55
  import type { ToolDependencies } from "../runtime.ts";
46
56
  import { SELECT_TESTS_DESCRIPTION } from "../texts/select-tests.ts";
57
+ import { controlsFor, ReviewReport, reportMetrics } from "./review-report.ts";
47
58
 
48
59
  export const selectTestsParameters = Type.Object(
49
60
  {
61
+ root: Type.Optional(Type.String()),
50
62
  base: Type.Optional(Type.String({ minLength: 1 })),
51
63
  paths: Type.Optional(Type.Array(Type.String({ minLength: 1 }))),
52
64
  witnesses: Type.Optional(
@@ -125,6 +137,13 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
125
137
  ctx: { cwd: string } & GuideContext,
126
138
  ) {
127
139
  const client = dependencies.client;
140
+ const evidence = await resolveEvidenceContext(ctx.cwd, args.root, {
141
+ exec: execute,
142
+ signal,
143
+ origin: dependencies.evidenceOrigin,
144
+ });
145
+ const evidenceContext = evidence.context;
146
+ const cwd = evidence.ok ? evidence.cwd : evidenceContext.authority.path;
128
147
  const started = performance.now();
129
148
  const exec = shareGitInventory(execute);
130
149
  const totals = {
@@ -134,10 +153,17 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
134
153
  cacheRequests: 0,
135
154
  usage: { inputTokens: 0, costUsd: 0 },
136
155
  };
156
+ const selectionEvidence = {
157
+ criteria: [] as { criterion: string; matches: string[] }[],
158
+ wideningTriggers: [] as string[],
159
+ inventory: [] as string[],
160
+ };
161
+ const report = new ReviewReport();
137
162
  let costKnown = false;
163
+ let unknownCost = false;
138
164
  let sent = 0;
139
165
  let budget: BudgetRefusal | undefined;
140
- const finish = (input: Omit<EnvelopeInput, "yield">) => {
166
+ const finish = (input: Omit<EnvelopeInput, "yield">, cause?: Cause) => {
141
167
  const limitKeys = new Set<string>();
142
168
  const uniqueLimits = input.limitations?.filter((limit) => {
143
169
  const key = JSON.stringify([limit.fact, limit.next]);
@@ -150,37 +176,97 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
150
176
  limitations: uniqueLimits,
151
177
  yield: {
152
178
  ...totals,
153
- costUsd: costKnown ? totals.usage.costUsd : undefined,
179
+ costUsd:
180
+ costKnown && !unknownCost ? totals.usage.costUsd : undefined,
154
181
  elapsedMs: performance.now() - started,
155
182
  },
156
183
  });
184
+ if (input.refusal) {
185
+ report.refusal = true;
186
+ report.diagnose(cause ?? "internal_error", input.refusal);
187
+ }
188
+ if (!report.items.size && !input.refusal)
189
+ report.diagnose(
190
+ "collection_empty",
191
+ "No test decision candidates in the discovered inventory; this is not proof of no affected tests.",
192
+ undefined,
193
+ [],
194
+ false,
195
+ );
196
+ const result = report.build(
197
+ "jev_select_tests",
198
+ evidenceContext,
199
+ reportMetrics(envelope),
200
+ );
157
201
  runtime.session.record(envelope);
158
202
  runtime.guide.deliver(ctx);
159
- const details: Judgment & { limitations?: readonly Limitation[] } = {
203
+ const details: Judgment & {
204
+ result: ResultReportV1;
205
+ limitations?: readonly Limitation[];
206
+ evidenceContext: EvidenceContext;
207
+ selectionEvidence: typeof selectionEvidence;
208
+ } = {
160
209
  ok: true,
161
210
  answers: {},
162
211
  ...totals,
212
+ evidenceContext,
213
+ selectionEvidence,
214
+ result,
163
215
  limitations: uniqueLimits,
164
216
  };
165
217
  return {
166
- content: [{ type: "text" as const, text: renderEnvelope(envelope) }],
218
+ content: [
219
+ {
220
+ type: "text" as const,
221
+ text: renderResultReport(result, { details: envelope }),
222
+ },
223
+ ],
167
224
  details,
168
225
  ...hostUsage(host.isOmp, costKnown ? totals.usage : undefined),
169
226
  };
170
227
  };
171
- const comparison = await resolveBase(exec, ctx.cwd, args.base, signal);
172
- if (!comparison.ok) return finish({ refusal: comparison.error });
228
+ if (!evidence.ok)
229
+ return finish(
230
+ { refusal: evidence.error },
231
+ evidence.cause ?? "invalid_root",
232
+ );
233
+ evidenceContext.requestedBase = args.base ?? "HEAD";
234
+ for (const path of args.paths ?? [])
235
+ if (
236
+ isAbsolute(path) ||
237
+ path.split(/[\\/]/).includes("..") ||
238
+ path.split(/[\\/]/).includes(".git")
239
+ )
240
+ return finish(
241
+ { refusal: `Path not permitted: ${path}` },
242
+ "forbidden_path",
243
+ );
244
+ const comparison = await resolveBase(exec, cwd, args.base, signal);
245
+ if (!comparison.ok)
246
+ return finish(
247
+ { refusal: comparison.error },
248
+ comparison.cause ?? "invalid_base",
249
+ );
250
+ evidenceContext.resolvedBase = comparison.base;
173
251
  const analysis = await createAnalysisContext();
174
252
  const [inventory, diff] = await Promise.all([
175
- collectTestInventory(exec, ctx.cwd, signal),
253
+ collectTestInventory(exec, cwd, signal),
176
254
  collectUnits(
177
255
  exec,
178
- { cwd: ctx.cwd, base: comparison.base, signal },
256
+ { cwd: cwd, base: comparison.base, signal },
179
257
  analysis.parser,
180
258
  ),
181
259
  ]);
182
- if (!inventory.ok) return finish({ refusal: inventory.error });
183
- if (!diff.ok) return finish({ refusal: diff.error });
260
+ if (!inventory.ok)
261
+ return finish(
262
+ { refusal: inventory.error },
263
+ inventory.cause ?? "file_unavailable",
264
+ );
265
+ if (!diff.ok)
266
+ return finish({ refusal: diff.error }, diff.cause ?? "git_failure");
267
+ if (evidenceContext.effectiveRoot)
268
+ evidenceContext.effectiveRoot.path = inventory.cwd;
269
+ selectionEvidence.inventory = [...inventory.paths];
184
270
  const known = new Set(inventory.paths);
185
271
  const sourceFiles = new Map<string, ImportSource>();
186
272
  const read = async (path: string): Promise<ImportSource | undefined> => {
@@ -275,6 +361,23 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
275
361
  !graph.edges.has(file.path) ||
276
362
  !/\.[cm]?[jt]sx?$|\.py$/.test(file.path),
277
363
  );
364
+ selectionEvidence.wideningTriggers = diff.files
365
+ .filter(
366
+ (file) =>
367
+ !graph.edges.has(file.path) ||
368
+ !/\.[cm]?[jt]sx?$|\.py$/.test(file.path),
369
+ )
370
+ .map((file) => file.path);
371
+ selectionEvidence.criteria = (args.paths ?? []).map((criterion) => ({
372
+ criterion,
373
+ matches: versionedEntries
374
+ .filter(
375
+ (entry) =>
376
+ matchesGlob(entry.path, criterion) ||
377
+ entry.path.startsWith(`${criterion.replace(/\/$/, "")}/`),
378
+ )
379
+ .map((entry) => entry.path),
380
+ }));
278
381
  const candidates = versionedEntries
279
382
  .filter(
280
383
  (entry) =>
@@ -306,6 +409,77 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
306
409
  outside,
307
410
  );
308
411
  });
412
+ report.inventories.push({
413
+ id: "test-candidates",
414
+ kind: "tests",
415
+ rules: [
416
+ "Tracked supported test declarations and literal project runner configuration",
417
+ "Import closure selection; conservative fallback preserves unjudged decisions",
418
+ ],
419
+ restrictions: args.paths ?? [],
420
+ discovered: reportKnown(versionedEntries.length),
421
+ considered: reportKnown(candidates.length),
422
+ scopeRestricted: Boolean(args.paths?.length),
423
+ criteria: selectionEvidence.criteria.map((criterion) => ({
424
+ criterion: criterion.criterion,
425
+ matches: reportKnown(criterion.matches.length),
426
+ outcome: criterion.matches.length ? "matched" : "no_match",
427
+ diagnosticIds: [],
428
+ })),
429
+ });
430
+ for (const candidate of candidates) {
431
+ const scenarios = candidate.entry.scenarios;
432
+ if (!scenarios.length)
433
+ report.expect(
434
+ `test:${candidate.entry.path}:unknown`,
435
+ candidate.entry.path,
436
+ "scenario",
437
+ );
438
+ for (const scenario of scenarios)
439
+ report.expect(
440
+ `test:${candidate.entry.path}:${scenario.id}`,
441
+ `${candidate.entry.path}: ${scenario.name}`,
442
+ "scenario",
443
+ `test:${candidate.entry.path}`,
444
+ );
445
+ }
446
+ for (const criterion of selectionEvidence.criteria)
447
+ if (!criterion.matches.length) {
448
+ report.diagnose(
449
+ "criteria_no_match",
450
+ `No discovered tests match criterion ${criterion.criterion}`,
451
+ criterion.criterion,
452
+ );
453
+ const excluded = inventory.limits.filter(
454
+ (limit) =>
455
+ matchesGlob(limit.path, criterion.criterion) ||
456
+ limit.path.startsWith(
457
+ `${criterion.criterion.replace(/\/$/, "")}/`,
458
+ ),
459
+ );
460
+ if (excluded.length) {
461
+ report.diagnose(
462
+ "outside_inventory",
463
+ "Requested criterion names entries excluded from the admitted inventory",
464
+ criterion.criterion,
465
+ [],
466
+ false,
467
+ excluded.map((limit) => limit.path),
468
+ );
469
+ const entry = report.inventories[0]?.criteria.find(
470
+ (entry) => entry.criterion === criterion.criterion,
471
+ );
472
+ if (entry) entry.outcome = "outside_inventory";
473
+ }
474
+ }
475
+ if (selectionEvidence.wideningTriggers.length)
476
+ report.diagnose(
477
+ "conservative_widening",
478
+ `Changed files outside supported dependency graph: ${selectionEvidence.wideningTriggers.join(", ")}`,
479
+ undefined,
480
+ [],
481
+ false,
482
+ );
309
483
  const selected: {
310
484
  entry: TestEntry;
311
485
  scenarioIds: readonly string[] | null;
@@ -371,9 +545,11 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
371
545
  questions: Record<string, Question>,
372
546
  witnessIds?: readonly string[],
373
547
  ) => {
548
+ state = withEvidenceContext(state, evidenceContext);
374
549
  if (JSON.stringify(state).length > STATE_MAX_CHARS)
375
550
  return {
376
551
  ok: false as const,
552
+ cause: "evidence_too_large" as const,
377
553
  error: `required evidence exceeds STATE_MAX_CHARS=${STATE_MAX_CHARS}`,
378
554
  };
379
555
  if (args.max_calls !== undefined && sent >= args.max_calls) {
@@ -387,6 +563,9 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
387
563
  const result = await client.judge(state, questions, {
388
564
  signal,
389
565
  witnesses: witnessIds,
566
+ ...runtime.session.requestGate(),
567
+ admissionCause: () =>
568
+ budget?.kind === "session" ? "session_budget" : "call_budget",
390
569
  beforeRequest: (questionCount) => {
391
570
  if (args.max_calls !== undefined && sent >= args.max_calls) {
392
571
  budget = {
@@ -409,6 +588,7 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
409
588
  totals.questions += result.questions ?? 0;
410
589
  totals.cacheHits += result.cacheHits ?? 0;
411
590
  totals.cacheRequests += result.cacheRequests ?? 0;
591
+ unknownCost ||= result.usage === undefined;
412
592
  if (result.usage) {
413
593
  costKnown = true;
414
594
  totals.usage.inputTokens += result.usage.inputTokens;
@@ -419,6 +599,27 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
419
599
  const pointerResults = await Promise.all(
420
600
  candidates.map(async (candidate) => {
421
601
  const { entry } = candidate;
602
+ const reportIds = [...report.items.values()]
603
+ .filter(
604
+ (item) =>
605
+ item.groupId === `test:${entry.path}` ||
606
+ item.id === `test:${entry.path}:unknown`,
607
+ )
608
+ .map((item) => item.id);
609
+ if (candidate.touched)
610
+ for (const id of reportIds) report.static(id, "touched", true);
611
+ if (
612
+ !candidate.units.length &&
613
+ !candidate.uncertain &&
614
+ !outside &&
615
+ !candidate.touched
616
+ )
617
+ for (const id of reportIds)
618
+ report.static(
619
+ id,
620
+ "static import closure cannot reach changed units",
621
+ false,
622
+ );
422
623
  if (candidate.touched)
423
624
  return {
424
625
  candidate,
@@ -437,6 +638,12 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
437
638
  if (
438
639
  units.some((unit) => unit.before === null && unit.after === null)
439
640
  ) {
641
+ report.diagnose(
642
+ "binary_or_non_utf8",
643
+ "Changed source unavailable",
644
+ entry.path,
645
+ reportIds,
646
+ );
440
647
  for (const unit of units) incompleteUnits.add(unit.id);
441
648
  unjudged.push({
442
649
  label: entry.path,
@@ -446,6 +653,13 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
446
653
  return { candidate, selected: null, touched: false };
447
654
  }
448
655
  if (!entry.scenarios.length) {
656
+ report.totalUnknown = true;
657
+ report.diagnose(
658
+ "unsupported_syntax",
659
+ "Scenario names or count unresolved",
660
+ entry.path,
661
+ reportIds,
662
+ );
449
663
  unjudged.push({
450
664
  label: entry.path,
451
665
  reason: "scenario names or count unresolved",
@@ -458,6 +672,9 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
458
672
  (item) => item.scenario.id,
459
673
  );
460
674
  for (const item of prepared.unjudged) {
675
+ report.diagnose("evidence_too_large", item.reason, entry.path, [
676
+ `test:${entry.path}:${item.scenario.id}`,
677
+ ]);
461
678
  for (const unit of units) incompleteUnits.add(unit.id);
462
679
  unjudged.push({
463
680
  label: `${entry.path}: ${item.scenario.name}`,
@@ -491,6 +708,41 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
491
708
  ]),
492
709
  );
493
710
  const result = await judge(batch.state, questions);
711
+ const batchIds = batch.scenarios.map(
712
+ (scenario) => `test:${entry.path}:${scenario.id}`,
713
+ );
714
+ if (!result.ok)
715
+ report.failure(
716
+ result,
717
+ batchIds,
718
+ budget
719
+ ? budget.kind === "session"
720
+ ? "session_budget"
721
+ : "call_budget"
722
+ : !client
723
+ ? "not_configured"
724
+ : undefined,
725
+ );
726
+ else
727
+ for (const scenario of batch.scenarios) {
728
+ const answer = result.answers[scenario.id];
729
+ const id = `test:${entry.path}:${scenario.id}`;
730
+ report.answer(id, answer, {
731
+ band: prepared.unjudged.length ? "unsure" : "verdict",
732
+ });
733
+ const item = report.items.get(id);
734
+ if (!item)
735
+ throw new Error(`Unregistered selection result ${id}`);
736
+ item.selection = {
737
+ selected:
738
+ answer?.type !== "choice" ||
739
+ 1 - (answer.probabilities.none ?? 0) >= SELECT_MIN,
740
+ reason:
741
+ answer?.type === "choice"
742
+ ? "changed-unit pointer selection threshold"
743
+ : "conservative_fallback",
744
+ };
745
+ }
494
746
  if (!result.ok) {
495
747
  fallback ||= !budget;
496
748
  for (const unit of units) incompleteUnits.add(unit.id);
@@ -584,6 +836,24 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
584
836
  preliminary.state,
585
837
  candidate.entry.scenarios,
586
838
  );
839
+ for (const unit of units)
840
+ for (const scenario of candidate.entry.scenarios)
841
+ report.expect(
842
+ `coverage:${candidate.entry.path}:${scenario.id}:${unit.id}`,
843
+ `${candidate.entry.path}: ${scenario.name} executes ${unit.name}`,
844
+ "scenario",
845
+ `coverage:${candidate.entry.path}:${scenario.id}`,
846
+ );
847
+ for (const omitted of prepared.unjudged)
848
+ for (const unit of units)
849
+ report.diagnose(
850
+ "evidence_too_large",
851
+ omitted.reason,
852
+ candidate.entry.path,
853
+ [
854
+ `coverage:${candidate.entry.path}:${omitted.scenario.id}:${unit.id}`,
855
+ ],
856
+ );
587
857
  if (prepared.unjudged.length || !candidate.entry.scenarios.length)
588
858
  for (const unit of units) incompleteUnits.add(unit.id);
589
859
  await Promise.all(
@@ -617,6 +887,72 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
617
887
  result,
618
888
  );
619
889
  limitations.push(...health.failures);
890
+ for (const failure of health.failures)
891
+ report.diagnose(
892
+ "control_failure",
893
+ failure.fact,
894
+ candidate.entry.path,
895
+ units.flatMap((unit) =>
896
+ batch.scenarios
897
+ .filter((scenario) =>
898
+ health.unhealthyQuestionIds.has(
899
+ `${scenario.id}_${unit.id}`,
900
+ ),
901
+ )
902
+ .map(
903
+ (scenario) =>
904
+ `coverage:${candidate.entry.path}:${scenario.id}:${unit.id}`,
905
+ ),
906
+ ),
907
+ false,
908
+ );
909
+ report.countControls(
910
+ result,
911
+ witnessQuestions.witnesses.map((witness) => witness.id),
912
+ );
913
+ for (const unit of units)
914
+ for (const scenario of batch.scenarios) {
915
+ const id = `coverage:${candidate.entry.path}:${scenario.id}:${unit.id}`;
916
+ report.expect(
917
+ id,
918
+ `${candidate.entry.path}: ${scenario.name} executes ${unit.name}`,
919
+ "scenario",
920
+ `coverage:${candidate.entry.path}:${scenario.id}`,
921
+ );
922
+ if (!result.ok)
923
+ report.failure(
924
+ result,
925
+ [id],
926
+ budget
927
+ ? budget.kind === "session"
928
+ ? "session_budget"
929
+ : "call_budget"
930
+ : !client
931
+ ? "not_configured"
932
+ : undefined,
933
+ );
934
+ else
935
+ report.answer(
936
+ id,
937
+ result.answers[`${scenario.id}_${unit.id}`],
938
+ {
939
+ band: health.unhealthyQuestionIds.has(
940
+ `${scenario.id}_${unit.id}`,
941
+ )
942
+ ? "unsure"
943
+ : "verdict",
944
+ reason: health.unhealthyQuestionIds.get(
945
+ `${scenario.id}_${unit.id}`,
946
+ ),
947
+ },
948
+ controlsFor(
949
+ `${scenario.id}_${unit.id}`,
950
+ result,
951
+ witnessQuestions.witnesses.map((witness) => witness.id),
952
+ ),
953
+ false,
954
+ );
955
+ }
620
956
  for (const unit of units) {
621
957
  if (!result.ok) {
622
958
  fallback ||= !budget;
@@ -682,16 +1018,34 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
682
1018
  evidenceLimits.length > 0 ||
683
1019
  Boolean(args.paths?.length);
684
1020
  for (const unit of residual)
685
- if ((coverage.get(unit.id) ?? 0) < SELECT_MIN)
686
- answers.push({
687
- label: `changed, run by no discovered test: ${unit.name} (${unit.file})`,
688
- value: {
689
- head: "within discovered inventory only",
690
- p: coverage.get(unit.id) ?? 0,
691
- },
692
- band:
693
- incomplete || incompleteUnits.has(unit.id) ? "unsure" : "verdict",
694
- });
1021
+ if ((coverage.get(unit.id) ?? 0) < SELECT_MIN) {
1022
+ const supportingIds = [...report.items.keys()].filter(
1023
+ (id) => id.startsWith("coverage:") && id.endsWith(`:${unit.id}`),
1024
+ );
1025
+ report.diagnose(
1026
+ incomplete || incompleteUnits.has(unit.id)
1027
+ ? "collection_omitted"
1028
+ : "conservative_widening",
1029
+ `Derived from retained coverage decisions: no discovered test established execution of ${unit.name} (${unit.file}) within the considered inventory only.${incomplete || incompleteUnits.has(unit.id) ? " Absence remains unsure because evidence, scope, or controls are incomplete." : " This is not a global coverage claim."}`,
1030
+ unit.file,
1031
+ supportingIds,
1032
+ false,
1033
+ );
1034
+ if (!supportingIds.length) {
1035
+ const diagnostic = report.diagnostics.at(-1);
1036
+ if (diagnostic)
1037
+ diagnostic.scope = {
1038
+ kind: "inventory",
1039
+ inventoryIds: ["test-candidates"],
1040
+ };
1041
+ const action = report.actions.at(-1);
1042
+ if (action)
1043
+ action.scope = {
1044
+ kind: "inventory",
1045
+ inventoryIds: ["test-candidates"],
1046
+ };
1047
+ }
1048
+ }
695
1049
  if (fallback) {
696
1050
  selected.length = 0;
697
1051
  selected.push(
@@ -700,6 +1054,13 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
700
1054
  scenarioIds: null,
701
1055
  })),
702
1056
  );
1057
+ report.diagnose(
1058
+ "conservative_widening",
1059
+ "Unavailable judgment conservatively selects all considered test candidates; static reachability exclusions do not narrow this fallback.",
1060
+ undefined,
1061
+ [],
1062
+ false,
1063
+ );
703
1064
  limitations.push({
704
1065
  fact: "fallback: all",
705
1066
  next: "run all discovered tests with their project runner",
@@ -716,11 +1077,111 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
716
1077
  })),
717
1078
  ),
718
1079
  );
1080
+ for (const limit of commands.limits)
1081
+ for (const path of limit.files)
1082
+ report.diagnose("unresolved_runner", limit.reason, path, [], false);
1083
+ for (const limit of discovery.limits)
1084
+ report.diagnose(
1085
+ "unresolved_runner",
1086
+ limit.kind,
1087
+ limit.path,
1088
+ [],
1089
+ limit.kind !== "local_runner_unproven" &&
1090
+ limit.kind !== "interactive_script_skipped",
1091
+ );
1092
+ for (const limit of inventory.limits)
1093
+ report.diagnose(
1094
+ limit.kind === "secret_pattern"
1095
+ ? "secret_pattern"
1096
+ : "collection_omitted",
1097
+ limit.kind,
1098
+ limit.path,
1099
+ );
1100
+ for (const limit of diff.limits)
1101
+ report.diagnose(
1102
+ limit.kind === "secret_pattern"
1103
+ ? "secret_pattern"
1104
+ : "collection_omitted",
1105
+ limit.kind,
1106
+ limit.file,
1107
+ );
1108
+ for (const limit of graph.limits)
1109
+ report.diagnose(
1110
+ "dynamic_dependency",
1111
+ `${limit.kind}${limit.specifier ? ` (${limit.specifier})` : ""}`,
1112
+ limit.path,
1113
+ [],
1114
+ false,
1115
+ );
1116
+ for (const item of report.items.values()) {
1117
+ const candidate = candidates.find(
1118
+ (candidate) =>
1119
+ item.id.startsWith(`test:${candidate.entry.path}:`) ||
1120
+ item.id.startsWith(`coverage:${candidate.entry.path}:`),
1121
+ );
1122
+ if (!candidate) continue;
1123
+ const plan = selected.find((plan) => plan.entry === candidate.entry);
1124
+ const scenario = candidate.entry.scenarios.find(
1125
+ (scenario) =>
1126
+ item.id === `test:${candidate.entry.path}:${scenario.id}` ||
1127
+ item.id.startsWith(
1128
+ `coverage:${candidate.entry.path}:${scenario.id}:`,
1129
+ ),
1130
+ );
1131
+ const isSelected = Boolean(
1132
+ plan &&
1133
+ (plan.scenarioIds === null ||
1134
+ (scenario && plan.scenarioIds.includes(scenario.id))),
1135
+ );
1136
+ item.selection = {
1137
+ selected: isSelected,
1138
+ reason: fallback
1139
+ ? "conservative_fallback"
1140
+ : item.treatment === "static"
1141
+ ? item.staticReason
1142
+ : item.treatment === "not_judged"
1143
+ ? "conservative_fallback"
1144
+ : "unchanged selection policy and whole-file widening",
1145
+ };
1146
+ }
1147
+ for (const criterion of report.inventories[0]?.criteria ?? [])
1148
+ criterion.diagnosticIds = report.diagnostics
1149
+ .filter(
1150
+ (diagnostic) =>
1151
+ diagnostic.cause === "criteria_no_match" &&
1152
+ diagnostic.target.status === "known" &&
1153
+ diagnostic.target.value === criterion.criterion,
1154
+ )
1155
+ .map((diagnostic) => diagnostic.id);
1156
+ report.actions.push({
1157
+ id: "execute-selection-plan",
1158
+ code: "execute_plan",
1159
+ target: reportKnown(inventory.cwd),
1160
+ scope: { kind: "call" },
1161
+ condition:
1162
+ "When verification is authorized and the listed runner/configuration is available.",
1163
+ instruction:
1164
+ "Execute the retained runner commands separately; this tool has not executed any test.",
1165
+ repeatUnchanged: false,
1166
+ });
719
1167
  if (skipped)
720
1168
  limitations.push({
721
1169
  fact: `${skipped} tests skipped because their imports cannot reach the diff`,
722
1170
  next: "an alias or dynamic import would have kept a test in",
723
1171
  });
1172
+ for (const criterion of selectionEvidence.criteria)
1173
+ if (!criterion.matches.length)
1174
+ limitations.push({
1175
+ path: criterion.criterion,
1176
+ cause: "selection criterion has no discovered test match",
1177
+ fact: `selection criterion without match: ${criterion.criterion}`,
1178
+ next: "Change the criterion or supply the missing test/configuration evidence; an empty scope does not prove absence of impact.",
1179
+ });
1180
+ if (selectionEvidence.wideningTriggers.length)
1181
+ limitations.push({
1182
+ fact: `selection widened conservatively: ${selectionEvidence.wideningTriggers.join(", ")}`,
1183
+ next: "Run the widened static plans; dependency reachability is not established for these changed files.",
1184
+ });
724
1185
  return finish({
725
1186
  answers,
726
1187
  unjudged,
@@ -735,7 +1196,7 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
735
1196
  lines: commands.commands.map((command) => ({
736
1197
  type: "command" as const,
737
1198
  ...command,
738
- cwd: relative(ctx.cwd, resolve(inventory.cwd, command.cwd)) || ".",
1199
+ cwd: relative(cwd, resolve(inventory.cwd, command.cwd)) || ".",
739
1200
  })),
740
1201
  });
741
1202
  },