jev-agent-tools 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +37 -1
- package/CONTRIBUTING.md +3 -0
- package/README.md +22 -14
- package/SECURITY.md +17 -1
- package/dist/adapters/ask-files.js +11 -2
- package/dist/adapters/ask-proof.js +63 -7
- package/dist/adapters/command.js +82 -29
- package/dist/adapters/docs.js +30 -10
- package/dist/adapters/evidence-context.js +119 -0
- package/dist/adapters/files.js +141 -16
- package/dist/adapters/find.js +34 -6
- package/dist/adapters/git-base.js +7 -1
- package/dist/adapters/git.js +51 -7
- package/dist/adapters/locate-file.js +47 -9
- package/dist/adapters/private-storage.js +14 -6
- package/dist/adapters/risk-callers.js +3 -0
- package/dist/adapters/shell.js +23 -7
- package/dist/adapters/test-inventory.js +10 -2
- package/dist/configuration.js +17 -7
- package/dist/constants.js +26 -5
- package/dist/core/ask-references.js +193 -109
- package/dist/core/asks.js +78 -7
- package/dist/core/locate.js +8 -8
- package/dist/core/output.js +17 -0
- package/dist/core/result-report.js +302 -0
- package/dist/core/secret-path.js +34 -0
- package/dist/core/state.js +8 -1
- package/dist/core/units.js +1 -1
- package/dist/jev/client.js +34 -12
- package/dist/mcp/protocol.js +50 -27
- package/dist/mcp/tools.js +20 -7
- package/dist/render.js +72 -0
- package/dist/report-schema.js +1356 -0
- package/dist/result-types.js +1 -0
- package/dist/texts/ask-files.js +3 -1
- package/dist/texts/ask.js +3 -1
- package/dist/texts/check-diff.js +7 -4
- package/dist/texts/find.js +7 -2
- package/dist/texts/guide.js +3 -16
- package/dist/texts/instructions.js +72 -0
- package/dist/texts/locate.js +7 -2
- package/dist/texts/select-tests.js +3 -1
- package/dist/tools/ask-files.js +248 -15
- package/dist/tools/ask.js +523 -62
- package/dist/tools/check-diff.js +222 -30
- package/dist/tools/docs-check.js +122 -13
- package/dist/tools/find.js +320 -27
- package/dist/tools/locate.js +317 -18
- package/dist/tools/review-report.js +230 -0
- package/dist/tools/select-tests.js +273 -19
- package/dist/tools/spec-check.js +119 -22
- package/docs/adr/0001-strict-typescript-pure-core-offline-tests.md +3 -3
- package/docs/agent-instructions.md +59 -30
- package/docs/design.md +13 -1
- package/docs/mcp.md +8 -6
- package/docs/tools/jev_ask.md +8 -5
- package/docs/tools/jev_ask_files.md +2 -1
- package/docs/tools/jev_check_diff.md +4 -1
- package/docs/tools/jev_find_files.md +2 -1
- package/docs/tools/jev_locate_in_file.md +5 -0
- package/docs/tools/jev_select_tests.md +4 -1
- package/package.json +1 -1
- package/rules/jev-ask.md +22 -1
- package/server.json +2 -2
- package/src/adapters/ask-files.ts +11 -3
- package/src/adapters/ask-proof.ts +69 -11
- package/src/adapters/command.ts +96 -33
- package/src/adapters/docs.ts +33 -14
- package/src/adapters/evidence-context.ts +169 -0
- package/src/adapters/files.ts +146 -16
- package/src/adapters/find.ts +37 -7
- package/src/adapters/git-base.ts +7 -1
- package/src/adapters/git.ts +61 -8
- package/src/adapters/locate-file.ts +51 -9
- package/src/adapters/private-storage.ts +17 -5
- package/src/adapters/risk-callers.ts +3 -0
- package/src/adapters/shell.ts +23 -7
- package/src/adapters/test-inventory.ts +12 -4
- package/src/configuration.ts +16 -2
- package/src/constants.ts +26 -5
- package/src/core/ask-references.ts +262 -146
- package/src/core/asks.ts +79 -7
- package/src/core/import-boundaries.ts +8 -3
- package/src/core/locate.ts +8 -5
- package/src/core/output.ts +34 -0
- package/src/core/result-report.ts +410 -0
- package/src/core/secret-path.ts +37 -0
- package/src/core/state.ts +8 -1
- package/src/core/units.ts +3 -2
- package/src/index.ts +3 -0
- package/src/jev/client.ts +54 -16
- package/src/jev/types.ts +18 -3
- package/src/mcp/protocol.ts +91 -41
- package/src/mcp/tools.ts +26 -13
- package/src/render.ts +109 -0
- package/src/report-schema.ts +1380 -0
- package/src/result-types.ts +234 -0
- package/src/result.ts +4 -1
- package/src/runtime.ts +6 -0
- package/src/texts/ask-files.ts +4 -1
- package/src/texts/ask.ts +8 -1
- package/src/texts/check-diff.ts +7 -4
- package/src/texts/find.ts +8 -2
- package/src/texts/guide.ts +8 -16
- package/src/texts/instructions.ts +98 -0
- package/src/texts/locate.ts +8 -2
- package/src/texts/run-end.ts +2 -2
- package/src/texts/select-tests.ts +4 -1
- package/src/tools/ask-files.ts +309 -14
- package/src/tools/ask.ts +700 -77
- package/src/tools/check-diff.ts +331 -28
- package/src/tools/docs-check.ts +241 -39
- package/src/tools/find.ts +386 -29
- package/src/tools/locate.ts +384 -19
- package/src/tools/review-report.ts +308 -0
- package/src/tools/select-tests.ts +479 -21
- package/src/tools/spec-check.ts +193 -19
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { matchesGlob, relative, resolve } from "node:path";
|
|
1
|
+
import { isAbsolute, matchesGlob, relative, resolve } from "node:path";
|
|
2
2
|
import { Type } from "@sinclair/typebox";
|
|
3
3
|
import { createAnalysisContext } from "../adapters/analysis-context.js";
|
|
4
|
-
import {
|
|
4
|
+
import { resolveEvidenceContext, withEvidenceContext, } from "../adapters/evidence-context.js";
|
|
5
5
|
import { collectUnits } from "../adapters/git.js";
|
|
6
6
|
import { resolveBase } from "../adapters/git-base.js";
|
|
7
7
|
import { shareGitInventory } from "../adapters/git-inventory.js";
|
|
@@ -11,15 +11,18 @@ import { hostUsage } from "../adapters/usage.js";
|
|
|
11
11
|
import { SELECT_MIN, STATE_MAX_CHARS, WITNESS_AUTO_MIN_CELLS, } from "../constants.js";
|
|
12
12
|
import { createImportGraphBuilder, importClosureLazy, } from "../core/imports.js";
|
|
13
13
|
import { buildEnvelope } from "../core/output.js";
|
|
14
|
+
import { known as reportKnown, } from "../core/result-report.js";
|
|
14
15
|
import { buildRunnerCommands } from "../core/test-commands.js";
|
|
15
16
|
import { prepareCoverageWitnesses } from "../core/test-coverage.js";
|
|
16
17
|
import { discoverTests, isTestConfiguration } from "../core/test-discovery.js";
|
|
17
18
|
import { prepareTestEvidence, } from "../core/test-evidence.js";
|
|
18
19
|
import { prepareTestStates } from "../core/test-state.js";
|
|
19
20
|
import { buildCoverageWitnessUnits, evaluateBatchWitnessHealth, } from "../presets/witnesses.js";
|
|
20
|
-
import {
|
|
21
|
+
import { renderResultReport } from "../render.js";
|
|
21
22
|
import { SELECT_TESTS_DESCRIPTION } from "../texts/select-tests.js";
|
|
23
|
+
import { controlsFor, ReviewReport, reportMetrics } from "./review-report.js";
|
|
22
24
|
export const selectTestsParameters = Type.Object({
|
|
25
|
+
root: Type.Optional(Type.String()),
|
|
23
26
|
base: Type.Optional(Type.String({ minLength: 1 })),
|
|
24
27
|
paths: Type.Optional(Type.Array(Type.String({ minLength: 1 }))),
|
|
25
28
|
witnesses: Type.Optional(Type.Union([
|
|
@@ -70,7 +73,13 @@ export function createSelectTestsTool(dependencies) {
|
|
|
70
73
|
}),
|
|
71
74
|
async execute(_id, args, signal, _update, ctx) {
|
|
72
75
|
const client = dependencies.client;
|
|
73
|
-
const
|
|
76
|
+
const evidence = await resolveEvidenceContext(ctx.cwd, args.root, {
|
|
77
|
+
exec: execute,
|
|
78
|
+
signal,
|
|
79
|
+
origin: dependencies.evidenceOrigin,
|
|
80
|
+
});
|
|
81
|
+
const evidenceContext = evidence.context;
|
|
82
|
+
const cwd = evidence.ok ? evidence.cwd : evidenceContext.authority.path;
|
|
74
83
|
const started = performance.now();
|
|
75
84
|
const exec = shareGitInventory(execute);
|
|
76
85
|
const totals = {
|
|
@@ -80,10 +89,17 @@ export function createSelectTestsTool(dependencies) {
|
|
|
80
89
|
cacheRequests: 0,
|
|
81
90
|
usage: { inputTokens: 0, costUsd: 0 },
|
|
82
91
|
};
|
|
92
|
+
const selectionEvidence = {
|
|
93
|
+
criteria: [],
|
|
94
|
+
wideningTriggers: [],
|
|
95
|
+
inventory: [],
|
|
96
|
+
};
|
|
97
|
+
const report = new ReviewReport();
|
|
83
98
|
let costKnown = false;
|
|
99
|
+
let unknownCost = false;
|
|
84
100
|
let sent = 0;
|
|
85
101
|
let budget;
|
|
86
|
-
const finish = (input) => {
|
|
102
|
+
const finish = (input, cause) => {
|
|
87
103
|
const limitKeys = new Set();
|
|
88
104
|
const uniqueLimits = input.limitations?.filter((limit) => {
|
|
89
105
|
const key = JSON.stringify([limit.fact, limit.next]);
|
|
@@ -97,36 +113,63 @@ export function createSelectTestsTool(dependencies) {
|
|
|
97
113
|
limitations: uniqueLimits,
|
|
98
114
|
yield: {
|
|
99
115
|
...totals,
|
|
100
|
-
costUsd: costKnown ? totals.usage.costUsd : undefined,
|
|
116
|
+
costUsd: costKnown && !unknownCost ? totals.usage.costUsd : undefined,
|
|
101
117
|
elapsedMs: performance.now() - started,
|
|
102
118
|
},
|
|
103
119
|
});
|
|
120
|
+
if (input.refusal) {
|
|
121
|
+
report.refusal = true;
|
|
122
|
+
report.diagnose(cause ?? "internal_error", input.refusal);
|
|
123
|
+
}
|
|
124
|
+
if (!report.items.size && !input.refusal)
|
|
125
|
+
report.diagnose("collection_empty", "No test decision candidates in the discovered inventory; this is not proof of no affected tests.", undefined, [], false);
|
|
126
|
+
const result = report.build("jev_select_tests", evidenceContext, reportMetrics(envelope));
|
|
104
127
|
runtime.session.record(envelope);
|
|
105
128
|
runtime.guide.deliver(ctx);
|
|
106
129
|
const details = {
|
|
107
130
|
ok: true,
|
|
108
131
|
answers: {},
|
|
109
132
|
...totals,
|
|
133
|
+
evidenceContext,
|
|
134
|
+
selectionEvidence,
|
|
135
|
+
result,
|
|
110
136
|
limitations: uniqueLimits,
|
|
111
137
|
};
|
|
112
138
|
return {
|
|
113
|
-
content: [
|
|
139
|
+
content: [
|
|
140
|
+
{
|
|
141
|
+
type: "text",
|
|
142
|
+
text: renderResultReport(result, { details: envelope }),
|
|
143
|
+
},
|
|
144
|
+
],
|
|
114
145
|
details,
|
|
115
146
|
...hostUsage(host.isOmp, costKnown ? totals.usage : undefined),
|
|
116
147
|
};
|
|
117
148
|
};
|
|
149
|
+
if (!evidence.ok)
|
|
150
|
+
return finish({ refusal: evidence.error }, evidence.cause ?? "invalid_root");
|
|
151
|
+
evidenceContext.requestedBase = args.base ?? "HEAD";
|
|
152
|
+
for (const path of args.paths ?? [])
|
|
153
|
+
if (isAbsolute(path) ||
|
|
154
|
+
path.split(/[\\/]/).includes("..") ||
|
|
155
|
+
path.split(/[\\/]/).includes(".git"))
|
|
156
|
+
return finish({ refusal: `Path not permitted: ${path}` }, "forbidden_path");
|
|
118
157
|
const comparison = await resolveBase(exec, cwd, args.base, signal);
|
|
119
158
|
if (!comparison.ok)
|
|
120
|
-
return finish({ refusal: comparison.error });
|
|
159
|
+
return finish({ refusal: comparison.error }, comparison.cause ?? "invalid_base");
|
|
160
|
+
evidenceContext.resolvedBase = comparison.base;
|
|
121
161
|
const analysis = await createAnalysisContext();
|
|
122
162
|
const [inventory, diff] = await Promise.all([
|
|
123
163
|
collectTestInventory(exec, cwd, signal),
|
|
124
164
|
collectUnits(exec, { cwd: cwd, base: comparison.base, signal }, analysis.parser),
|
|
125
165
|
]);
|
|
126
166
|
if (!inventory.ok)
|
|
127
|
-
return finish({ refusal: inventory.error });
|
|
167
|
+
return finish({ refusal: inventory.error }, inventory.cause ?? "file_unavailable");
|
|
128
168
|
if (!diff.ok)
|
|
129
|
-
return finish({ refusal: diff.error });
|
|
169
|
+
return finish({ refusal: diff.error }, diff.cause ?? "git_failure");
|
|
170
|
+
if (evidenceContext.effectiveRoot)
|
|
171
|
+
evidenceContext.effectiveRoot.path = inventory.cwd;
|
|
172
|
+
selectionEvidence.inventory = [...inventory.paths];
|
|
130
173
|
const known = new Set(inventory.paths);
|
|
131
174
|
const sourceFiles = new Map();
|
|
132
175
|
const read = async (path) => {
|
|
@@ -180,6 +223,17 @@ export function createSelectTestsTool(dependencies) {
|
|
|
180
223
|
const evidenceLimits = [];
|
|
181
224
|
const outside = diff.files.some((file) => !graph.edges.has(file.path) ||
|
|
182
225
|
!/\.[cm]?[jt]sx?$|\.py$/.test(file.path));
|
|
226
|
+
selectionEvidence.wideningTriggers = diff.files
|
|
227
|
+
.filter((file) => !graph.edges.has(file.path) ||
|
|
228
|
+
!/\.[cm]?[jt]sx?$|\.py$/.test(file.path))
|
|
229
|
+
.map((file) => file.path);
|
|
230
|
+
selectionEvidence.criteria = (args.paths ?? []).map((criterion) => ({
|
|
231
|
+
criterion,
|
|
232
|
+
matches: versionedEntries
|
|
233
|
+
.filter((entry) => matchesGlob(entry.path, criterion) ||
|
|
234
|
+
entry.path.startsWith(`${criterion.replace(/\/$/, "")}/`))
|
|
235
|
+
.map((entry) => entry.path),
|
|
236
|
+
}));
|
|
183
237
|
const candidates = versionedEntries
|
|
184
238
|
.filter((entry) => !args.paths?.length ||
|
|
185
239
|
args.paths.some((path) => matchesGlob(entry.path, path) ||
|
|
@@ -197,6 +251,45 @@ export function createSelectTestsTool(dependencies) {
|
|
|
197
251
|
? { ...evidence, units: diff.units }
|
|
198
252
|
: evidence, changed, outside);
|
|
199
253
|
});
|
|
254
|
+
report.inventories.push({
|
|
255
|
+
id: "test-candidates",
|
|
256
|
+
kind: "tests",
|
|
257
|
+
rules: [
|
|
258
|
+
"Tracked supported test declarations and literal project runner configuration",
|
|
259
|
+
"Import closure selection; conservative fallback preserves unjudged decisions",
|
|
260
|
+
],
|
|
261
|
+
restrictions: args.paths ?? [],
|
|
262
|
+
discovered: reportKnown(versionedEntries.length),
|
|
263
|
+
considered: reportKnown(candidates.length),
|
|
264
|
+
scopeRestricted: Boolean(args.paths?.length),
|
|
265
|
+
criteria: selectionEvidence.criteria.map((criterion) => ({
|
|
266
|
+
criterion: criterion.criterion,
|
|
267
|
+
matches: reportKnown(criterion.matches.length),
|
|
268
|
+
outcome: criterion.matches.length ? "matched" : "no_match",
|
|
269
|
+
diagnosticIds: [],
|
|
270
|
+
})),
|
|
271
|
+
});
|
|
272
|
+
for (const candidate of candidates) {
|
|
273
|
+
const scenarios = candidate.entry.scenarios;
|
|
274
|
+
if (!scenarios.length)
|
|
275
|
+
report.expect(`test:${candidate.entry.path}:unknown`, candidate.entry.path, "scenario");
|
|
276
|
+
for (const scenario of scenarios)
|
|
277
|
+
report.expect(`test:${candidate.entry.path}:${scenario.id}`, `${candidate.entry.path}: ${scenario.name}`, "scenario", `test:${candidate.entry.path}`);
|
|
278
|
+
}
|
|
279
|
+
for (const criterion of selectionEvidence.criteria)
|
|
280
|
+
if (!criterion.matches.length) {
|
|
281
|
+
report.diagnose("criteria_no_match", `No discovered tests match criterion ${criterion.criterion}`, criterion.criterion);
|
|
282
|
+
const excluded = inventory.limits.filter((limit) => matchesGlob(limit.path, criterion.criterion) ||
|
|
283
|
+
limit.path.startsWith(`${criterion.criterion.replace(/\/$/, "")}/`));
|
|
284
|
+
if (excluded.length) {
|
|
285
|
+
report.diagnose("outside_inventory", "Requested criterion names entries excluded from the admitted inventory", criterion.criterion, [], false, excluded.map((limit) => limit.path));
|
|
286
|
+
const entry = report.inventories[0]?.criteria.find((entry) => entry.criterion === criterion.criterion);
|
|
287
|
+
if (entry)
|
|
288
|
+
entry.outcome = "outside_inventory";
|
|
289
|
+
}
|
|
290
|
+
}
|
|
291
|
+
if (selectionEvidence.wideningTriggers.length)
|
|
292
|
+
report.diagnose("conservative_widening", `Changed files outside supported dependency graph: ${selectionEvidence.wideningTriggers.join(", ")}`, undefined, [], false);
|
|
200
293
|
const selected = [];
|
|
201
294
|
const answers = [];
|
|
202
295
|
const unjudged = [];
|
|
@@ -253,9 +346,11 @@ export function createSelectTestsTool(dependencies) {
|
|
|
253
346
|
let fallback = false;
|
|
254
347
|
let skipped = 0;
|
|
255
348
|
const judge = async (state, questions, witnessIds) => {
|
|
349
|
+
state = withEvidenceContext(state, evidenceContext);
|
|
256
350
|
if (JSON.stringify(state).length > STATE_MAX_CHARS)
|
|
257
351
|
return {
|
|
258
352
|
ok: false,
|
|
353
|
+
cause: "evidence_too_large",
|
|
259
354
|
error: `required evidence exceeds STATE_MAX_CHARS=${STATE_MAX_CHARS}`,
|
|
260
355
|
};
|
|
261
356
|
if (args.max_calls !== undefined && sent >= args.max_calls) {
|
|
@@ -271,6 +366,7 @@ export function createSelectTestsTool(dependencies) {
|
|
|
271
366
|
signal,
|
|
272
367
|
witnesses: witnessIds,
|
|
273
368
|
...runtime.session.requestGate(),
|
|
369
|
+
admissionCause: () => budget?.kind === "session" ? "session_budget" : "call_budget",
|
|
274
370
|
beforeRequest: (questionCount) => {
|
|
275
371
|
if (args.max_calls !== undefined && sent >= args.max_calls) {
|
|
276
372
|
budget = {
|
|
@@ -293,6 +389,7 @@ export function createSelectTestsTool(dependencies) {
|
|
|
293
389
|
totals.questions += result.questions ?? 0;
|
|
294
390
|
totals.cacheHits += result.cacheHits ?? 0;
|
|
295
391
|
totals.cacheRequests += result.cacheRequests ?? 0;
|
|
392
|
+
unknownCost ||= result.usage === undefined;
|
|
296
393
|
if (result.usage) {
|
|
297
394
|
costKnown = true;
|
|
298
395
|
totals.usage.inputTokens += result.usage.inputTokens;
|
|
@@ -302,6 +399,19 @@ export function createSelectTestsTool(dependencies) {
|
|
|
302
399
|
};
|
|
303
400
|
const pointerResults = await Promise.all(candidates.map(async (candidate) => {
|
|
304
401
|
const { entry } = candidate;
|
|
402
|
+
const reportIds = [...report.items.values()]
|
|
403
|
+
.filter((item) => item.groupId === `test:${entry.path}` ||
|
|
404
|
+
item.id === `test:${entry.path}:unknown`)
|
|
405
|
+
.map((item) => item.id);
|
|
406
|
+
if (candidate.touched)
|
|
407
|
+
for (const id of reportIds)
|
|
408
|
+
report.static(id, "touched", true);
|
|
409
|
+
if (!candidate.units.length &&
|
|
410
|
+
!candidate.uncertain &&
|
|
411
|
+
!outside &&
|
|
412
|
+
!candidate.touched)
|
|
413
|
+
for (const id of reportIds)
|
|
414
|
+
report.static(id, "static import closure cannot reach changed units", false);
|
|
305
415
|
if (candidate.touched)
|
|
306
416
|
return {
|
|
307
417
|
candidate,
|
|
@@ -318,6 +428,7 @@ export function createSelectTestsTool(dependencies) {
|
|
|
318
428
|
}
|
|
319
429
|
const units = candidate.units;
|
|
320
430
|
if (units.some((unit) => unit.before === null && unit.after === null)) {
|
|
431
|
+
report.diagnose("binary_or_non_utf8", "Changed source unavailable", entry.path, reportIds);
|
|
321
432
|
for (const unit of units)
|
|
322
433
|
incompleteUnits.add(unit.id);
|
|
323
434
|
unjudged.push({
|
|
@@ -328,6 +439,8 @@ export function createSelectTestsTool(dependencies) {
|
|
|
328
439
|
return { candidate, selected: null, touched: false };
|
|
329
440
|
}
|
|
330
441
|
if (!entry.scenarios.length) {
|
|
442
|
+
report.totalUnknown = true;
|
|
443
|
+
report.diagnose("unsupported_syntax", "Scenario names or count unresolved", entry.path, reportIds);
|
|
331
444
|
unjudged.push({
|
|
332
445
|
label: entry.path,
|
|
333
446
|
reason: "scenario names or count unresolved",
|
|
@@ -338,6 +451,9 @@ export function createSelectTestsTool(dependencies) {
|
|
|
338
451
|
const prepared = prepareTestStates(candidate.state, entry.scenarios);
|
|
339
452
|
const ids = prepared.unjudged.map((item) => item.scenario.id);
|
|
340
453
|
for (const item of prepared.unjudged) {
|
|
454
|
+
report.diagnose("evidence_too_large", item.reason, entry.path, [
|
|
455
|
+
`test:${entry.path}:${item.scenario.id}`,
|
|
456
|
+
]);
|
|
341
457
|
for (const unit of units)
|
|
342
458
|
incompleteUnits.add(unit.id);
|
|
343
459
|
unjudged.push({
|
|
@@ -369,6 +485,33 @@ export function createSelectTestsTool(dependencies) {
|
|
|
369
485
|
},
|
|
370
486
|
]));
|
|
371
487
|
const result = await judge(batch.state, questions);
|
|
488
|
+
const batchIds = batch.scenarios.map((scenario) => `test:${entry.path}:${scenario.id}`);
|
|
489
|
+
if (!result.ok)
|
|
490
|
+
report.failure(result, batchIds, budget
|
|
491
|
+
? budget.kind === "session"
|
|
492
|
+
? "session_budget"
|
|
493
|
+
: "call_budget"
|
|
494
|
+
: !client
|
|
495
|
+
? "not_configured"
|
|
496
|
+
: undefined);
|
|
497
|
+
else
|
|
498
|
+
for (const scenario of batch.scenarios) {
|
|
499
|
+
const answer = result.answers[scenario.id];
|
|
500
|
+
const id = `test:${entry.path}:${scenario.id}`;
|
|
501
|
+
report.answer(id, answer, {
|
|
502
|
+
band: prepared.unjudged.length ? "unsure" : "verdict",
|
|
503
|
+
});
|
|
504
|
+
const item = report.items.get(id);
|
|
505
|
+
if (!item)
|
|
506
|
+
throw new Error(`Unregistered selection result ${id}`);
|
|
507
|
+
item.selection = {
|
|
508
|
+
selected: answer?.type !== "choice" ||
|
|
509
|
+
1 - (answer.probabilities.none ?? 0) >= SELECT_MIN,
|
|
510
|
+
reason: answer?.type === "choice"
|
|
511
|
+
? "changed-unit pointer selection threshold"
|
|
512
|
+
: "conservative_fallback",
|
|
513
|
+
};
|
|
514
|
+
}
|
|
372
515
|
if (!result.ok) {
|
|
373
516
|
fallback ||= !budget;
|
|
374
517
|
for (const unit of units)
|
|
@@ -443,6 +586,14 @@ export function createSelectTestsTool(dependencies) {
|
|
|
443
586
|
: buildCoverageWitnessUnits(units);
|
|
444
587
|
const preliminary = prepareCoverageWitnesses(candidate.state, {}, witnessUnits);
|
|
445
588
|
const prepared = prepareTestStates(preliminary.state, candidate.entry.scenarios);
|
|
589
|
+
for (const unit of units)
|
|
590
|
+
for (const scenario of candidate.entry.scenarios)
|
|
591
|
+
report.expect(`coverage:${candidate.entry.path}:${scenario.id}:${unit.id}`, `${candidate.entry.path}: ${scenario.name} executes ${unit.name}`, "scenario", `coverage:${candidate.entry.path}:${scenario.id}`);
|
|
592
|
+
for (const omitted of prepared.unjudged)
|
|
593
|
+
for (const unit of units)
|
|
594
|
+
report.diagnose("evidence_too_large", omitted.reason, candidate.entry.path, [
|
|
595
|
+
`coverage:${candidate.entry.path}:${omitted.scenario.id}:${unit.id}`,
|
|
596
|
+
]);
|
|
446
597
|
if (prepared.unjudged.length || !candidate.entry.scenarios.length)
|
|
447
598
|
for (const unit of units)
|
|
448
599
|
incompleteUnits.add(unit.id);
|
|
@@ -461,6 +612,31 @@ export function createSelectTestsTool(dependencies) {
|
|
|
461
612
|
const result = await judge(batch.state, { ...questions, ...witnessQuestions.questions }, witnessQuestions.witnesses.map((witness) => witness.id));
|
|
462
613
|
const health = evaluateBatchWitnessHealth(realIds, witnessQuestions.witnesses, result);
|
|
463
614
|
limitations.push(...health.failures);
|
|
615
|
+
for (const failure of health.failures)
|
|
616
|
+
report.diagnose("control_failure", failure.fact, candidate.entry.path, units.flatMap((unit) => batch.scenarios
|
|
617
|
+
.filter((scenario) => health.unhealthyQuestionIds.has(`${scenario.id}_${unit.id}`))
|
|
618
|
+
.map((scenario) => `coverage:${candidate.entry.path}:${scenario.id}:${unit.id}`)), false);
|
|
619
|
+
report.countControls(result, witnessQuestions.witnesses.map((witness) => witness.id));
|
|
620
|
+
for (const unit of units)
|
|
621
|
+
for (const scenario of batch.scenarios) {
|
|
622
|
+
const id = `coverage:${candidate.entry.path}:${scenario.id}:${unit.id}`;
|
|
623
|
+
report.expect(id, `${candidate.entry.path}: ${scenario.name} executes ${unit.name}`, "scenario", `coverage:${candidate.entry.path}:${scenario.id}`);
|
|
624
|
+
if (!result.ok)
|
|
625
|
+
report.failure(result, [id], budget
|
|
626
|
+
? budget.kind === "session"
|
|
627
|
+
? "session_budget"
|
|
628
|
+
: "call_budget"
|
|
629
|
+
: !client
|
|
630
|
+
? "not_configured"
|
|
631
|
+
: undefined);
|
|
632
|
+
else
|
|
633
|
+
report.answer(id, result.answers[`${scenario.id}_${unit.id}`], {
|
|
634
|
+
band: health.unhealthyQuestionIds.has(`${scenario.id}_${unit.id}`)
|
|
635
|
+
? "unsure"
|
|
636
|
+
: "verdict",
|
|
637
|
+
reason: health.unhealthyQuestionIds.get(`${scenario.id}_${unit.id}`),
|
|
638
|
+
}, controlsFor(`${scenario.id}_${unit.id}`, result, witnessQuestions.witnesses.map((witness) => witness.id)), false);
|
|
639
|
+
}
|
|
464
640
|
for (const unit of units) {
|
|
465
641
|
if (!result.ok) {
|
|
466
642
|
fallback ||= !budget;
|
|
@@ -513,21 +689,33 @@ export function createSelectTestsTool(dependencies) {
|
|
|
513
689
|
evidenceLimits.length > 0 ||
|
|
514
690
|
Boolean(args.paths?.length);
|
|
515
691
|
for (const unit of residual)
|
|
516
|
-
if ((coverage.get(unit.id) ?? 0) < SELECT_MIN)
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
692
|
+
if ((coverage.get(unit.id) ?? 0) < SELECT_MIN) {
|
|
693
|
+
const supportingIds = [...report.items.keys()].filter((id) => id.startsWith("coverage:") && id.endsWith(`:${unit.id}`));
|
|
694
|
+
report.diagnose(incomplete || incompleteUnits.has(unit.id)
|
|
695
|
+
? "collection_omitted"
|
|
696
|
+
: "conservative_widening", `Derived from retained coverage decisions: no discovered test established execution of ${unit.name} (${unit.file}) within the considered inventory only.${incomplete || incompleteUnits.has(unit.id) ? " Absence remains unsure because evidence, scope, or controls are incomplete." : " This is not a global coverage claim."}`, unit.file, supportingIds, false);
|
|
697
|
+
if (!supportingIds.length) {
|
|
698
|
+
const diagnostic = report.diagnostics.at(-1);
|
|
699
|
+
if (diagnostic)
|
|
700
|
+
diagnostic.scope = {
|
|
701
|
+
kind: "inventory",
|
|
702
|
+
inventoryIds: ["test-candidates"],
|
|
703
|
+
};
|
|
704
|
+
const action = report.actions.at(-1);
|
|
705
|
+
if (action)
|
|
706
|
+
action.scope = {
|
|
707
|
+
kind: "inventory",
|
|
708
|
+
inventoryIds: ["test-candidates"],
|
|
709
|
+
};
|
|
710
|
+
}
|
|
711
|
+
}
|
|
525
712
|
if (fallback) {
|
|
526
713
|
selected.length = 0;
|
|
527
714
|
selected.push(...candidates.map((candidate) => ({
|
|
528
715
|
entry: candidate.entry,
|
|
529
716
|
scenarioIds: null,
|
|
530
717
|
})));
|
|
718
|
+
report.diagnose("conservative_widening", "Unavailable judgment conservatively selects all considered test candidates; static reachability exclusions do not narrow this fallback.", undefined, [], false);
|
|
531
719
|
limitations.push({
|
|
532
720
|
fact: "fallback: all",
|
|
533
721
|
next: "run all discovered tests with their project runner",
|
|
@@ -540,11 +728,77 @@ export function createSelectTestsTool(dependencies) {
|
|
|
540
728
|
fact: `${limit.reason}: ${path}`,
|
|
541
729
|
next: limit.action,
|
|
542
730
|
}))));
|
|
731
|
+
for (const limit of commands.limits)
|
|
732
|
+
for (const path of limit.files)
|
|
733
|
+
report.diagnose("unresolved_runner", limit.reason, path, [], false);
|
|
734
|
+
for (const limit of discovery.limits)
|
|
735
|
+
report.diagnose("unresolved_runner", limit.kind, limit.path, [], limit.kind !== "local_runner_unproven" &&
|
|
736
|
+
limit.kind !== "interactive_script_skipped");
|
|
737
|
+
for (const limit of inventory.limits)
|
|
738
|
+
report.diagnose(limit.kind === "secret_pattern"
|
|
739
|
+
? "secret_pattern"
|
|
740
|
+
: "collection_omitted", limit.kind, limit.path);
|
|
741
|
+
for (const limit of diff.limits)
|
|
742
|
+
report.diagnose(limit.kind === "secret_pattern"
|
|
743
|
+
? "secret_pattern"
|
|
744
|
+
: "collection_omitted", limit.kind, limit.file);
|
|
745
|
+
for (const limit of graph.limits)
|
|
746
|
+
report.diagnose("dynamic_dependency", `${limit.kind}${limit.specifier ? ` (${limit.specifier})` : ""}`, limit.path, [], false);
|
|
747
|
+
for (const item of report.items.values()) {
|
|
748
|
+
const candidate = candidates.find((candidate) => item.id.startsWith(`test:${candidate.entry.path}:`) ||
|
|
749
|
+
item.id.startsWith(`coverage:${candidate.entry.path}:`));
|
|
750
|
+
if (!candidate)
|
|
751
|
+
continue;
|
|
752
|
+
const plan = selected.find((plan) => plan.entry === candidate.entry);
|
|
753
|
+
const scenario = candidate.entry.scenarios.find((scenario) => item.id === `test:${candidate.entry.path}:${scenario.id}` ||
|
|
754
|
+
item.id.startsWith(`coverage:${candidate.entry.path}:${scenario.id}:`));
|
|
755
|
+
const isSelected = Boolean(plan &&
|
|
756
|
+
(plan.scenarioIds === null ||
|
|
757
|
+
(scenario && plan.scenarioIds.includes(scenario.id))));
|
|
758
|
+
item.selection = {
|
|
759
|
+
selected: isSelected,
|
|
760
|
+
reason: fallback
|
|
761
|
+
? "conservative_fallback"
|
|
762
|
+
: item.treatment === "static"
|
|
763
|
+
? item.staticReason
|
|
764
|
+
: item.treatment === "not_judged"
|
|
765
|
+
? "conservative_fallback"
|
|
766
|
+
: "unchanged selection policy and whole-file widening",
|
|
767
|
+
};
|
|
768
|
+
}
|
|
769
|
+
for (const criterion of report.inventories[0]?.criteria ?? [])
|
|
770
|
+
criterion.diagnosticIds = report.diagnostics
|
|
771
|
+
.filter((diagnostic) => diagnostic.cause === "criteria_no_match" &&
|
|
772
|
+
diagnostic.target.status === "known" &&
|
|
773
|
+
diagnostic.target.value === criterion.criterion)
|
|
774
|
+
.map((diagnostic) => diagnostic.id);
|
|
775
|
+
report.actions.push({
|
|
776
|
+
id: "execute-selection-plan",
|
|
777
|
+
code: "execute_plan",
|
|
778
|
+
target: reportKnown(inventory.cwd),
|
|
779
|
+
scope: { kind: "call" },
|
|
780
|
+
condition: "When verification is authorized and the listed runner/configuration is available.",
|
|
781
|
+
instruction: "Execute the retained runner commands separately; this tool has not executed any test.",
|
|
782
|
+
repeatUnchanged: false,
|
|
783
|
+
});
|
|
543
784
|
if (skipped)
|
|
544
785
|
limitations.push({
|
|
545
786
|
fact: `${skipped} tests skipped because their imports cannot reach the diff`,
|
|
546
787
|
next: "an alias or dynamic import would have kept a test in",
|
|
547
788
|
});
|
|
789
|
+
for (const criterion of selectionEvidence.criteria)
|
|
790
|
+
if (!criterion.matches.length)
|
|
791
|
+
limitations.push({
|
|
792
|
+
path: criterion.criterion,
|
|
793
|
+
cause: "selection criterion has no discovered test match",
|
|
794
|
+
fact: `selection criterion without match: ${criterion.criterion}`,
|
|
795
|
+
next: "Change the criterion or supply the missing test/configuration evidence; an empty scope does not prove absence of impact.",
|
|
796
|
+
});
|
|
797
|
+
if (selectionEvidence.wideningTriggers.length)
|
|
798
|
+
limitations.push({
|
|
799
|
+
fact: `selection widened conservatively: ${selectionEvidence.wideningTriggers.join(", ")}`,
|
|
800
|
+
next: "Run the widened static plans; dependency reachability is not established for these changed files.",
|
|
801
|
+
});
|
|
548
802
|
return finish({
|
|
549
803
|
answers,
|
|
550
804
|
unjudged,
|