jev-agent-tools 0.1.4 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +106 -1
- package/CONTRIBUTING.md +43 -0
- package/README.md +58 -17
- package/SECURITY.md +43 -0
- package/dist/adapters/analysis-context.js +75 -0
- package/dist/adapters/ask-files.js +198 -0
- package/dist/adapters/ask-proof.js +200 -0
- package/dist/adapters/ask-syntax.js +385 -0
- package/dist/adapters/canonical-path.js +17 -0
- package/dist/adapters/command.js +234 -0
- package/dist/adapters/docs.js +192 -0
- package/dist/adapters/evidence-context.js +119 -0
- package/dist/adapters/exec.js +207 -0
- package/dist/adapters/files.js +418 -0
- package/dist/adapters/find.js +150 -0
- package/dist/adapters/git-base.js +32 -0
- package/dist/adapters/git-inventory.js +71 -0
- package/dist/adapters/git.js +483 -0
- package/dist/adapters/locate-file.js +197 -0
- package/dist/adapters/output-lines.js +46 -0
- package/dist/adapters/private-storage.js +106 -0
- package/dist/adapters/risk-callers.js +429 -0
- package/dist/adapters/runner-version.js +78 -0
- package/dist/adapters/shell.js +92 -0
- package/dist/adapters/syntax.js +187 -0
- package/dist/adapters/test-inventory.js +139 -0
- package/dist/adapters/usage.js +20 -0
- package/dist/adapters/utf8.js +47 -0
- package/dist/configuration.js +267 -0
- package/dist/constants.js +140 -0
- package/dist/core/ask-closure.js +282 -0
- package/dist/core/ask-proof.js +1 -0
- package/dist/core/ask-references.js +278 -0
- package/dist/core/asks.js +507 -0
- package/dist/core/batches.js +65 -0
- package/dist/core/command-output.js +224 -0
- package/dist/core/diff.js +178 -0
- package/dist/core/docs.js +302 -0
- package/dist/core/find.js +108 -0
- package/dist/core/git.js +1 -0
- package/dist/core/imports.js +550 -0
- package/dist/core/integrity.js +45 -0
- package/dist/core/lexical.js +132 -0
- package/dist/core/locate.js +169 -0
- package/dist/core/output.js +137 -0
- package/dist/core/pointer.js +29 -0
- package/dist/core/result-report.js +302 -0
- package/dist/core/risk-callers.js +851 -0
- package/dist/core/runner-version.js +45 -0
- package/dist/core/secret-path.js +34 -0
- package/dist/core/sections.js +230 -0
- package/dist/core/state.js +51 -0
- package/dist/core/syntax.js +1 -0
- package/dist/core/test-commands.js +334 -0
- package/dist/core/test-coverage.js +74 -0
- package/dist/core/test-discovery.js +1382 -0
- package/dist/core/test-evidence.js +527 -0
- package/dist/core/test-state.js +81 -0
- package/dist/core/truncate.js +12 -0
- package/dist/core/units.js +349 -0
- package/dist/describe.js +23 -0
- package/dist/guide.js +33 -0
- package/dist/host.js +24 -0
- package/dist/jev/client.js +456 -0
- package/dist/jev/pool.js +54 -0
- package/dist/jev/types.js +1 -0
- package/dist/mcp/main.js +124 -0
- package/dist/mcp/protocol.js +210 -0
- package/dist/mcp/tools.js +129 -0
- package/dist/presets/docs.js +62 -0
- package/dist/presets/risk.js +179 -0
- package/dist/presets/spec.js +81 -0
- package/dist/presets/witnesses.js +249 -0
- package/dist/render.js +114 -0
- package/dist/report-schema.js +1356 -0
- package/dist/result-types.js +1 -0
- package/dist/result.js +3 -0
- package/dist/runtime.js +1 -0
- package/dist/session.js +147 -0
- package/dist/texts/ask-files.js +3 -0
- package/dist/texts/ask.js +4 -0
- package/dist/texts/check-diff.js +20 -0
- package/dist/texts/configuration.js +1 -0
- package/dist/texts/find.js +19 -0
- package/dist/texts/guide.js +3 -0
- package/dist/texts/instructions.js +72 -0
- package/dist/texts/locate.js +15 -0
- package/dist/texts/select-tests.js +4 -0
- package/dist/tools/ask-files.js +450 -0
- package/dist/tools/ask-schema.js +70 -0
- package/dist/tools/ask.js +1147 -0
- package/dist/tools/check-diff.js +594 -0
- package/dist/tools/docs-check.js +408 -0
- package/dist/tools/find.js +682 -0
- package/dist/tools/locate.js +602 -0
- package/dist/tools/review-report.js +230 -0
- package/dist/tools/select-tests.js +821 -0
- package/dist/tools/spec-check.js +263 -0
- package/docs/adr/0001-strict-typescript-pure-core-offline-tests.md +31 -0
- package/docs/adr/0002-one-http-protocol-across-hosts.md +17 -0
- package/docs/adr/0003-explicit-scope-conservative-automation.md +19 -0
- package/docs/adr/0004-compiled-typed-intents.md +19 -0
- package/docs/adr/0005-evidence-construction-before-judgment.md +19 -0
- package/docs/adr/0006-visible-uncertainty-constrained-controls.md +21 -0
- package/docs/adr/0007-bounded-evidence-visible-limits.md +21 -0
- package/docs/adr/0008-static-test-discovery-conservative-plans.md +19 -0
- package/docs/adr/0009-session-cache-requested-model-identity.md +17 -0
- package/docs/adr/0010-mcp-server-thin-host.md +23 -0
- package/docs/agent-instructions.md +120 -0
- package/docs/design.md +16 -4
- package/docs/mcp.md +233 -0
- package/docs/tools/jev_ask.md +8 -5
- package/docs/tools/jev_ask_files.md +2 -1
- package/docs/tools/jev_check_diff.md +4 -1
- package/docs/tools/jev_find_files.md +2 -1
- package/docs/tools/jev_locate_in_file.md +5 -0
- package/docs/tools/jev_select_tests.md +4 -1
- package/package.json +19 -4
- package/rules/jev-ask.md +22 -1
- package/server.json +57 -0
- package/src/adapters/ask-files.ts +11 -3
- package/src/adapters/ask-proof.ts +69 -11
- package/src/adapters/canonical-path.ts +18 -0
- package/src/adapters/command.ts +102 -36
- package/src/adapters/docs.ts +33 -14
- package/src/adapters/evidence-context.ts +169 -0
- package/src/adapters/exec.ts +226 -0
- package/src/adapters/files.ts +146 -16
- package/src/adapters/find.ts +37 -7
- package/src/adapters/git-base.ts +7 -1
- package/src/adapters/git.ts +61 -8
- package/src/adapters/locate-file.ts +51 -9
- package/src/adapters/private-storage.ts +155 -0
- package/src/adapters/risk-callers.ts +7 -2
- package/src/adapters/shell.ts +113 -0
- package/src/adapters/test-inventory.ts +12 -4
- package/src/configuration.ts +55 -14
- package/src/constants.ts +37 -5
- package/src/core/ask-references.ts +262 -146
- package/src/core/asks.ts +79 -7
- package/src/core/command-output.ts +17 -1
- package/src/core/import-boundaries.ts +8 -3
- package/src/core/locate.ts +8 -5
- package/src/core/output.ts +34 -0
- package/src/core/result-report.ts +410 -0
- package/src/core/secret-path.ts +37 -0
- package/src/core/state.ts +8 -1
- package/src/core/units.ts +3 -2
- package/src/host.ts +11 -0
- package/src/index.ts +3 -0
- package/src/jev/client.ts +66 -16
- package/src/jev/types.ts +24 -3
- package/src/mcp/main.ts +135 -0
- package/src/mcp/protocol.ts +332 -0
- package/src/mcp/tools.ts +179 -0
- package/src/render.ts +109 -0
- package/src/report-schema.ts +1380 -0
- package/src/result-types.ts +234 -0
- package/src/result.ts +4 -1
- package/src/runtime.ts +6 -0
- package/src/session.ts +59 -0
- package/src/setup.ts +13 -5
- package/src/texts/ask-files.ts +4 -1
- package/src/texts/ask.ts +8 -1
- package/src/texts/check-diff.ts +7 -4
- package/src/texts/find.ts +8 -2
- package/src/texts/guide.ts +8 -16
- package/src/texts/instructions.ts +98 -0
- package/src/texts/locate.ts +8 -2
- package/src/texts/run-end.ts +2 -2
- package/src/texts/select-tests.ts +4 -1
- package/src/tools/ask-files.ts +311 -18
- package/src/tools/ask.ts +722 -95
- package/src/tools/check-diff.ts +337 -31
- package/src/tools/docs-check.ts +241 -38
- package/src/tools/find.ts +389 -29
- package/src/tools/locate.ts +387 -25
- package/src/tools/review-report.ts +308 -0
- package/src/tools/select-tests.ts +484 -23
- package/src/tools/spec-check.ts +194 -19
|
@@ -1,7 +1,12 @@
|
|
|
1
|
-
import { matchesGlob, relative, resolve } from "node:path";
|
|
1
|
+
import { isAbsolute, matchesGlob, relative, resolve } from "node:path";
|
|
2
2
|
import type { ToolDefinition } from "@earendil-works/pi-coding-agent";
|
|
3
3
|
import { type Static, Type } from "@sinclair/typebox";
|
|
4
4
|
import { createAnalysisContext } from "../adapters/analysis-context.ts";
|
|
5
|
+
import {
|
|
6
|
+
type EvidenceContext,
|
|
7
|
+
resolveEvidenceContext,
|
|
8
|
+
withEvidenceContext,
|
|
9
|
+
} from "../adapters/evidence-context.ts";
|
|
5
10
|
import { collectUnits } from "../adapters/git.ts";
|
|
6
11
|
import { resolveBase } from "../adapters/git-base.ts";
|
|
7
12
|
import { shareGitInventory } from "../adapters/git-inventory.ts";
|
|
@@ -25,6 +30,11 @@ import type {
|
|
|
25
30
|
Limitation,
|
|
26
31
|
} from "../core/output.ts";
|
|
27
32
|
import { buildEnvelope } from "../core/output.ts";
|
|
33
|
+
import {
|
|
34
|
+
type Cause,
|
|
35
|
+
type ResultReportV1,
|
|
36
|
+
known as reportKnown,
|
|
37
|
+
} from "../core/result-report.ts";
|
|
28
38
|
import { buildRunnerCommands } from "../core/test-commands.ts";
|
|
29
39
|
import { prepareCoverageWitnesses } from "../core/test-coverage.ts";
|
|
30
40
|
import type { TestEntry } from "../core/test-discovery.ts";
|
|
@@ -41,12 +51,14 @@ import {
|
|
|
41
51
|
buildCoverageWitnessUnits,
|
|
42
52
|
evaluateBatchWitnessHealth,
|
|
43
53
|
} from "../presets/witnesses.ts";
|
|
44
|
-
import {
|
|
54
|
+
import { renderResultReport } from "../render.ts";
|
|
45
55
|
import type { ToolDependencies } from "../runtime.ts";
|
|
46
56
|
import { SELECT_TESTS_DESCRIPTION } from "../texts/select-tests.ts";
|
|
57
|
+
import { controlsFor, ReviewReport, reportMetrics } from "./review-report.ts";
|
|
47
58
|
|
|
48
59
|
export const selectTestsParameters = Type.Object(
|
|
49
60
|
{
|
|
61
|
+
root: Type.Optional(Type.String()),
|
|
50
62
|
base: Type.Optional(Type.String({ minLength: 1 })),
|
|
51
63
|
paths: Type.Optional(Type.Array(Type.String({ minLength: 1 }))),
|
|
52
64
|
witnesses: Type.Optional(
|
|
@@ -125,6 +137,13 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
125
137
|
ctx: { cwd: string } & GuideContext,
|
|
126
138
|
) {
|
|
127
139
|
const client = dependencies.client;
|
|
140
|
+
const evidence = await resolveEvidenceContext(ctx.cwd, args.root, {
|
|
141
|
+
exec: execute,
|
|
142
|
+
signal,
|
|
143
|
+
origin: dependencies.evidenceOrigin,
|
|
144
|
+
});
|
|
145
|
+
const evidenceContext = evidence.context;
|
|
146
|
+
const cwd = evidence.ok ? evidence.cwd : evidenceContext.authority.path;
|
|
128
147
|
const started = performance.now();
|
|
129
148
|
const exec = shareGitInventory(execute);
|
|
130
149
|
const totals = {
|
|
@@ -134,10 +153,17 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
134
153
|
cacheRequests: 0,
|
|
135
154
|
usage: { inputTokens: 0, costUsd: 0 },
|
|
136
155
|
};
|
|
156
|
+
const selectionEvidence = {
|
|
157
|
+
criteria: [] as { criterion: string; matches: string[] }[],
|
|
158
|
+
wideningTriggers: [] as string[],
|
|
159
|
+
inventory: [] as string[],
|
|
160
|
+
};
|
|
161
|
+
const report = new ReviewReport();
|
|
137
162
|
let costKnown = false;
|
|
163
|
+
let unknownCost = false;
|
|
138
164
|
let sent = 0;
|
|
139
165
|
let budget: BudgetRefusal | undefined;
|
|
140
|
-
const finish = (input: Omit<EnvelopeInput, "yield"
|
|
166
|
+
const finish = (input: Omit<EnvelopeInput, "yield">, cause?: Cause) => {
|
|
141
167
|
const limitKeys = new Set<string>();
|
|
142
168
|
const uniqueLimits = input.limitations?.filter((limit) => {
|
|
143
169
|
const key = JSON.stringify([limit.fact, limit.next]);
|
|
@@ -150,37 +176,97 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
150
176
|
limitations: uniqueLimits,
|
|
151
177
|
yield: {
|
|
152
178
|
...totals,
|
|
153
|
-
costUsd:
|
|
179
|
+
costUsd:
|
|
180
|
+
costKnown && !unknownCost ? totals.usage.costUsd : undefined,
|
|
154
181
|
elapsedMs: performance.now() - started,
|
|
155
182
|
},
|
|
156
183
|
});
|
|
184
|
+
if (input.refusal) {
|
|
185
|
+
report.refusal = true;
|
|
186
|
+
report.diagnose(cause ?? "internal_error", input.refusal);
|
|
187
|
+
}
|
|
188
|
+
if (!report.items.size && !input.refusal)
|
|
189
|
+
report.diagnose(
|
|
190
|
+
"collection_empty",
|
|
191
|
+
"No test decision candidates in the discovered inventory; this is not proof of no affected tests.",
|
|
192
|
+
undefined,
|
|
193
|
+
[],
|
|
194
|
+
false,
|
|
195
|
+
);
|
|
196
|
+
const result = report.build(
|
|
197
|
+
"jev_select_tests",
|
|
198
|
+
evidenceContext,
|
|
199
|
+
reportMetrics(envelope),
|
|
200
|
+
);
|
|
157
201
|
runtime.session.record(envelope);
|
|
158
202
|
runtime.guide.deliver(ctx);
|
|
159
|
-
const details: Judgment & {
|
|
203
|
+
const details: Judgment & {
|
|
204
|
+
result: ResultReportV1;
|
|
205
|
+
limitations?: readonly Limitation[];
|
|
206
|
+
evidenceContext: EvidenceContext;
|
|
207
|
+
selectionEvidence: typeof selectionEvidence;
|
|
208
|
+
} = {
|
|
160
209
|
ok: true,
|
|
161
210
|
answers: {},
|
|
162
211
|
...totals,
|
|
212
|
+
evidenceContext,
|
|
213
|
+
selectionEvidence,
|
|
214
|
+
result,
|
|
163
215
|
limitations: uniqueLimits,
|
|
164
216
|
};
|
|
165
217
|
return {
|
|
166
|
-
content: [
|
|
218
|
+
content: [
|
|
219
|
+
{
|
|
220
|
+
type: "text" as const,
|
|
221
|
+
text: renderResultReport(result, { details: envelope }),
|
|
222
|
+
},
|
|
223
|
+
],
|
|
167
224
|
details,
|
|
168
225
|
...hostUsage(host.isOmp, costKnown ? totals.usage : undefined),
|
|
169
226
|
};
|
|
170
227
|
};
|
|
171
|
-
|
|
172
|
-
|
|
228
|
+
if (!evidence.ok)
|
|
229
|
+
return finish(
|
|
230
|
+
{ refusal: evidence.error },
|
|
231
|
+
evidence.cause ?? "invalid_root",
|
|
232
|
+
);
|
|
233
|
+
evidenceContext.requestedBase = args.base ?? "HEAD";
|
|
234
|
+
for (const path of args.paths ?? [])
|
|
235
|
+
if (
|
|
236
|
+
isAbsolute(path) ||
|
|
237
|
+
path.split(/[\\/]/).includes("..") ||
|
|
238
|
+
path.split(/[\\/]/).includes(".git")
|
|
239
|
+
)
|
|
240
|
+
return finish(
|
|
241
|
+
{ refusal: `Path not permitted: ${path}` },
|
|
242
|
+
"forbidden_path",
|
|
243
|
+
);
|
|
244
|
+
const comparison = await resolveBase(exec, cwd, args.base, signal);
|
|
245
|
+
if (!comparison.ok)
|
|
246
|
+
return finish(
|
|
247
|
+
{ refusal: comparison.error },
|
|
248
|
+
comparison.cause ?? "invalid_base",
|
|
249
|
+
);
|
|
250
|
+
evidenceContext.resolvedBase = comparison.base;
|
|
173
251
|
const analysis = await createAnalysisContext();
|
|
174
252
|
const [inventory, diff] = await Promise.all([
|
|
175
|
-
collectTestInventory(exec,
|
|
253
|
+
collectTestInventory(exec, cwd, signal),
|
|
176
254
|
collectUnits(
|
|
177
255
|
exec,
|
|
178
|
-
{ cwd:
|
|
256
|
+
{ cwd: cwd, base: comparison.base, signal },
|
|
179
257
|
analysis.parser,
|
|
180
258
|
),
|
|
181
259
|
]);
|
|
182
|
-
if (!inventory.ok)
|
|
183
|
-
|
|
260
|
+
if (!inventory.ok)
|
|
261
|
+
return finish(
|
|
262
|
+
{ refusal: inventory.error },
|
|
263
|
+
inventory.cause ?? "file_unavailable",
|
|
264
|
+
);
|
|
265
|
+
if (!diff.ok)
|
|
266
|
+
return finish({ refusal: diff.error }, diff.cause ?? "git_failure");
|
|
267
|
+
if (evidenceContext.effectiveRoot)
|
|
268
|
+
evidenceContext.effectiveRoot.path = inventory.cwd;
|
|
269
|
+
selectionEvidence.inventory = [...inventory.paths];
|
|
184
270
|
const known = new Set(inventory.paths);
|
|
185
271
|
const sourceFiles = new Map<string, ImportSource>();
|
|
186
272
|
const read = async (path: string): Promise<ImportSource | undefined> => {
|
|
@@ -275,6 +361,23 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
275
361
|
!graph.edges.has(file.path) ||
|
|
276
362
|
!/\.[cm]?[jt]sx?$|\.py$/.test(file.path),
|
|
277
363
|
);
|
|
364
|
+
selectionEvidence.wideningTriggers = diff.files
|
|
365
|
+
.filter(
|
|
366
|
+
(file) =>
|
|
367
|
+
!graph.edges.has(file.path) ||
|
|
368
|
+
!/\.[cm]?[jt]sx?$|\.py$/.test(file.path),
|
|
369
|
+
)
|
|
370
|
+
.map((file) => file.path);
|
|
371
|
+
selectionEvidence.criteria = (args.paths ?? []).map((criterion) => ({
|
|
372
|
+
criterion,
|
|
373
|
+
matches: versionedEntries
|
|
374
|
+
.filter(
|
|
375
|
+
(entry) =>
|
|
376
|
+
matchesGlob(entry.path, criterion) ||
|
|
377
|
+
entry.path.startsWith(`${criterion.replace(/\/$/, "")}/`),
|
|
378
|
+
)
|
|
379
|
+
.map((entry) => entry.path),
|
|
380
|
+
}));
|
|
278
381
|
const candidates = versionedEntries
|
|
279
382
|
.filter(
|
|
280
383
|
(entry) =>
|
|
@@ -306,6 +409,77 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
306
409
|
outside,
|
|
307
410
|
);
|
|
308
411
|
});
|
|
412
|
+
report.inventories.push({
|
|
413
|
+
id: "test-candidates",
|
|
414
|
+
kind: "tests",
|
|
415
|
+
rules: [
|
|
416
|
+
"Tracked supported test declarations and literal project runner configuration",
|
|
417
|
+
"Import closure selection; conservative fallback preserves unjudged decisions",
|
|
418
|
+
],
|
|
419
|
+
restrictions: args.paths ?? [],
|
|
420
|
+
discovered: reportKnown(versionedEntries.length),
|
|
421
|
+
considered: reportKnown(candidates.length),
|
|
422
|
+
scopeRestricted: Boolean(args.paths?.length),
|
|
423
|
+
criteria: selectionEvidence.criteria.map((criterion) => ({
|
|
424
|
+
criterion: criterion.criterion,
|
|
425
|
+
matches: reportKnown(criterion.matches.length),
|
|
426
|
+
outcome: criterion.matches.length ? "matched" : "no_match",
|
|
427
|
+
diagnosticIds: [],
|
|
428
|
+
})),
|
|
429
|
+
});
|
|
430
|
+
for (const candidate of candidates) {
|
|
431
|
+
const scenarios = candidate.entry.scenarios;
|
|
432
|
+
if (!scenarios.length)
|
|
433
|
+
report.expect(
|
|
434
|
+
`test:${candidate.entry.path}:unknown`,
|
|
435
|
+
candidate.entry.path,
|
|
436
|
+
"scenario",
|
|
437
|
+
);
|
|
438
|
+
for (const scenario of scenarios)
|
|
439
|
+
report.expect(
|
|
440
|
+
`test:${candidate.entry.path}:${scenario.id}`,
|
|
441
|
+
`${candidate.entry.path}: ${scenario.name}`,
|
|
442
|
+
"scenario",
|
|
443
|
+
`test:${candidate.entry.path}`,
|
|
444
|
+
);
|
|
445
|
+
}
|
|
446
|
+
for (const criterion of selectionEvidence.criteria)
|
|
447
|
+
if (!criterion.matches.length) {
|
|
448
|
+
report.diagnose(
|
|
449
|
+
"criteria_no_match",
|
|
450
|
+
`No discovered tests match criterion ${criterion.criterion}`,
|
|
451
|
+
criterion.criterion,
|
|
452
|
+
);
|
|
453
|
+
const excluded = inventory.limits.filter(
|
|
454
|
+
(limit) =>
|
|
455
|
+
matchesGlob(limit.path, criterion.criterion) ||
|
|
456
|
+
limit.path.startsWith(
|
|
457
|
+
`${criterion.criterion.replace(/\/$/, "")}/`,
|
|
458
|
+
),
|
|
459
|
+
);
|
|
460
|
+
if (excluded.length) {
|
|
461
|
+
report.diagnose(
|
|
462
|
+
"outside_inventory",
|
|
463
|
+
"Requested criterion names entries excluded from the admitted inventory",
|
|
464
|
+
criterion.criterion,
|
|
465
|
+
[],
|
|
466
|
+
false,
|
|
467
|
+
excluded.map((limit) => limit.path),
|
|
468
|
+
);
|
|
469
|
+
const entry = report.inventories[0]?.criteria.find(
|
|
470
|
+
(entry) => entry.criterion === criterion.criterion,
|
|
471
|
+
);
|
|
472
|
+
if (entry) entry.outcome = "outside_inventory";
|
|
473
|
+
}
|
|
474
|
+
}
|
|
475
|
+
if (selectionEvidence.wideningTriggers.length)
|
|
476
|
+
report.diagnose(
|
|
477
|
+
"conservative_widening",
|
|
478
|
+
`Changed files outside supported dependency graph: ${selectionEvidence.wideningTriggers.join(", ")}`,
|
|
479
|
+
undefined,
|
|
480
|
+
[],
|
|
481
|
+
false,
|
|
482
|
+
);
|
|
309
483
|
const selected: {
|
|
310
484
|
entry: TestEntry;
|
|
311
485
|
scenarioIds: readonly string[] | null;
|
|
@@ -371,9 +545,11 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
371
545
|
questions: Record<string, Question>,
|
|
372
546
|
witnessIds?: readonly string[],
|
|
373
547
|
) => {
|
|
548
|
+
state = withEvidenceContext(state, evidenceContext);
|
|
374
549
|
if (JSON.stringify(state).length > STATE_MAX_CHARS)
|
|
375
550
|
return {
|
|
376
551
|
ok: false as const,
|
|
552
|
+
cause: "evidence_too_large" as const,
|
|
377
553
|
error: `required evidence exceeds STATE_MAX_CHARS=${STATE_MAX_CHARS}`,
|
|
378
554
|
};
|
|
379
555
|
if (args.max_calls !== undefined && sent >= args.max_calls) {
|
|
@@ -387,6 +563,9 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
387
563
|
const result = await client.judge(state, questions, {
|
|
388
564
|
signal,
|
|
389
565
|
witnesses: witnessIds,
|
|
566
|
+
...runtime.session.requestGate(),
|
|
567
|
+
admissionCause: () =>
|
|
568
|
+
budget?.kind === "session" ? "session_budget" : "call_budget",
|
|
390
569
|
beforeRequest: (questionCount) => {
|
|
391
570
|
if (args.max_calls !== undefined && sent >= args.max_calls) {
|
|
392
571
|
budget = {
|
|
@@ -409,6 +588,7 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
409
588
|
totals.questions += result.questions ?? 0;
|
|
410
589
|
totals.cacheHits += result.cacheHits ?? 0;
|
|
411
590
|
totals.cacheRequests += result.cacheRequests ?? 0;
|
|
591
|
+
unknownCost ||= result.usage === undefined;
|
|
412
592
|
if (result.usage) {
|
|
413
593
|
costKnown = true;
|
|
414
594
|
totals.usage.inputTokens += result.usage.inputTokens;
|
|
@@ -419,6 +599,27 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
419
599
|
const pointerResults = await Promise.all(
|
|
420
600
|
candidates.map(async (candidate) => {
|
|
421
601
|
const { entry } = candidate;
|
|
602
|
+
const reportIds = [...report.items.values()]
|
|
603
|
+
.filter(
|
|
604
|
+
(item) =>
|
|
605
|
+
item.groupId === `test:${entry.path}` ||
|
|
606
|
+
item.id === `test:${entry.path}:unknown`,
|
|
607
|
+
)
|
|
608
|
+
.map((item) => item.id);
|
|
609
|
+
if (candidate.touched)
|
|
610
|
+
for (const id of reportIds) report.static(id, "touched", true);
|
|
611
|
+
if (
|
|
612
|
+
!candidate.units.length &&
|
|
613
|
+
!candidate.uncertain &&
|
|
614
|
+
!outside &&
|
|
615
|
+
!candidate.touched
|
|
616
|
+
)
|
|
617
|
+
for (const id of reportIds)
|
|
618
|
+
report.static(
|
|
619
|
+
id,
|
|
620
|
+
"static import closure cannot reach changed units",
|
|
621
|
+
false,
|
|
622
|
+
);
|
|
422
623
|
if (candidate.touched)
|
|
423
624
|
return {
|
|
424
625
|
candidate,
|
|
@@ -437,6 +638,12 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
437
638
|
if (
|
|
438
639
|
units.some((unit) => unit.before === null && unit.after === null)
|
|
439
640
|
) {
|
|
641
|
+
report.diagnose(
|
|
642
|
+
"binary_or_non_utf8",
|
|
643
|
+
"Changed source unavailable",
|
|
644
|
+
entry.path,
|
|
645
|
+
reportIds,
|
|
646
|
+
);
|
|
440
647
|
for (const unit of units) incompleteUnits.add(unit.id);
|
|
441
648
|
unjudged.push({
|
|
442
649
|
label: entry.path,
|
|
@@ -446,6 +653,13 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
446
653
|
return { candidate, selected: null, touched: false };
|
|
447
654
|
}
|
|
448
655
|
if (!entry.scenarios.length) {
|
|
656
|
+
report.totalUnknown = true;
|
|
657
|
+
report.diagnose(
|
|
658
|
+
"unsupported_syntax",
|
|
659
|
+
"Scenario names or count unresolved",
|
|
660
|
+
entry.path,
|
|
661
|
+
reportIds,
|
|
662
|
+
);
|
|
449
663
|
unjudged.push({
|
|
450
664
|
label: entry.path,
|
|
451
665
|
reason: "scenario names or count unresolved",
|
|
@@ -458,6 +672,9 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
458
672
|
(item) => item.scenario.id,
|
|
459
673
|
);
|
|
460
674
|
for (const item of prepared.unjudged) {
|
|
675
|
+
report.diagnose("evidence_too_large", item.reason, entry.path, [
|
|
676
|
+
`test:${entry.path}:${item.scenario.id}`,
|
|
677
|
+
]);
|
|
461
678
|
for (const unit of units) incompleteUnits.add(unit.id);
|
|
462
679
|
unjudged.push({
|
|
463
680
|
label: `${entry.path}: ${item.scenario.name}`,
|
|
@@ -491,6 +708,41 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
491
708
|
]),
|
|
492
709
|
);
|
|
493
710
|
const result = await judge(batch.state, questions);
|
|
711
|
+
const batchIds = batch.scenarios.map(
|
|
712
|
+
(scenario) => `test:${entry.path}:${scenario.id}`,
|
|
713
|
+
);
|
|
714
|
+
if (!result.ok)
|
|
715
|
+
report.failure(
|
|
716
|
+
result,
|
|
717
|
+
batchIds,
|
|
718
|
+
budget
|
|
719
|
+
? budget.kind === "session"
|
|
720
|
+
? "session_budget"
|
|
721
|
+
: "call_budget"
|
|
722
|
+
: !client
|
|
723
|
+
? "not_configured"
|
|
724
|
+
: undefined,
|
|
725
|
+
);
|
|
726
|
+
else
|
|
727
|
+
for (const scenario of batch.scenarios) {
|
|
728
|
+
const answer = result.answers[scenario.id];
|
|
729
|
+
const id = `test:${entry.path}:${scenario.id}`;
|
|
730
|
+
report.answer(id, answer, {
|
|
731
|
+
band: prepared.unjudged.length ? "unsure" : "verdict",
|
|
732
|
+
});
|
|
733
|
+
const item = report.items.get(id);
|
|
734
|
+
if (!item)
|
|
735
|
+
throw new Error(`Unregistered selection result ${id}`);
|
|
736
|
+
item.selection = {
|
|
737
|
+
selected:
|
|
738
|
+
answer?.type !== "choice" ||
|
|
739
|
+
1 - (answer.probabilities.none ?? 0) >= SELECT_MIN,
|
|
740
|
+
reason:
|
|
741
|
+
answer?.type === "choice"
|
|
742
|
+
? "changed-unit pointer selection threshold"
|
|
743
|
+
: "conservative_fallback",
|
|
744
|
+
};
|
|
745
|
+
}
|
|
494
746
|
if (!result.ok) {
|
|
495
747
|
fallback ||= !budget;
|
|
496
748
|
for (const unit of units) incompleteUnits.add(unit.id);
|
|
@@ -584,6 +836,24 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
584
836
|
preliminary.state,
|
|
585
837
|
candidate.entry.scenarios,
|
|
586
838
|
);
|
|
839
|
+
for (const unit of units)
|
|
840
|
+
for (const scenario of candidate.entry.scenarios)
|
|
841
|
+
report.expect(
|
|
842
|
+
`coverage:${candidate.entry.path}:${scenario.id}:${unit.id}`,
|
|
843
|
+
`${candidate.entry.path}: ${scenario.name} executes ${unit.name}`,
|
|
844
|
+
"scenario",
|
|
845
|
+
`coverage:${candidate.entry.path}:${scenario.id}`,
|
|
846
|
+
);
|
|
847
|
+
for (const omitted of prepared.unjudged)
|
|
848
|
+
for (const unit of units)
|
|
849
|
+
report.diagnose(
|
|
850
|
+
"evidence_too_large",
|
|
851
|
+
omitted.reason,
|
|
852
|
+
candidate.entry.path,
|
|
853
|
+
[
|
|
854
|
+
`coverage:${candidate.entry.path}:${omitted.scenario.id}:${unit.id}`,
|
|
855
|
+
],
|
|
856
|
+
);
|
|
587
857
|
if (prepared.unjudged.length || !candidate.entry.scenarios.length)
|
|
588
858
|
for (const unit of units) incompleteUnits.add(unit.id);
|
|
589
859
|
await Promise.all(
|
|
@@ -617,6 +887,72 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
617
887
|
result,
|
|
618
888
|
);
|
|
619
889
|
limitations.push(...health.failures);
|
|
890
|
+
for (const failure of health.failures)
|
|
891
|
+
report.diagnose(
|
|
892
|
+
"control_failure",
|
|
893
|
+
failure.fact,
|
|
894
|
+
candidate.entry.path,
|
|
895
|
+
units.flatMap((unit) =>
|
|
896
|
+
batch.scenarios
|
|
897
|
+
.filter((scenario) =>
|
|
898
|
+
health.unhealthyQuestionIds.has(
|
|
899
|
+
`${scenario.id}_${unit.id}`,
|
|
900
|
+
),
|
|
901
|
+
)
|
|
902
|
+
.map(
|
|
903
|
+
(scenario) =>
|
|
904
|
+
`coverage:${candidate.entry.path}:${scenario.id}:${unit.id}`,
|
|
905
|
+
),
|
|
906
|
+
),
|
|
907
|
+
false,
|
|
908
|
+
);
|
|
909
|
+
report.countControls(
|
|
910
|
+
result,
|
|
911
|
+
witnessQuestions.witnesses.map((witness) => witness.id),
|
|
912
|
+
);
|
|
913
|
+
for (const unit of units)
|
|
914
|
+
for (const scenario of batch.scenarios) {
|
|
915
|
+
const id = `coverage:${candidate.entry.path}:${scenario.id}:${unit.id}`;
|
|
916
|
+
report.expect(
|
|
917
|
+
id,
|
|
918
|
+
`${candidate.entry.path}: ${scenario.name} executes ${unit.name}`,
|
|
919
|
+
"scenario",
|
|
920
|
+
`coverage:${candidate.entry.path}:${scenario.id}`,
|
|
921
|
+
);
|
|
922
|
+
if (!result.ok)
|
|
923
|
+
report.failure(
|
|
924
|
+
result,
|
|
925
|
+
[id],
|
|
926
|
+
budget
|
|
927
|
+
? budget.kind === "session"
|
|
928
|
+
? "session_budget"
|
|
929
|
+
: "call_budget"
|
|
930
|
+
: !client
|
|
931
|
+
? "not_configured"
|
|
932
|
+
: undefined,
|
|
933
|
+
);
|
|
934
|
+
else
|
|
935
|
+
report.answer(
|
|
936
|
+
id,
|
|
937
|
+
result.answers[`${scenario.id}_${unit.id}`],
|
|
938
|
+
{
|
|
939
|
+
band: health.unhealthyQuestionIds.has(
|
|
940
|
+
`${scenario.id}_${unit.id}`,
|
|
941
|
+
)
|
|
942
|
+
? "unsure"
|
|
943
|
+
: "verdict",
|
|
944
|
+
reason: health.unhealthyQuestionIds.get(
|
|
945
|
+
`${scenario.id}_${unit.id}`,
|
|
946
|
+
),
|
|
947
|
+
},
|
|
948
|
+
controlsFor(
|
|
949
|
+
`${scenario.id}_${unit.id}`,
|
|
950
|
+
result,
|
|
951
|
+
witnessQuestions.witnesses.map((witness) => witness.id),
|
|
952
|
+
),
|
|
953
|
+
false,
|
|
954
|
+
);
|
|
955
|
+
}
|
|
620
956
|
for (const unit of units) {
|
|
621
957
|
if (!result.ok) {
|
|
622
958
|
fallback ||= !budget;
|
|
@@ -682,16 +1018,34 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
682
1018
|
evidenceLimits.length > 0 ||
|
|
683
1019
|
Boolean(args.paths?.length);
|
|
684
1020
|
for (const unit of residual)
|
|
685
|
-
if ((coverage.get(unit.id) ?? 0) < SELECT_MIN)
|
|
686
|
-
|
|
687
|
-
|
|
688
|
-
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
1021
|
+
if ((coverage.get(unit.id) ?? 0) < SELECT_MIN) {
|
|
1022
|
+
const supportingIds = [...report.items.keys()].filter(
|
|
1023
|
+
(id) => id.startsWith("coverage:") && id.endsWith(`:${unit.id}`),
|
|
1024
|
+
);
|
|
1025
|
+
report.diagnose(
|
|
1026
|
+
incomplete || incompleteUnits.has(unit.id)
|
|
1027
|
+
? "collection_omitted"
|
|
1028
|
+
: "conservative_widening",
|
|
1029
|
+
`Derived from retained coverage decisions: no discovered test established execution of ${unit.name} (${unit.file}) within the considered inventory only.${incomplete || incompleteUnits.has(unit.id) ? " Absence remains unsure because evidence, scope, or controls are incomplete." : " This is not a global coverage claim."}`,
|
|
1030
|
+
unit.file,
|
|
1031
|
+
supportingIds,
|
|
1032
|
+
false,
|
|
1033
|
+
);
|
|
1034
|
+
if (!supportingIds.length) {
|
|
1035
|
+
const diagnostic = report.diagnostics.at(-1);
|
|
1036
|
+
if (diagnostic)
|
|
1037
|
+
diagnostic.scope = {
|
|
1038
|
+
kind: "inventory",
|
|
1039
|
+
inventoryIds: ["test-candidates"],
|
|
1040
|
+
};
|
|
1041
|
+
const action = report.actions.at(-1);
|
|
1042
|
+
if (action)
|
|
1043
|
+
action.scope = {
|
|
1044
|
+
kind: "inventory",
|
|
1045
|
+
inventoryIds: ["test-candidates"],
|
|
1046
|
+
};
|
|
1047
|
+
}
|
|
1048
|
+
}
|
|
695
1049
|
if (fallback) {
|
|
696
1050
|
selected.length = 0;
|
|
697
1051
|
selected.push(
|
|
@@ -700,6 +1054,13 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
700
1054
|
scenarioIds: null,
|
|
701
1055
|
})),
|
|
702
1056
|
);
|
|
1057
|
+
report.diagnose(
|
|
1058
|
+
"conservative_widening",
|
|
1059
|
+
"Unavailable judgment conservatively selects all considered test candidates; static reachability exclusions do not narrow this fallback.",
|
|
1060
|
+
undefined,
|
|
1061
|
+
[],
|
|
1062
|
+
false,
|
|
1063
|
+
);
|
|
703
1064
|
limitations.push({
|
|
704
1065
|
fact: "fallback: all",
|
|
705
1066
|
next: "run all discovered tests with their project runner",
|
|
@@ -716,11 +1077,111 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
716
1077
|
})),
|
|
717
1078
|
),
|
|
718
1079
|
);
|
|
1080
|
+
for (const limit of commands.limits)
|
|
1081
|
+
for (const path of limit.files)
|
|
1082
|
+
report.diagnose("unresolved_runner", limit.reason, path, [], false);
|
|
1083
|
+
for (const limit of discovery.limits)
|
|
1084
|
+
report.diagnose(
|
|
1085
|
+
"unresolved_runner",
|
|
1086
|
+
limit.kind,
|
|
1087
|
+
limit.path,
|
|
1088
|
+
[],
|
|
1089
|
+
limit.kind !== "local_runner_unproven" &&
|
|
1090
|
+
limit.kind !== "interactive_script_skipped",
|
|
1091
|
+
);
|
|
1092
|
+
for (const limit of inventory.limits)
|
|
1093
|
+
report.diagnose(
|
|
1094
|
+
limit.kind === "secret_pattern"
|
|
1095
|
+
? "secret_pattern"
|
|
1096
|
+
: "collection_omitted",
|
|
1097
|
+
limit.kind,
|
|
1098
|
+
limit.path,
|
|
1099
|
+
);
|
|
1100
|
+
for (const limit of diff.limits)
|
|
1101
|
+
report.diagnose(
|
|
1102
|
+
limit.kind === "secret_pattern"
|
|
1103
|
+
? "secret_pattern"
|
|
1104
|
+
: "collection_omitted",
|
|
1105
|
+
limit.kind,
|
|
1106
|
+
limit.file,
|
|
1107
|
+
);
|
|
1108
|
+
for (const limit of graph.limits)
|
|
1109
|
+
report.diagnose(
|
|
1110
|
+
"dynamic_dependency",
|
|
1111
|
+
`${limit.kind}${limit.specifier ? ` (${limit.specifier})` : ""}`,
|
|
1112
|
+
limit.path,
|
|
1113
|
+
[],
|
|
1114
|
+
false,
|
|
1115
|
+
);
|
|
1116
|
+
for (const item of report.items.values()) {
|
|
1117
|
+
const candidate = candidates.find(
|
|
1118
|
+
(candidate) =>
|
|
1119
|
+
item.id.startsWith(`test:${candidate.entry.path}:`) ||
|
|
1120
|
+
item.id.startsWith(`coverage:${candidate.entry.path}:`),
|
|
1121
|
+
);
|
|
1122
|
+
if (!candidate) continue;
|
|
1123
|
+
const plan = selected.find((plan) => plan.entry === candidate.entry);
|
|
1124
|
+
const scenario = candidate.entry.scenarios.find(
|
|
1125
|
+
(scenario) =>
|
|
1126
|
+
item.id === `test:${candidate.entry.path}:${scenario.id}` ||
|
|
1127
|
+
item.id.startsWith(
|
|
1128
|
+
`coverage:${candidate.entry.path}:${scenario.id}:`,
|
|
1129
|
+
),
|
|
1130
|
+
);
|
|
1131
|
+
const isSelected = Boolean(
|
|
1132
|
+
plan &&
|
|
1133
|
+
(plan.scenarioIds === null ||
|
|
1134
|
+
(scenario && plan.scenarioIds.includes(scenario.id))),
|
|
1135
|
+
);
|
|
1136
|
+
item.selection = {
|
|
1137
|
+
selected: isSelected,
|
|
1138
|
+
reason: fallback
|
|
1139
|
+
? "conservative_fallback"
|
|
1140
|
+
: item.treatment === "static"
|
|
1141
|
+
? item.staticReason
|
|
1142
|
+
: item.treatment === "not_judged"
|
|
1143
|
+
? "conservative_fallback"
|
|
1144
|
+
: "unchanged selection policy and whole-file widening",
|
|
1145
|
+
};
|
|
1146
|
+
}
|
|
1147
|
+
for (const criterion of report.inventories[0]?.criteria ?? [])
|
|
1148
|
+
criterion.diagnosticIds = report.diagnostics
|
|
1149
|
+
.filter(
|
|
1150
|
+
(diagnostic) =>
|
|
1151
|
+
diagnostic.cause === "criteria_no_match" &&
|
|
1152
|
+
diagnostic.target.status === "known" &&
|
|
1153
|
+
diagnostic.target.value === criterion.criterion,
|
|
1154
|
+
)
|
|
1155
|
+
.map((diagnostic) => diagnostic.id);
|
|
1156
|
+
report.actions.push({
|
|
1157
|
+
id: "execute-selection-plan",
|
|
1158
|
+
code: "execute_plan",
|
|
1159
|
+
target: reportKnown(inventory.cwd),
|
|
1160
|
+
scope: { kind: "call" },
|
|
1161
|
+
condition:
|
|
1162
|
+
"When verification is authorized and the listed runner/configuration is available.",
|
|
1163
|
+
instruction:
|
|
1164
|
+
"Execute the retained runner commands separately; this tool has not executed any test.",
|
|
1165
|
+
repeatUnchanged: false,
|
|
1166
|
+
});
|
|
719
1167
|
if (skipped)
|
|
720
1168
|
limitations.push({
|
|
721
1169
|
fact: `${skipped} tests skipped because their imports cannot reach the diff`,
|
|
722
1170
|
next: "an alias or dynamic import would have kept a test in",
|
|
723
1171
|
});
|
|
1172
|
+
for (const criterion of selectionEvidence.criteria)
|
|
1173
|
+
if (!criterion.matches.length)
|
|
1174
|
+
limitations.push({
|
|
1175
|
+
path: criterion.criterion,
|
|
1176
|
+
cause: "selection criterion has no discovered test match",
|
|
1177
|
+
fact: `selection criterion without match: ${criterion.criterion}`,
|
|
1178
|
+
next: "Change the criterion or supply the missing test/configuration evidence; an empty scope does not prove absence of impact.",
|
|
1179
|
+
});
|
|
1180
|
+
if (selectionEvidence.wideningTriggers.length)
|
|
1181
|
+
limitations.push({
|
|
1182
|
+
fact: `selection widened conservatively: ${selectionEvidence.wideningTriggers.join(", ")}`,
|
|
1183
|
+
next: "Run the widened static plans; dependency reachability is not established for these changed files.",
|
|
1184
|
+
});
|
|
724
1185
|
return finish({
|
|
725
1186
|
answers,
|
|
726
1187
|
unjudged,
|
|
@@ -735,7 +1196,7 @@ export function createSelectTestsTool(dependencies: ToolDependencies) {
|
|
|
735
1196
|
lines: commands.commands.map((command) => ({
|
|
736
1197
|
type: "command" as const,
|
|
737
1198
|
...command,
|
|
738
|
-
cwd: relative(
|
|
1199
|
+
cwd: relative(cwd, resolve(inventory.cwd, command.cwd)) || ".",
|
|
739
1200
|
})),
|
|
740
1201
|
});
|
|
741
1202
|
},
|