jev-agent-tools 0.1.4 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +70 -1
- package/CONTRIBUTING.md +40 -0
- package/README.md +42 -9
- package/SECURITY.md +27 -0
- package/dist/adapters/analysis-context.js +75 -0
- package/dist/adapters/ask-files.js +189 -0
- package/dist/adapters/ask-proof.js +144 -0
- package/dist/adapters/ask-syntax.js +385 -0
- package/dist/adapters/canonical-path.js +17 -0
- package/dist/adapters/command.js +181 -0
- package/dist/adapters/docs.js +172 -0
- package/dist/adapters/exec.js +207 -0
- package/dist/adapters/files.js +293 -0
- package/dist/adapters/find.js +122 -0
- package/dist/adapters/git-base.js +26 -0
- package/dist/adapters/git-inventory.js +71 -0
- package/dist/adapters/git.js +439 -0
- package/dist/adapters/locate-file.js +159 -0
- package/dist/adapters/output-lines.js +46 -0
- package/dist/adapters/private-storage.js +98 -0
- package/dist/adapters/risk-callers.js +426 -0
- package/dist/adapters/runner-version.js +78 -0
- package/dist/adapters/shell.js +76 -0
- package/dist/adapters/syntax.js +187 -0
- package/dist/adapters/test-inventory.js +131 -0
- package/dist/adapters/usage.js +20 -0
- package/dist/adapters/utf8.js +47 -0
- package/dist/configuration.js +257 -0
- package/dist/constants.js +119 -0
- package/dist/core/ask-closure.js +282 -0
- package/dist/core/ask-proof.js +1 -0
- package/dist/core/ask-references.js +194 -0
- package/dist/core/asks.js +436 -0
- package/dist/core/batches.js +65 -0
- package/dist/core/command-output.js +224 -0
- package/dist/core/diff.js +178 -0
- package/dist/core/docs.js +302 -0
- package/dist/core/find.js +108 -0
- package/dist/core/git.js +1 -0
- package/dist/core/imports.js +550 -0
- package/dist/core/integrity.js +45 -0
- package/dist/core/lexical.js +132 -0
- package/dist/core/locate.js +169 -0
- package/dist/core/output.js +120 -0
- package/dist/core/pointer.js +29 -0
- package/dist/core/risk-callers.js +851 -0
- package/dist/core/runner-version.js +45 -0
- package/dist/core/sections.js +230 -0
- package/dist/core/state.js +44 -0
- package/dist/core/syntax.js +1 -0
- package/dist/core/test-commands.js +334 -0
- package/dist/core/test-coverage.js +74 -0
- package/dist/core/test-discovery.js +1382 -0
- package/dist/core/test-evidence.js +527 -0
- package/dist/core/test-state.js +81 -0
- package/dist/core/truncate.js +12 -0
- package/dist/core/units.js +349 -0
- package/dist/describe.js +23 -0
- package/dist/guide.js +33 -0
- package/dist/host.js +24 -0
- package/dist/jev/client.js +434 -0
- package/dist/jev/pool.js +54 -0
- package/dist/jev/types.js +1 -0
- package/dist/mcp/main.js +124 -0
- package/dist/mcp/protocol.js +187 -0
- package/dist/mcp/tools.js +116 -0
- package/dist/presets/docs.js +62 -0
- package/dist/presets/risk.js +179 -0
- package/dist/presets/spec.js +81 -0
- package/dist/presets/witnesses.js +249 -0
- package/dist/render.js +42 -0
- package/dist/result.js +3 -0
- package/dist/runtime.js +1 -0
- package/dist/session.js +147 -0
- package/dist/texts/ask-files.js +1 -0
- package/dist/texts/ask.js +2 -0
- package/dist/texts/check-diff.js +17 -0
- package/dist/texts/configuration.js +1 -0
- package/dist/texts/find.js +14 -0
- package/dist/texts/guide.js +16 -0
- package/dist/texts/locate.js +10 -0
- package/dist/texts/select-tests.js +2 -0
- package/dist/tools/ask-files.js +217 -0
- package/dist/tools/ask-schema.js +70 -0
- package/dist/tools/ask.js +686 -0
- package/dist/tools/check-diff.js +402 -0
- package/dist/tools/docs-check.js +299 -0
- package/dist/tools/find.js +389 -0
- package/dist/tools/locate.js +303 -0
- package/dist/tools/select-tests.js +567 -0
- package/dist/tools/spec-check.js +166 -0
- package/docs/adr/0001-strict-typescript-pure-core-offline-tests.md +31 -0
- package/docs/adr/0002-one-http-protocol-across-hosts.md +17 -0
- package/docs/adr/0003-explicit-scope-conservative-automation.md +19 -0
- package/docs/adr/0004-compiled-typed-intents.md +19 -0
- package/docs/adr/0005-evidence-construction-before-judgment.md +19 -0
- package/docs/adr/0006-visible-uncertainty-constrained-controls.md +21 -0
- package/docs/adr/0007-bounded-evidence-visible-limits.md +21 -0
- package/docs/adr/0008-static-test-discovery-conservative-plans.md +19 -0
- package/docs/adr/0009-session-cache-requested-model-identity.md +17 -0
- package/docs/adr/0010-mcp-server-thin-host.md +23 -0
- package/docs/agent-instructions.md +91 -0
- package/docs/design.md +3 -3
- package/docs/mcp.md +231 -0
- package/package.json +19 -4
- package/server.json +57 -0
- package/src/adapters/canonical-path.ts +18 -0
- package/src/adapters/command.ts +7 -4
- package/src/adapters/exec.ts +226 -0
- package/src/adapters/private-storage.ts +143 -0
- package/src/adapters/risk-callers.ts +4 -2
- package/src/adapters/shell.ts +97 -0
- package/src/configuration.ts +39 -12
- package/src/constants.ts +11 -0
- package/src/core/command-output.ts +17 -1
- package/src/host.ts +11 -0
- package/src/jev/client.ts +12 -0
- package/src/jev/types.ts +6 -0
- package/src/mcp/main.ts +135 -0
- package/src/mcp/protocol.ts +282 -0
- package/src/mcp/tools.ts +166 -0
- package/src/session.ts +59 -0
- package/src/setup.ts +13 -5
- package/src/tools/ask-files.ts +5 -7
- package/src/tools/ask.ts +26 -22
- package/src/tools/check-diff.ts +8 -5
- package/src/tools/docs-check.ts +1 -0
- package/src/tools/find.ts +5 -2
- package/src/tools/locate.ts +5 -8
- package/src/tools/select-tests.ts +7 -4
- package/src/tools/spec-check.ts +1 -0
|
@@ -0,0 +1,567 @@
|
|
|
1
|
+
import { matchesGlob, relative, resolve } from "node:path";
|
|
2
|
+
import { Type } from "@sinclair/typebox";
|
|
3
|
+
import { createAnalysisContext } from "../adapters/analysis-context.js";
|
|
4
|
+
import { canonicalPath } from "../adapters/canonical-path.js";
|
|
5
|
+
import { collectUnits } from "../adapters/git.js";
|
|
6
|
+
import { resolveBase } from "../adapters/git-base.js";
|
|
7
|
+
import { shareGitInventory } from "../adapters/git-inventory.js";
|
|
8
|
+
import { attachRunnerVersions } from "../adapters/runner-version.js";
|
|
9
|
+
import { collectTestInventory } from "../adapters/test-inventory.js";
|
|
10
|
+
import { hostUsage } from "../adapters/usage.js";
|
|
11
|
+
import { SELECT_MIN, STATE_MAX_CHARS, WITNESS_AUTO_MIN_CELLS, } from "../constants.js";
|
|
12
|
+
import { createImportGraphBuilder, importClosureLazy, } from "../core/imports.js";
|
|
13
|
+
import { buildEnvelope } from "../core/output.js";
|
|
14
|
+
import { buildRunnerCommands } from "../core/test-commands.js";
|
|
15
|
+
import { prepareCoverageWitnesses } from "../core/test-coverage.js";
|
|
16
|
+
import { discoverTests, isTestConfiguration } from "../core/test-discovery.js";
|
|
17
|
+
import { prepareTestEvidence, } from "../core/test-evidence.js";
|
|
18
|
+
import { prepareTestStates } from "../core/test-state.js";
|
|
19
|
+
import { buildCoverageWitnessUnits, evaluateBatchWitnessHealth, } from "../presets/witnesses.js";
|
|
20
|
+
import { renderEnvelope } from "../render.js";
|
|
21
|
+
import { SELECT_TESTS_DESCRIPTION } from "../texts/select-tests.js";
|
|
22
|
+
export const selectTestsParameters = Type.Object({
|
|
23
|
+
base: Type.Optional(Type.String({ minLength: 1 })),
|
|
24
|
+
paths: Type.Optional(Type.Array(Type.String({ minLength: 1 }))),
|
|
25
|
+
witnesses: Type.Optional(Type.Union([
|
|
26
|
+
Type.Literal("off"),
|
|
27
|
+
Type.Literal("auto"),
|
|
28
|
+
Type.Literal("on"),
|
|
29
|
+
])),
|
|
30
|
+
max_calls: Type.Optional(Type.Integer({ minimum: 0 })),
|
|
31
|
+
}, { additionalProperties: false });
|
|
32
|
+
function makeCandidate(entry, files, evidence, changed, outside) {
|
|
33
|
+
const enriched = evidence.units.map((unit) => ({
|
|
34
|
+
id: unit.id,
|
|
35
|
+
kind: unit.kind,
|
|
36
|
+
file: unit.file,
|
|
37
|
+
name: unit.name,
|
|
38
|
+
exported: unit.exported,
|
|
39
|
+
before: unit.before,
|
|
40
|
+
after: unit.after,
|
|
41
|
+
usedBy: (evidence.usedBy.get(unit.id) ?? []).map(({ path, text }) => ({
|
|
42
|
+
path,
|
|
43
|
+
text,
|
|
44
|
+
})),
|
|
45
|
+
}));
|
|
46
|
+
return {
|
|
47
|
+
entry,
|
|
48
|
+
paths: evidence.paths,
|
|
49
|
+
units: evidence.units,
|
|
50
|
+
state: {
|
|
51
|
+
testFile: { path: entry.path, text: files.get(entry.path)?.text ?? "" },
|
|
52
|
+
changedUnits: enriched,
|
|
53
|
+
imports: evidence.imports.map(({ path, text }) => ({ path, text })),
|
|
54
|
+
},
|
|
55
|
+
touched: changed.has(entry.path),
|
|
56
|
+
uncertain: evidence.uncertain || outside,
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
export function createSelectTestsTool(dependencies) {
|
|
60
|
+
const { host, runtime, exec: execute } = dependencies;
|
|
61
|
+
return {
|
|
62
|
+
name: "jev_select_tests",
|
|
63
|
+
label: "Jev select tests",
|
|
64
|
+
description: SELECT_TESTS_DESCRIPTION,
|
|
65
|
+
parameters: selectTestsParameters,
|
|
66
|
+
...(host.isOmp
|
|
67
|
+
? { approval: "read", loadMode: "essential" }
|
|
68
|
+
: {
|
|
69
|
+
promptSnippet: "Pick the tests the diff can affect and the runner command for only those",
|
|
70
|
+
}),
|
|
71
|
+
async execute(_id, args, signal, _update, ctx) {
|
|
72
|
+
const client = dependencies.client;
|
|
73
|
+
const cwd = await canonicalPath(ctx.cwd);
|
|
74
|
+
const started = performance.now();
|
|
75
|
+
const exec = shareGitInventory(execute);
|
|
76
|
+
const totals = {
|
|
77
|
+
calls: 0,
|
|
78
|
+
questions: 0,
|
|
79
|
+
cacheHits: 0,
|
|
80
|
+
cacheRequests: 0,
|
|
81
|
+
usage: { inputTokens: 0, costUsd: 0 },
|
|
82
|
+
};
|
|
83
|
+
let costKnown = false;
|
|
84
|
+
let sent = 0;
|
|
85
|
+
let budget;
|
|
86
|
+
const finish = (input) => {
|
|
87
|
+
const limitKeys = new Set();
|
|
88
|
+
const uniqueLimits = input.limitations?.filter((limit) => {
|
|
89
|
+
const key = JSON.stringify([limit.fact, limit.next]);
|
|
90
|
+
if (limitKeys.has(key))
|
|
91
|
+
return false;
|
|
92
|
+
limitKeys.add(key);
|
|
93
|
+
return true;
|
|
94
|
+
});
|
|
95
|
+
const envelope = buildEnvelope({
|
|
96
|
+
...input,
|
|
97
|
+
limitations: uniqueLimits,
|
|
98
|
+
yield: {
|
|
99
|
+
...totals,
|
|
100
|
+
costUsd: costKnown ? totals.usage.costUsd : undefined,
|
|
101
|
+
elapsedMs: performance.now() - started,
|
|
102
|
+
},
|
|
103
|
+
});
|
|
104
|
+
runtime.session.record(envelope);
|
|
105
|
+
runtime.guide.deliver(ctx);
|
|
106
|
+
const details = {
|
|
107
|
+
ok: true,
|
|
108
|
+
answers: {},
|
|
109
|
+
...totals,
|
|
110
|
+
limitations: uniqueLimits,
|
|
111
|
+
};
|
|
112
|
+
return {
|
|
113
|
+
content: [{ type: "text", text: renderEnvelope(envelope) }],
|
|
114
|
+
details,
|
|
115
|
+
...hostUsage(host.isOmp, costKnown ? totals.usage : undefined),
|
|
116
|
+
};
|
|
117
|
+
};
|
|
118
|
+
const comparison = await resolveBase(exec, cwd, args.base, signal);
|
|
119
|
+
if (!comparison.ok)
|
|
120
|
+
return finish({ refusal: comparison.error });
|
|
121
|
+
const analysis = await createAnalysisContext();
|
|
122
|
+
const [inventory, diff] = await Promise.all([
|
|
123
|
+
collectTestInventory(exec, cwd, signal),
|
|
124
|
+
collectUnits(exec, { cwd: cwd, base: comparison.base, signal }, analysis.parser),
|
|
125
|
+
]);
|
|
126
|
+
if (!inventory.ok)
|
|
127
|
+
return finish({ refusal: inventory.error });
|
|
128
|
+
if (!diff.ok)
|
|
129
|
+
return finish({ refusal: diff.error });
|
|
130
|
+
const known = new Set(inventory.paths);
|
|
131
|
+
const sourceFiles = new Map();
|
|
132
|
+
const read = async (path) => {
|
|
133
|
+
const source = await inventory.read(path);
|
|
134
|
+
if (source)
|
|
135
|
+
sourceFiles.set(path, source);
|
|
136
|
+
return source;
|
|
137
|
+
};
|
|
138
|
+
const configurationPaths = inventory.paths.filter((path) => isTestConfiguration(path) ||
|
|
139
|
+
/(?:^|\/)tsconfig[^/]*\.json$/.test(path));
|
|
140
|
+
for (let offset = 0; offset < configurationPaths.length; offset += 16)
|
|
141
|
+
await Promise.all(configurationPaths.slice(offset, offset + 16).map(read));
|
|
142
|
+
const candidatePaths = new Set();
|
|
143
|
+
const configurationSet = new Set(configurationPaths);
|
|
144
|
+
for (;;) {
|
|
145
|
+
candidatePaths.clear();
|
|
146
|
+
const references = new Set();
|
|
147
|
+
discoverTests(inventory.paths.map((path) => sourceFiles.get(path) ?? { path, text: "" }), candidatePaths, analysis, references);
|
|
148
|
+
const pending = [...references].filter((path) => known.has(path) && !configurationSet.has(path));
|
|
149
|
+
if (!pending.length)
|
|
150
|
+
break;
|
|
151
|
+
for (const path of pending)
|
|
152
|
+
configurationSet.add(path);
|
|
153
|
+
configurationPaths.push(...pending);
|
|
154
|
+
for (let offset = 0; offset < pending.length; offset += 16)
|
|
155
|
+
await Promise.all(pending.slice(offset, offset + 16).map(read));
|
|
156
|
+
}
|
|
157
|
+
const planned = [...candidatePaths];
|
|
158
|
+
for (let offset = 0; offset < planned.length; offset += 16)
|
|
159
|
+
await Promise.all(planned.slice(offset, offset + 16).map(read));
|
|
160
|
+
const discovery = discoverTests([...sourceFiles.values()], undefined, analysis);
|
|
161
|
+
const versionedEntries = await attachRunnerVersions(inventory.cwd, discovery.entries, inventory.read, known);
|
|
162
|
+
const changed = new Set(diff.files.flatMap((file) => [file.path, file.oldPath]));
|
|
163
|
+
const build = createImportGraphBuilder([...sourceFiles.values()].filter((file) => configurationPaths.includes(file.path)), known, analysis);
|
|
164
|
+
const cache = new Map();
|
|
165
|
+
const roots = [
|
|
166
|
+
...new Set([
|
|
167
|
+
...discovery.entries.map((entry) => entry.path),
|
|
168
|
+
...diff.units.map((unit) => unit.file),
|
|
169
|
+
...inventory.paths.filter((path) => /(?:^|\/)conftest\.py$/.test(path)),
|
|
170
|
+
]),
|
|
171
|
+
];
|
|
172
|
+
for (const path of roots)
|
|
173
|
+
await importClosureLazy(read, path, { known, build, cache });
|
|
174
|
+
const graphs = await Promise.all(cache.values());
|
|
175
|
+
const graph = {
|
|
176
|
+
edges: new Map(graphs.flatMap((part) => [...part.edges])),
|
|
177
|
+
limits: graphs.flatMap((part) => part.limits),
|
|
178
|
+
};
|
|
179
|
+
const evidenceFor = prepareTestEvidence([...sourceFiles.values()], graph, diff.units, analysis.parser);
|
|
180
|
+
const evidenceLimits = [];
|
|
181
|
+
const outside = diff.files.some((file) => !graph.edges.has(file.path) ||
|
|
182
|
+
!/\.[cm]?[jt]sx?$|\.py$/.test(file.path));
|
|
183
|
+
const candidates = versionedEntries
|
|
184
|
+
.filter((entry) => !args.paths?.length ||
|
|
185
|
+
args.paths.some((path) => matchesGlob(entry.path, path) ||
|
|
186
|
+
entry.path.startsWith(`${path.replace(/\/$/, "")}/`)))
|
|
187
|
+
.map((entry) => {
|
|
188
|
+
const evidence = evidenceFor(entry);
|
|
189
|
+
evidenceLimits.push(...evidence.limits.map((limit) => ({
|
|
190
|
+
cause: limit.kind,
|
|
191
|
+
path: limit.path,
|
|
192
|
+
dependency: limit.specifier,
|
|
193
|
+
fact: `${limit.kind}: ${limit.path}${limit.specifier ? ` (${limit.specifier})` : ""}`,
|
|
194
|
+
next: "read the unresolved call chain or run this test",
|
|
195
|
+
})));
|
|
196
|
+
return makeCandidate(entry, sourceFiles, outside || evidence.uncertain
|
|
197
|
+
? { ...evidence, units: diff.units }
|
|
198
|
+
: evidence, changed, outside);
|
|
199
|
+
});
|
|
200
|
+
const selected = [];
|
|
201
|
+
const answers = [];
|
|
202
|
+
const unjudged = [];
|
|
203
|
+
const limitations = [
|
|
204
|
+
...evidenceLimits,
|
|
205
|
+
...discovery.limits.map((limit) => ({
|
|
206
|
+
cause: `runner ${limit.kind}${limit.framework ? ` (${limit.framework})` : ""}`,
|
|
207
|
+
path: limit.path,
|
|
208
|
+
fact: limit.kind === "types"
|
|
209
|
+
? `type tests separate from runtime — ${limit.path}`
|
|
210
|
+
: limit.kind === "local_runner_unproven"
|
|
211
|
+
? `local runner not proven: ${limit.framework} — ${limit.path}`
|
|
212
|
+
: limit.kind === "interactive_script_skipped"
|
|
213
|
+
? `interactive runner script skipped — ${limit.path}`
|
|
214
|
+
: limit.kind === "unsupported"
|
|
215
|
+
? `framework unsupported: ${limit.framework} — ${limit.path}`
|
|
216
|
+
: limit.kind === "unknown"
|
|
217
|
+
? `framework unknown — ${limit.path}`
|
|
218
|
+
: `runner discovery incomplete — ${limit.path}`,
|
|
219
|
+
next: limit.kind === "types"
|
|
220
|
+
? "run the identified project type-check command"
|
|
221
|
+
: limit.kind === "local_runner_unproven"
|
|
222
|
+
? "verify the project runner executable"
|
|
223
|
+
: limit.kind === "interactive_script_skipped"
|
|
224
|
+
? "use the non-interactive native runner command"
|
|
225
|
+
: limit.kind === "unsupported" || limit.kind === "unknown"
|
|
226
|
+
? "run the listed files with the project runner"
|
|
227
|
+
: "read the config or collect with the project runner",
|
|
228
|
+
})),
|
|
229
|
+
...inventory.limits.map((limit) => ({
|
|
230
|
+
cause: limit.kind,
|
|
231
|
+
path: limit.path,
|
|
232
|
+
fact: `${limit.kind}: ${limit.path}`,
|
|
233
|
+
next: "read the missing file or collect with the project runner",
|
|
234
|
+
})),
|
|
235
|
+
];
|
|
236
|
+
for (const limit of graph.limits)
|
|
237
|
+
limitations.push({
|
|
238
|
+
cause: `import ${limit.kind}`,
|
|
239
|
+
path: limit.path,
|
|
240
|
+
dependency: limit.specifier,
|
|
241
|
+
fact: `${limit.kind === "dynamic" ? "dynamic import unresolved" : `import ${limit.kind}`}: ${limit.path} (${limit.specifier})`,
|
|
242
|
+
next: "read the missing dependency or collect with the project runner",
|
|
243
|
+
});
|
|
244
|
+
for (const limit of diff.limits)
|
|
245
|
+
limitations.push({
|
|
246
|
+
cause: limit.kind,
|
|
247
|
+
path: limit.file,
|
|
248
|
+
fact: `${limit.kind}: ${limit.file}`,
|
|
249
|
+
next: "read the complete changed source",
|
|
250
|
+
});
|
|
251
|
+
const coverage = new Map();
|
|
252
|
+
const incompleteUnits = new Set();
|
|
253
|
+
let fallback = false;
|
|
254
|
+
let skipped = 0;
|
|
255
|
+
const judge = async (state, questions, witnessIds) => {
|
|
256
|
+
if (JSON.stringify(state).length > STATE_MAX_CHARS)
|
|
257
|
+
return {
|
|
258
|
+
ok: false,
|
|
259
|
+
error: `required evidence exceeds STATE_MAX_CHARS=${STATE_MAX_CHARS}`,
|
|
260
|
+
};
|
|
261
|
+
if (args.max_calls !== undefined && sent >= args.max_calls) {
|
|
262
|
+
budget = {
|
|
263
|
+
kind: "max_calls",
|
|
264
|
+
message: `max_calls=${args.max_calls} reached`,
|
|
265
|
+
};
|
|
266
|
+
return { ok: false, error: budget.message };
|
|
267
|
+
}
|
|
268
|
+
if (!client)
|
|
269
|
+
return { ok: false, error: "Jev unavailable" };
|
|
270
|
+
const result = await client.judge(state, questions, {
|
|
271
|
+
signal,
|
|
272
|
+
witnesses: witnessIds,
|
|
273
|
+
...runtime.session.requestGate(),
|
|
274
|
+
beforeRequest: (questionCount) => {
|
|
275
|
+
if (args.max_calls !== undefined && sent >= args.max_calls) {
|
|
276
|
+
budget = {
|
|
277
|
+
kind: "max_calls",
|
|
278
|
+
message: `max_calls=${args.max_calls} reached`,
|
|
279
|
+
};
|
|
280
|
+
return { ok: false, error: budget.message };
|
|
281
|
+
}
|
|
282
|
+
const admitted = runtime.session.admit(questionCount);
|
|
283
|
+
if (!admitted.ok) {
|
|
284
|
+
budget = { kind: "session", message: admitted.error };
|
|
285
|
+
return admitted;
|
|
286
|
+
}
|
|
287
|
+
sent++;
|
|
288
|
+
return admitted;
|
|
289
|
+
},
|
|
290
|
+
onUsage: (usage) => runtime.session.recordUsage(usage),
|
|
291
|
+
});
|
|
292
|
+
totals.calls += result.calls ?? 0;
|
|
293
|
+
totals.questions += result.questions ?? 0;
|
|
294
|
+
totals.cacheHits += result.cacheHits ?? 0;
|
|
295
|
+
totals.cacheRequests += result.cacheRequests ?? 0;
|
|
296
|
+
if (result.usage) {
|
|
297
|
+
costKnown = true;
|
|
298
|
+
totals.usage.inputTokens += result.usage.inputTokens;
|
|
299
|
+
totals.usage.costUsd += result.usage.costUsd;
|
|
300
|
+
}
|
|
301
|
+
return result;
|
|
302
|
+
};
|
|
303
|
+
const pointerResults = await Promise.all(candidates.map(async (candidate) => {
|
|
304
|
+
const { entry } = candidate;
|
|
305
|
+
if (candidate.touched)
|
|
306
|
+
return {
|
|
307
|
+
candidate,
|
|
308
|
+
selected: null,
|
|
309
|
+
touched: true,
|
|
310
|
+
};
|
|
311
|
+
if (!candidate.units.length && !candidate.uncertain && !outside) {
|
|
312
|
+
skipped++;
|
|
313
|
+
return {
|
|
314
|
+
candidate,
|
|
315
|
+
selected: [],
|
|
316
|
+
touched: false,
|
|
317
|
+
};
|
|
318
|
+
}
|
|
319
|
+
const units = candidate.units;
|
|
320
|
+
if (units.some((unit) => unit.before === null && unit.after === null)) {
|
|
321
|
+
for (const unit of units)
|
|
322
|
+
incompleteUnits.add(unit.id);
|
|
323
|
+
unjudged.push({
|
|
324
|
+
label: entry.path,
|
|
325
|
+
reason: "changed source not UTF-8 text or unavailable",
|
|
326
|
+
next: "run the whole file with the project runner",
|
|
327
|
+
});
|
|
328
|
+
return { candidate, selected: null, touched: false };
|
|
329
|
+
}
|
|
330
|
+
if (!entry.scenarios.length) {
|
|
331
|
+
unjudged.push({
|
|
332
|
+
label: entry.path,
|
|
333
|
+
reason: "scenario names or count unresolved",
|
|
334
|
+
next: "run the whole file with the project runner",
|
|
335
|
+
});
|
|
336
|
+
return { candidate, selected: null, touched: false };
|
|
337
|
+
}
|
|
338
|
+
const prepared = prepareTestStates(candidate.state, entry.scenarios);
|
|
339
|
+
const ids = prepared.unjudged.map((item) => item.scenario.id);
|
|
340
|
+
for (const item of prepared.unjudged) {
|
|
341
|
+
for (const unit of units)
|
|
342
|
+
incompleteUnits.add(unit.id);
|
|
343
|
+
unjudged.push({
|
|
344
|
+
label: `${entry.path}: ${item.scenario.name}`,
|
|
345
|
+
reason: item.reason,
|
|
346
|
+
next: "run this test with the project runner",
|
|
347
|
+
});
|
|
348
|
+
limitations.push({
|
|
349
|
+
cause: "required test evidence omitted",
|
|
350
|
+
path: entry.path,
|
|
351
|
+
dependency: item.scenario.name,
|
|
352
|
+
fact: `required test evidence omitted: ${entry.path}: ${item.scenario.name}`,
|
|
353
|
+
next: "read the complete test body and required imports",
|
|
354
|
+
});
|
|
355
|
+
}
|
|
356
|
+
await Promise.all(prepared.batches.map(async (batch) => {
|
|
357
|
+
const questions = Object.fromEntries(batch.scenarios.map((scenario) => [
|
|
358
|
+
scenario.id,
|
|
359
|
+
{
|
|
360
|
+
type: "choice",
|
|
361
|
+
instructions: `When the test named ${scenario.name} runs, which changed unit does it execute or read, directly or through the functions it calls?`,
|
|
362
|
+
criteria: Object.fromEntries([
|
|
363
|
+
...units.map((unit) => [
|
|
364
|
+
unit.id,
|
|
365
|
+
`${unit.name} (${unit.file})`,
|
|
366
|
+
]),
|
|
367
|
+
["none", "No changed unit executes or gets read"],
|
|
368
|
+
]),
|
|
369
|
+
},
|
|
370
|
+
]));
|
|
371
|
+
const result = await judge(batch.state, questions);
|
|
372
|
+
if (!result.ok) {
|
|
373
|
+
fallback ||= !budget;
|
|
374
|
+
for (const unit of units)
|
|
375
|
+
incompleteUnits.add(unit.id);
|
|
376
|
+
ids.push(...batch.scenarios.map((scenario) => scenario.id));
|
|
377
|
+
unjudged.push({
|
|
378
|
+
label: entry.path,
|
|
379
|
+
reason: result.error,
|
|
380
|
+
next: budget ? "raise max_calls" : "run all discovered tests",
|
|
381
|
+
});
|
|
382
|
+
return;
|
|
383
|
+
}
|
|
384
|
+
for (const scenario of batch.scenarios) {
|
|
385
|
+
const answer = result.answers[scenario.id];
|
|
386
|
+
if (answer?.type !== "choice" ||
|
|
387
|
+
typeof answer.probabilities.none !== "number" ||
|
|
388
|
+
!Number.isFinite(answer.probabilities.none)) {
|
|
389
|
+
ids.push(scenario.id);
|
|
390
|
+
for (const unit of units)
|
|
391
|
+
incompleteUnits.add(unit.id);
|
|
392
|
+
unjudged.push({
|
|
393
|
+
label: `${entry.path}: ${scenario.name}`,
|
|
394
|
+
reason: answer?.type === "unjudged"
|
|
395
|
+
? answer.reason
|
|
396
|
+
: "missing valid pointer",
|
|
397
|
+
next: "run this test",
|
|
398
|
+
});
|
|
399
|
+
continue;
|
|
400
|
+
}
|
|
401
|
+
const p = 1 - answer.probabilities.none;
|
|
402
|
+
if (p >= SELECT_MIN) {
|
|
403
|
+
ids.push(scenario.id);
|
|
404
|
+
coverage.set(answer.choice, Math.max(coverage.get(answer.choice) ?? 0, p));
|
|
405
|
+
answers.push({
|
|
406
|
+
label: `${entry.path}: ${scenario.name}`,
|
|
407
|
+
value: {
|
|
408
|
+
head: `runs ${units.find((unit) => unit.id === answer.choice)?.name ?? answer.choice}`,
|
|
409
|
+
p,
|
|
410
|
+
},
|
|
411
|
+
band: prepared.unjudged.length ? "unsure" : "verdict",
|
|
412
|
+
});
|
|
413
|
+
}
|
|
414
|
+
}
|
|
415
|
+
}));
|
|
416
|
+
return { candidate, selected: ids, touched: false };
|
|
417
|
+
}));
|
|
418
|
+
for (const result of pointerResults) {
|
|
419
|
+
if (result.selected === null || result.selected.length)
|
|
420
|
+
selected.push({
|
|
421
|
+
entry: result.candidate.entry,
|
|
422
|
+
scenarioIds: result.selected,
|
|
423
|
+
});
|
|
424
|
+
if (result.touched)
|
|
425
|
+
answers.push({
|
|
426
|
+
label: result.candidate.entry.path,
|
|
427
|
+
value: { head: "touched", p: 1 },
|
|
428
|
+
band: "verdict",
|
|
429
|
+
});
|
|
430
|
+
}
|
|
431
|
+
const residual = diff.units.filter((unit) => unit.exported && (coverage.get(unit.id) ?? 0) < SELECT_MIN);
|
|
432
|
+
await Promise.all(candidates.map(async (candidate) => {
|
|
433
|
+
const units = residual.filter((unit) => candidate.units.some((reached) => reached.id === unit.id) ||
|
|
434
|
+
candidate.uncertain ||
|
|
435
|
+
outside);
|
|
436
|
+
if (!units.length)
|
|
437
|
+
return;
|
|
438
|
+
const witnessUnits = args.witnesses === "off" ||
|
|
439
|
+
(args.witnesses !== "on" &&
|
|
440
|
+
units.length * candidate.entry.scenarios.length <
|
|
441
|
+
WITNESS_AUTO_MIN_CELLS)
|
|
442
|
+
? []
|
|
443
|
+
: buildCoverageWitnessUnits(units);
|
|
444
|
+
const preliminary = prepareCoverageWitnesses(candidate.state, {}, witnessUnits);
|
|
445
|
+
const prepared = prepareTestStates(preliminary.state, candidate.entry.scenarios);
|
|
446
|
+
if (prepared.unjudged.length || !candidate.entry.scenarios.length)
|
|
447
|
+
for (const unit of units)
|
|
448
|
+
incompleteUnits.add(unit.id);
|
|
449
|
+
await Promise.all(prepared.batches.map(async (batch) => {
|
|
450
|
+
const questions = Object.fromEntries(units.flatMap((unit) => batch.scenarios.map((scenario) => [
|
|
451
|
+
`${scenario.id}_${unit.id}`,
|
|
452
|
+
{
|
|
453
|
+
type: "bool",
|
|
454
|
+
instructions: `When test ${scenario.name} runs, does the after version of unit ${unit.id} (${unit.name}) execute or get read, directly or through functions it calls?`,
|
|
455
|
+
},
|
|
456
|
+
])));
|
|
457
|
+
if (!Object.keys(questions).length)
|
|
458
|
+
return;
|
|
459
|
+
const witnessQuestions = prepareCoverageWitnesses(candidate.state, {}, witnessUnits);
|
|
460
|
+
const realIds = Object.keys(questions);
|
|
461
|
+
const result = await judge(batch.state, { ...questions, ...witnessQuestions.questions }, witnessQuestions.witnesses.map((witness) => witness.id));
|
|
462
|
+
const health = evaluateBatchWitnessHealth(realIds, witnessQuestions.witnesses, result);
|
|
463
|
+
limitations.push(...health.failures);
|
|
464
|
+
for (const unit of units) {
|
|
465
|
+
if (!result.ok) {
|
|
466
|
+
fallback ||= !budget;
|
|
467
|
+
incompleteUnits.add(unit.id);
|
|
468
|
+
continue;
|
|
469
|
+
}
|
|
470
|
+
for (const scenario of batch.scenarios) {
|
|
471
|
+
const id = `${scenario.id}_${unit.id}`;
|
|
472
|
+
const answer = result.answers[id];
|
|
473
|
+
if (health.unhealthyQuestionIds.has(id)) {
|
|
474
|
+
incompleteUnits.add(unit.id);
|
|
475
|
+
continue;
|
|
476
|
+
}
|
|
477
|
+
if (answer?.type === "bool") {
|
|
478
|
+
coverage.set(unit.id, Math.max(coverage.get(unit.id) ?? 0, answer.p));
|
|
479
|
+
if (answer.p >= SELECT_MIN) {
|
|
480
|
+
const existing = selected.find((item) => item.entry === candidate.entry);
|
|
481
|
+
if (!existing)
|
|
482
|
+
selected.push({
|
|
483
|
+
entry: candidate.entry,
|
|
484
|
+
scenarioIds: [scenario.id],
|
|
485
|
+
});
|
|
486
|
+
else if (existing.scenarioIds !== null &&
|
|
487
|
+
!existing.scenarioIds.includes(scenario.id))
|
|
488
|
+
existing.scenarioIds = [
|
|
489
|
+
...existing.scenarioIds,
|
|
490
|
+
scenario.id,
|
|
491
|
+
];
|
|
492
|
+
answers.push({
|
|
493
|
+
label: `${candidate.entry.path}: ${scenario.name}`,
|
|
494
|
+
value: { head: `runs ${unit.name}`, p: answer.p },
|
|
495
|
+
band: "verdict",
|
|
496
|
+
});
|
|
497
|
+
}
|
|
498
|
+
}
|
|
499
|
+
else
|
|
500
|
+
incompleteUnits.add(unit.id);
|
|
501
|
+
}
|
|
502
|
+
}
|
|
503
|
+
}));
|
|
504
|
+
}));
|
|
505
|
+
const incomplete = discovery.limits.some((limit) => limit.kind === "types"
|
|
506
|
+
? !discovery.entries.some((entry) => entry.path === limit.path &&
|
|
507
|
+
entry.invocation?.kind === "typecheck")
|
|
508
|
+
: limit.kind !== "local_runner_unproven" &&
|
|
509
|
+
limit.kind !== "interactive_script_skipped") ||
|
|
510
|
+
inventory.limits.length > 0 ||
|
|
511
|
+
diff.limits.length > 0 ||
|
|
512
|
+
graph.limits.length > 0 ||
|
|
513
|
+
evidenceLimits.length > 0 ||
|
|
514
|
+
Boolean(args.paths?.length);
|
|
515
|
+
for (const unit of residual)
|
|
516
|
+
if ((coverage.get(unit.id) ?? 0) < SELECT_MIN)
|
|
517
|
+
answers.push({
|
|
518
|
+
label: `changed, run by no discovered test: ${unit.name} (${unit.file})`,
|
|
519
|
+
value: {
|
|
520
|
+
head: "within discovered inventory only",
|
|
521
|
+
p: coverage.get(unit.id) ?? 0,
|
|
522
|
+
},
|
|
523
|
+
band: incomplete || incompleteUnits.has(unit.id) ? "unsure" : "verdict",
|
|
524
|
+
});
|
|
525
|
+
if (fallback) {
|
|
526
|
+
selected.length = 0;
|
|
527
|
+
selected.push(...candidates.map((candidate) => ({
|
|
528
|
+
entry: candidate.entry,
|
|
529
|
+
scenarioIds: null,
|
|
530
|
+
})));
|
|
531
|
+
limitations.push({
|
|
532
|
+
fact: "fallback: all",
|
|
533
|
+
next: "run all discovered tests with their project runner",
|
|
534
|
+
});
|
|
535
|
+
}
|
|
536
|
+
const commands = buildRunnerCommands(selected);
|
|
537
|
+
limitations.push(...commands.limits.flatMap((limit) => limit.files.map((path) => ({
|
|
538
|
+
cause: limit.reason,
|
|
539
|
+
path,
|
|
540
|
+
fact: `${limit.reason}: ${path}`,
|
|
541
|
+
next: limit.action,
|
|
542
|
+
}))));
|
|
543
|
+
if (skipped)
|
|
544
|
+
limitations.push({
|
|
545
|
+
fact: `${skipped} tests skipped because their imports cannot reach the diff`,
|
|
546
|
+
next: "an alias or dynamic import would have kept a test in",
|
|
547
|
+
});
|
|
548
|
+
return finish({
|
|
549
|
+
answers,
|
|
550
|
+
unjudged,
|
|
551
|
+
limitations,
|
|
552
|
+
limitationPriorityPaths: [
|
|
553
|
+
...new Set([
|
|
554
|
+
...selected.map(({ entry }) => entry.path),
|
|
555
|
+
...diff.units.map((unit) => unit.file),
|
|
556
|
+
]),
|
|
557
|
+
],
|
|
558
|
+
...(budget ? { budget } : {}),
|
|
559
|
+
lines: commands.commands.map((command) => ({
|
|
560
|
+
type: "command",
|
|
561
|
+
...command,
|
|
562
|
+
cwd: relative(cwd, resolve(inventory.cwd, command.cwd)) || ".",
|
|
563
|
+
})),
|
|
564
|
+
});
|
|
565
|
+
},
|
|
566
|
+
};
|
|
567
|
+
}
|