supercov 0.0.43 → 0.0.44
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/analyzers/typescript/README.md +4 -0
- package/analyzers/typescript/bin/identity.mjs +4 -0
- package/analyzers/typescript/dist/analyze.js +1352 -51
- package/analyzers/typescript/dist/archive.js +31 -3
- package/analyzers/typescript/dist/awaited-observations.js +376 -0
- package/analyzers/typescript/dist/build-identity.json +1 -1
- package/analyzers/typescript/dist/mock-counts.js +2517 -0
- package/analyzers/typescript/dist/pragmas.js +59 -16
- package/analyzers/typescript/src/analyze.ts +1738 -96
- package/analyzers/typescript/src/archive.ts +36 -3
- package/analyzers/typescript/src/awaited-observations.ts +561 -0
- package/analyzers/typescript/src/mock-counts.ts +3219 -0
- package/analyzers/typescript/src/pragmas.ts +90 -24
- package/docs/agent-loop.md +116 -31
- package/docs/assertion-evidence.md +560 -1
- package/docs/cli.md +15 -15
- package/docs/code-verification.md +3 -181
- package/docs/coverage-model.md +6 -6
- package/docs/evidence.md +6 -6
- package/docs/getting-started.md +60 -72
- package/docs/performance.md +4 -4
- package/docs/troubleshooting.md +8 -8
- package/docs/verification.md +2 -2
- package/docs/workspace-isolation.md +1 -1
- package/package.json +11 -9
- package/runtime/javascript/nodeAssertAdapter.mjs +32 -8
- package/runtime/javascript/nodeTest.mjs +13 -5
- package/runtime/javascript/runnerEvidence.mjs +33 -11
- package/runtime/javascript/runtime.mjs +22 -3
|
@@ -1,5 +1,13 @@
|
|
|
1
1
|
import type ts from "typescript";
|
|
2
2
|
import type { Site } from "./types.js";
|
|
3
|
+
import type { AwaitedObservationSource } from "./awaited-observations.js";
|
|
4
|
+
import type {
|
|
5
|
+
CallOmissionEvidence,
|
|
6
|
+
CountSensitivityEvidence,
|
|
7
|
+
PayloadSensitivityEvidence,
|
|
8
|
+
DirectReturnSensitivityEvidence,
|
|
9
|
+
CompletionSensitivityEvidence,
|
|
10
|
+
} from "./mock-counts.js";
|
|
3
11
|
|
|
4
12
|
export type AssertionPhase = { source?: string; op: string; status?: string };
|
|
5
13
|
export function assertionWitnessIssue(
|
|
@@ -31,6 +39,7 @@ interface Attachment {
|
|
|
31
39
|
source: string;
|
|
32
40
|
method: string;
|
|
33
41
|
inert: boolean;
|
|
42
|
+
awaitedObservation?: AwaitedObservationSource;
|
|
34
43
|
}
|
|
35
44
|
interface Comment {
|
|
36
45
|
id: string;
|
|
@@ -40,6 +49,7 @@ interface Comment {
|
|
|
40
49
|
candidateSites: string[];
|
|
41
50
|
issue?: string;
|
|
42
51
|
attachments: Attachment[];
|
|
52
|
+
check?: "missing-call" | "count" | "value" | "completion";
|
|
43
53
|
}
|
|
44
54
|
export interface PragmaHint {
|
|
45
55
|
id: string;
|
|
@@ -53,6 +63,13 @@ export interface PragmaHint {
|
|
|
53
63
|
assertionMethod?: string;
|
|
54
64
|
witness: "passed" | "unavailable";
|
|
55
65
|
witnessIssue?: string;
|
|
66
|
+
awaitedObservation?: AwaitedObservationSource;
|
|
67
|
+
check?: "missing-call" | "count" | "value" | "completion";
|
|
68
|
+
callOmission?: CallOmissionEvidence;
|
|
69
|
+
countSensitivity?: CountSensitivityEvidence;
|
|
70
|
+
payloadSensitivity?: PayloadSensitivityEvidence;
|
|
71
|
+
directReturnSensitivity?: DirectReturnSensitivityEvidence;
|
|
72
|
+
completionSensitivity?: CompletionSensitivityEvidence;
|
|
56
73
|
}
|
|
57
74
|
|
|
58
75
|
/** Source hints are kept OUT of observations. A comment never adds a boundary. */
|
|
@@ -74,8 +91,21 @@ export function collectPragmas(
|
|
|
74
91
|
const raw = sf.text.slice(range.pos, range.end);
|
|
75
92
|
if (!/^\/\/\s*observes:/.test(raw) || comments.has(key(sf, range.pos)))
|
|
76
93
|
continue;
|
|
94
|
+
const parts = raw.split(/;\s*check\s+/);
|
|
95
|
+
const check =
|
|
96
|
+
parts.length === 2 && parts[1].trim() === "missing call"
|
|
97
|
+
? ("missing-call" as const)
|
|
98
|
+
: parts.length === 2 && parts[1].trim() === "count"
|
|
99
|
+
? ("count" as const)
|
|
100
|
+
: parts.length === 2 && parts[1].trim() === "value"
|
|
101
|
+
? ("value" as const)
|
|
102
|
+
: parts.length === 2 && parts[1].trim() === "completion"
|
|
103
|
+
? ("completion" as const)
|
|
104
|
+
: undefined;
|
|
105
|
+
const checkIssue =
|
|
106
|
+
parts.length > 1 && !check ? "unsupported-check-recipe" : undefined;
|
|
77
107
|
const parsed = /^\/\/\s*observes:\s*(\S+?)#(\S+)(?:\s+(.+?))?\s*$/.exec(
|
|
78
|
-
|
|
108
|
+
parts[0],
|
|
79
109
|
);
|
|
80
110
|
const suffix = parsed?.[3]?.split(/(?:^|\s+)via\s+/);
|
|
81
111
|
const target = parsed
|
|
@@ -92,31 +122,51 @@ export function collectPragmas(
|
|
|
92
122
|
target.file
|
|
93
123
|
.split("/")
|
|
94
124
|
.every((part) => part && part !== "." && part !== "..");
|
|
95
|
-
const
|
|
96
|
-
? sites
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
(
|
|
102
|
-
|
|
103
|
-
|
|
125
|
+
const matchingSites = validPath
|
|
126
|
+
? sites.filter(
|
|
127
|
+
(s) =>
|
|
128
|
+
s.file === target.file &&
|
|
129
|
+
// The recipe selects the emission, not a containing callback-return site.
|
|
130
|
+
(!check ||
|
|
131
|
+
(check === "missing-call"
|
|
132
|
+
? s.category === "log"
|
|
133
|
+
: check === "completion"
|
|
134
|
+
? s.category === "return" || s.category === "throw"
|
|
135
|
+
: s.kind === "decision" || s.category === "return")) &&
|
|
136
|
+
(s.owner === target.function || s.fn === target.function) &&
|
|
137
|
+
(!target.snippet || s.text.includes(target.snippet)),
|
|
138
|
+
)
|
|
104
139
|
: [];
|
|
140
|
+
// An exact whole-site selector is more specific than a containing return
|
|
141
|
+
// whose display text happens to include that expression. Equal sibling
|
|
142
|
+
// sites remain ambiguous. This only selects a node; it proves no behavior.
|
|
143
|
+
const exact =
|
|
144
|
+
check && target?.snippet
|
|
145
|
+
? matchingSites.filter(
|
|
146
|
+
(s) => s.text.trim() === target.snippet!.trim(),
|
|
147
|
+
)
|
|
148
|
+
: [];
|
|
149
|
+
const candidates = (exact.length ? exact : matchingSites).map(
|
|
150
|
+
(s) => s.id,
|
|
151
|
+
);
|
|
105
152
|
comments.set(key(sf, range.pos), {
|
|
106
153
|
id: location(sf, range.pos),
|
|
107
154
|
where: location(sf, range.pos),
|
|
108
155
|
raw,
|
|
109
156
|
target,
|
|
157
|
+
...(check ? { check } : {}),
|
|
110
158
|
candidateSites: candidates,
|
|
111
|
-
issue:
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
? "invalid-
|
|
115
|
-
: !
|
|
116
|
-
? "target-
|
|
117
|
-
: candidates.length
|
|
118
|
-
? "
|
|
119
|
-
:
|
|
159
|
+
issue:
|
|
160
|
+
checkIssue ??
|
|
161
|
+
(!target
|
|
162
|
+
? "invalid-syntax"
|
|
163
|
+
: !validPath
|
|
164
|
+
? "invalid-target-path"
|
|
165
|
+
: !candidates.length
|
|
166
|
+
? "target-not-in-inventory"
|
|
167
|
+
: candidates.length > 1
|
|
168
|
+
? "ambiguous-target"
|
|
169
|
+
: undefined),
|
|
120
170
|
attachments: [],
|
|
121
171
|
});
|
|
122
172
|
}
|
|
@@ -129,11 +179,23 @@ export function collectPragmas(
|
|
|
129
179
|
visit(sf);
|
|
130
180
|
}
|
|
131
181
|
return {
|
|
182
|
+
hasHint(node: ts.Node) {
|
|
183
|
+
let statement = node;
|
|
184
|
+
while (statement.parent && !compiler.isStatement(statement))
|
|
185
|
+
statement = statement.parent;
|
|
186
|
+
if (!compiler.isExpressionStatement(statement)) return false;
|
|
187
|
+
const sf = node.getSourceFile();
|
|
188
|
+
return (
|
|
189
|
+
compiler.getLeadingCommentRanges(sf.text, statement.getFullStart()) ??
|
|
190
|
+
[]
|
|
191
|
+
).some((range) => comments.has(key(sf, range.pos)));
|
|
192
|
+
},
|
|
132
193
|
register(
|
|
133
194
|
node: ts.CallExpression,
|
|
134
195
|
method: string,
|
|
135
196
|
testKey: string,
|
|
136
197
|
inert: boolean,
|
|
198
|
+
awaitedObservation?: AwaitedObservationSource,
|
|
137
199
|
) {
|
|
138
200
|
let statement: ts.Node = node;
|
|
139
201
|
while (statement.parent && !compiler.isStatement(statement))
|
|
@@ -152,6 +214,7 @@ export function collectPragmas(
|
|
|
152
214
|
source: location(sf, node.getStart(sf)),
|
|
153
215
|
method,
|
|
154
216
|
inert,
|
|
217
|
+
...(awaitedObservation ? { awaitedObservation } : {}),
|
|
155
218
|
};
|
|
156
219
|
if (
|
|
157
220
|
!comment.attachments.some(
|
|
@@ -193,11 +256,11 @@ export function collectPragmas(
|
|
|
193
256
|
},
|
|
194
257
|
];
|
|
195
258
|
return matched.map(({ a, b }) => {
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
259
|
+
// A source-checked poll is not an explicit assertion phase. Neither
|
|
260
|
+
// a matching phase name nor a passing test can supply its read receipt.
|
|
261
|
+
const witnessIssue = a.awaitedObservation
|
|
262
|
+
? "observation-capture-unavailable"
|
|
263
|
+
: assertionWitnessIssue(b.phases, a.source, a.method);
|
|
201
264
|
return {
|
|
202
265
|
...comment,
|
|
203
266
|
id: `${comment.id}@${b.id}`,
|
|
@@ -205,6 +268,9 @@ export function collectPragmas(
|
|
|
205
268
|
test: b.id,
|
|
206
269
|
assertionSource: a.source,
|
|
207
270
|
assertionMethod: a.method,
|
|
271
|
+
...(a.awaitedObservation
|
|
272
|
+
? { awaitedObservation: a.awaitedObservation }
|
|
273
|
+
: {}),
|
|
208
274
|
witness: witnessIssue
|
|
209
275
|
? ("unavailable" as const)
|
|
210
276
|
: ("passed" as const),
|
package/docs/agent-loop.md
CHANGED
|
@@ -1,41 +1,33 @@
|
|
|
1
1
|
# Agent workflow
|
|
2
2
|
|
|
3
|
-
Supercov
|
|
4
|
-
|
|
3
|
+
Use Supercov with your coding agent and the test suite you already have.
|
|
4
|
+
Supercov reports coverage and gaps. Your agent writes a test, reruns the suite,
|
|
5
|
+
and checks what improved.
|
|
5
6
|
|
|
6
|
-
|
|
7
|
-
run the suite → choose a gap → write one test → rerun → compare
|
|
8
|
-
↑ |
|
|
9
|
-
└────────────────────────────────────────────────────────────┘
|
|
10
|
-
```
|
|
11
|
-
|
|
12
|
-
Supercov supplies the coverage signal and evidence. Your coding agent writes
|
|
13
|
-
the tests.
|
|
7
|
+
## Start with one test
|
|
14
8
|
|
|
15
|
-
|
|
9
|
+
Open your own repository in your coding agent and paste this prompt. You don't
|
|
10
|
+
need to install Supercov first; the agent can handle that.
|
|
16
11
|
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
on coverage. Only edit tests. Rerun the complete suite and report what improved.
|
|
12
|
+
```text supercov-prompt
|
|
13
|
+
Measure code coverage with npx supercov and write one missing test.
|
|
14
|
+
Only change tests. Rerun the full test suite and show me the test you
|
|
15
|
+
added and the before-and-after coverage.
|
|
22
16
|
```
|
|
23
17
|
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
Use `npx supercov` to improve coverage. Only write tests. Keep going while
|
|
28
|
-
useful gaps remain. Never weaken assertions or change application code to make
|
|
29
|
-
coverage easier. Stop at a measurement limit, unreachable behavior, or the end
|
|
30
|
-
of the available time budget. Report the run ids compared and what improved.
|
|
31
|
-
```
|
|
18
|
+
If the project has several test commands, tell the agent which full suite to
|
|
19
|
+
use. Let it run the commands and edit the tests, approving those actions if
|
|
20
|
+
your agent asks.
|
|
32
21
|
|
|
33
|
-
The
|
|
34
|
-
|
|
22
|
+
The result is a normal test-file change and a coverage comparison in the
|
|
23
|
+
conversation. Ask separately if you want a commit or pull request.
|
|
35
24
|
|
|
36
25
|
## One safe pass
|
|
37
26
|
|
|
38
|
-
|
|
27
|
+
The agent should run the suite, inspect a gap, write a test, then rerun the
|
|
28
|
+
same suite and compare. These are the commands it can use:
|
|
29
|
+
|
|
30
|
+
```sh supercov-example
|
|
39
31
|
# 1. Establish a baseline.
|
|
40
32
|
npx supercov -- npm test
|
|
41
33
|
|
|
@@ -52,8 +44,10 @@ npx supercov -- npm test
|
|
|
52
44
|
npx supercov diff <previous-run-id> latest
|
|
53
45
|
```
|
|
54
46
|
|
|
55
|
-
|
|
56
|
-
and
|
|
47
|
+
Everything after `--` is your project's test command. Use your actual command
|
|
48
|
+
and file paths in place of the examples. For example,
|
|
49
|
+
Rust projects can use `cargo test`, Python projects `pytest`, and Ruby projects
|
|
50
|
+
`bundle exec rspec`. Keep the baseline and verification commands identical.
|
|
57
51
|
|
|
58
52
|
The `line` query is useful before writing a test because it shows which tests
|
|
59
53
|
already reach that line. Extending a nearby test is often better than adding a
|
|
@@ -72,9 +66,100 @@ test gaps separate from analysis limits. Follow the returned evidence pointers
|
|
|
72
66
|
and `pagination.nextOffset`, pinning `--analysis` and the run id while paging.
|
|
73
67
|
See [assertion evidence](assertion-evidence.md) for requirements and examples.
|
|
74
68
|
|
|
75
|
-
##
|
|
69
|
+
## Example
|
|
70
|
+
|
|
71
|
+
Here's a recorded Codex run in a JavaScript project, using the first prompt.
|
|
72
|
+
The files are from our checkout example; you don't need to add them to your
|
|
73
|
+
project.
|
|
74
|
+
|
|
75
|
+
[`src/session.js`](https://github.com/supercorp-ai/supercov/blob/main/examples/checkout-verification/starter/src/session.js)
|
|
76
|
+
allows checkout only when the customer is signed in and their session has
|
|
77
|
+
not expired:
|
|
78
|
+
|
|
79
|
+
```js
|
|
80
|
+
export function canCheckout(signedIn, expired) {
|
|
81
|
+
if (signedIn && !expired) return true;
|
|
82
|
+
return false;
|
|
83
|
+
}
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
The two tests in
|
|
87
|
+
[`tests/session.test.js`](https://github.com/supercorp-ai/supercov/blob/main/examples/checkout-verification/starter/tests/session.test.js)
|
|
88
|
+
check a valid session and a signed-out visitor:
|
|
89
|
+
|
|
90
|
+
```js
|
|
91
|
+
assert.equal(canCheckout(true, false), true);
|
|
92
|
+
assert.equal(canCheckout(false, false), false);
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
The agent ran `npx supercov -- npm test`. Both tests passed, and the summary
|
|
96
|
+
from `npx supercov runs latest` showed:
|
|
76
97
|
|
|
77
98
|
```text
|
|
99
|
+
Coverage
|
|
100
|
+
Lines 100.00% (3/3)
|
|
101
|
+
Branches 100.00% (2/2)
|
|
102
|
+
MC/DC 50.00% (1/2)
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
It listed the gaps and inspected the file. You can open those views with:
|
|
106
|
+
|
|
107
|
+
```sh
|
|
108
|
+
npx supercov runs latest gaps
|
|
109
|
+
npx supercov runs latest file src/session.js
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
The file query explained the gap:
|
|
113
|
+
|
|
114
|
+
```text
|
|
115
|
+
LINE STATUS SOURCE
|
|
116
|
+
2 PARTIAL signedIn && !expired
|
|
117
|
+
Unobserved: no witness pair shows `!expired` independently changing the decision result
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
Both return paths had run, but neither test checked an expired session. MC/DC
|
|
121
|
+
checks whether each condition has been shown to affect the decision
|
|
122
|
+
independently. Here, `signedIn` had; `!expired` had not.
|
|
123
|
+
|
|
124
|
+
The agent added one test to `tests/session.test.js`, leaving the application
|
|
125
|
+
code and existing tests unchanged:
|
|
126
|
+
|
|
127
|
+
```js
|
|
128
|
+
test('a signed-in visitor with an expired session cannot check out', () => {
|
|
129
|
+
assert.equal(canCheckout(true, true), false);
|
|
130
|
+
});
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
It reran the same full suite. All three tests passed, and MC/DC reached 100%.
|
|
134
|
+
The comparison from `npx supercov diff <before-run-id> latest` showed:
|
|
135
|
+
|
|
136
|
+
```text
|
|
137
|
+
lines +0pp, branches +0pp, MC/DC +50pp
|
|
138
|
+
gained: 0 lines, 0 branches, 1 MC/DC conditions
|
|
139
|
+
lost: 0 lines, 0 branches, 0 MC/DC conditions
|
|
140
|
+
+ MC/DC src/session.js:2 C2 !expired
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
The new assertion checks that checkout is denied when a signed-in customer's
|
|
144
|
+
session has expired. Removing the expiry check makes this test fail; the
|
|
145
|
+
original two tests still pass.
|
|
146
|
+
|
|
147
|
+
In your project, look for the same evidence: the test checks the behavior the
|
|
148
|
+
agent identified, and the full suite passes. One useful test won't necessarily
|
|
149
|
+
take coverage to 100%.
|
|
150
|
+
|
|
151
|
+
To try these exact files, [download the starter](https://supercov.com/downloads/supercov-tutorial.zip),
|
|
152
|
+
extract it, open the `supercov-tutorial` folder in your agent, and run `npm ci`.
|
|
153
|
+
Then use the JavaScript prompt above. The completed test is not included in the
|
|
154
|
+
download. The [recorded run](https://github.com/supercorp-ai/supercov/tree/main/examples/checkout-verification/agent-run)
|
|
155
|
+
includes the commands, full output, and completed test.
|
|
156
|
+
|
|
157
|
+
## A complete prompt for longer runs
|
|
158
|
+
|
|
159
|
+
Once you've reviewed the first test, use this prompt to continue through
|
|
160
|
+
useful gaps—for example, during an overnight run:
|
|
161
|
+
|
|
162
|
+
```text supercov-prompt
|
|
78
163
|
Use `npx supercov` to improve coverage. Only write tests. Keep going while
|
|
79
164
|
useful gaps remain.
|
|
80
165
|
|
|
@@ -110,7 +195,7 @@ reason to manufacture a test.
|
|
|
110
195
|
|
|
111
196
|
If the repository separates test levels, narrow the view:
|
|
112
197
|
|
|
113
|
-
```sh
|
|
198
|
+
```sh supercov
|
|
114
199
|
npx supercov runs latest gaps --kind e2e --limit 10
|
|
115
200
|
```
|
|
116
201
|
|