supercov 0.0.43 → 0.0.44

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,13 @@
1
1
  import type ts from "typescript";
2
2
  import type { Site } from "./types.js";
3
+ import type { AwaitedObservationSource } from "./awaited-observations.js";
4
+ import type {
5
+ CallOmissionEvidence,
6
+ CountSensitivityEvidence,
7
+ PayloadSensitivityEvidence,
8
+ DirectReturnSensitivityEvidence,
9
+ CompletionSensitivityEvidence,
10
+ } from "./mock-counts.js";
3
11
 
4
12
  export type AssertionPhase = { source?: string; op: string; status?: string };
5
13
  export function assertionWitnessIssue(
@@ -31,6 +39,7 @@ interface Attachment {
31
39
  source: string;
32
40
  method: string;
33
41
  inert: boolean;
42
+ awaitedObservation?: AwaitedObservationSource;
34
43
  }
35
44
  interface Comment {
36
45
  id: string;
@@ -40,6 +49,7 @@ interface Comment {
40
49
  candidateSites: string[];
41
50
  issue?: string;
42
51
  attachments: Attachment[];
52
+ check?: "missing-call" | "count" | "value" | "completion";
43
53
  }
44
54
  export interface PragmaHint {
45
55
  id: string;
@@ -53,6 +63,13 @@ export interface PragmaHint {
53
63
  assertionMethod?: string;
54
64
  witness: "passed" | "unavailable";
55
65
  witnessIssue?: string;
66
+ awaitedObservation?: AwaitedObservationSource;
67
+ check?: "missing-call" | "count" | "value" | "completion";
68
+ callOmission?: CallOmissionEvidence;
69
+ countSensitivity?: CountSensitivityEvidence;
70
+ payloadSensitivity?: PayloadSensitivityEvidence;
71
+ directReturnSensitivity?: DirectReturnSensitivityEvidence;
72
+ completionSensitivity?: CompletionSensitivityEvidence;
56
73
  }
57
74
 
58
75
  /** Source hints are kept OUT of observations. A comment never adds a boundary. */
@@ -74,8 +91,21 @@ export function collectPragmas(
74
91
  const raw = sf.text.slice(range.pos, range.end);
75
92
  if (!/^\/\/\s*observes:/.test(raw) || comments.has(key(sf, range.pos)))
76
93
  continue;
94
+ const parts = raw.split(/;\s*check\s+/);
95
+ const check =
96
+ parts.length === 2 && parts[1].trim() === "missing call"
97
+ ? ("missing-call" as const)
98
+ : parts.length === 2 && parts[1].trim() === "count"
99
+ ? ("count" as const)
100
+ : parts.length === 2 && parts[1].trim() === "value"
101
+ ? ("value" as const)
102
+ : parts.length === 2 && parts[1].trim() === "completion"
103
+ ? ("completion" as const)
104
+ : undefined;
105
+ const checkIssue =
106
+ parts.length > 1 && !check ? "unsupported-check-recipe" : undefined;
77
107
  const parsed = /^\/\/\s*observes:\s*(\S+?)#(\S+)(?:\s+(.+?))?\s*$/.exec(
78
- raw,
108
+ parts[0],
79
109
  );
80
110
  const suffix = parsed?.[3]?.split(/(?:^|\s+)via\s+/);
81
111
  const target = parsed
@@ -92,31 +122,51 @@ export function collectPragmas(
92
122
  target.file
93
123
  .split("/")
94
124
  .every((part) => part && part !== "." && part !== "..");
95
- const candidates = validPath
96
- ? sites
97
- .filter(
98
- (s) =>
99
- s.file === target.file &&
100
- (s.owner === target.function || s.fn === target.function) &&
101
- (!target.snippet || s.text.includes(target.snippet)),
102
- )
103
- .map((s) => s.id)
125
+ const matchingSites = validPath
126
+ ? sites.filter(
127
+ (s) =>
128
+ s.file === target.file &&
129
+ // The recipe selects the emission, not a containing callback-return site.
130
+ (!check ||
131
+ (check === "missing-call"
132
+ ? s.category === "log"
133
+ : check === "completion"
134
+ ? s.category === "return" || s.category === "throw"
135
+ : s.kind === "decision" || s.category === "return")) &&
136
+ (s.owner === target.function || s.fn === target.function) &&
137
+ (!target.snippet || s.text.includes(target.snippet)),
138
+ )
104
139
  : [];
140
+ // An exact whole-site selector is more specific than a containing return
141
+ // whose display text happens to include that expression. Equal sibling
142
+ // sites remain ambiguous. This only selects a node; it proves no behavior.
143
+ const exact =
144
+ check && target?.snippet
145
+ ? matchingSites.filter(
146
+ (s) => s.text.trim() === target.snippet!.trim(),
147
+ )
148
+ : [];
149
+ const candidates = (exact.length ? exact : matchingSites).map(
150
+ (s) => s.id,
151
+ );
105
152
  comments.set(key(sf, range.pos), {
106
153
  id: location(sf, range.pos),
107
154
  where: location(sf, range.pos),
108
155
  raw,
109
156
  target,
157
+ ...(check ? { check } : {}),
110
158
  candidateSites: candidates,
111
- issue: !target
112
- ? "invalid-syntax"
113
- : !validPath
114
- ? "invalid-target-path"
115
- : !candidates.length
116
- ? "target-not-in-inventory"
117
- : candidates.length > 1
118
- ? "ambiguous-target"
119
- : undefined,
159
+ issue:
160
+ checkIssue ??
161
+ (!target
162
+ ? "invalid-syntax"
163
+ : !validPath
164
+ ? "invalid-target-path"
165
+ : !candidates.length
166
+ ? "target-not-in-inventory"
167
+ : candidates.length > 1
168
+ ? "ambiguous-target"
169
+ : undefined),
120
170
  attachments: [],
121
171
  });
122
172
  }
@@ -129,11 +179,23 @@ export function collectPragmas(
129
179
  visit(sf);
130
180
  }
131
181
  return {
182
+ hasHint(node: ts.Node) {
183
+ let statement = node;
184
+ while (statement.parent && !compiler.isStatement(statement))
185
+ statement = statement.parent;
186
+ if (!compiler.isExpressionStatement(statement)) return false;
187
+ const sf = node.getSourceFile();
188
+ return (
189
+ compiler.getLeadingCommentRanges(sf.text, statement.getFullStart()) ??
190
+ []
191
+ ).some((range) => comments.has(key(sf, range.pos)));
192
+ },
132
193
  register(
133
194
  node: ts.CallExpression,
134
195
  method: string,
135
196
  testKey: string,
136
197
  inert: boolean,
198
+ awaitedObservation?: AwaitedObservationSource,
137
199
  ) {
138
200
  let statement: ts.Node = node;
139
201
  while (statement.parent && !compiler.isStatement(statement))
@@ -152,6 +214,7 @@ export function collectPragmas(
152
214
  source: location(sf, node.getStart(sf)),
153
215
  method,
154
216
  inert,
217
+ ...(awaitedObservation ? { awaitedObservation } : {}),
155
218
  };
156
219
  if (
157
220
  !comment.attachments.some(
@@ -193,11 +256,11 @@ export function collectPragmas(
193
256
  },
194
257
  ];
195
258
  return matched.map(({ a, b }) => {
196
- const witnessIssue = assertionWitnessIssue(
197
- b.phases,
198
- a.source,
199
- a.method,
200
- );
259
+ // A source-checked poll is not an explicit assertion phase. Neither
260
+ // a matching phase name nor a passing test can supply its read receipt.
261
+ const witnessIssue = a.awaitedObservation
262
+ ? "observation-capture-unavailable"
263
+ : assertionWitnessIssue(b.phases, a.source, a.method);
201
264
  return {
202
265
  ...comment,
203
266
  id: `${comment.id}@${b.id}`,
@@ -205,6 +268,9 @@ export function collectPragmas(
205
268
  test: b.id,
206
269
  assertionSource: a.source,
207
270
  assertionMethod: a.method,
271
+ ...(a.awaitedObservation
272
+ ? { awaitedObservation: a.awaitedObservation }
273
+ : {}),
208
274
  witness: witnessIssue
209
275
  ? ("unavailable" as const)
210
276
  : ("passed" as const),
@@ -1,41 +1,33 @@
1
1
  # Agent workflow
2
2
 
3
- Supercov works best as a small, repeatable loop: run the suite, choose one useful
4
- gap, write one test, rerun, and prove what improved.
3
+ Use Supercov with your coding agent and the test suite you already have.
4
+ Supercov reports coverage and gaps. Your agent writes a test, reruns the suite,
5
+ and checks what improved.
5
6
 
6
- ```text
7
- run the suite → choose a gap → write one test → rerun → compare
8
- ↑ |
9
- └────────────────────────────────────────────────────────────┘
10
- ```
11
-
12
- Supercov supplies the coverage signal and evidence. Your coding agent writes
13
- the tests.
7
+ ## Start with one test
14
8
 
15
- ## Choose the job
9
+ Open your own repository in your coding agent and paste this prompt. You don't
10
+ need to install Supercov first; the agent can handle that.
16
11
 
17
- For one careful first pass, ask:
18
-
19
- ```text
20
- Measure code coverage with `npx supercov` and write the first useful test based
21
- on coverage. Only edit tests. Rerun the complete suite and report what improved.
12
+ ```text supercov-prompt
13
+ Measure code coverage with npx supercov and write one missing test.
14
+ Only change tests. Rerun the full test suite and show me the test you
15
+ added and the before-and-after coverage.
22
16
  ```
23
17
 
24
- For an overnight run or leftover token budget, ask:
25
-
26
- ```text
27
- Use `npx supercov` to improve coverage. Only write tests. Keep going while
28
- useful gaps remain. Never weaken assertions or change application code to make
29
- coverage easier. Stop at a measurement limit, unreachable behavior, or the end
30
- of the available time budget. Report the run ids compared and what improved.
31
- ```
18
+ If the project has several test commands, tell the agent which full suite to
19
+ use. Let it run the commands and edit the tests, approving those actions if
20
+ your agent asks.
32
21
 
33
- The second prompt is intentionally open-ended, but 100% is a direction rather
34
- than permission to write meaningless tests or reshape application code.
22
+ The result is a normal test-file change and a coverage comparison in the
23
+ conversation. Ask separately if you want a commit or pull request.
35
24
 
36
25
  ## One safe pass
37
26
 
38
- ```sh
27
+ The agent should run the suite, inspect a gap, write a test, then rerun the
28
+ same suite and compare. These are the commands it can use:
29
+
30
+ ```sh supercov-example
39
31
  # 1. Establish a baseline.
40
32
  npx supercov -- npm test
41
33
 
@@ -52,8 +44,10 @@ npx supercov -- npm test
52
44
  npx supercov diff <previous-run-id> latest
53
45
  ```
54
46
 
55
- For Rust, use `cargo test` or `cargo nextest run` in both runs. Keep the baseline
56
- and verification commands identical.
47
+ Everything after `--` is your project's test command. Use your actual command
48
+ and file paths in place of the examples. For example,
49
+ Rust projects can use `cargo test`, Python projects `pytest`, and Ruby projects
50
+ `bundle exec rspec`. Keep the baseline and verification commands identical.
57
51
 
58
52
  The `line` query is useful before writing a test because it shows which tests
59
53
  already reach that line. Extending a nearby test is often better than adding a
@@ -72,9 +66,100 @@ test gaps separate from analysis limits. Follow the returned evidence pointers
72
66
  and `pagination.nextOffset`, pinning `--analysis` and the run id while paging.
73
67
  See [assertion evidence](assertion-evidence.md) for requirements and examples.
74
68
 
75
- ## A complete prompt for longer runs
69
+ ## Example
70
+
71
+ Here's a recorded Codex run in a JavaScript project, using the first prompt.
72
+ The files are from our checkout example; you don't need to add them to your
73
+ project.
74
+
75
+ [`src/session.js`](https://github.com/supercorp-ai/supercov/blob/main/examples/checkout-verification/starter/src/session.js)
76
+ allows checkout only when the customer is signed in and their session has
77
+ not expired:
78
+
79
+ ```js
80
+ export function canCheckout(signedIn, expired) {
81
+ if (signedIn && !expired) return true;
82
+ return false;
83
+ }
84
+ ```
85
+
86
+ The two tests in
87
+ [`tests/session.test.js`](https://github.com/supercorp-ai/supercov/blob/main/examples/checkout-verification/starter/tests/session.test.js)
88
+ check a valid session and a signed-out visitor:
89
+
90
+ ```js
91
+ assert.equal(canCheckout(true, false), true);
92
+ assert.equal(canCheckout(false, false), false);
93
+ ```
94
+
95
+ The agent ran `npx supercov -- npm test`. Both tests passed, and the summary
96
+ from `npx supercov runs latest` showed:
76
97
 
77
98
  ```text
99
+ Coverage
100
+ Lines 100.00% (3/3)
101
+ Branches 100.00% (2/2)
102
+ MC/DC 50.00% (1/2)
103
+ ```
104
+
105
+ It listed the gaps and inspected the file. You can open those views with:
106
+
107
+ ```sh
108
+ npx supercov runs latest gaps
109
+ npx supercov runs latest file src/session.js
110
+ ```
111
+
112
+ The file query explained the gap:
113
+
114
+ ```text
115
+ LINE STATUS SOURCE
116
+ 2 PARTIAL signedIn && !expired
117
+ Unobserved: no witness pair shows `!expired` independently changing the decision result
118
+ ```
119
+
120
+ Both return paths had run, but neither test checked an expired session. MC/DC
121
+ checks whether each condition has been shown to affect the decision
122
+ independently. Here, `signedIn` had; `!expired` had not.
123
+
124
+ The agent added one test to `tests/session.test.js`, leaving the application
125
+ code and existing tests unchanged:
126
+
127
+ ```js
128
+ test('a signed-in visitor with an expired session cannot check out', () => {
129
+ assert.equal(canCheckout(true, true), false);
130
+ });
131
+ ```
132
+
133
+ It reran the same full suite. All three tests passed, and MC/DC reached 100%.
134
+ The comparison from `npx supercov diff <before-run-id> latest` showed:
135
+
136
+ ```text
137
+ lines +0pp, branches +0pp, MC/DC +50pp
138
+ gained: 0 lines, 0 branches, 1 MC/DC conditions
139
+ lost: 0 lines, 0 branches, 0 MC/DC conditions
140
+ + MC/DC src/session.js:2 C2 !expired
141
+ ```
142
+
143
+ The new assertion checks that checkout is denied when a signed-in customer's
144
+ session has expired. Removing the expiry check makes this test fail; the
145
+ original two tests still pass.
146
+
147
+ In your project, look for the same evidence: the test checks the behavior the
148
+ agent identified, and the full suite passes. One useful test won't necessarily
149
+ take coverage to 100%.
150
+
151
+ To try these exact files, [download the starter](https://supercov.com/downloads/supercov-tutorial.zip),
152
+ extract it, open the `supercov-tutorial` folder in your agent, and run `npm ci`.
153
+ Then use the JavaScript prompt above. The completed test is not included in the
154
+ download. The [recorded run](https://github.com/supercorp-ai/supercov/tree/main/examples/checkout-verification/agent-run)
155
+ includes the commands, full output, and completed test.
156
+
157
+ ## A complete prompt for longer runs
158
+
159
+ Once you've reviewed the first test, use this prompt to continue through
160
+ useful gaps—for example, during an overnight run:
161
+
162
+ ```text supercov-prompt
78
163
  Use `npx supercov` to improve coverage. Only write tests. Keep going while
79
164
  useful gaps remain.
80
165
 
@@ -110,7 +195,7 @@ reason to manufacture a test.
110
195
 
111
196
  If the repository separates test levels, narrow the view:
112
197
 
113
- ```sh
198
+ ```sh supercov
114
199
  npx supercov runs latest gaps --kind e2e --limit 10
115
200
  ```
116
201