minitest-impact 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE.txt +21 -0
- data/README.md +205 -0
- data/eval/ci_failures.rb +72 -0
- data/exe/minitest-impact +7 -0
- data/lib/minitest/impact/cli.rb +190 -0
- data/lib/minitest/impact/diff.rb +47 -0
- data/lib/minitest/impact/evaluation.rb +80 -0
- data/lib/minitest/impact/jev/client.rb +88 -0
- data/lib/minitest/impact/jev/questions.rb +58 -0
- data/lib/minitest/impact/jev/ranker.rb +121 -0
- data/lib/minitest/impact/locale_keys.rb +39 -0
- data/lib/minitest/impact/map.rb +115 -0
- data/lib/minitest/impact/methods.rb +61 -0
- data/lib/minitest/impact/railtie.rb +13 -0
- data/lib/minitest/impact/rake_task.rb +30 -0
- data/lib/minitest/impact/record_bootstrap.rb +19 -0
- data/lib/minitest/impact/recorder.rb +104 -0
- data/lib/minitest/impact/recording_hooks.rb +21 -0
- data/lib/minitest/impact/repo.rb +80 -0
- data/lib/minitest/impact/rules.rb +127 -0
- data/lib/minitest/impact/selector.rb +337 -0
- data/lib/minitest/impact/version.rb +7 -0
- data/lib/minitest/impact.rb +22 -0
- metadata +71 -0
checksums.yaml
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
---
|
|
2
|
+
SHA256:
|
|
3
|
+
metadata.gz: 729696ea279a7dac126254382c8761f9e26ae5972e0047c1961b5c0b57e6037c
|
|
4
|
+
data.tar.gz: d9abac4812e8a11862e87bc4c69c6b6b96f368c3b8b269b2f643c9547479557b
|
|
5
|
+
SHA512:
|
|
6
|
+
metadata.gz: 5b5cb7ba14512aead7ede47003cd271ccbbe749f9c0880666b0cb372d2a6f69f5a28aed4ca96e6c998b7ee30e4b12df0e6a1e7be8b32896f9d541fe150fb23a6
|
|
7
|
+
data.tar.gz: 0d83d45f8d06b7762b375e7428ea63b85f20189dce5bcf2d9d4bb08039995c165eda2855d323710d5b09809fc953c72c724734041ed222d02975939c6348ff3f
|
data/LICENSE.txt
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
The MIT License (MIT)
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Bruno Costanzo
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in
|
|
13
|
+
all copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
|
21
|
+
THE SOFTWARE.
|
data/README.md
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
# minitest-impact
|
|
2
|
+
|
|
3
|
+
Pick the Minitest test files a change is most likely to break, or that were written to verify it,
|
|
4
|
+
so a coding agent runs a handful of tests inside its loop instead of the whole suite. The full
|
|
5
|
+
suite stays the final gate: this gem decides what to run *while working*, never what is allowed
|
|
6
|
+
to ship.
|
|
7
|
+
|
|
8
|
+
```console
|
|
9
|
+
$ minitest-impact select --since main
|
|
10
|
+
Confidence: medium
|
|
11
|
+
exact app/models/invoice.rb
|
|
12
|
+
convention config/locales/es.yml (keys: invoices.show.title)
|
|
13
|
+
4 test files, most likely first:
|
|
14
|
+
1.00 test/models/invoice_test.rb - runs Invoice#total; is the test for app/models/invoice.rb
|
|
15
|
+
0.95 test/controllers/invoices_controller_test.rb - uses the key invoices.show.title
|
|
16
|
+
...
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
It works in two layers:
|
|
20
|
+
|
|
21
|
+
1. **A coverage map, exact and free.** One recording run of your suite notes which project lines
|
|
22
|
+
every test file executed (Ruby's `Coverage`, with `eval: true` so ERB views count too). Given a
|
|
23
|
+
diff, a changed line is traced to the method around it, the method is found by name in the
|
|
24
|
+
recorded version of the file (so moved lines still match), and the tests that ran that method
|
|
25
|
+
are selected. A change outside any method (a constant, a `validates`, a `has_many`) selects
|
|
26
|
+
every test that loaded the file, weighted down when the file is loaded by nearly everything.
|
|
27
|
+
2. **Rules for what the map cannot see**, all in `lib/minitest/impact/rules.rb`: new files (the
|
|
28
|
+
conventional test path, and tests that mention the new constant), locale keys (tests that use
|
|
29
|
+
the key, views that render it), routes (their controllers), migrations and `schema.rb` (the
|
|
30
|
+
models of the changed tables), fixtures, Stimulus controllers (the views that use them), files a
|
|
31
|
+
test reads by path, and files that need the whole suite (`Gemfile.lock`, `test_helper.rb`, boot
|
|
32
|
+
configuration).
|
|
33
|
+
3. **Jev, optionally**, when the map and the rules are not enough: some file could not be traced,
|
|
34
|
+
or the selection is too large to run in a loop. See [Jev](#jev) below.
|
|
35
|
+
|
|
36
|
+
The output is an ordered list of test files, each with its reasons, and an overall confidence:
|
|
37
|
+
|
|
38
|
+
- **high**: every changed file was traced exactly: a changed test, a changed method the map saw
|
|
39
|
+
run, or a file no test reads.
|
|
40
|
+
- **medium**: some file was traced by the whole file or by convention.
|
|
41
|
+
- **low**: some file could not be traced, or the change needs the whole suite. Run the suite.
|
|
42
|
+
|
|
43
|
+
## Install
|
|
44
|
+
|
|
45
|
+
```ruby
|
|
46
|
+
# Gemfile
|
|
47
|
+
group :development, :test do
|
|
48
|
+
gem "minitest-impact"
|
|
49
|
+
end
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Ruby 3.3 or newer (it uses the Prism parser from the standard library). No runtime dependencies.
|
|
53
|
+
|
|
54
|
+
## Record the map
|
|
55
|
+
|
|
56
|
+
```console
|
|
57
|
+
$ minitest-impact record -- bin/rails test
|
|
58
|
+
$ minitest-impact record -- sh -c 'bin/rails test; bin/rails test:system'
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
`record` runs the command with `RUBYOPT` loading a small recorder before your app boots, so
|
|
62
|
+
nothing changes in your test helper. Each test process (Rails' forked parallel workers included)
|
|
63
|
+
appends one JSON line per test to its own file; when the command ends they are merged into
|
|
64
|
+
`tmp/minitest-impact/map.json` (`--map PATH` to change it), stamped with the commit it describes.
|
|
65
|
+
|
|
66
|
+
- **Turn SimpleCov off while recording** (`SimpleCov.start unless ENV["MINITEST_IMPACT_RECORD"]`,
|
|
67
|
+
or your app's own switch). Ruby allows one coverage setup per process.
|
|
68
|
+
- Recording is slower than a normal run (about 2 to 3 times on a Rails app), because every test
|
|
69
|
+
reads and clears the coverage counters. Record on a quiet machine, or in CI, and refresh the map
|
|
70
|
+
when it drifts: a map a few hundred commits old still works, because methods are matched by name.
|
|
71
|
+
- Record from a clean working tree; the map is stamped with `HEAD`.
|
|
72
|
+
|
|
73
|
+
## Select and run
|
|
74
|
+
|
|
75
|
+
```console
|
|
76
|
+
$ minitest-impact select # the uncommitted change
|
|
77
|
+
$ minitest-impact select --since main # everything since main
|
|
78
|
+
$ minitest-impact select --since main --format json
|
|
79
|
+
$ bin/rails test $(minitest-impact select --since main --format paths)
|
|
80
|
+
$ minitest-impact run --since main # select, then bin/rails test the selection
|
|
81
|
+
$ bin/rails test:impact SINCE=main # the same, as a Rake task (added by a Railtie)
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
`--format paths` exits with status 10, and prints nothing, when the change needs the whole suite.
|
|
85
|
+
`run` and `test:impact` run the whole suite themselves in that case. `--max N` keeps the N most
|
|
86
|
+
likely files.
|
|
87
|
+
|
|
88
|
+
### For coding agents
|
|
89
|
+
|
|
90
|
+
Put this in the agent's instructions (`CLAUDE.md`, `AGENTS.md`):
|
|
91
|
+
|
|
92
|
+
> While you work, run `bin/rails test:impact SINCE=main` instead of the whole suite. When it says
|
|
93
|
+
> "Confidence: low", or when you are done, the full gate runs; you do not run it yourself.
|
|
94
|
+
|
|
95
|
+
and run the full gate as code after the agent says it is done, feeding back only failures.
|
|
96
|
+
|
|
97
|
+
## Jev
|
|
98
|
+
|
|
99
|
+
[Jev](https://docs.typesafe.ai) is TypeSafe's System One model: a fast, cheap classifier that
|
|
100
|
+
answers typed questions (yes/no, choice, score) about a state, with calibrated probabilities. It
|
|
101
|
+
is used here the way TypeSafe's own guidance says to use it:
|
|
102
|
+
|
|
103
|
+
- **Rules stay in code.** Jev never decides what a test file is, which files need the whole suite,
|
|
104
|
+
or anything else a path can tell.
|
|
105
|
+
- **One narrow question per judgment, all in one request.** The state is the change (paths, a
|
|
106
|
+
trimmed diff, your `--intent`) and up to 48 candidate test files with their test names. The
|
|
107
|
+
questions: one yes/no per candidate ("do these tests call, render or assert on something the
|
|
108
|
+
change modifies?"), one choice of the candidate most directly written for the change, **with a
|
|
109
|
+
"none" option**, and one yes/no for "does every test depend on this?".
|
|
110
|
+
- **Fast search first, then re-rank**, as in TypeSafe's re-ranking cookbook: the candidates are
|
|
111
|
+
the map's selection plus test files whose paths and test names share words with the change.
|
|
112
|
+
- **Exact map hits are never dropped.** Jev can add tests, drop weak non-exact ones and reorder,
|
|
113
|
+
but a test the map saw run the changed method stays.
|
|
114
|
+
- **Thresholds live in one file** (`lib/minitest/impact/jev/questions.rb`) and **the model version
|
|
115
|
+
is pinned** (`jev-1.13.0`), because a threshold tuned on one version does not carry over.
|
|
116
|
+
|
|
117
|
+
Set `TYPESAFE_API_KEY` to turn it on (`TYPESAFE_BASE_URL` for another endpoint,
|
|
118
|
+
`MINITEST_IMPACT_JEV_MODEL` to move the pin). Without a key, or with `--no-jev`, everything runs
|
|
119
|
+
offline on the map and the rules. If Jev fails, the map's selection is used and the error is
|
|
120
|
+
reported; the agent's loop never breaks on it.
|
|
121
|
+
|
|
122
|
+
Cost: Jev bills input tokens only, $0.042 per million (docs.typesafe.ai/models). A request with
|
|
123
|
+
48 candidates and a 12,000-character diff is under 10,000 tokens: about $0.0004.
|
|
124
|
+
|
|
125
|
+
### Tuning
|
|
126
|
+
|
|
127
|
+
The thresholds ship **untuned** (0.5 to keep a candidate, 0.5 confidence for the "most direct"
|
|
128
|
+
choice, 0.8 for "whole suite"). Tune them on your own history before trusting them:
|
|
129
|
+
|
|
130
|
+
```console
|
|
131
|
+
$ minitest-impact eval --cases cases.json --jev --format json
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
and move the constants in `questions.rb` to the values that give the recall you need at the
|
|
135
|
+
smallest selection.
|
|
136
|
+
|
|
137
|
+
## Measure it on your history
|
|
138
|
+
|
|
139
|
+
```console
|
|
140
|
+
$ minitest-impact eval --co-changed 200 # commits that changed code and its tests together
|
|
141
|
+
$ ruby eval/ci_failures.rb --repo . --out cases.json # failed CI runs, via the gh CLI
|
|
142
|
+
$ minitest-impact eval --cases cases.json
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
Two kinds of labelled cases:
|
|
146
|
+
|
|
147
|
+
- **Tests a change broke** (`eval/ci_failures.rb`): for each failed GitHub Actions run on the
|
|
148
|
+
default branch, the change since the last green run, and the test files the failed jobs
|
|
149
|
+
reported. A test already failing in the run before is not counted again.
|
|
150
|
+
- **Tests written for a change** (`--co-changed N`): commits that changed code and existing test
|
|
151
|
+
files together. The selector sees only the code; the tests the commit edited are the answer.
|
|
152
|
+
|
|
153
|
+
Reported per case and on average: whether any expected test was selected (`caught`), whether all
|
|
154
|
+
were (`all_caught`), recall, precision, and the share of the suite selected.
|
|
155
|
+
|
|
156
|
+
### First numbers: Piou Piou, map only (2026-09-28)
|
|
157
|
+
|
|
158
|
+
The map was recorded at one commit: 396 test files (2,673 unit and 88 system tests), 580 KB of
|
|
159
|
+
JSON (72 KB gzipped). Recording made the unit suite about 2 to 3 times slower. The evaluation made
|
|
160
|
+
no Jev calls.
|
|
161
|
+
|
|
162
|
+
| Cases | Caught (any expected test selected) | All caught | Recall | Precision | Share of suite selected | Share of suite time |
|
|
163
|
+
|---|---:|---:|---:|---:|---:|---:|
|
|
164
|
+
| 150 commits that changed code and tests together, method-level | 94.0% | 90.7% | 0.93 | 0.17 | 12.1% | 24.0% |
|
|
165
|
+
| The same 150, file-level | 94.0% | 90.7% | 0.93 | 0.15 | 12.3% | 24.4% |
|
|
166
|
+
| 17 failed CI runs on main | 70.6% | 64.7% | 0.68 | 0.07 | 38.5% | 42.2% |
|
|
167
|
+
| The same, without 4 runs where only a flaky system test failed | 12 of 13 | 11 of 13 | 0.88 | 0.10 | 50.2% | 55.1% |
|
|
168
|
+
|
|
169
|
+
How to read them:
|
|
170
|
+
|
|
171
|
+
- Six of the 13 real CI breaks changed `Gemfile.lock`, `ci.yml`-tested config or `test_helper.rb`,
|
|
172
|
+
so the rule selected the whole suite. That is correct but costly. On the other seven, the
|
|
173
|
+
selection was 7.5% of the suite.
|
|
174
|
+
- The one real break missed: a `config/piou.yml` change that failed
|
|
175
|
+
`test/services/sandbox_container_test.rb`, which reads the setting through the app and never
|
|
176
|
+
names the file.
|
|
177
|
+
- Method-level tracing barely beats file-level on this history. Most changes land in small,
|
|
178
|
+
focused files, where the two agree.
|
|
179
|
+
|
|
180
|
+
## Prior art
|
|
181
|
+
|
|
182
|
+
Nothing did most of this for Minitest, offline, when this gem was written (September 2026):
|
|
183
|
+
|
|
184
|
+
| Project | What it is | What we took |
|
|
185
|
+
|---|---|---|
|
|
186
|
+
| [Crystalball](https://github.com/toptal/crystalball) (and GitLab's fork) | Coverage-map test selection for RSpec | The shape: record per-test coverage, predict from the diff; views, locales and schema as separate strategies |
|
|
187
|
+
| [affected_tests](https://rubygems.org/gems/affected_tests), [test_impact](https://rubygems.org/gems/test_impact) | 2026 map-based selectors, RSpec only | `Coverage.result(clear: true)` per test; exit code for "run everything"; a staleness warning |
|
|
188
|
+
| [fast_cov](https://github.com/Gusto/fast_cov) | A C-extension file tracker that leaves `Coverage` to SimpleCov | A candidate recorder backend if recording beside SimpleCov matters more than line numbers |
|
|
189
|
+
| [Datadog Test Impact Analysis](https://docs.datadoghq.com/tests/test_impact_analysis/) | Minitest support, but the decision needs Datadog's service | Nothing adopted |
|
|
190
|
+
| [Launchable / CloudBees Smart Tests](https://docs.cloudbees.com/docs/cloudbees-smart-tests/latest/features/predictive-test-selection) | Predictive selection as a service | Nothing adopted |
|
|
191
|
+
| Google TAP (Memon et al., 2017), Facebook's predictive test selection (Machalica et al., 2019), Ekstazi (Gligoric et al., 2015) | Research | File-level dynamic selection is safe and cheap; most failures are close to the change |
|
|
192
|
+
| [`jev`](https://rubygems.org/gems/jev) 0.2.0 | A Ruby client for Jev | Not used yet: it always sends `jev-latest` to one fixed URL, and thresholds need a pinned version |
|
|
193
|
+
|
|
194
|
+
## Limits
|
|
195
|
+
|
|
196
|
+
- Code that runs only at boot (initializers, `config/*.rb`) is not in the map; a change there is
|
|
197
|
+
traced by rules or reported as low confidence.
|
|
198
|
+
- Views are traced by file, not by line: compiled templates' line numbers are not the ERB's.
|
|
199
|
+
- Line coverage cannot see what a test *would* run after the change (a new branch, a new
|
|
200
|
+
callback). That is what the conventional test and Jev are for, and why the full suite stays the
|
|
201
|
+
final gate.
|
|
202
|
+
|
|
203
|
+
## License
|
|
204
|
+
|
|
205
|
+
MIT.
|
data/eval/ci_failures.rb
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
#!/usr/bin/env ruby
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
# Builds evaluation cases from a repository's failed GitHub Actions runs on its default branch:
|
|
5
|
+
# for each failed run, the change is everything since the last green run before it, and the
|
|
6
|
+
# expected tests are the test files the failed jobs reported. Logs are cached, so a second run is
|
|
7
|
+
# offline.
|
|
8
|
+
#
|
|
9
|
+
# ruby eval/ci_failures.rb --repo ~/app --workflow ci.yml --out cases.json [--cache DIR] [--limit 200]
|
|
10
|
+
#
|
|
11
|
+
# Needs the gh CLI, authenticated for the repository. Costs nothing but GitHub API calls.
|
|
12
|
+
|
|
13
|
+
require "json"
|
|
14
|
+
require "open3"
|
|
15
|
+
require "optparse"
|
|
16
|
+
require "fileutils"
|
|
17
|
+
|
|
18
|
+
options = { workflow: "ci.yml", limit: 300, cache: "tmp/minitest-impact/ci-logs", branch: "main" }
|
|
19
|
+
OptionParser.new do |parser|
|
|
20
|
+
parser.on("--repo DIR") { |v| options[:repo] = File.expand_path(v) }
|
|
21
|
+
parser.on("--workflow FILE") { |v| options[:workflow] = v }
|
|
22
|
+
parser.on("--branch NAME") { |v| options[:branch] = v }
|
|
23
|
+
parser.on("--out FILE") { |v| options[:out] = v }
|
|
24
|
+
parser.on("--cache DIR") { |v| options[:cache] = v }
|
|
25
|
+
parser.on("--limit N", Integer) { |v| options[:limit] = v }
|
|
26
|
+
end.parse!
|
|
27
|
+
abort "usage: ci_failures.rb --repo DIR --out FILE" unless options[:repo] && options[:out]
|
|
28
|
+
|
|
29
|
+
def sh(*cmd, dir:)
|
|
30
|
+
out, err, status = Open3.capture3(*cmd, chdir: dir)
|
|
31
|
+
raise "#{cmd.join(" ")}: #{err}" unless status.success?
|
|
32
|
+
|
|
33
|
+
out
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
FileUtils.mkdir_p(options[:cache])
|
|
37
|
+
runs = JSON.parse(sh("gh", "run", "list", "--workflow", options[:workflow], "--branch", options[:branch], "--event", "push",
|
|
38
|
+
"--limit", options[:limit].to_s, "--json", "databaseId,headSha,conclusion,createdAt", dir: options[:repo]))
|
|
39
|
+
runs.sort_by! { |run| run["createdAt"] }
|
|
40
|
+
|
|
41
|
+
# Rails prints a rerun line per failure ("bin/rails test test/models/user_test.rb:12"), and
|
|
42
|
+
# Minitest a location in brackets ("[test/models/user_test.rb:12]").
|
|
43
|
+
FAILING = %r{(?:bin/rails test |\[)(test/[\w/.-]+_test\.rb):\d+}
|
|
44
|
+
|
|
45
|
+
last_green = nil
|
|
46
|
+
still_failing = []
|
|
47
|
+
cases = []
|
|
48
|
+
runs.each do |run|
|
|
49
|
+
if run["conclusion"] == "success"
|
|
50
|
+
last_green = run["headSha"]
|
|
51
|
+
still_failing = []
|
|
52
|
+
next
|
|
53
|
+
end
|
|
54
|
+
next unless run["conclusion"] == "failure" && last_green
|
|
55
|
+
|
|
56
|
+
cache = File.join(options[:cache], "#{run["databaseId"]}.log")
|
|
57
|
+
unless File.exist?(cache)
|
|
58
|
+
log, _err, status = Open3.capture3("gh", "run", "view", run["databaseId"].to_s, "--log-failed", chdir: options[:repo])
|
|
59
|
+
File.write(cache, status.success? ? log : "")
|
|
60
|
+
end
|
|
61
|
+
reported = File.read(cache).scan(FAILING).flatten.uniq.sort
|
|
62
|
+
# A test that already failed in the run before was broken by an earlier push, not this one.
|
|
63
|
+
failing = reported - still_failing
|
|
64
|
+
still_failing |= reported
|
|
65
|
+
next if failing.empty?
|
|
66
|
+
|
|
67
|
+
cases << { "id" => "run-#{run["databaseId"]}", "base" => last_green, "head" => run["headSha"], "expected" => failing,
|
|
68
|
+
"source" => "ci_failure" }
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
File.write(options[:out], JSON.pretty_generate(cases))
|
|
72
|
+
puts "#{cases.size} cases from #{runs.count { |r| r["conclusion"] == "failure" }} failed runs into #{options[:out]}"
|
data/exe/minitest-impact
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
require "optparse"
|
|
5
|
+
require "tmpdir"
|
|
6
|
+
|
|
7
|
+
module Minitest
|
|
8
|
+
module Impact
|
|
9
|
+
# minitest-impact record | select | run | eval
|
|
10
|
+
class CLI
|
|
11
|
+
WHOLE_SUITE_EXIT = 10
|
|
12
|
+
|
|
13
|
+
def self.start(argv, out: $stdout, err: $stderr, env: ENV) = new(out: out, err: err, env: env).call(argv)
|
|
14
|
+
|
|
15
|
+
def initialize(out:, err:, env:)
|
|
16
|
+
@out = out
|
|
17
|
+
@err = err
|
|
18
|
+
@env = env
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def call(argv)
|
|
22
|
+
command = argv.shift
|
|
23
|
+
case command
|
|
24
|
+
when "record" then record(argv)
|
|
25
|
+
when "select" then select(argv)
|
|
26
|
+
when "run" then run(argv)
|
|
27
|
+
when "eval" then evaluate(argv)
|
|
28
|
+
else
|
|
29
|
+
@out.puts(USAGE)
|
|
30
|
+
command.nil? || %w[-h --help help].include?(command) ? 0 : 1
|
|
31
|
+
end
|
|
32
|
+
rescue Error, OptionParser::ParseError => error
|
|
33
|
+
@err.puts("minitest-impact: #{error.message}")
|
|
34
|
+
1
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
USAGE = <<~TEXT
|
|
38
|
+
Usage:
|
|
39
|
+
minitest-impact record [--map PATH] -- COMMAND... Run your suite and record the coverage map
|
|
40
|
+
minitest-impact select [--since REV] [options] Print the tests a change should run
|
|
41
|
+
minitest-impact run [--since REV] [options] Select, then run them with bin/rails test
|
|
42
|
+
minitest-impact eval (--cases FILE | --co-changed N) [--jev] [--granularity method|file]
|
|
43
|
+
Measure selection on history
|
|
44
|
+
|
|
45
|
+
Select options:
|
|
46
|
+
--since REV compare against REV (default HEAD: the uncommitted change)
|
|
47
|
+
--head REV compare REV instead of the working tree
|
|
48
|
+
--intent TEXT what the change is for, in words (used by Jev only)
|
|
49
|
+
--map PATH coverage map (default #{Map::DEFAULT_PATH})
|
|
50
|
+
--format FORMAT text (default), json or paths
|
|
51
|
+
--max N keep the N most likely test files
|
|
52
|
+
--no-jev map and rules only, even with TYPESAFE_API_KEY set
|
|
53
|
+
|
|
54
|
+
Exit status 10 from select/run with --format paths means: run the whole suite.
|
|
55
|
+
TEXT
|
|
56
|
+
|
|
57
|
+
private
|
|
58
|
+
|
|
59
|
+
def record(argv)
|
|
60
|
+
map_path = Map::DEFAULT_PATH
|
|
61
|
+
if argv.first == "--map"
|
|
62
|
+
argv.shift
|
|
63
|
+
map_path = argv.shift
|
|
64
|
+
end
|
|
65
|
+
argv.shift if argv.first == "--"
|
|
66
|
+
raise Error, "record needs a command, e.g. minitest-impact record -- bin/rails test" if argv.empty?
|
|
67
|
+
|
|
68
|
+
repo = Repo.new
|
|
69
|
+
Dir.mktmpdir("minitest-impact") do |dir|
|
|
70
|
+
lib = File.expand_path("../..", __dir__)
|
|
71
|
+
bootstrap = File.join(__dir__, "record_bootstrap.rb")
|
|
72
|
+
env = {
|
|
73
|
+
"MINITEST_IMPACT_RECORD" => dir,
|
|
74
|
+
"MINITEST_IMPACT_ROOT" => repo.root,
|
|
75
|
+
"RUBYOPT" => ["-I#{lib}", "-r#{bootstrap}", @env["RUBYOPT"]].compact.join(" ")
|
|
76
|
+
}
|
|
77
|
+
ok = system(env, *argv)
|
|
78
|
+
map = Map.merge(File.join(dir, Recorder::PARTS), commit: repo.head)
|
|
79
|
+
raise Error, "no test recorded anything; is the command a Minitest run?" if map.tests.empty?
|
|
80
|
+
|
|
81
|
+
map.save(map_path)
|
|
82
|
+
dirty = !repo.git("status", "--porcelain", "--untracked-files=no").empty?
|
|
83
|
+
@err.puts("minitest-impact: the working tree has changes; the map describes them, not #{repo.head[0, 10]}") if dirty
|
|
84
|
+
@out.puts("Recorded #{map.tests.size} test files at #{repo.head[0, 10]} into #{map_path}")
|
|
85
|
+
ok ? 0 : 1
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
def select_options(argv)
|
|
90
|
+
options = { since: "HEAD", format: "text", map: Map::DEFAULT_PATH, jev: true }
|
|
91
|
+
OptionParser.new do |parser|
|
|
92
|
+
parser.on("--since REV") { |v| options[:since] = v }
|
|
93
|
+
parser.on("--head REV") { |v| options[:head] = v }
|
|
94
|
+
parser.on("--intent TEXT") { |v| options[:intent] = v }
|
|
95
|
+
parser.on("--map PATH") { |v| options[:map] = v }
|
|
96
|
+
parser.on("--format FORMAT", %w[text json paths]) { |v| options[:format] = v }
|
|
97
|
+
parser.on("--max N", Integer) { |v| options[:max] = v }
|
|
98
|
+
parser.on("--[no-]jev") { |v| options[:jev] = v }
|
|
99
|
+
end.parse!(argv)
|
|
100
|
+
options
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def selection(options)
|
|
104
|
+
raise Error, "no coverage map at #{options[:map]}; run: minitest-impact record -- bin/rails test" unless File.exist?(options[:map])
|
|
105
|
+
|
|
106
|
+
repo = Repo.new
|
|
107
|
+
ranker = options[:jev] ? jev_ranker : nil
|
|
108
|
+
selection = Selector.new(repo: repo, map: Map.load(options[:map]), base: options[:since], head: options[:head],
|
|
109
|
+
intent: options[:intent], ranker: ranker).call
|
|
110
|
+
selection.picks = selection.picks.first(options[:max]) if options[:max]
|
|
111
|
+
selection
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
def jev_ranker
|
|
115
|
+
client = Jev::Client.from_env(@env)
|
|
116
|
+
client && Jev::Ranker.new(client: client)
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
def select(argv)
|
|
120
|
+
options = select_options(argv)
|
|
121
|
+
result = selection(options)
|
|
122
|
+
case options[:format]
|
|
123
|
+
when "json" then @out.puts(JSON.pretty_generate(result.to_h))
|
|
124
|
+
when "paths"
|
|
125
|
+
return WHOLE_SUITE_EXIT if result.whole_suite
|
|
126
|
+
|
|
127
|
+
result.tests.each { |test| @out.puts(test) }
|
|
128
|
+
else print_text(result)
|
|
129
|
+
end
|
|
130
|
+
0
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def run(argv)
|
|
134
|
+
result = selection(select_options(argv))
|
|
135
|
+
print_text(result)
|
|
136
|
+
return WHOLE_SUITE_EXIT if result.whole_suite
|
|
137
|
+
return 0 if result.tests.empty?
|
|
138
|
+
|
|
139
|
+
runner = File.exist?("bin/rails") ? ["bin/rails", "test"] : ["ruby", "-Itest", "-e", "ARGV.each { |f| require File.expand_path(f) }"]
|
|
140
|
+
system(*runner, *result.tests) ? 0 : 1
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
def print_text(result)
|
|
144
|
+
@out.puts("Confidence: #{result.confidence}#{result.whole_suite ? " (run the whole suite)" : ""}")
|
|
145
|
+
result.warnings.each { |warning| @out.puts("Warning: #{warning}") }
|
|
146
|
+
@out.puts("Jev: #{result.jev.map { |k, v| "#{k}=#{v}" }.join(" ")}") if result.jev
|
|
147
|
+
result.resolutions.each do |resolution|
|
|
148
|
+
@out.puts(" #{resolution.how.to_s.ljust(12)} #{resolution.path}#{resolution.note ? " (#{resolution.note})" : ""}")
|
|
149
|
+
end
|
|
150
|
+
@out.puts(result.picks.empty? ? "No tests selected." : "#{result.picks.size} test files, most likely first:")
|
|
151
|
+
result.picks.each do |pick|
|
|
152
|
+
@out.puts(format(" %.2f %s - %s", pick.score, pick.test, pick.reasons.first(2).join("; ")))
|
|
153
|
+
end
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
def evaluate(argv)
|
|
157
|
+
options = { map: Map::DEFAULT_PATH, jev: false, format: "text", granularity: :method }
|
|
158
|
+
OptionParser.new do |parser|
|
|
159
|
+
parser.on("--cases FILE") { |v| options[:cases] = v }
|
|
160
|
+
parser.on("--co-changed N", Integer) { |v| options[:co_changed] = v }
|
|
161
|
+
parser.on("--map PATH") { |v| options[:map] = v }
|
|
162
|
+
parser.on("--[no-]jev") { |v| options[:jev] = v }
|
|
163
|
+
parser.on("--format FORMAT", %w[text json]) { |v| options[:format] = v }
|
|
164
|
+
parser.on("--granularity LEVEL", %w[method file]) { |v| options[:granularity] = v.to_sym }
|
|
165
|
+
end.parse!(argv)
|
|
166
|
+
repo = Repo.new
|
|
167
|
+
map = Map.load(options[:map])
|
|
168
|
+
cases = if options[:cases]
|
|
169
|
+
JSON.parse(File.read(options[:cases]))
|
|
170
|
+
elsif options[:co_changed]
|
|
171
|
+
Evaluation.co_changed_cases(repo, map, limit: options[:co_changed])
|
|
172
|
+
else
|
|
173
|
+
raise Error, "eval needs --cases FILE or --co-changed N"
|
|
174
|
+
end
|
|
175
|
+
results = Evaluation.new(repo: repo, map: map, ranker: options[:jev] ? jev_ranker : nil, granularity: options[:granularity]).run(cases)
|
|
176
|
+
summary = Evaluation.summary(results)
|
|
177
|
+
if options[:format] == "json"
|
|
178
|
+
@out.puts(JSON.pretty_generate("summary" => summary, "cases" => results.map { |r| r.to_h.merge(recall: r.recall, precision: r.precision) }))
|
|
179
|
+
else
|
|
180
|
+
results.each do |r|
|
|
181
|
+
@out.puts(format("%-12s caught=%-5s recall=%.2f precision=%.2f selected=%d/%d %s", r.id, r.caught?, r.recall, r.precision,
|
|
182
|
+
r.selected.size, r.suite_size, r.confidence))
|
|
183
|
+
end
|
|
184
|
+
@out.puts(JSON.pretty_generate(summary))
|
|
185
|
+
end
|
|
186
|
+
0
|
|
187
|
+
end
|
|
188
|
+
end
|
|
189
|
+
end
|
|
190
|
+
end
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Minitest
|
|
4
|
+
module Impact
|
|
5
|
+
# One file in a change. Line numbers are on the old side of the diff, because the coverage
|
|
6
|
+
# map was recorded against old code; an insertion is placed on the line it follows.
|
|
7
|
+
ChangedFile = Struct.new(:path, :old_path, :status, :old_lines, :new_lines, :added_text, :removed_text, keyword_init: true) do
|
|
8
|
+
def added? = status == :added
|
|
9
|
+
def deleted? = status == :deleted
|
|
10
|
+
def text = [removed_text, added_text].join("\n")
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
module Diff
|
|
14
|
+
HUNK = /\A@@ -(\d+)(?:,(\d+))? \+(\d+)(?:,(\d+))? @@/
|
|
15
|
+
|
|
16
|
+
# Parses `git diff --unified=0` output.
|
|
17
|
+
def self.parse(text)
|
|
18
|
+
files = []
|
|
19
|
+
current = nil
|
|
20
|
+
text.each_line do |line|
|
|
21
|
+
line = line.chomp
|
|
22
|
+
case line
|
|
23
|
+
when /\Adiff --git a\/(.+) b\/(.+)\z/
|
|
24
|
+
current = ChangedFile.new(path: Regexp.last_match(2), old_path: Regexp.last_match(1), status: :modified,
|
|
25
|
+
old_lines: [], new_lines: [], added_text: +"", removed_text: +"")
|
|
26
|
+
files << current
|
|
27
|
+
when /\Anew file mode/ then current.status = :added
|
|
28
|
+
when /\Adeleted file mode/ then current.status = :deleted
|
|
29
|
+
when /\Arename from (.+)\z/ then current.old_path = Regexp.last_match(1)
|
|
30
|
+
when /\Arename to (.+)\z/ then current.status = :renamed if current.status == :modified
|
|
31
|
+
when HUNK
|
|
32
|
+
start = Regexp.last_match(1).to_i
|
|
33
|
+
count = (Regexp.last_match(2) || "1").to_i
|
|
34
|
+
current.old_lines.concat(count.zero? ? [[start, 1].max] : (start...(start + count)).to_a)
|
|
35
|
+
new_start = Regexp.last_match(3).to_i
|
|
36
|
+
new_count = (Regexp.last_match(4) || "1").to_i
|
|
37
|
+
current.new_lines.concat((new_start...(new_start + new_count)).to_a)
|
|
38
|
+
when /\A\+\+\+ |\A--- / then next
|
|
39
|
+
when /\A\+(.*)\z/ then current&.added_text&.<<(Regexp.last_match(1) + "\n")
|
|
40
|
+
when /\A-(.*)\z/ then current&.removed_text&.<<(Regexp.last_match(1) + "\n")
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
files.each { |file| file.old_path = nil if file.status == :added }
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
end
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
module Minitest
|
|
6
|
+
module Impact
|
|
7
|
+
# Replays a selection on past changes whose right answer is known and measures it.
|
|
8
|
+
#
|
|
9
|
+
# A case is {"id", "base", "head", "expected" => [test files]}: the tests a change broke (from
|
|
10
|
+
# CI failures) or the tests written for it (from commits that changed code and its tests
|
|
11
|
+
# together, with the test edits hidden from the selector).
|
|
12
|
+
class Evaluation
|
|
13
|
+
Result = Struct.new(:id, :expected, :selected, :hit, :suite_size, :seconds_share, :whole_suite, :confidence, keyword_init: true) do
|
|
14
|
+
def caught? = hit.any?
|
|
15
|
+
def all_caught? = (expected - hit).empty?
|
|
16
|
+
def recall = hit.size.to_f / expected.size
|
|
17
|
+
def precision = selected.empty? ? 0.0 : hit.size.to_f / selected.size
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
# Cases from commits that changed code and existing tests together. The tests they edited
|
|
21
|
+
# are the ones written to verify the change.
|
|
22
|
+
def self.co_changed_cases(repo, map, limit: 200)
|
|
23
|
+
shas = repo.git("log", "--no-merges", "--format=%H", "-n", (limit * 10).to_s).split("\n")
|
|
24
|
+
shas.each_with_object([]) do |sha, cases|
|
|
25
|
+
break cases if cases.size >= limit
|
|
26
|
+
|
|
27
|
+
files = Diff.parse(repo.git("diff", "--no-color", "-M", "--unified=0", "#{sha}^", sha, allow_failure: true).to_s)
|
|
28
|
+
tests = files.select { |f| Rules.test_file?(f.path) && f.status == :modified && map.tests.key?(f.path) }.map(&:path)
|
|
29
|
+
code = files.reject { |f| f.path.start_with?(Rules::TEST_ROOT) || Rules.quiet?(f.path) }
|
|
30
|
+
next if tests.empty? || code.empty?
|
|
31
|
+
|
|
32
|
+
cases << { "id" => sha[0, 10], "base" => "#{sha}^", "head" => sha, "expected" => tests, "hide_tests" => true }
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
def initialize(repo:, map:, ranker: nil, granularity: :method)
|
|
37
|
+
@repo = repo
|
|
38
|
+
@map = map
|
|
39
|
+
@ranker = ranker
|
|
40
|
+
@granularity = granularity
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def run(cases)
|
|
44
|
+
cases.filter_map do |kase|
|
|
45
|
+
expected = kase.fetch("expected") & @map.test_files
|
|
46
|
+
next if expected.empty? || !@repo.commit?(kase.fetch("head"))
|
|
47
|
+
|
|
48
|
+
exclude = kase["hide_tests"] ? ->(path) { path.start_with?(Rules::TEST_ROOT) } : nil
|
|
49
|
+
selection = Selector.new(repo: @repo, map: @map, base: kase.fetch("base"), head: kase.fetch("head"),
|
|
50
|
+
intent: kase["intent"], ranker: @ranker, exclude: exclude, granularity: @granularity).call
|
|
51
|
+
selected = selection.whole_suite ? @map.test_files : selection.tests
|
|
52
|
+
Result.new(id: kase.fetch("id"), expected: expected, selected: selected, hit: expected & selected,
|
|
53
|
+
suite_size: @map.test_files.size, seconds_share: seconds_share(selected), whole_suite: selection.whole_suite,
|
|
54
|
+
confidence: selection.confidence)
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def seconds_share(selected)
|
|
59
|
+
total = @map.tests.values.sum(&:seconds)
|
|
60
|
+
total.zero? ? 0.0 : selected.sum { |test| @map.seconds(test).to_f } / total
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def self.summary(results)
|
|
64
|
+
return {} if results.empty?
|
|
65
|
+
|
|
66
|
+
n = results.size.to_f
|
|
67
|
+
{
|
|
68
|
+
"cases" => results.size,
|
|
69
|
+
"caught" => (results.count(&:caught?) / n).round(3),
|
|
70
|
+
"all_caught" => (results.count(&:all_caught?) / n).round(3),
|
|
71
|
+
"recall" => (results.sum(&:recall) / n).round(3),
|
|
72
|
+
"precision" => (results.sum(&:precision) / n).round(3),
|
|
73
|
+
"selected_share" => (results.sum { |r| r.selected.size.to_f / r.suite_size } / n).round(3),
|
|
74
|
+
"seconds_share" => (results.sum(&:seconds_share) / n).round(3),
|
|
75
|
+
"whole_suite" => results.count(&:whole_suite)
|
|
76
|
+
}
|
|
77
|
+
end
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
end
|