pytest-timing 0.1.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/.github/workflows/ci.yml +9 -8
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/.github/workflows/release.yml +2 -2
- pytest_timing-0.3.0/ARCHITECTURE.md +507 -0
- pytest_timing-0.3.0/CHANGELOG.md +88 -0
- pytest_timing-0.3.0/PKG-INFO +369 -0
- pytest_timing-0.3.0/README.md +343 -0
- pytest_timing-0.3.0/benchmarks/cpu_bench.py +376 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/examples/test_demo.py +16 -1
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/pyproject.toml +3 -3
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/__init__.py +5 -1
- pytest_timing-0.3.0/src/pytest_timing/admission.py +210 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/collector.py +48 -9
- pytest_timing-0.3.0/src/pytest_timing/demand.py +483 -0
- pytest_timing-0.3.0/src/pytest_timing/fixtures.py +229 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/model.py +147 -13
- pytest_timing-0.3.0/src/pytest_timing/plugin.py +733 -0
- pytest_timing-0.3.0/src/pytest_timing/schedule.py +577 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/static/report.html +0 -10
- pytest_timing-0.3.0/src/pytest_timing/telemetry.py +834 -0
- pytest_timing-0.3.0/src/pytest_timing/xdist_compat.py +337 -0
- pytest_timing-0.3.0/src/pytest_timing/xdist_scheduler.py +1131 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/conftest.py +50 -0
- pytest_timing-0.3.0/tests/test_admission.py +194 -0
- pytest_timing-0.3.0/tests/test_benchmarks.py +22 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_cli.py +0 -2
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_collector.py +26 -4
- pytest_timing-0.3.0/tests/test_cpu.py +789 -0
- pytest_timing-0.3.0/tests/test_fixtures.py +362 -0
- pytest_timing-0.3.0/tests/test_memory.py +216 -0
- pytest_timing-0.3.0/tests/test_memory_gate.py +272 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_plugin.py +1 -15
- pytest_timing-0.3.0/tests/test_runtime_admission.py +712 -0
- pytest_timing-0.3.0/tests/test_schedule.py +1435 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/uv.lock +3 -3
- pytest_timing-0.1.0/CHANGELOG.md +0 -10
- pytest_timing-0.1.0/PKG-INFO +0 -161
- pytest_timing-0.1.0/README.md +0 -135
- pytest_timing-0.1.0/src/pytest_timing/plugin.py +0 -382
- pytest_timing-0.1.0/src/pytest_timing/xdist_compat.py +0 -158
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/.gitignore +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/LICENSE +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/docs/report.png +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/cli.py +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/outputs.py +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/render/__init__.py +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/render/ascii.py +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/render/html.py +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/render/trace.py +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/static/__init__.py +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/fixtures/legacy_complete_false.json +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/fixtures/legacy_complete_true.json +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/fixtures/legacy_no_flag.json +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_ascii.py +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_html.py +0 -0
- {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_trace.py +0 -0
|
@@ -7,7 +7,7 @@ on:
|
|
|
7
7
|
|
|
8
8
|
jobs:
|
|
9
9
|
test:
|
|
10
|
-
name: py${{ matrix.python }} / pytest ${{ matrix.pytest }}
|
|
10
|
+
name: ${{ matrix.os }} / py${{ matrix.python }} / pytest ${{ matrix.pytest }}${{ matrix.xdist && format(' / xdist {0}', matrix.xdist) || '' }}
|
|
11
11
|
runs-on: ${{ matrix.os }}
|
|
12
12
|
strategy:
|
|
13
13
|
fail-fast: false
|
|
@@ -16,14 +16,14 @@ jobs:
|
|
|
16
16
|
python: ["3.10", "3.11", "3.12", "3.13", "3.14"]
|
|
17
17
|
pytest: ["latest"]
|
|
18
18
|
include:
|
|
19
|
-
- { os: ubuntu-latest, python: "3.10", pytest: "7.3" }
|
|
19
|
+
- { os: ubuntu-latest, python: "3.10", pytest: "7.3", xdist: "3.7.0" }
|
|
20
20
|
- { os: ubuntu-latest, python: "3.11", pytest: "7.4" }
|
|
21
21
|
- { os: ubuntu-latest, python: "3.12", pytest: "8.0" }
|
|
22
22
|
- { os: windows-latest, python: "3.12", pytest: "latest" }
|
|
23
23
|
- { os: macos-latest, python: "3.13", pytest: "latest" }
|
|
24
24
|
steps:
|
|
25
|
-
- uses: actions/checkout@
|
|
26
|
-
- uses: astral-sh/setup-uv@
|
|
25
|
+
- uses: actions/checkout@v7
|
|
26
|
+
- uses: astral-sh/setup-uv@v10.1.0
|
|
27
27
|
with:
|
|
28
28
|
python-version: ${{ matrix.python }}
|
|
29
29
|
- name: Install
|
|
@@ -31,6 +31,7 @@ jobs:
|
|
|
31
31
|
uv sync --no-dev
|
|
32
32
|
uv pip install pytest-xdist pytest-rerunfailures hypothesis
|
|
33
33
|
if [ "${{ matrix.pytest }}" != "latest" ]; then uv pip install "pytest~=${{ matrix.pytest }}.0"; fi
|
|
34
|
+
if [ -n "${{ matrix.xdist }}" ]; then uv pip install "pytest-xdist==${{ matrix.xdist }}"; fi
|
|
34
35
|
shell: bash
|
|
35
36
|
- name: Test
|
|
36
37
|
run: uv run --no-sync pytest -v -p no:cacheprovider
|
|
@@ -39,8 +40,8 @@ jobs:
|
|
|
39
40
|
name: without pytest-xdist
|
|
40
41
|
runs-on: ubuntu-latest
|
|
41
42
|
steps:
|
|
42
|
-
- uses: actions/checkout@
|
|
43
|
-
- uses: astral-sh/setup-uv@
|
|
43
|
+
- uses: actions/checkout@v7
|
|
44
|
+
- uses: astral-sh/setup-uv@v10.1.0
|
|
44
45
|
with:
|
|
45
46
|
python-version: "3.12"
|
|
46
47
|
- run: uv sync --no-dev
|
|
@@ -50,8 +51,8 @@ jobs:
|
|
|
50
51
|
lint:
|
|
51
52
|
runs-on: ubuntu-latest
|
|
52
53
|
steps:
|
|
53
|
-
- uses: actions/checkout@
|
|
54
|
-
- uses: astral-sh/setup-uv@
|
|
54
|
+
- uses: actions/checkout@v7
|
|
55
|
+
- uses: astral-sh/setup-uv@v10.1.0
|
|
55
56
|
with:
|
|
56
57
|
python-version: "3.12"
|
|
57
58
|
- run: uv sync
|
|
@@ -0,0 +1,507 @@
|
|
|
1
|
+
# Architecture
|
|
2
|
+
|
|
3
|
+
How pytest-timing captures timings, times shared fixtures inside xdist workers, and
|
|
4
|
+
schedules a run from the previous one. For usage, see the [README](README.md).
|
|
5
|
+
|
|
6
|
+
## Modules
|
|
7
|
+
|
|
8
|
+
| Module | Role |
|
|
9
|
+
|---|---|
|
|
10
|
+
| `plugin.py` | pytest hooks on the controller: options, settings, capture, terminal summary, outputs, and the `pytest_xdist_make_scheduler` hook. |
|
|
11
|
+
| `collector.py` | Folds phase reports and worker events into a `Run`. Knows nothing about pytest objects. |
|
|
12
|
+
| `model.py` | The recorded run: `Run`, `Worker`, `TestSpan`, `Phase`, and the JSON representation. |
|
|
13
|
+
| `outputs.py` | The ASCII, HTML and trace renderers, in one table shared by the plugin and the CLI. |
|
|
14
|
+
| `cli.py` | `pytest-timing render` and `pytest-timing merge`. |
|
|
15
|
+
| `fixtures.py` | Times shared fixture set-up where tests run and lists dependencies on reports. |
|
|
16
|
+
| `schedule.py` | Duration estimates, the fixture-aware cost model and the planner. Free of xdist. |
|
|
17
|
+
| `demand.py` | Declared CPU demand: the `timing_cpu` marker, the fixture decorator, the declarations a worker sends, and the meter that records each test's CPU work and resident memory. |
|
|
18
|
+
| `admission.py` | One host's CPU budget, reservations, fair waiting line and pressure feedback. Free of xdist and platform reads. |
|
|
19
|
+
| `telemetry.py` | Platform reads: affinity and cgroup quota, CPU time and resident memory of a process tree, pressure and throttling. |
|
|
20
|
+
| `xdist_scheduler.py` | The xdist scheduler that drives workers with the planner and the admission gate, in `load` and `worksteal` mode. |
|
|
21
|
+
| `xdist_compat.py` | Everything that reads pytest-xdist internals (versions, worker detection, why the controller stopped, the custom worker event and controller command). |
|
|
22
|
+
|
|
23
|
+
## Capturing timings
|
|
24
|
+
|
|
25
|
+
Since pytest 7.3 every `TestReport` carries wall-clock `start` and `stop` timestamps.
|
|
26
|
+
xdist serialises reports to the controller unchanged and attaches the worker to each,
|
|
27
|
+
so phase aggregation uses controller-side hooks: `pytest_runtest_logreport` for the
|
|
28
|
+
setup / call / teardown phases, plus xdist's node-ready, collection-finished and
|
|
29
|
+
node-down hooks for the worker lifecycle. CPU admission adds custom events and commands
|
|
30
|
+
through the xdist adapter; timing reports use pytest's normal report transport.
|
|
31
|
+
|
|
32
|
+
Each lane in the report shows boot (worker start-up until it is ready), collection,
|
|
33
|
+
tests, idle gaps and the point the worker shut down. Session-scoped fixture set-up is
|
|
34
|
+
attributed to the first test that needs that instance on each worker and its teardown
|
|
35
|
+
to the test that finalizes it, exactly as pytest reports it.
|
|
36
|
+
|
|
37
|
+
Timing collection is enabled by `--timing`, an output, scheduling or CPU-budget
|
|
38
|
+
setting. Formatting settings alone do not enable it. The marker is always registered;
|
|
39
|
+
when enabled, an xdist worker also registers the fixture timer and CPU meter.
|
|
40
|
+
|
|
41
|
+
## Timing shared fixtures in the workers
|
|
42
|
+
|
|
43
|
+
pytest charges a class, module, package or session fixture's set-up to the setup phase
|
|
44
|
+
of the first test that needs it on each worker, so a test's recorded duration depends
|
|
45
|
+
on where it happened to land. The scheduler needs the two apart: how long the test
|
|
46
|
+
itself takes, and which shared fixtures it needs at what price.
|
|
47
|
+
|
|
48
|
+
`fixtures.py` wraps `pytest_fixture_setup` for the shared scopes and times each
|
|
49
|
+
set-up, excluding any set-up nested inside it through `getfixturevalue` (that one is
|
|
50
|
+
timed on its own). When a test's setup and call reports are built, the timer attaches
|
|
51
|
+
`timing_fixtures` to them: a mapping from fixture key to the seconds this test spent
|
|
52
|
+
setting the fixture up, or `None` when the test only uses it. Extra report attributes
|
|
53
|
+
survive xdist's serialisation, and the collector merges the two reports into
|
|
54
|
+
`TestSpan.fixtures`.
|
|
55
|
+
|
|
56
|
+
A test's fixtures are gathered from three sources, because none is complete alone:
|
|
57
|
+
|
|
58
|
+
- the static fixture closure, which names what the signatures ask for;
|
|
59
|
+
- the request's resolved fixture definitions, which add what `getfixturevalue` pulled
|
|
60
|
+
in for this test, including inside the test body (hence the call report);
|
|
61
|
+
- a per-worker memo of what each shared fixture's set-up requested dynamically. A
|
|
62
|
+
test served from the cache never sees those requests, but it depends on them just
|
|
63
|
+
the same.
|
|
64
|
+
|
|
65
|
+
A fixture key is `<scope>:<scope node id>:<defined at>::<name>[<params>]`. The
|
|
66
|
+
scope node is what pytest shares the instance under: the session (empty), the module,
|
|
67
|
+
the class, or the matching package collector. A package fixture without a matching
|
|
68
|
+
collector belongs to the session. "Defined
|
|
69
|
+
at" tells overrides of the same name apart: the fixture's `baseid` (its conftest
|
|
70
|
+
directory or test module), or, for a plugin's fixture, whose `baseid` is empty just
|
|
71
|
+
like a root conftest's on pytest 7, the defining module and qualified function name.
|
|
72
|
+
This also distinguishes a plugin class registered from that same conftest. The parameter part,
|
|
73
|
+
taken from the item's
|
|
74
|
+
callspec, tells the instances of a parametrized fixture apart whether the fixture or
|
|
75
|
+
the test (`indirect=True`) carries the parameters, since pytest keeps only one
|
|
76
|
+
instance alive at a time: `base[1]`. A fixture that depends on a parametrized one,
|
|
77
|
+
directly or through other fixtures, is torn down with the instance it was built on
|
|
78
|
+
and built again for the next parameter, so those indices are part of its key too:
|
|
79
|
+
`derived[base=1]`, or `mixed[0,base=1]` for a fixture with parameters of its own. It
|
|
80
|
+
is an index, not a value, so two modules that parametrize the same session fixture
|
|
81
|
+
with different values at the same index share a key; the planner is slightly
|
|
82
|
+
optimistic there.
|
|
83
|
+
|
|
84
|
+
## Estimates
|
|
85
|
+
|
|
86
|
+
`Estimates.from_run` subtracts shared set-up and runtime admission waits from each
|
|
87
|
+
attempt's duration to estimate the test's *own* time. Crashed attempts and nonpositive
|
|
88
|
+
results are excluded. For each node id it ranks the attempts clean before contended
|
|
89
|
+
and passing before failing, and takes the median of the best rank present: a failed
|
|
90
|
+
attempt usually stops early, and the longest attempt grows with the number of
|
|
91
|
+
attempts, so a history merged from several runs would drift upward. Tests without a
|
|
92
|
+
usable estimate get the mean of those estimates, or zero if none exist. Fixture
|
|
93
|
+
dependencies are combined across attempts, and each fixture key's set-up cost is the
|
|
94
|
+
median of the recorded ones.
|
|
95
|
+
|
|
96
|
+
A missing or unreadable history file leaves xdist's scheduler in place unless an
|
|
97
|
+
explicit CPU budget was requested. With that budget the custom scheduler still runs,
|
|
98
|
+
using equal 1 ms test estimates. It supports only `load` and `worksteal`.
|
|
99
|
+
|
|
100
|
+
## Cost model
|
|
101
|
+
|
|
102
|
+
A shared fixture makes the cost of a test depend on the worker. Every test has a
|
|
103
|
+
*family*, the set of shared fixtures it needs. Fixtures cheaper than a millisecond
|
|
104
|
+
are ignored unless they declare CPU demand, in which case they are retained even
|
|
105
|
+
without recorded set-up time. A queue's cost is its tests' own durations plus every
|
|
106
|
+
set-up the queue forces:
|
|
107
|
+
|
|
108
|
+
- session fixtures are paid once per worker, as are package fixtures without a
|
|
109
|
+
matching package collector;
|
|
110
|
+
- class, module and package fixtures last until the worker leaves their scope,
|
|
111
|
+
including through tests that do not use them. Re-entering the scope pays again;
|
|
112
|
+
- a parameter change replaces the fixture and its dependents.
|
|
113
|
+
|
|
114
|
+
A `Lane` holds one worker's projected finish (`free`), fixture instances at that
|
|
115
|
+
point (`fixtures`), and unsent plan. `Costs.charge` computes a single item's duration,
|
|
116
|
+
CPU work, peak demand and fixture transition. `Costs.project` folds these into a
|
|
117
|
+
`Projection` for a whole sequence. Time and CPU work therefore use one traversal.
|
|
118
|
+
|
|
119
|
+
The scheduler keeps one `Charge` per dispatched item. It records the holds already
|
|
120
|
+
included in the peak, so actual live holds supplement it without double counting.
|
|
121
|
+
Charged duration, CPU work and the pre-dispatch fixture checkpoint remain historical
|
|
122
|
+
facts even when a runtime request changes the reservation. Stealing refunds the same
|
|
123
|
+
record and restores that checkpoint.
|
|
124
|
+
Ordinary transfers and steals compare finish times with the same CPU-work floor.
|
|
125
|
+
|
|
126
|
+
## Planning
|
|
127
|
+
|
|
128
|
+
`plan` places nonempty fixture families first, then the family without shared
|
|
129
|
+
fixtures. Families are ordered by cold CPU work (slot-seconds, including set-ups).
|
|
130
|
+
Members are ordered by declared slots, then own duration, both descending. Each
|
|
131
|
+
family is tried on 1..n lanes: members are greedily split in that order, putting each
|
|
132
|
+
on the chunk with the least own time so far, and the chunks go to lanes chosen by
|
|
133
|
+
projected finish and fixture cost.
|
|
134
|
+
|
|
135
|
+
The placement score is the maximum of the longest lane after placement, the average
|
|
136
|
+
lane including work still unplaced, and, with admission enabled, total slot-seconds
|
|
137
|
+
divided by the sum of domain admission limits. The average avoids splitting an early
|
|
138
|
+
family just because lanes are still empty; the CPU floor counts duplicated heavy set-ups.
|
|
139
|
+
A split of a nonempty family must improve the score by more than 1% (`SPLIT_GAIN`).
|
|
140
|
+
The family without shared fixtures is spread more evenly on near ties instead.
|
|
141
|
+
|
|
142
|
+
A bounded balance pass then moves single tests off the tail of the longest lane while
|
|
143
|
+
that clearly helps, using the same CPU-work floor. Finally each lane is ordered by
|
|
144
|
+
fixture-bound work, declared slots and own duration, then grouped by family in order
|
|
145
|
+
of first appearance. This prioritizes fixture reuse and heavy work while leaving
|
|
146
|
+
lighter work available at the tail for other workers.
|
|
147
|
+
|
|
148
|
+
Lane free times are clock readings. The percentage tolerance is applied to time left
|
|
149
|
+
from the earliest of them, never to the reading itself, or a large monotonic clock
|
|
150
|
+
value would swallow every gain.
|
|
151
|
+
|
|
152
|
+
`transfer` moves a tail chunk of another lane's unsent plan to a lane that ran dry:
|
|
153
|
+
the whole tail run of one family, half of it, or its last test, from every other lane,
|
|
154
|
+
keeping the move with the best predicted finish. A move that leaves the prediction
|
|
155
|
+
unchanged is still taken, because the receiver is idle and the estimates are only
|
|
156
|
+
estimates.
|
|
157
|
+
|
|
158
|
+
Admission delays themselves are not modelled, so projected finish times remain
|
|
159
|
+
approximate while workers wait for slots.
|
|
160
|
+
|
|
161
|
+
## Driving xdist
|
|
162
|
+
|
|
163
|
+
`DurationScheduling` subclasses xdist's `LoadScheduling` and keeps its lifecycle
|
|
164
|
+
(collection check, crash handling, restarts, shutdown), replacing what goes out and
|
|
165
|
+
when. Two constraints of xdist's worker protocol shape it:
|
|
166
|
+
|
|
167
|
+
- a worker runs a test only once it knows the following one, or that there is none
|
|
168
|
+
(it needs `nextitem` for fixture teardown). A queue's last test waits for another
|
|
169
|
+
test or a shutdown command. CPU admission deliberately uses that wait to keep a
|
|
170
|
+
test from starting until slots are available;
|
|
171
|
+
- a worker completes tests in the order it received them, and the controller only
|
|
172
|
+
learns about completions after the fact.
|
|
173
|
+
|
|
174
|
+
Dispatch charges each test with exactly what the lane's cost function says (test plus
|
|
175
|
+
set-ups it forces), and completion refunds exactly that, so the projection of when a
|
|
176
|
+
worker will be free stays honest. A worker projected to be done by now is late and is
|
|
177
|
+
assumed busy with all its dispatched work from now; one still projected to be busy is
|
|
178
|
+
never assumed free before it has run the tests it has not started yet, however long
|
|
179
|
+
the running one is taking.
|
|
180
|
+
|
|
181
|
+
**`--dist load`.** Once every worker has collected, the planner lays out a lane per
|
|
182
|
+
worker. Each worker gets its next tests from its own plan: at least two, plus a small
|
|
183
|
+
time budget of estimated seconds so runs of trivial tests are batched. The budget is
|
|
184
|
+
a tenth of a second at most and shrinks for short suites. Refills happen with
|
|
185
|
+
hysteresis (below half the budget, filled to the whole budget) so tiny tests do not
|
|
186
|
+
cost a command each. A worker whose plan runs dry takes from other lanes' unsent
|
|
187
|
+
plans through `transfer`, which accepts moves that do not worsen the predicted
|
|
188
|
+
finish. An idle worker can shut down when no suitable work remains; admission may
|
|
189
|
+
keep it waiting with its fixtures alive. Everything not yet sent stays on the
|
|
190
|
+
controller, where it can still move.
|
|
191
|
+
|
|
192
|
+
**`--dist worksteal`.** Without CPU admission, each worker's whole plan is dispatched
|
|
193
|
+
in one command, as xdist's own work-stealing scheduler does. With admission, dispatch
|
|
194
|
+
is segmented at demand increases as described below. A worker down to its last test
|
|
195
|
+
is put on a waiting list, and the controller picks a donor and a tail chunk using the
|
|
196
|
+
same cost function as `transfer`. Candidates are the last family in the queue, half
|
|
197
|
+
that family, and a chunk sized to the dispatch time budget (or the remaining family
|
|
198
|
+
if shorter). Chunks smaller than that batch candidate must shorten the run; larger
|
|
199
|
+
ones may leave the prediction unchanged. Ties favor balancing donor and thief.
|
|
200
|
+
The chunk is requested through xdist's `steal` command, which is all or nothing from
|
|
201
|
+
3.7 on: the worker refuses
|
|
202
|
+
unless it still holds every requested test. The running test, the one after it, and
|
|
203
|
+
one more as a margin for reports in flight are never requested. One steal is in
|
|
204
|
+
flight at a time; a refusal (the donor had moved on) leads to a fresh choice against
|
|
205
|
+
the updated bookkeeping, and a receiver that would be left with a single test asks
|
|
206
|
+
again or is shut down so it can run it. `tests_finished` stays false while a steal is
|
|
207
|
+
in flight.
|
|
208
|
+
|
|
209
|
+
**Restarts.** When a worker dies, its dispatched-but-unrun tests go back to the pool
|
|
210
|
+
and are planned over the surviving lanes together with its unsent plan. If no worker
|
|
211
|
+
is left, the tests stay pending without a lane and are planned onto the replacement
|
|
212
|
+
when it joins.
|
|
213
|
+
|
|
214
|
+
## CPU demand
|
|
215
|
+
|
|
216
|
+
A test's declared weight comes from the closest `timing_cpu` marker (function, class,
|
|
217
|
+
module); an invalid marker counts as one slot and fails the test at set-up with the
|
|
218
|
+
reason, so a typo cannot silently change the schedule. A fixture's demand is an
|
|
219
|
+
attribute the `pytest_timing.cpu` decorator sets on the fixture function, in either
|
|
220
|
+
decorator order (pytest 8.4 wraps fixture functions in a definition object; both get
|
|
221
|
+
the attribute). The decorator stores metadata directly on the fixture rather than
|
|
222
|
+
applying a pytest mark to it.
|
|
223
|
+
|
|
224
|
+
The controller never sees items, only node ids, and xdist's collection report carries
|
|
225
|
+
nothing else. So after collecting, every worker builds a `Declarations` record (slots
|
|
226
|
+
per item, the keys of CPU-declared shared fixtures in each item's static closure,
|
|
227
|
+
each such fixture's set-up and hold slots, and the host's detected CPU environment)
|
|
228
|
+
and sends it as one custom event on its channel, ahead of xdist's own collection
|
|
229
|
+
event. xdist raises on unknown events, so `pytest_configure_node` wraps each node's
|
|
230
|
+
`process_from_remote` on the instance, before the channel callback is registered, to
|
|
231
|
+
take that one event off the stream. The handler runs on execnet's receiver thread and
|
|
232
|
+
only stores the record under a lock; the scheduler reads it when it lays out the run,
|
|
233
|
+
which happens after the collection event that follows it on the same channel.
|
|
234
|
+
|
|
235
|
+
`Costs` folds the declarations into the cost model: `slots` per test, set-up and hold
|
|
236
|
+
slots per fixture definition (keyed without the parameter index), and declared
|
|
237
|
+
fixtures added to their tests' families even before any run has timed them. The
|
|
238
|
+
`demand` of a test on a worker is the most it may need at any point there, its own
|
|
239
|
+
slots or a set-up it has to pay, plus the holds of the fixtures alive around it. A
|
|
240
|
+
function-scoped fixture's declaration is folded into its tests' own slots, since its
|
|
241
|
+
set-up runs inside the test's reservation: the test weighs its heaviest stage, each
|
|
242
|
+
such set-up next to what the other function-scoped fixtures may hold by then, or
|
|
243
|
+
the call with all of them alive. Fixtures no signature names (requested with
|
|
244
|
+
`getfixturevalue`) are not in any collected family; the worker still sends every
|
|
245
|
+
declared shared fixture definition it knows (`<scope>:<defined at>::<name>`), so a
|
|
246
|
+
family recorded by an earlier run and loaded through `--timing-schedule` finds the
|
|
247
|
+
demand of such a fixture by definition. Without a recorded run the planner cannot
|
|
248
|
+
see them, and admission happens at run time instead (next section).
|
|
249
|
+
|
|
250
|
+
## Admission
|
|
251
|
+
|
|
252
|
+
xdist's worker protocol is the gate. A worker runs a test only once it has been told
|
|
253
|
+
the following one, or been shut down. So every test sent to a worker is *queued* but
|
|
254
|
+
not *granted* until that next message goes out, and the controller sends it only when
|
|
255
|
+
the worker's reservation covers everything the worker could then run without another
|
|
256
|
+
word: the tests before the tail, or the whole queue after a shutdown. Ordinary test
|
|
257
|
+
admission uses existing xdist commands; runtime fixture requests and shutdown
|
|
258
|
+
cancellation use the custom messages described below. Pytest's `nextitem` teardown
|
|
259
|
+
semantics are preserved for executed tests.
|
|
260
|
+
|
|
261
|
+
The states of a test:
|
|
262
|
+
|
|
263
|
+
- *planned* on the controller, in a lane's unsent plan;
|
|
264
|
+
- *sent*, the ungranted tail of a worker's queue;
|
|
265
|
+
- *granted*, once the message after it went out; the worker's reservation covers it;
|
|
266
|
+
- *waiting*, when it is the head of its worker's queue and the reservation cannot be
|
|
267
|
+
raised to cover it: the worker blocks in xdist's queue with fixtures alive, and
|
|
268
|
+
`tests_finished` stays false;
|
|
269
|
+
- *parked*, when it is next in a worker's plan but could never fit next to what the
|
|
270
|
+
other workers' fixtures hold: it stays on the controller, the worker runs what
|
|
271
|
+
else it has or waits idle, and the test is tried again at every release and exit;
|
|
272
|
+
- *completed* at `runtest_protocol_complete`, after teardown, when the reservation is
|
|
273
|
+
recomputed from what is left;
|
|
274
|
+
- *cancelled* when shutdown withdraws an unadmitted test (see below). Steals return
|
|
275
|
+
tests to a plan; worker death reports a started test as crashed and requeues the
|
|
276
|
+
unstarted tail.
|
|
277
|
+
|
|
278
|
+
A worker's reservation covers the largest recorded peak among its granted tests,
|
|
279
|
+
plus live fixture holds not already included in that peak, or only the holds while
|
|
280
|
+
it is idle. Raising it uses admission; lowering it always goes through. A rise in
|
|
281
|
+
demand behind lighter tests is deferred: the batch stops before the message that
|
|
282
|
+
would grant it, the worker runs down to it, and the grant is tried when it is the head. So in
|
|
283
|
+
`--dist worksteal` a plan is dispatched in segments that end at the next rise, and a
|
|
284
|
+
queue never reserves for its heaviest member while lighter ones run. Stolen tests go
|
|
285
|
+
to the front of the taker's plan and out through the same gate.
|
|
286
|
+
|
|
287
|
+
The waiting line (`Admission`) is ordered by arrival. The head's shortfall is pledged:
|
|
288
|
+
other workers may raise their reservations only out of what is left, or when the work
|
|
289
|
+
behind the request is projected to end before the head could have had its slots
|
|
290
|
+
anyway, computed from the projected release times of the reservations in the way
|
|
291
|
+
(EASY backfilling). A request above the limit is clamped to the whole limit and
|
|
292
|
+
reported, so an oversized declaration can still run.
|
|
293
|
+
|
|
294
|
+
Fixtures can keep slots occupied while their worker waits, so they need their own
|
|
295
|
+
rule. Before a test is dispatched, its demand on that worker is checked against
|
|
296
|
+
what the other workers of the domain hold; if it could never fit while they live it
|
|
297
|
+
is passed over (parked) and the worker takes the next test of its plan, or waits idle
|
|
298
|
+
when there is none. Ordinary admission respects the limit. After every retry of
|
|
299
|
+
the waiting line the scheduler looks for a domain at a standstill, where nothing
|
|
300
|
+
runs and nobody in line fits, and decides what gives: a parked worker is shut down
|
|
301
|
+
first, so its exit lets go of its own holds and its plan is handed to the workers
|
|
302
|
+
that stay at that moment (a worker told to shut down takes nothing more); failing
|
|
303
|
+
that, the oldest queued head or runtime request is forced, unless a draining worker
|
|
304
|
+
can still release the needed holds. Forced admission can exceed the limit and is
|
|
305
|
+
counted in the summary. Budgets are per resource domain: `local` for popen workers,
|
|
306
|
+
the host named in the execnet spec otherwise. An explicit integer sets each domain's
|
|
307
|
+
budget; `auto` uses the smallest budget detected by its workers. When declarations
|
|
308
|
+
enable admission without an explicit budget, the detected budget is raised to at
|
|
309
|
+
least the domain's worker count. This floor does not prevent heavy tests or fixture
|
|
310
|
+
holds from delaying plain tests.
|
|
311
|
+
|
|
312
|
+
When xdist shuts every worker down itself (`-x`, `--maxfail`, the restart budget for
|
|
313
|
+
crashed workers), it calls each node's `shutdown` directly. The scheduler wraps that
|
|
314
|
+
method on every node it is given, so the call passes through it first: what the
|
|
315
|
+
worker holds is granted if it fits, and otherwise the tail of its queue, the one test
|
|
316
|
+
the worker cannot start without another word, is withdrawn with a `timing_cancel`
|
|
317
|
+
command, and the worker-side plugin skips that test's protocol. The worker takes
|
|
318
|
+
commands of ours off xdist's command dispatcher on its receiver thread, wrapped per
|
|
319
|
+
instance like the event on the controller. A worker that dies while holding a single
|
|
320
|
+
unadmitted test had not started it: the test goes back to the pool rather than being
|
|
321
|
+
reported as crashed.
|
|
322
|
+
|
|
323
|
+
A worker blocked inside a runtime fixture request has already started its test;
|
|
324
|
+
if it dies there, the test is reported as crashed rather than silently replayed.
|
|
325
|
+
|
|
326
|
+
A fixture reached only at run time may be absent from the planned reservation.
|
|
327
|
+
Before setting up a dynamic fixture that declares multiple slots or a hold, the
|
|
328
|
+
worker-side meter sends a `timing_request` event (test, fixture key, request id,
|
|
329
|
+
set-up and hold slots) and blocks
|
|
330
|
+
until a `timing_grant` command names that request. Request ids are unique within the
|
|
331
|
+
worker, so a late grant cannot satisfy a retry of the same fixture. On the controller
|
|
332
|
+
the event is re-posted as an in-process event, so the run loop handles it on the main
|
|
333
|
+
thread like xdist's own. The scheduler treats it as a rise: while it waits the worker
|
|
334
|
+
runs nothing, so its reservation drops to what its fixtures hold. The request
|
|
335
|
+
reserves the set-up's slots next to those, through the same gate, waiting line and backfilling as
|
|
336
|
+
a queued test; the instance then counts as alive on that worker, its hold added to
|
|
337
|
+
everything queued there. Two workers asking at once therefore take turns instead of
|
|
338
|
+
deadlocking on the slots of the tests they are blocked in. Each request also carries
|
|
339
|
+
the total holds of all live fixtures. The worker tracks successful setup and
|
|
340
|
+
finalization at every scope, also sending `timing_holds` events on those transitions.
|
|
341
|
+
These actual lifetimes are separate from `Lane.fixtures`, which projects through
|
|
342
|
+
prefetched tests and can already describe a different scope or
|
|
343
|
+
parameter. Failed setup adds no live hold; a granted setup is not yet a live instance.
|
|
344
|
+
Function-scoped holds end with the current test and never become shared fixture
|
|
345
|
+
costs. A draining shutdown keeps a blocked test behind the gate, but a live draining
|
|
346
|
+
worker's request still participates in deadlock recovery. Workers that can release
|
|
347
|
+
holds by finishing are allowed to do so before forcing another request. Worker exit
|
|
348
|
+
releases its reservation even when xdist skips scheduler removal after maxfail, so a request
|
|
349
|
+
cannot remain blocked by a departed worker. After five minutes a blocked request
|
|
350
|
+
cancels its fixture setup and asks to recover the test's original reservation.
|
|
351
|
+
Only after that grant does it fail and unwind pytest's incomplete fixture instance;
|
|
352
|
+
cleanup and retries therefore retain a reservation and valid fixture bookkeeping.
|
|
353
|
+
Collection includes declarations on dynamic-only fixtures, including function
|
|
354
|
+
fixtures, when deciding to enable the gate. Declarations alone do not select this
|
|
355
|
+
scheduler: a usable history file or an explicit CPU budget is also required.
|
|
356
|
+
|
|
357
|
+
The scheduler records how long each head waited. Runtime requests and phase reports
|
|
358
|
+
carry a private collection index and attempt token, so the plugin sums waits onto
|
|
359
|
+
the exact execution's `cpu.wait`, including repeated requests, retries and duplicate
|
|
360
|
+
selections. A wait before execution belongs to the first attempt. If the worker dies
|
|
361
|
+
before sending any phase report, its unreported crash retains that wait. The worker
|
|
362
|
+
also records the part inside the test span
|
|
363
|
+
as `runtime_wait`. All open fixture clocks pause during that wait, and test estimates
|
|
364
|
+
subtract it along with shared setup. CPU `elapsed` and `work` exclude the wait and
|
|
365
|
+
its process-tree CPU work, so the measured rate covers execution. Older records
|
|
366
|
+
without `runtime_wait` default to zero; pre-start waits are not subtracted again.
|
|
367
|
+
Event handlers are registered before each worker starts, even when another plugin
|
|
368
|
+
selects the scheduler: without an admission gate a request is granted immediately.
|
|
369
|
+
The cgroup quota behind `auto` is read from the process's own group, walking up to
|
|
370
|
+
the mount for the tightest limit, on cgroup v2 (`cpu.max`) and v1 (`cpu.cfs_quota_us`).
|
|
371
|
+
Which hierarchy holds the `cpu` controller decides: on a hybrid system a v1 line
|
|
372
|
+
naming it in `/proc/self/cgroup` wins over the `0::` v2 line, whose hierarchy has no CPU
|
|
373
|
+
controller there. Throttling is read from the same group's `cpu.stat` (`throttled_usec`
|
|
374
|
+
on v2, `throttled_time` in nanoseconds on v1).
|
|
375
|
+
|
|
376
|
+
## Memory admission
|
|
377
|
+
|
|
378
|
+
Memory goes through the same gate as CPU, as a second `Admission` per domain counted
|
|
379
|
+
in bytes, and a test is granted only when its reservation fits both: `_raise` checks
|
|
380
|
+
every gate with `fits` first and reserves on all of them or none, and a refused worker
|
|
381
|
+
waits in every line with its respective need. Reservations, releases, forced
|
|
382
|
+
admissions, withdrawals and stall resolution act on both gates in step, so their
|
|
383
|
+
`busy` and `idle` states never disagree. The rules that let CPU admission exceed its
|
|
384
|
+
limit are safe for memory for the same reasons they are safe for CPU: backfilling
|
|
385
|
+
never exceeds the limit (it lends out slots pledged to the head), a request above the
|
|
386
|
+
limit is clamped so the test runs alone rather than never, and a forced admission
|
|
387
|
+
happens only when nothing runs in the domain, so nothing else's memory is at risk.
|
|
388
|
+
Pressure feedback moves only the CPU limit.
|
|
389
|
+
|
|
390
|
+
Demand comes from history, never from declarations. `Estimates.from_run` keeps, per
|
|
391
|
+
test, the largest `peak - base` any attempt showed (the worst case is what an
|
|
392
|
+
out-of-memory kill depends on), and per shared fixture key the memory the attempt that
|
|
393
|
+
paid its set-up kept resident (`after - base`, split evenly when one attempt paid for
|
|
394
|
+
several). `Costs.charge` adds a `memory` to each `Charge`: the test's rise plus what
|
|
395
|
+
the fixtures alive around it keep, projected through the lane's fixture state exactly
|
|
396
|
+
as CPU holds are, since workers report nothing about memory at run time. A fixture with
|
|
397
|
+
recorded memory belongs to its tests' families even when its set-up was too quick to
|
|
398
|
+
matter for time. `_need_memory` is the twin of `_need`; an idle worker reserves what
|
|
399
|
+
its fixtures keep, and a worker whose next test could never fit next to what the
|
|
400
|
+
other workers' fixtures keep is parked like one blocked by CPU holds. A fixture
|
|
401
|
+
reached at run time adds its recorded memory to the queued charges when its
|
|
402
|
+
request is granted.
|
|
403
|
+
|
|
404
|
+
The budget is `--timing-memory`: a size, or `auto`, which takes 80% of the smallest
|
|
405
|
+
memory the domain's workers report (physical memory capped by the cgroup limit,
|
|
406
|
+
`memory.max` on v2 and `memory.limit_in_bytes` on v1, read like the CPU quota). The
|
|
407
|
+
share is deliberate: estimates are rises above each worker's footprint, and the
|
|
408
|
+
footprints, the controller and the rest of the host are not in them. Without a
|
|
409
|
+
recorded run every estimate is zero and the gate admits everything; the summary says
|
|
410
|
+
so. The planner ignores memory: with tests kept apart by the gate, balancing lanes by
|
|
411
|
+
memory would add little, and the lane plan can still move.
|
|
412
|
+
|
|
413
|
+
## Measuring CPU work
|
|
414
|
+
|
|
415
|
+
`ProcessTreeClock` uses `os.times()` for worker CPU time and, on POSIX, reaped child
|
|
416
|
+
CPU time. Where available, it adds live descendants through
|
|
417
|
+
`/proc/<pid>/task/*/children` and `/proc/<pid>/stat` on Linux, through `libproc` on
|
|
418
|
+
macOS (`proc_listchildpids` to find them, `proc_pidinfo` for their task times, in
|
|
419
|
+
Mach time units converted with `mach_timebase_info`), or psutil otherwise. psutil is
|
|
420
|
+
the last resort because its `children` scans the whole process table, about ten
|
|
421
|
+
milliseconds on macOS, and the clock is read several times per test: with psutil
|
|
422
|
+
installed, 3,000 trivial tests went from under a second to over a minute.
|
|
423
|
+
Readings account for a waited-for child moving from the live total into the reaped
|
|
424
|
+
total. Process discovery is a snapshot, so exits during traversal or descendants
|
|
425
|
+
that outlive or detach from their parents can leave gaps. Every record reports its
|
|
426
|
+
coverage: `tree`, `reaped` (worker plus waited-for children), or `self` (worker only,
|
|
427
|
+
including Windows without psutil).
|
|
428
|
+
|
|
429
|
+
The meter takes a reading at the start of setup and when the teardown report is made,
|
|
430
|
+
and the fixture timer takes readings around shared set-ups, so a record carries the
|
|
431
|
+
test's total `work` and the `setup_work` inside set-ups charged to it. Incomplete
|
|
432
|
+
tests without a teardown report may have no CPU record. Each record also carries
|
|
433
|
+
the host's PSI `some` share and whether the cgroup was throttled during
|
|
434
|
+
the test; `Estimates.from_run` treats such attempts as contended and prefers a clean
|
|
435
|
+
attempt of the same test when one exists, keeping the contended ones in the file.
|
|
436
|
+
|
|
437
|
+
## Measuring memory
|
|
438
|
+
|
|
439
|
+
Memory is recorded so that a later run can keep tests that need a lot of it from
|
|
440
|
+
running at the same time; nothing schedules on it yet. `ResidentMemory` reads the
|
|
441
|
+
resident set size of the process running the tests and, where they can be listed, of
|
|
442
|
+
its live descendants: `/proc/self/statm` and the `/proc` walk shared with CPU
|
|
443
|
+
measurement on Linux, the `libproc` reader shared with the CPU clock on macOS,
|
|
444
|
+
`GetProcessMemoryInfo` on Windows for the worker alone. psutil fills in what the
|
|
445
|
+
platform readers cannot do, today descendants on Windows, and is never preferred
|
|
446
|
+
over a native reader. Descendants are summed, so pages they share count more than
|
|
447
|
+
once; the total errs on the large side.
|
|
448
|
+
|
|
449
|
+
A high-water mark such as `ru_maxrss` never comes down, so it cannot say what one
|
|
450
|
+
test needed: only the test that first raised the worker's peak would show anything.
|
|
451
|
+
`MemorySampler` therefore polls from a daemon thread while a window is open and
|
|
452
|
+
keeps the highest total. The worker's own size is read every 20 ms; it costs about a
|
|
453
|
+
microsecond on macOS and ten on Linux. Descendants are listed at most every 100 ms,
|
|
454
|
+
and while none are found the interval doubles up to a second, since listing costs an
|
|
455
|
+
order of magnitude more (and far more with psutil). Opening or closing a window never
|
|
456
|
+
lists descendants by itself, so a fast test costs two readings of its own process, a
|
|
457
|
+
few microseconds. The meter opens the window at `pytest_runtest_setup` and closes it
|
|
458
|
+
when the teardown report is made, so shared fixture set-ups are charged to the test
|
|
459
|
+
that paid for them, as their time is. The window goes on the teardown report as
|
|
460
|
+
`timing_memory` and into the JSON as `memory`: `base`, `peak` and `after` in bytes,
|
|
461
|
+
and the coverage (`tree` or `self`). A platform with no reading records nothing.
|
|
462
|
+
The sampler sleeps between windows and is closed at `pytest_unconfigure`.
|
|
463
|
+
|
|
464
|
+
A test shorter than the sampling interval is seen only at its edges: its record is
|
|
465
|
+
what was resident before and after it, and a buffer allocated and freed inside it
|
|
466
|
+
is missed. Faulting in enough memory to matter takes longer than one interval.
|
|
467
|
+
|
|
468
|
+
`peak - base` is the attempt's rise: what it needed on top of the worker's footprint.
|
|
469
|
+
`after - base` is what stayed resident, which for the first test of a session fixture
|
|
470
|
+
is roughly the fixture. Both are biased by allocator behaviour: a heap that already
|
|
471
|
+
grew for an earlier test can serve a later one without raising RSS, so a rise can
|
|
472
|
+
undercount a test whose allocations reuse freed heap, and a freed buffer the allocator
|
|
473
|
+
keeps can leave `after` high. Large buffers and subprocess memory, the usual causes of
|
|
474
|
+
an out-of-memory kill, are mapped and unmapped directly and measure well.
|
|
475
|
+
|
|
476
|
+
## Feedback
|
|
477
|
+
|
|
478
|
+
Admission's pressure feedback moves a domain's limit, never its budget, on evidence
|
|
479
|
+
of contention: increasing cgroup quota throttling, or a PSI `some` share at or above
|
|
480
|
+
25% while the run's own measured CPU rate is below three quarters of the limit. That
|
|
481
|
+
rate sums the latest measured rate of each local worker currently running a test;
|
|
482
|
+
idle, departed and remote workers contribute nothing. High PSI alone does not lower
|
|
483
|
+
the limit when the run's own measured throughput is high enough.
|
|
484
|
+
Three contended samples in a row lower the limit by one slot; eight clean
|
|
485
|
+
samples in a row raise it by one, up to the budget; a ten-second cooldown separates
|
|
486
|
+
moves; samples are taken at most once a second, on the controller's host only, and
|
|
487
|
+
only when the host has the signals. Running tests are never touched: a lower limit
|
|
488
|
+
defers the next admissions. Without PSI or cgroup statistics the limit stays at the
|
|
489
|
+
budget.
|
|
490
|
+
|
|
491
|
+
## Evaluation
|
|
492
|
+
|
|
493
|
+
`tests/test_schedule.py` covers estimates, fixture lifetimes, planning, rebalancing
|
|
494
|
+
and gated dispatch with fake workers, plus pytester runs with real workers.
|
|
495
|
+
`tests/test_admission.py` exercises reservations, the waiting line, backfilling and
|
|
496
|
+
pressure feedback. `tests/test_cpu.py` runs subprocess workloads, cold fixture
|
|
497
|
+
set-ups, held slots, crashes and `-x`; `tests/test_runtime_admission.py` covers dynamic
|
|
498
|
+
fixtures, runtime waits, cancellation, retries and shutdown recovery.
|
|
499
|
+
|
|
500
|
+
`benchmarks/cpu_bench.py` generates suites of tiny tests, single-CPU tests, internally
|
|
501
|
+
parallel tests with and without deadlines, mixed weights, an expensive session
|
|
502
|
+
fixture and background load. It compares configurations using whole-process wall
|
|
503
|
+
time (including pytest startup and reporting), slowest heavy test, failures, fixture
|
|
504
|
+
set-ups, throttling and pressure. For configurations producing a timing report, its
|
|
505
|
+
JSON also includes pytest session time; otherwise that field uses process wall
|
|
506
|
+
time. Results depend on the machine and load; the script can also run under a
|
|
507
|
+
container CPU quota.
|