pytest-timing 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/.github/workflows/ci.yml +9 -8
  2. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/.github/workflows/release.yml +2 -2
  3. pytest_timing-0.2.0/ARCHITECTURE.md +423 -0
  4. pytest_timing-0.2.0/CHANGELOG.md +60 -0
  5. pytest_timing-0.2.0/PKG-INFO +324 -0
  6. pytest_timing-0.2.0/README.md +298 -0
  7. pytest_timing-0.2.0/benchmarks/cpu_bench.py +376 -0
  8. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/examples/test_demo.py +16 -1
  9. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/pyproject.toml +3 -3
  10. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/__init__.py +5 -1
  11. pytest_timing-0.2.0/src/pytest_timing/admission.py +210 -0
  12. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/collector.py +35 -8
  13. pytest_timing-0.2.0/src/pytest_timing/demand.py +458 -0
  14. pytest_timing-0.2.0/src/pytest_timing/fixtures.py +229 -0
  15. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/model.py +93 -13
  16. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/plugin.py +288 -26
  17. pytest_timing-0.2.0/src/pytest_timing/schedule.py +518 -0
  18. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/static/report.html +0 -10
  19. pytest_timing-0.2.0/src/pytest_timing/telemetry.py +346 -0
  20. pytest_timing-0.2.0/src/pytest_timing/xdist_compat.py +337 -0
  21. pytest_timing-0.2.0/src/pytest_timing/xdist_scheduler.py +988 -0
  22. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/tests/conftest.py +50 -0
  23. pytest_timing-0.2.0/tests/test_admission.py +194 -0
  24. pytest_timing-0.2.0/tests/test_benchmarks.py +22 -0
  25. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/tests/test_cli.py +0 -2
  26. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/tests/test_collector.py +26 -4
  27. pytest_timing-0.2.0/tests/test_cpu.py +789 -0
  28. pytest_timing-0.2.0/tests/test_fixtures.py +362 -0
  29. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/tests/test_plugin.py +1 -15
  30. pytest_timing-0.2.0/tests/test_runtime_admission.py +712 -0
  31. pytest_timing-0.2.0/tests/test_schedule.py +1407 -0
  32. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/uv.lock +3 -3
  33. pytest_timing-0.1.0/CHANGELOG.md +0 -10
  34. pytest_timing-0.1.0/PKG-INFO +0 -161
  35. pytest_timing-0.1.0/README.md +0 -135
  36. pytest_timing-0.1.0/src/pytest_timing/xdist_compat.py +0 -158
  37. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/.gitignore +0 -0
  38. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/LICENSE +0 -0
  39. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/docs/report.png +0 -0
  40. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/cli.py +0 -0
  41. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/outputs.py +0 -0
  42. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/render/__init__.py +0 -0
  43. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/render/ascii.py +0 -0
  44. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/render/html.py +0 -0
  45. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/render/trace.py +0 -0
  46. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/src/pytest_timing/static/__init__.py +0 -0
  47. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/tests/fixtures/legacy_complete_false.json +0 -0
  48. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/tests/fixtures/legacy_complete_true.json +0 -0
  49. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/tests/fixtures/legacy_no_flag.json +0 -0
  50. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/tests/test_ascii.py +0 -0
  51. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/tests/test_html.py +0 -0
  52. {pytest_timing-0.1.0 → pytest_timing-0.2.0}/tests/test_trace.py +0 -0
@@ -7,7 +7,7 @@ on:
7
7
 
8
8
  jobs:
9
9
  test:
10
- name: py${{ matrix.python }} / pytest ${{ matrix.pytest }}
10
+ name: ${{ matrix.os }} / py${{ matrix.python }} / pytest ${{ matrix.pytest }}${{ matrix.xdist && format(' / xdist {0}', matrix.xdist) || '' }}
11
11
  runs-on: ${{ matrix.os }}
12
12
  strategy:
13
13
  fail-fast: false
@@ -16,14 +16,14 @@ jobs:
16
16
  python: ["3.10", "3.11", "3.12", "3.13", "3.14"]
17
17
  pytest: ["latest"]
18
18
  include:
19
- - { os: ubuntu-latest, python: "3.10", pytest: "7.3" }
19
+ - { os: ubuntu-latest, python: "3.10", pytest: "7.3", xdist: "3.7.0" }
20
20
  - { os: ubuntu-latest, python: "3.11", pytest: "7.4" }
21
21
  - { os: ubuntu-latest, python: "3.12", pytest: "8.0" }
22
22
  - { os: windows-latest, python: "3.12", pytest: "latest" }
23
23
  - { os: macos-latest, python: "3.13", pytest: "latest" }
24
24
  steps:
25
- - uses: actions/checkout@v4
26
- - uses: astral-sh/setup-uv@v5
25
+ - uses: actions/checkout@v7
26
+ - uses: astral-sh/setup-uv@v10.1.0
27
27
  with:
28
28
  python-version: ${{ matrix.python }}
29
29
  - name: Install
@@ -31,6 +31,7 @@ jobs:
31
31
  uv sync --no-dev
32
32
  uv pip install pytest-xdist pytest-rerunfailures hypothesis
33
33
  if [ "${{ matrix.pytest }}" != "latest" ]; then uv pip install "pytest~=${{ matrix.pytest }}.0"; fi
34
+ if [ -n "${{ matrix.xdist }}" ]; then uv pip install "pytest-xdist==${{ matrix.xdist }}"; fi
34
35
  shell: bash
35
36
  - name: Test
36
37
  run: uv run --no-sync pytest -v -p no:cacheprovider
@@ -39,8 +40,8 @@ jobs:
39
40
  name: without pytest-xdist
40
41
  runs-on: ubuntu-latest
41
42
  steps:
42
- - uses: actions/checkout@v4
43
- - uses: astral-sh/setup-uv@v5
43
+ - uses: actions/checkout@v7
44
+ - uses: astral-sh/setup-uv@v10.1.0
44
45
  with:
45
46
  python-version: "3.12"
46
47
  - run: uv sync --no-dev
@@ -50,8 +51,8 @@ jobs:
50
51
  lint:
51
52
  runs-on: ubuntu-latest
52
53
  steps:
53
- - uses: actions/checkout@v4
54
- - uses: astral-sh/setup-uv@v5
54
+ - uses: actions/checkout@v7
55
+ - uses: astral-sh/setup-uv@v10.1.0
55
56
  with:
56
57
  python-version: "3.12"
57
58
  - run: uv sync
@@ -11,7 +11,7 @@ jobs:
11
11
  permissions:
12
12
  id-token: write
13
13
  steps:
14
- - uses: actions/checkout@v4
15
- - uses: astral-sh/setup-uv@v5
14
+ - uses: actions/checkout@v7
15
+ - uses: astral-sh/setup-uv@v10.1.0
16
16
  - run: uv build
17
17
  - run: uv publish
@@ -0,0 +1,423 @@
1
+ # Architecture
2
+
3
+ How pytest-timing captures timings, times shared fixtures inside xdist workers, and
4
+ schedules a run from the previous one. For usage, see the [README](README.md).
5
+
6
+ ## Modules
7
+
8
+ | Module | Role |
9
+ |---|---|
10
+ | `plugin.py` | pytest hooks on the controller: options, settings, capture, terminal summary, outputs, and the `pytest_xdist_make_scheduler` hook. |
11
+ | `collector.py` | Folds phase reports and worker events into a `Run`. Knows nothing about pytest objects. |
12
+ | `model.py` | The recorded run: `Run`, `Worker`, `TestSpan`, `Phase`, and the JSON representation. |
13
+ | `outputs.py` | The ASCII, HTML and trace renderers, in one table shared by the plugin and the CLI. |
14
+ | `cli.py` | `pytest-timing render` and `pytest-timing merge`. |
15
+ | `fixtures.py` | Times shared fixture set-up where tests run and lists dependencies on reports. |
16
+ | `schedule.py` | Duration estimates, the fixture-aware cost model and the planner. Free of xdist. |
17
+ | `demand.py` | Declared CPU demand: the `timing_cpu` marker, the fixture decorator, the declarations a worker sends, and the meter that records each test's CPU work. |
18
+ | `admission.py` | One host's CPU budget, reservations, fair waiting line and pressure feedback. Free of xdist and platform reads. |
19
+ | `telemetry.py` | Platform reads: affinity and cgroup quota, CPU time of a process tree, pressure and throttling. |
20
+ | `xdist_scheduler.py` | The xdist scheduler that drives workers with the planner and the admission gate, in `load` and `worksteal` mode. |
21
+ | `xdist_compat.py` | Everything that reads pytest-xdist internals (versions, worker detection, why the controller stopped, the custom worker event and controller command). |
22
+
23
+ ## Capturing timings
24
+
25
+ Since pytest 7.3 every `TestReport` carries wall-clock `start` and `stop` timestamps.
26
+ xdist serialises reports to the controller unchanged and attaches the worker to each,
27
+ so phase aggregation uses controller-side hooks: `pytest_runtest_logreport` for the
28
+ setup / call / teardown phases, plus xdist's node-ready, collection-finished and
29
+ node-down hooks for the worker lifecycle. CPU admission adds custom events and commands
30
+ through the xdist adapter; timing reports use pytest's normal report transport.
31
+
32
+ Each lane in the report shows boot (worker start-up until it is ready), collection,
33
+ tests, idle gaps and the point the worker shut down. Session-scoped fixture set-up is
34
+ attributed to the first test that needs that instance on each worker and its teardown
35
+ to the test that finalizes it, exactly as pytest reports it.
36
+
37
+ Timing collection is enabled by `--timing`, an output, scheduling or CPU-budget
38
+ setting. Formatting settings alone do not enable it. The marker is always registered;
39
+ when enabled, an xdist worker also registers the fixture timer and CPU meter.
40
+
41
+ ## Timing shared fixtures in the workers
42
+
43
+ pytest charges a class, module, package or session fixture's set-up to the setup phase
44
+ of the first test that needs it on each worker, so a test's recorded duration depends
45
+ on where it happened to land. The scheduler needs the two apart: how long the test
46
+ itself takes, and which shared fixtures it needs at what price.
47
+
48
+ `fixtures.py` wraps `pytest_fixture_setup` for the shared scopes and times each
49
+ set-up, excluding any set-up nested inside it through `getfixturevalue` (that one is
50
+ timed on its own). When a test's setup and call reports are built, the timer attaches
51
+ `timing_fixtures` to them: a mapping from fixture key to the seconds this test spent
52
+ setting the fixture up, or `None` when the test only uses it. Extra report attributes
53
+ survive xdist's serialisation, and the collector merges the two reports into
54
+ `TestSpan.fixtures`.
55
+
56
+ A test's fixtures are gathered from three sources, because none is complete alone:
57
+
58
+ - the static fixture closure, which names what the signatures ask for;
59
+ - the request's resolved fixture definitions, which add what `getfixturevalue` pulled
60
+ in for this test, including inside the test body (hence the call report);
61
+ - a per-worker memo of what each shared fixture's set-up requested dynamically. A
62
+ test served from the cache never sees those requests, but it depends on them just
63
+ the same.
64
+
65
+ A fixture key is `<scope>:<scope node id>:<defined at>::<name>[<params>]`. The
66
+ scope node is what pytest shares the instance under: the session (empty), the module,
67
+ the class, or the matching package collector. A package fixture without a matching
68
+ collector belongs to the session. "Defined
69
+ at" tells overrides of the same name apart: the fixture's `baseid` (its conftest
70
+ directory or test module), or, for a plugin's fixture, whose `baseid` is empty just
71
+ like a root conftest's on pytest 7, the defining module and qualified function name.
72
+ This also distinguishes a plugin class registered from that same conftest. The parameter part,
73
+ taken from the item's
74
+ callspec, tells the instances of a parametrized fixture apart whether the fixture or
75
+ the test (`indirect=True`) carries the parameters, since pytest keeps only one
76
+ instance alive at a time: `base[1]`. A fixture that depends on a parametrized one,
77
+ directly or through other fixtures, is torn down with the instance it was built on
78
+ and built again for the next parameter, so those indices are part of its key too:
79
+ `derived[base=1]`, or `mixed[0,base=1]` for a fixture with parameters of its own. It
80
+ is an index, not a value, so two modules that parametrize the same session fixture
81
+ with different values at the same index share a key; the planner is slightly
82
+ optimistic there.
83
+
84
+ ## Estimates
85
+
86
+ `Estimates.from_run` subtracts shared set-up and runtime admission waits from each
87
+ attempt's duration to estimate the test's *own* time. Crashed attempts and nonpositive
88
+ results are excluded. For each node id it uses the longest clean attempt, or the
89
+ longest contended attempt when no clean one exists. Tests without a usable estimate
90
+ get the mean of those estimates, or zero if none exist. Fixture dependencies are
91
+ combined across attempts, and each fixture key's set-up cost is the longest recorded.
92
+
93
+ A missing or unreadable history file leaves xdist's scheduler in place unless an
94
+ explicit CPU budget was requested. With that budget the custom scheduler still runs,
95
+ using equal 1 ms test estimates. It supports only `load` and `worksteal`.
96
+
97
+ ## Cost model
98
+
99
+ A shared fixture makes the cost of a test depend on the worker. Every test has a
100
+ *family*, the set of shared fixtures it needs. Fixtures cheaper than a millisecond
101
+ are ignored unless they declare CPU demand, in which case they are retained even
102
+ without recorded set-up time. A queue's cost is its tests' own durations plus every
103
+ set-up the queue forces:
104
+
105
+ - session fixtures are paid once per worker, as are package fixtures without a
106
+ matching package collector;
107
+ - class, module and package fixtures last until the worker leaves their scope,
108
+ including through tests that do not use them. Re-entering the scope pays again;
109
+ - a parameter change replaces the fixture and its dependents.
110
+
111
+ A `Lane` holds one worker's projected finish (`free`), fixture instances at that
112
+ point (`fixtures`), and unsent plan. `Costs.charge` computes a single item's duration,
113
+ CPU work, peak demand and fixture transition. `Costs.project` folds these into a
114
+ `Projection` for a whole sequence. Time and CPU work therefore use one traversal.
115
+
116
+ The scheduler keeps one `Charge` per dispatched item. It records the holds already
117
+ included in the peak, so actual live holds supplement it without double counting.
118
+ Charged duration, CPU work and the pre-dispatch fixture checkpoint remain historical
119
+ facts even when a runtime request changes the reservation. Stealing refunds the same
120
+ record and restores that checkpoint.
121
+ Ordinary transfers and steals compare finish times with the same CPU-work floor.
122
+
123
+ ## Planning
124
+
125
+ `plan` places nonempty fixture families first, then the family without shared
126
+ fixtures. Families are ordered by cold CPU work (slot-seconds, including set-ups).
127
+ Members are ordered by declared slots, then own duration, both descending. Each
128
+ family is tried on 1..n lanes: members are greedily split in that order, putting each
129
+ on the chunk with the least own time so far, and the chunks go to lanes chosen by
130
+ projected finish and fixture cost.
131
+
132
+ The placement score is the maximum of the longest lane after placement, the average
133
+ lane including work still unplaced, and, with admission enabled, total slot-seconds
134
+ divided by the sum of domain admission limits. The average avoids splitting an early
135
+ family just because lanes are still empty; the CPU floor counts duplicated heavy set-ups.
136
+ A split of a nonempty family must improve the score by more than 1% (`SPLIT_GAIN`).
137
+ The family without shared fixtures is spread more evenly on near ties instead.
138
+
139
+ A bounded balance pass then moves single tests off the tail of the longest lane while
140
+ that clearly helps, using the same CPU-work floor. Finally each lane is ordered by
141
+ fixture-bound work, declared slots and own duration, then grouped by family in order
142
+ of first appearance. This prioritizes fixture reuse and heavy work while leaving
143
+ lighter work available at the tail for other workers.
144
+
145
+ Lane free times are clock readings. The percentage tolerance is applied to time left
146
+ from the earliest of them, never to the reading itself, or a large monotonic clock
147
+ value would swallow every gain.
148
+
149
+ `transfer` moves a tail chunk of another lane's unsent plan to a lane that ran dry:
150
+ the whole tail run of one family, half of it, or its last test, from every other lane,
151
+ keeping the move with the best predicted finish. A move that leaves the prediction
152
+ unchanged is still taken, because the receiver is idle and the estimates are only
153
+ estimates.
154
+
155
+ Admission delays themselves are not modelled, so projected finish times remain
156
+ approximate while workers wait for slots.
157
+
158
+ ## Driving xdist
159
+
160
+ `DurationScheduling` subclasses xdist's `LoadScheduling` and keeps its lifecycle
161
+ (collection check, crash handling, restarts, shutdown), replacing what goes out and
162
+ when. Two constraints of xdist's worker protocol shape it:
163
+
164
+ - a worker runs a test only once it knows the following one, or that there is none
165
+ (it needs `nextitem` for fixture teardown). A queue's last test waits for another
166
+ test or a shutdown command. CPU admission deliberately uses that wait to keep a
167
+ test from starting until slots are available;
168
+ - a worker completes tests in the order it received them, and the controller only
169
+ learns about completions after the fact.
170
+
171
+ Dispatch charges each test with exactly what the lane's cost function says (test plus
172
+ set-ups it forces), and completion refunds exactly that, so the projection of when a
173
+ worker will be free stays honest. A worker projected to be done by now is late and is
174
+ assumed busy with all its dispatched work from now; one still projected to be busy is
175
+ never assumed free before it has run the tests it has not started yet, however long
176
+ the running one is taking.
177
+
178
+ **`--dist load`.** Once every worker has collected, the planner lays out a lane per
179
+ worker. Each worker gets its next tests from its own plan: at least two, plus a small
180
+ time budget of estimated seconds so runs of trivial tests are batched. The budget is
181
+ a tenth of a second at most and shrinks for short suites. Refills happen with
182
+ hysteresis (below half the budget, filled to the whole budget) so tiny tests do not
183
+ cost a command each. A worker whose plan runs dry takes from other lanes' unsent
184
+ plans through `transfer`, which accepts moves that do not worsen the predicted
185
+ finish. An idle worker can shut down when no suitable work remains; admission may
186
+ keep it waiting with its fixtures alive. Everything not yet sent stays on the
187
+ controller, where it can still move.
188
+
189
+ **`--dist worksteal`.** Without CPU admission, each worker's whole plan is dispatched
190
+ in one command, as xdist's own work-stealing scheduler does. With admission, dispatch
191
+ is segmented at demand increases as described below. A worker down to its last test
192
+ is put on a waiting list, and the controller picks a donor and a tail chunk using the
193
+ same cost function as `transfer`. Candidates are the last family in the queue, half
194
+ that family, and a chunk sized to the dispatch time budget (or the remaining family
195
+ if shorter). Chunks smaller than that batch candidate must shorten the run; larger
196
+ ones may leave the prediction unchanged. Ties favor balancing donor and thief.
197
+ The chunk is requested through xdist's `steal` command, which is all or nothing from
198
+ 3.7 on: the worker refuses
199
+ unless it still holds every requested test. The running test, the one after it, and
200
+ one more as a margin for reports in flight are never requested. One steal is in
201
+ flight at a time; a refusal (the donor had moved on) leads to a fresh choice against
202
+ the updated bookkeeping, and a receiver that would be left with a single test asks
203
+ again or is shut down so it can run it. `tests_finished` stays false while a steal is
204
+ in flight.
205
+
206
+ **Restarts.** When a worker dies, its dispatched-but-unrun tests go back to the pool
207
+ and are planned over the surviving lanes together with its unsent plan. If no worker
208
+ is left, the tests stay pending without a lane and are planned onto the replacement
209
+ when it joins.
210
+
211
+ ## CPU demand
212
+
213
+ A test's declared weight comes from the closest `timing_cpu` marker (function, class,
214
+ module); an invalid marker counts as one slot and fails the test at set-up with the
215
+ reason, so a typo cannot silently change the schedule. A fixture's demand is an
216
+ attribute the `pytest_timing.cpu` decorator sets on the fixture function, in either
217
+ decorator order (pytest 8.4 wraps fixture functions in a definition object; both get
218
+ the attribute). The decorator stores metadata directly on the fixture rather than
219
+ applying a pytest mark to it.
220
+
221
+ The controller never sees items, only node ids, and xdist's collection report carries
222
+ nothing else. So after collecting, every worker builds a `Declarations` record (slots
223
+ per item, the keys of CPU-declared shared fixtures in each item's static closure,
224
+ each such fixture's set-up and hold slots, and the host's detected CPU environment)
225
+ and sends it as one custom event on its channel, ahead of xdist's own collection
226
+ event. xdist raises on unknown events, so `pytest_configure_node` wraps each node's
227
+ `process_from_remote` on the instance, before the channel callback is registered, to
228
+ take that one event off the stream. The handler runs on execnet's receiver thread and
229
+ only stores the record under a lock; the scheduler reads it when it lays out the run,
230
+ which happens after the collection event that follows it on the same channel.
231
+
232
+ `Costs` folds the declarations into the cost model: `slots` per test, set-up and hold
233
+ slots per fixture definition (keyed without the parameter index), and declared
234
+ fixtures added to their tests' families even before any run has timed them. The
235
+ `demand` of a test on a worker is the most it may need at any point there, its own
236
+ slots or a set-up it has to pay, plus the holds of the fixtures alive around it. A
237
+ function-scoped fixture's declaration is folded into its tests' own slots, since its
238
+ set-up runs inside the test's reservation: the test weighs its heaviest stage, each
239
+ such set-up next to what the other function-scoped fixtures may hold by then, or
240
+ the call with all of them alive. Fixtures no signature names (requested with
241
+ `getfixturevalue`) are not in any collected family; the worker still sends every
242
+ declared shared fixture definition it knows (`<scope>:<defined at>::<name>`), so a
243
+ family recorded by an earlier run and loaded through `--timing-schedule` finds the
244
+ demand of such a fixture by definition. Without a recorded run the planner cannot
245
+ see them, and admission happens at run time instead (next section).
246
+
247
+ ## Admission
248
+
249
+ xdist's worker protocol is the gate. A worker runs a test only once it has been told
250
+ the following one, or been shut down. So every test sent to a worker is *queued* but
251
+ not *granted* until that next message goes out, and the controller sends it only when
252
+ the worker's reservation covers everything the worker could then run without another
253
+ word: the tests before the tail, or the whole queue after a shutdown. Ordinary test
254
+ admission uses existing xdist commands; runtime fixture requests and shutdown
255
+ cancellation use the custom messages described below. Pytest's `nextitem` teardown
256
+ semantics are preserved for executed tests.
257
+
258
+ The states of a test:
259
+
260
+ - *planned* on the controller, in a lane's unsent plan;
261
+ - *sent*, the ungranted tail of a worker's queue;
262
+ - *granted*, once the message after it went out; the worker's reservation covers it;
263
+ - *waiting*, when it is the head of its worker's queue and the reservation cannot be
264
+ raised to cover it: the worker blocks in xdist's queue with fixtures alive, and
265
+ `tests_finished` stays false;
266
+ - *parked*, when it is next in a worker's plan but could never fit next to what the
267
+ other workers' fixtures hold: it stays on the controller, the worker runs what
268
+ else it has or waits idle, and the test is tried again at every release and exit;
269
+ - *completed* at `runtest_protocol_complete`, after teardown, when the reservation is
270
+ recomputed from what is left;
271
+ - *cancelled* when shutdown withdraws an unadmitted test (see below). Steals return
272
+ tests to a plan; worker death reports a started test as crashed and requeues the
273
+ unstarted tail.
274
+
275
+ A worker's reservation covers the largest recorded peak among its granted tests,
276
+ plus live fixture holds not already included in that peak, or only the holds while
277
+ it is idle. Raising it uses admission; lowering it always goes through. A rise in
278
+ demand behind lighter tests is deferred: the batch stops before the message that
279
+ would grant it, the worker runs down to it, and the grant is tried when it is the head. So in
280
+ `--dist worksteal` a plan is dispatched in segments that end at the next rise, and a
281
+ queue never reserves for its heaviest member while lighter ones run. Stolen tests go
282
+ to the front of the taker's plan and out through the same gate.
283
+
284
+ The waiting line (`Admission`) is ordered by arrival. The head's shortfall is pledged:
285
+ other workers may raise their reservations only out of what is left, or when the work
286
+ behind the request is projected to end before the head could have had its slots
287
+ anyway, computed from the projected release times of the reservations in the way
288
+ (EASY backfilling). A request above the limit is clamped to the whole limit and
289
+ reported, so an oversized declaration can still run.
290
+
291
+ Fixtures can keep slots occupied while their worker waits, so they need their own
292
+ rule. Before a test is dispatched, its demand on that worker is checked against
293
+ what the other workers of the domain hold; if it could never fit while they live it
294
+ is passed over (parked) and the worker takes the next test of its plan, or waits idle
295
+ when there is none. Ordinary admission respects the limit. After every retry of
296
+ the waiting line the scheduler looks for a domain at a standstill, where nothing
297
+ runs and nobody in line fits, and decides what gives: a parked worker is shut down
298
+ first, so its exit lets go of its own holds and its plan is handed to the workers
299
+ that stay at that moment (a worker told to shut down takes nothing more); failing
300
+ that, the oldest queued head or runtime request is forced, unless a draining worker
301
+ can still release the needed holds. Forced admission can exceed the limit and is
302
+ counted in the summary. Budgets are per resource domain: `local` for popen workers,
303
+ the host named in the execnet spec otherwise. An explicit integer sets each domain's
304
+ budget; `auto` uses the smallest budget detected by its workers. When declarations
305
+ enable admission without an explicit budget, the detected budget is raised to at
306
+ least the domain's worker count. This floor does not prevent heavy tests or fixture
307
+ holds from delaying plain tests.
308
+
309
+ When xdist shuts every worker down itself (`-x`, `--maxfail`, the restart budget for
310
+ crashed workers), it calls each node's `shutdown` directly. The scheduler wraps that
311
+ method on every node it is given, so the call passes through it first: what the
312
+ worker holds is granted if it fits, and otherwise the tail of its queue, the one test
313
+ the worker cannot start without another word, is withdrawn with a `timing_cancel`
314
+ command, and the worker-side plugin skips that test's protocol. The worker takes
315
+ commands of ours off xdist's command dispatcher on its receiver thread, wrapped per
316
+ instance like the event on the controller. A worker that dies while holding a single
317
+ unadmitted test had not started it: the test goes back to the pool rather than being
318
+ reported as crashed.
319
+
320
+ A worker blocked inside a runtime fixture request has already started its test;
321
+ if it dies there, the test is reported as crashed rather than silently replayed.
322
+
323
+ A fixture reached only at run time may be absent from the planned reservation.
324
+ Before setting up a dynamic fixture that declares multiple slots or a hold, the
325
+ worker-side meter sends a `timing_request` event (test, fixture key, request id,
326
+ set-up and hold slots) and blocks
327
+ until a `timing_grant` command names that request. Request ids are unique within the
328
+ worker, so a late grant cannot satisfy a retry of the same fixture. On the controller
329
+ the event is re-posted as an in-process event, so the run loop handles it on the main
330
+ thread like xdist's own. The scheduler treats it as a rise: while it waits the worker
331
+ runs nothing, so its reservation drops to what its fixtures hold. The request
332
+ reserves the set-up's slots next to those, through the same gate, waiting line and backfilling as
333
+ a queued test; the instance then counts as alive on that worker, its hold added to
334
+ everything queued there. Two workers asking at once therefore take turns instead of
335
+ deadlocking on the slots of the tests they are blocked in. Each request also carries
336
+ the total holds of all live fixtures. The worker tracks successful setup and
337
+ finalization at every scope, also sending `timing_holds` events on those transitions.
338
+ These actual lifetimes are separate from `Lane.fixtures`, which projects through
339
+ prefetched tests and can already describe a different scope or
340
+ parameter. Failed setup adds no live hold; a granted setup is not yet a live instance.
341
+ Function-scoped holds end with the current test and never become shared fixture
342
+ costs. A draining shutdown keeps a blocked test behind the gate, but a live draining
343
+ worker's request still participates in deadlock recovery. Workers that can release
344
+ holds by finishing are allowed to do so before forcing another request. Worker exit
345
+ releases its reservation even when xdist skips scheduler removal after maxfail, so a request
346
+ cannot remain blocked by a departed worker. After five minutes a blocked request
347
+ cancels its fixture setup and asks to recover the test's original reservation.
348
+ Only after that grant does it fail and unwind pytest's incomplete fixture instance;
349
+ cleanup and retries therefore retain a reservation and valid fixture bookkeeping.
350
+ Collection includes declarations on dynamic-only fixtures, including function
351
+ fixtures, when deciding to enable the gate. Declarations alone do not select this
352
+ scheduler: a usable history file or an explicit CPU budget is also required.
353
+
354
+ The scheduler records how long each head waited. Runtime requests and phase reports
355
+ carry a private collection index and attempt token, so the plugin sums waits onto
356
+ the exact execution's `cpu.wait`, including repeated requests, retries and duplicate
357
+ selections. A wait before execution belongs to the first attempt. If the worker dies
358
+ before sending any phase report, its unreported crash retains that wait. The worker
359
+ also records the part inside the test span
360
+ as `runtime_wait`. All open fixture clocks pause during that wait, and test estimates
361
+ subtract it along with shared setup. CPU `elapsed` and `work` exclude the wait and
362
+ its process-tree CPU work, so the measured rate covers execution. Older records
363
+ without `runtime_wait` default to zero; pre-start waits are not subtracted again.
364
+ Event handlers are registered before each worker starts, even when another plugin
365
+ selects the scheduler: without an admission gate a request is granted immediately.
366
+ The cgroup quota behind `auto` is read from the process's own group, walking up to
367
+ the mount for the tightest limit, on cgroup v2 (`cpu.max`) and v1 (`cpu.cfs_quota_us`).
368
+ Which hierarchy holds the `cpu` controller decides: on a hybrid system a v1 line
369
+ naming it in `/proc/self/cgroup` wins over the `0::` v2 line, whose hierarchy has no CPU
370
+ controller there. Throttling is read from the same group's `cpu.stat` (`throttled_usec`
371
+ on v2, `throttled_time` in nanoseconds on v1).
372
+
373
+ ## Measuring CPU work
374
+
375
+ `ProcessTreeClock` uses `os.times()` for worker CPU time and, on POSIX, reaped child
376
+ CPU time. Where available, it adds live descendants through
377
+ `/proc/<pid>/task/*/children` and `/proc/<pid>/stat` on Linux, or psutil otherwise.
378
+ Readings account for a waited-for child moving from the live total into the reaped
379
+ total. Process discovery is a snapshot, so exits during traversal or descendants
380
+ that outlive or detach from their parents can leave gaps. Every record reports its
381
+ coverage: `tree`, `reaped` (worker plus waited-for children), or `self` (worker only,
382
+ including Windows without psutil).
383
+
384
+ The meter takes a reading at the start of setup and when the teardown report is made,
385
+ and the fixture timer takes readings around shared set-ups, so a record carries the
386
+ test's total `work` and the `setup_work` inside set-ups charged to it. Incomplete
387
+ tests without a teardown report may have no CPU record. Each record also carries
388
+ the host's PSI `some` share and whether the cgroup was throttled during
389
+ the test; `Estimates.from_run` treats such attempts as contended and prefers a clean
390
+ attempt of the same test when one exists, keeping the contended ones in the file.
391
+
392
+ ## Feedback
393
+
394
+ Admission's pressure feedback moves a domain's limit, never its budget, on evidence
395
+ of contention: increasing cgroup quota throttling, or a PSI `some` share at or above
396
+ 25% while the run's own measured CPU rate is below three quarters of the limit. That
397
+ rate sums the latest measured rate of each local worker currently running a test;
398
+ idle, departed and remote workers contribute nothing. High PSI alone does not lower
399
+ the limit when the run's own measured throughput is high enough.
400
+ Three contended samples in a row lower the limit by one slot; eight clean
401
+ samples in a row raise it by one, up to the budget; a ten-second cooldown separates
402
+ moves; samples are taken at most once a second, on the controller's host only, and
403
+ only when the host has the signals. Running tests are never touched: a lower limit
404
+ defers the next admissions. Without PSI or cgroup statistics the limit stays at the
405
+ budget.
406
+
407
+ ## Evaluation
408
+
409
+ `tests/test_schedule.py` covers estimates, fixture lifetimes, planning, rebalancing
410
+ and gated dispatch with fake workers, plus pytester runs with real workers.
411
+ `tests/test_admission.py` exercises reservations, the waiting line, backfilling and
412
+ pressure feedback. `tests/test_cpu.py` runs subprocess workloads, cold fixture
413
+ set-ups, held slots, crashes and `-x`; `tests/test_runtime_admission.py` covers dynamic
414
+ fixtures, runtime waits, cancellation, retries and shutdown recovery.
415
+
416
+ `benchmarks/cpu_bench.py` generates suites of tiny tests, single-CPU tests, internally
417
+ parallel tests with and without deadlines, mixed weights, an expensive session
418
+ fixture and background load. It compares configurations using whole-process wall
419
+ time (including pytest startup and reporting), slowest heavy test, failures, fixture
420
+ set-ups, throttling and pressure. For configurations producing a timing report, its
421
+ JSON also includes pytest session time; otherwise that field uses process wall
422
+ time. Results depend on the machine and load; the script can also run under a
423
+ container CPU quota.
@@ -0,0 +1,60 @@
1
+ # Changelog
2
+
3
+ ## 0.2.0
4
+
5
+ - Raise the minimum supported pytest-xdist version to 3.7. Running without xdist
6
+ remains supported.
7
+ - CPU-aware scheduling under pytest-xdist. `@pytest.mark.timing_cpu(N)` declares that a
8
+ test's workload, subprocesses included, needs `N` CPU slots (class and module markers
9
+ set defaults, the closest wins); `@pytest_timing.cpu(N, hold=M)` declares a fixture's
10
+ set-up demand and what it keeps busy while alive. `--timing-cpus N|auto` (ini
11
+ `timing_cpus`, env `PYTEST_TIMING_CPUS`) sets the budget per host, `auto` from the
12
+ affinity mask and cgroup quota (v2 and v1, the process's own group and its
13
+ ancestors), enforced even below the number of workers. In `load` and `worksteal`
14
+ mode, the controller normally admits a test when its reservation fits; a worker
15
+ whose next test does not fit waits with its fixtures alive. The oldest request
16
+ has priority, with backfilling for short work. A test that cannot fit next to other
17
+ workers' fixture holds stays on the controller until those holds end, and demand above
18
+ the admission limit is clamped to that limit. Deadlock recovery can force a request
19
+ above the limit, counted in the summary. Under `-x`/`--maxfail` a test a worker holds
20
+ without having been admitted is withdrawn rather than run. A declared fixture
21
+ reached only through `getfixturevalue` is admitted at run time: the worker asks
22
+ for its slots before the set-up and waits for them. A fixture built on a
23
+ parametrized one is keyed by that parameter too (`derived[base=1]`), since pytest
24
+ sets it up again for every parameter. A plugin's fixture is keyed by its qualified
25
+ function so it never collides with a root conftest override. Test JSON records now include
26
+ `cpu` when available: elapsed time, CPU time of the worker and its
27
+ descendants, declared demand, measurement coverage, host pressure and throttling, and
28
+ the time it waited for slots. Runtime waits are recorded separately and excluded
29
+ from test and fixture duration estimates and CPU rate; a request timeout cancels
30
+ setup and recovers the test's reservation before cleanup. Admission tracks holds
31
+ through unused tests in the same scope, expires package state when leaving the
32
+ package, and releases exited workers' reservations under `--maxfail`.
33
+ On the controller's Linux host, sustained quota throttling or CPU pressure with
34
+ low throughput lowers the limit for new admissions one slot at a time, with gradual
35
+ recovery. Running reservations are unchanged.
36
+ - `--timing-schedule PATH` (ini `timing_schedule`, env `PYTEST_TIMING_SCHEDULE`) plans
37
+ pytest-xdist workers' queues from a previous run's JSON: test duration and declared
38
+ CPU demand balanced across workers, shared fixture set-ups counted where they will
39
+ be paid, families of tests that share expensive fixtures kept together and split
40
+ only when that finishes sooner, and unsent work moved
41
+ between workers as the run drifts from the plan. With `--dist worksteal` the plan is
42
+ dispatched whole unless CPU admission requires segments, and rebalanced through
43
+ xdist's steal command. A missing or unreadable history file falls back to xdist's
44
+ scheduling unless an explicit CPU budget keeps admission active with equal test
45
+ estimates. Other distribution modes always keep xdist's scheduler, with a note in
46
+ the summary.
47
+ - Shared (class, module, package, session) fixture set-ups are timed in the process that
48
+ runs the tests and recorded per test in the JSON as `fixtures`, keyed by scope, scope
49
+ node, definition and parameter, with the seconds spent on set-ups charged to that test.
50
+ Fixtures requested with `getfixturevalue` count for every test that uses them, and
51
+ `parametrize(..., indirect=True)` parameters are told apart like a fixture's own.
52
+
53
+ ## 0.1.0
54
+
55
+ - Output options are boolean flags (`--timing-json`) with separate `--timing-json-file PATH`
56
+ options, so an output flag can never swallow a test path.
57
+ - Runs record their termination (`finished`, `interrupted`, `aborted`, ...) explicitly.
58
+ - Initial release: controller-side timing capture with and without pytest-xdist,
59
+ ASCII Gantt chart in the terminal summary, self-contained HTML report, JSON output,
60
+ Chrome trace output, and a `pytest-timing` CLI with `render` and `merge` commands.