pytest-timing 0.1.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/.github/workflows/ci.yml +9 -8
  2. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/.github/workflows/release.yml +2 -2
  3. pytest_timing-0.3.0/ARCHITECTURE.md +507 -0
  4. pytest_timing-0.3.0/CHANGELOG.md +88 -0
  5. pytest_timing-0.3.0/PKG-INFO +369 -0
  6. pytest_timing-0.3.0/README.md +343 -0
  7. pytest_timing-0.3.0/benchmarks/cpu_bench.py +376 -0
  8. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/examples/test_demo.py +16 -1
  9. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/pyproject.toml +3 -3
  10. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/__init__.py +5 -1
  11. pytest_timing-0.3.0/src/pytest_timing/admission.py +210 -0
  12. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/collector.py +48 -9
  13. pytest_timing-0.3.0/src/pytest_timing/demand.py +483 -0
  14. pytest_timing-0.3.0/src/pytest_timing/fixtures.py +229 -0
  15. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/model.py +147 -13
  16. pytest_timing-0.3.0/src/pytest_timing/plugin.py +733 -0
  17. pytest_timing-0.3.0/src/pytest_timing/schedule.py +577 -0
  18. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/static/report.html +0 -10
  19. pytest_timing-0.3.0/src/pytest_timing/telemetry.py +834 -0
  20. pytest_timing-0.3.0/src/pytest_timing/xdist_compat.py +337 -0
  21. pytest_timing-0.3.0/src/pytest_timing/xdist_scheduler.py +1131 -0
  22. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/conftest.py +50 -0
  23. pytest_timing-0.3.0/tests/test_admission.py +194 -0
  24. pytest_timing-0.3.0/tests/test_benchmarks.py +22 -0
  25. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_cli.py +0 -2
  26. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_collector.py +26 -4
  27. pytest_timing-0.3.0/tests/test_cpu.py +789 -0
  28. pytest_timing-0.3.0/tests/test_fixtures.py +362 -0
  29. pytest_timing-0.3.0/tests/test_memory.py +216 -0
  30. pytest_timing-0.3.0/tests/test_memory_gate.py +272 -0
  31. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_plugin.py +1 -15
  32. pytest_timing-0.3.0/tests/test_runtime_admission.py +712 -0
  33. pytest_timing-0.3.0/tests/test_schedule.py +1435 -0
  34. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/uv.lock +3 -3
  35. pytest_timing-0.1.0/CHANGELOG.md +0 -10
  36. pytest_timing-0.1.0/PKG-INFO +0 -161
  37. pytest_timing-0.1.0/README.md +0 -135
  38. pytest_timing-0.1.0/src/pytest_timing/plugin.py +0 -382
  39. pytest_timing-0.1.0/src/pytest_timing/xdist_compat.py +0 -158
  40. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/.gitignore +0 -0
  41. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/LICENSE +0 -0
  42. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/docs/report.png +0 -0
  43. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/cli.py +0 -0
  44. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/outputs.py +0 -0
  45. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/render/__init__.py +0 -0
  46. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/render/ascii.py +0 -0
  47. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/render/html.py +0 -0
  48. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/render/trace.py +0 -0
  49. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/src/pytest_timing/static/__init__.py +0 -0
  50. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/fixtures/legacy_complete_false.json +0 -0
  51. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/fixtures/legacy_complete_true.json +0 -0
  52. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/fixtures/legacy_no_flag.json +0 -0
  53. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_ascii.py +0 -0
  54. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_html.py +0 -0
  55. {pytest_timing-0.1.0 → pytest_timing-0.3.0}/tests/test_trace.py +0 -0
@@ -7,7 +7,7 @@ on:
7
7
 
8
8
  jobs:
9
9
  test:
10
- name: py${{ matrix.python }} / pytest ${{ matrix.pytest }}
10
+ name: ${{ matrix.os }} / py${{ matrix.python }} / pytest ${{ matrix.pytest }}${{ matrix.xdist && format(' / xdist {0}', matrix.xdist) || '' }}
11
11
  runs-on: ${{ matrix.os }}
12
12
  strategy:
13
13
  fail-fast: false
@@ -16,14 +16,14 @@ jobs:
16
16
  python: ["3.10", "3.11", "3.12", "3.13", "3.14"]
17
17
  pytest: ["latest"]
18
18
  include:
19
- - { os: ubuntu-latest, python: "3.10", pytest: "7.3" }
19
+ - { os: ubuntu-latest, python: "3.10", pytest: "7.3", xdist: "3.7.0" }
20
20
  - { os: ubuntu-latest, python: "3.11", pytest: "7.4" }
21
21
  - { os: ubuntu-latest, python: "3.12", pytest: "8.0" }
22
22
  - { os: windows-latest, python: "3.12", pytest: "latest" }
23
23
  - { os: macos-latest, python: "3.13", pytest: "latest" }
24
24
  steps:
25
- - uses: actions/checkout@v4
26
- - uses: astral-sh/setup-uv@v5
25
+ - uses: actions/checkout@v7
26
+ - uses: astral-sh/setup-uv@v10.1.0
27
27
  with:
28
28
  python-version: ${{ matrix.python }}
29
29
  - name: Install
@@ -31,6 +31,7 @@ jobs:
31
31
  uv sync --no-dev
32
32
  uv pip install pytest-xdist pytest-rerunfailures hypothesis
33
33
  if [ "${{ matrix.pytest }}" != "latest" ]; then uv pip install "pytest~=${{ matrix.pytest }}.0"; fi
34
+ if [ -n "${{ matrix.xdist }}" ]; then uv pip install "pytest-xdist==${{ matrix.xdist }}"; fi
34
35
  shell: bash
35
36
  - name: Test
36
37
  run: uv run --no-sync pytest -v -p no:cacheprovider
@@ -39,8 +40,8 @@ jobs:
39
40
  name: without pytest-xdist
40
41
  runs-on: ubuntu-latest
41
42
  steps:
42
- - uses: actions/checkout@v4
43
- - uses: astral-sh/setup-uv@v5
43
+ - uses: actions/checkout@v7
44
+ - uses: astral-sh/setup-uv@v10.1.0
44
45
  with:
45
46
  python-version: "3.12"
46
47
  - run: uv sync --no-dev
@@ -50,8 +51,8 @@ jobs:
50
51
  lint:
51
52
  runs-on: ubuntu-latest
52
53
  steps:
53
- - uses: actions/checkout@v4
54
- - uses: astral-sh/setup-uv@v5
54
+ - uses: actions/checkout@v7
55
+ - uses: astral-sh/setup-uv@v10.1.0
55
56
  with:
56
57
  python-version: "3.12"
57
58
  - run: uv sync
@@ -11,7 +11,7 @@ jobs:
11
11
  permissions:
12
12
  id-token: write
13
13
  steps:
14
- - uses: actions/checkout@v4
15
- - uses: astral-sh/setup-uv@v5
14
+ - uses: actions/checkout@v7
15
+ - uses: astral-sh/setup-uv@v10.1.0
16
16
  - run: uv build
17
17
  - run: uv publish
@@ -0,0 +1,507 @@
1
+ # Architecture
2
+
3
+ How pytest-timing captures timings, times shared fixtures inside xdist workers, and
4
+ schedules a run from the previous one. For usage, see the [README](README.md).
5
+
6
+ ## Modules
7
+
8
+ | Module | Role |
9
+ |---|---|
10
+ | `plugin.py` | pytest hooks on the controller: options, settings, capture, terminal summary, outputs, and the `pytest_xdist_make_scheduler` hook. |
11
+ | `collector.py` | Folds phase reports and worker events into a `Run`. Knows nothing about pytest objects. |
12
+ | `model.py` | The recorded run: `Run`, `Worker`, `TestSpan`, `Phase`, and the JSON representation. |
13
+ | `outputs.py` | The ASCII, HTML and trace renderers, in one table shared by the plugin and the CLI. |
14
+ | `cli.py` | `pytest-timing render` and `pytest-timing merge`. |
15
+ | `fixtures.py` | Times shared fixture set-up where tests run and lists dependencies on reports. |
16
+ | `schedule.py` | Duration estimates, the fixture-aware cost model and the planner. Free of xdist. |
17
+ | `demand.py` | Declared CPU demand: the `timing_cpu` marker, the fixture decorator, the declarations a worker sends, and the meter that records each test's CPU work and resident memory. |
18
+ | `admission.py` | One host's CPU budget, reservations, fair waiting line and pressure feedback. Free of xdist and platform reads. |
19
+ | `telemetry.py` | Platform reads: affinity and cgroup quota, CPU time and resident memory of a process tree, pressure and throttling. |
20
+ | `xdist_scheduler.py` | The xdist scheduler that drives workers with the planner and the admission gate, in `load` and `worksteal` mode. |
21
+ | `xdist_compat.py` | Everything that reads pytest-xdist internals (versions, worker detection, why the controller stopped, the custom worker event and controller command). |
22
+
23
+ ## Capturing timings
24
+
25
+ Since pytest 7.3 every `TestReport` carries wall-clock `start` and `stop` timestamps.
26
+ xdist serialises reports to the controller unchanged and attaches the worker to each,
27
+ so phase aggregation uses controller-side hooks: `pytest_runtest_logreport` for the
28
+ setup / call / teardown phases, plus xdist's node-ready, collection-finished and
29
+ node-down hooks for the worker lifecycle. CPU admission adds custom events and commands
30
+ through the xdist adapter; timing reports use pytest's normal report transport.
31
+
32
+ Each lane in the report shows boot (worker start-up until it is ready), collection,
33
+ tests, idle gaps and the point the worker shut down. Session-scoped fixture set-up is
34
+ attributed to the first test that needs that instance on each worker and its teardown
35
+ to the test that finalizes it, exactly as pytest reports it.
36
+
37
+ Timing collection is enabled by `--timing`, an output, scheduling or CPU-budget
38
+ setting. Formatting settings alone do not enable it. The marker is always registered;
39
+ when enabled, an xdist worker also registers the fixture timer and CPU meter.
40
+
41
+ ## Timing shared fixtures in the workers
42
+
43
+ pytest charges a class, module, package or session fixture's set-up to the setup phase
44
+ of the first test that needs it on each worker, so a test's recorded duration depends
45
+ on where it happened to land. The scheduler needs the two apart: how long the test
46
+ itself takes, and which shared fixtures it needs at what price.
47
+
48
+ `fixtures.py` wraps `pytest_fixture_setup` for the shared scopes and times each
49
+ set-up, excluding any set-up nested inside it through `getfixturevalue` (that one is
50
+ timed on its own). When a test's setup and call reports are built, the timer attaches
51
+ `timing_fixtures` to them: a mapping from fixture key to the seconds this test spent
52
+ setting the fixture up, or `None` when the test only uses it. Extra report attributes
53
+ survive xdist's serialisation, and the collector merges the two reports into
54
+ `TestSpan.fixtures`.
55
+
56
+ A test's fixtures are gathered from three sources, because none is complete alone:
57
+
58
+ - the static fixture closure, which names what the signatures ask for;
59
+ - the request's resolved fixture definitions, which add what `getfixturevalue` pulled
60
+ in for this test, including inside the test body (hence the call report);
61
+ - a per-worker memo of what each shared fixture's set-up requested dynamically. A
62
+ test served from the cache never sees those requests, but it depends on them just
63
+ the same.
64
+
65
+ A fixture key is `<scope>:<scope node id>:<defined at>::<name>[<params>]`. The
66
+ scope node is what pytest shares the instance under: the session (empty), the module,
67
+ the class, or the matching package collector. A package fixture without a matching
68
+ collector belongs to the session. "Defined
69
+ at" tells overrides of the same name apart: the fixture's `baseid` (its conftest
70
+ directory or test module), or, for a plugin's fixture, whose `baseid` is empty just
71
+ like a root conftest's on pytest 7, the defining module and qualified function name.
72
+ This also distinguishes a plugin class registered from that same conftest. The parameter part,
73
+ taken from the item's
74
+ callspec, tells the instances of a parametrized fixture apart whether the fixture or
75
+ the test (`indirect=True`) carries the parameters, since pytest keeps only one
76
+ instance alive at a time: `base[1]`. A fixture that depends on a parametrized one,
77
+ directly or through other fixtures, is torn down with the instance it was built on
78
+ and built again for the next parameter, so those indices are part of its key too:
79
+ `derived[base=1]`, or `mixed[0,base=1]` for a fixture with parameters of its own. It
80
+ is an index, not a value, so two modules that parametrize the same session fixture
81
+ with different values at the same index share a key; the planner is slightly
82
+ optimistic there.
83
+
84
+ ## Estimates
85
+
86
+ `Estimates.from_run` subtracts shared set-up and runtime admission waits from each
87
+ attempt's duration to estimate the test's *own* time. Crashed attempts and nonpositive
88
+ results are excluded. For each node id it ranks the attempts clean before contended
89
+ and passing before failing, and takes the median of the best rank present: a failed
90
+ attempt usually stops early, and the longest attempt grows with the number of
91
+ attempts, so a history merged from several runs would drift upward. Tests without a
92
+ usable estimate get the mean of those estimates, or zero if none exist. Fixture
93
+ dependencies are combined across attempts, and each fixture key's set-up cost is the
94
+ median of the recorded ones.
95
+
96
+ A missing or unreadable history file leaves xdist's scheduler in place unless an
97
+ explicit CPU budget was requested. With that budget the custom scheduler still runs,
98
+ using equal 1 ms test estimates. It supports only `load` and `worksteal`.
99
+
100
+ ## Cost model
101
+
102
+ A shared fixture makes the cost of a test depend on the worker. Every test has a
103
+ *family*, the set of shared fixtures it needs. Fixtures cheaper than a millisecond
104
+ are ignored unless they declare CPU demand, in which case they are retained even
105
+ without recorded set-up time. A queue's cost is its tests' own durations plus every
106
+ set-up the queue forces:
107
+
108
+ - session fixtures are paid once per worker, as are package fixtures without a
109
+ matching package collector;
110
+ - class, module and package fixtures last until the worker leaves their scope,
111
+ including through tests that do not use them. Re-entering the scope pays again;
112
+ - a parameter change replaces the fixture and its dependents.
113
+
114
+ A `Lane` holds one worker's projected finish (`free`), fixture instances at that
115
+ point (`fixtures`), and unsent plan. `Costs.charge` computes a single item's duration,
116
+ CPU work, peak demand and fixture transition. `Costs.project` folds these into a
117
+ `Projection` for a whole sequence. Time and CPU work therefore use one traversal.
118
+
119
+ The scheduler keeps one `Charge` per dispatched item. It records the holds already
120
+ included in the peak, so actual live holds supplement it without double counting.
121
+ Charged duration, CPU work and the pre-dispatch fixture checkpoint remain historical
122
+ facts even when a runtime request changes the reservation. Stealing refunds the same
123
+ record and restores that checkpoint.
124
+ Ordinary transfers and steals compare finish times with the same CPU-work floor.
125
+
126
+ ## Planning
127
+
128
+ `plan` places nonempty fixture families first, then the family without shared
129
+ fixtures. Families are ordered by cold CPU work (slot-seconds, including set-ups).
130
+ Members are ordered by declared slots, then own duration, both descending. Each
131
+ family is tried on 1..n lanes: members are greedily split in that order, putting each
132
+ on the chunk with the least own time so far, and the chunks go to lanes chosen by
133
+ projected finish and fixture cost.
134
+
135
+ The placement score is the maximum of the longest lane after placement, the average
136
+ lane including work still unplaced, and, with admission enabled, total slot-seconds
137
+ divided by the sum of domain admission limits. The average avoids splitting an early
138
+ family just because lanes are still empty; the CPU floor counts duplicated heavy set-ups.
139
+ A split of a nonempty family must improve the score by more than 1% (`SPLIT_GAIN`).
140
+ The family without shared fixtures is spread more evenly on near ties instead.
141
+
142
+ A bounded balance pass then moves single tests off the tail of the longest lane while
143
+ that clearly helps, using the same CPU-work floor. Finally each lane is ordered by
144
+ fixture-bound work, declared slots and own duration, then grouped by family in order
145
+ of first appearance. This prioritizes fixture reuse and heavy work while leaving
146
+ lighter work available at the tail for other workers.
147
+
148
+ Lane free times are clock readings. The percentage tolerance is applied to time left
149
+ from the earliest of them, never to the reading itself, or a large monotonic clock
150
+ value would swallow every gain.
151
+
152
+ `transfer` moves a tail chunk of another lane's unsent plan to a lane that ran dry:
153
+ the whole tail run of one family, half of it, or its last test, from every other lane,
154
+ keeping the move with the best predicted finish. A move that leaves the prediction
155
+ unchanged is still taken, because the receiver is idle and the estimates are only
156
+ estimates.
157
+
158
+ Admission delays themselves are not modelled, so projected finish times remain
159
+ approximate while workers wait for slots.
160
+
161
+ ## Driving xdist
162
+
163
+ `DurationScheduling` subclasses xdist's `LoadScheduling` and keeps its lifecycle
164
+ (collection check, crash handling, restarts, shutdown), replacing what goes out and
165
+ when. Two constraints of xdist's worker protocol shape it:
166
+
167
+ - a worker runs a test only once it knows the following one, or that there is none
168
+ (it needs `nextitem` for fixture teardown). A queue's last test waits for another
169
+ test or a shutdown command. CPU admission deliberately uses that wait to keep a
170
+ test from starting until slots are available;
171
+ - a worker completes tests in the order it received them, and the controller only
172
+ learns about completions after the fact.
173
+
174
+ Dispatch charges each test with exactly what the lane's cost function says (test plus
175
+ set-ups it forces), and completion refunds exactly that, so the projection of when a
176
+ worker will be free stays honest. A worker projected to be done by now is late and is
177
+ assumed busy with all its dispatched work from now; one still projected to be busy is
178
+ never assumed free before it has run the tests it has not started yet, however long
179
+ the running one is taking.
180
+
181
+ **`--dist load`.** Once every worker has collected, the planner lays out a lane per
182
+ worker. Each worker gets its next tests from its own plan: at least two, plus a small
183
+ time budget of estimated seconds so runs of trivial tests are batched. The budget is
184
+ a tenth of a second at most and shrinks for short suites. Refills happen with
185
+ hysteresis (below half the budget, filled to the whole budget) so tiny tests do not
186
+ cost a command each. A worker whose plan runs dry takes from other lanes' unsent
187
+ plans through `transfer`, which accepts moves that do not worsen the predicted
188
+ finish. An idle worker can shut down when no suitable work remains; admission may
189
+ keep it waiting with its fixtures alive. Everything not yet sent stays on the
190
+ controller, where it can still move.
191
+
192
+ **`--dist worksteal`.** Without CPU admission, each worker's whole plan is dispatched
193
+ in one command, as xdist's own work-stealing scheduler does. With admission, dispatch
194
+ is segmented at demand increases as described below. A worker down to its last test
195
+ is put on a waiting list, and the controller picks a donor and a tail chunk using the
196
+ same cost function as `transfer`. Candidates are the last family in the queue, half
197
+ that family, and a chunk sized to the dispatch time budget (or the remaining family
198
+ if shorter). Chunks smaller than that batch candidate must shorten the run; larger
199
+ ones may leave the prediction unchanged. Ties favor balancing donor and thief.
200
+ The chunk is requested through xdist's `steal` command, which is all or nothing from
201
+ 3.7 on: the worker refuses
202
+ unless it still holds every requested test. The running test, the one after it, and
203
+ one more as a margin for reports in flight are never requested. One steal is in
204
+ flight at a time; a refusal (the donor had moved on) leads to a fresh choice against
205
+ the updated bookkeeping, and a receiver that would be left with a single test asks
206
+ again or is shut down so it can run it. `tests_finished` stays false while a steal is
207
+ in flight.
208
+
209
+ **Restarts.** When a worker dies, its dispatched-but-unrun tests go back to the pool
210
+ and are planned over the surviving lanes together with its unsent plan. If no worker
211
+ is left, the tests stay pending without a lane and are planned onto the replacement
212
+ when it joins.
213
+
214
+ ## CPU demand
215
+
216
+ A test's declared weight comes from the closest `timing_cpu` marker (function, class,
217
+ module); an invalid marker counts as one slot and fails the test at set-up with the
218
+ reason, so a typo cannot silently change the schedule. A fixture's demand is an
219
+ attribute the `pytest_timing.cpu` decorator sets on the fixture function, in either
220
+ decorator order (pytest 8.4 wraps fixture functions in a definition object; both get
221
+ the attribute). The decorator stores metadata directly on the fixture rather than
222
+ applying a pytest mark to it.
223
+
224
+ The controller never sees items, only node ids, and xdist's collection report carries
225
+ nothing else. So after collecting, every worker builds a `Declarations` record (slots
226
+ per item, the keys of CPU-declared shared fixtures in each item's static closure,
227
+ each such fixture's set-up and hold slots, and the host's detected CPU environment)
228
+ and sends it as one custom event on its channel, ahead of xdist's own collection
229
+ event. xdist raises on unknown events, so `pytest_configure_node` wraps each node's
230
+ `process_from_remote` on the instance, before the channel callback is registered, to
231
+ take that one event off the stream. The handler runs on execnet's receiver thread and
232
+ only stores the record under a lock; the scheduler reads it when it lays out the run,
233
+ which happens after the collection event that follows it on the same channel.
234
+
235
+ `Costs` folds the declarations into the cost model: `slots` per test, set-up and hold
236
+ slots per fixture definition (keyed without the parameter index), and declared
237
+ fixtures added to their tests' families even before any run has timed them. The
238
+ `demand` of a test on a worker is the most it may need at any point there, its own
239
+ slots or a set-up it has to pay, plus the holds of the fixtures alive around it. A
240
+ function-scoped fixture's declaration is folded into its tests' own slots, since its
241
+ set-up runs inside the test's reservation: the test weighs its heaviest stage, each
242
+ such set-up next to what the other function-scoped fixtures may hold by then, or
243
+ the call with all of them alive. Fixtures no signature names (requested with
244
+ `getfixturevalue`) are not in any collected family; the worker still sends every
245
+ declared shared fixture definition it knows (`<scope>:<defined at>::<name>`), so a
246
+ family recorded by an earlier run and loaded through `--timing-schedule` finds the
247
+ demand of such a fixture by definition. Without a recorded run the planner cannot
248
+ see them, and admission happens at run time instead (next section).
249
+
250
+ ## Admission
251
+
252
+ xdist's worker protocol is the gate. A worker runs a test only once it has been told
253
+ the following one, or been shut down. So every test sent to a worker is *queued* but
254
+ not *granted* until that next message goes out, and the controller sends it only when
255
+ the worker's reservation covers everything the worker could then run without another
256
+ word: the tests before the tail, or the whole queue after a shutdown. Ordinary test
257
+ admission uses existing xdist commands; runtime fixture requests and shutdown
258
+ cancellation use the custom messages described below. Pytest's `nextitem` teardown
259
+ semantics are preserved for executed tests.
260
+
261
+ The states of a test:
262
+
263
+ - *planned* on the controller, in a lane's unsent plan;
264
+ - *sent*, the ungranted tail of a worker's queue;
265
+ - *granted*, once the message after it went out; the worker's reservation covers it;
266
+ - *waiting*, when it is the head of its worker's queue and the reservation cannot be
267
+ raised to cover it: the worker blocks in xdist's queue with fixtures alive, and
268
+ `tests_finished` stays false;
269
+ - *parked*, when it is next in a worker's plan but could never fit next to what the
270
+ other workers' fixtures hold: it stays on the controller, the worker runs what
271
+ else it has or waits idle, and the test is tried again at every release and exit;
272
+ - *completed* at `runtest_protocol_complete`, after teardown, when the reservation is
273
+ recomputed from what is left;
274
+ - *cancelled* when shutdown withdraws an unadmitted test (see below). Steals return
275
+ tests to a plan; worker death reports a started test as crashed and requeues the
276
+ unstarted tail.
277
+
278
+ A worker's reservation covers the largest recorded peak among its granted tests,
279
+ plus live fixture holds not already included in that peak, or only the holds while
280
+ it is idle. Raising it uses admission; lowering it always goes through. A rise in
281
+ demand behind lighter tests is deferred: the batch stops before the message that
282
+ would grant it, the worker runs down to it, and the grant is tried when it is the head. So in
283
+ `--dist worksteal` a plan is dispatched in segments that end at the next rise, and a
284
+ queue never reserves for its heaviest member while lighter ones run. Stolen tests go
285
+ to the front of the taker's plan and out through the same gate.
286
+
287
+ The waiting line (`Admission`) is ordered by arrival. The head's shortfall is pledged:
288
+ other workers may raise their reservations only out of what is left, or when the work
289
+ behind the request is projected to end before the head could have had its slots
290
+ anyway, computed from the projected release times of the reservations in the way
291
+ (EASY backfilling). A request above the limit is clamped to the whole limit and
292
+ reported, so an oversized declaration can still run.
293
+
294
+ Fixtures can keep slots occupied while their worker waits, so they need their own
295
+ rule. Before a test is dispatched, its demand on that worker is checked against
296
+ what the other workers of the domain hold; if it could never fit while they live it
297
+ is passed over (parked) and the worker takes the next test of its plan, or waits idle
298
+ when there is none. Ordinary admission respects the limit. After every retry of
299
+ the waiting line the scheduler looks for a domain at a standstill, where nothing
300
+ runs and nobody in line fits, and decides what gives: a parked worker is shut down
301
+ first, so its exit lets go of its own holds and its plan is handed to the workers
302
+ that stay at that moment (a worker told to shut down takes nothing more); failing
303
+ that, the oldest queued head or runtime request is forced, unless a draining worker
304
+ can still release the needed holds. Forced admission can exceed the limit and is
305
+ counted in the summary. Budgets are per resource domain: `local` for popen workers,
306
+ the host named in the execnet spec otherwise. An explicit integer sets each domain's
307
+ budget; `auto` uses the smallest budget detected by its workers. When declarations
308
+ enable admission without an explicit budget, the detected budget is raised to at
309
+ least the domain's worker count. This floor does not prevent heavy tests or fixture
310
+ holds from delaying plain tests.
311
+
312
+ When xdist shuts every worker down itself (`-x`, `--maxfail`, the restart budget for
313
+ crashed workers), it calls each node's `shutdown` directly. The scheduler wraps that
314
+ method on every node it is given, so the call passes through it first: what the
315
+ worker holds is granted if it fits, and otherwise the tail of its queue, the one test
316
+ the worker cannot start without another word, is withdrawn with a `timing_cancel`
317
+ command, and the worker-side plugin skips that test's protocol. The worker takes
318
+ commands of ours off xdist's command dispatcher on its receiver thread, wrapped per
319
+ instance like the event on the controller. A worker that dies while holding a single
320
+ unadmitted test had not started it: the test goes back to the pool rather than being
321
+ reported as crashed.
322
+
323
+ A worker blocked inside a runtime fixture request has already started its test;
324
+ if it dies there, the test is reported as crashed rather than silently replayed.
325
+
326
+ A fixture reached only at run time may be absent from the planned reservation.
327
+ Before setting up a dynamic fixture that declares multiple slots or a hold, the
328
+ worker-side meter sends a `timing_request` event (test, fixture key, request id,
329
+ set-up and hold slots) and blocks
330
+ until a `timing_grant` command names that request. Request ids are unique within the
331
+ worker, so a late grant cannot satisfy a retry of the same fixture. On the controller
332
+ the event is re-posted as an in-process event, so the run loop handles it on the main
333
+ thread like xdist's own. The scheduler treats it as a rise: while it waits the worker
334
+ runs nothing, so its reservation drops to what its fixtures hold. The request
335
+ reserves the set-up's slots next to those, through the same gate, waiting line and backfilling as
336
+ a queued test; the instance then counts as alive on that worker, its hold added to
337
+ everything queued there. Two workers asking at once therefore take turns instead of
338
+ deadlocking on the slots of the tests they are blocked in. Each request also carries
339
+ the total holds of all live fixtures. The worker tracks successful setup and
340
+ finalization at every scope, also sending `timing_holds` events on those transitions.
341
+ These actual lifetimes are separate from `Lane.fixtures`, which projects through
342
+ prefetched tests and can already describe a different scope or
343
+ parameter. Failed setup adds no live hold; a granted setup is not yet a live instance.
344
+ Function-scoped holds end with the current test and never become shared fixture
345
+ costs. A draining shutdown keeps a blocked test behind the gate, but a live draining
346
+ worker's request still participates in deadlock recovery. Workers that can release
347
+ holds by finishing are allowed to do so before forcing another request. Worker exit
348
+ releases its reservation even when xdist skips scheduler removal after maxfail, so a request
349
+ cannot remain blocked by a departed worker. After five minutes a blocked request
350
+ cancels its fixture setup and asks to recover the test's original reservation.
351
+ Only after that grant does it fail and unwind pytest's incomplete fixture instance;
352
+ cleanup and retries therefore retain a reservation and valid fixture bookkeeping.
353
+ Collection includes declarations on dynamic-only fixtures, including function
354
+ fixtures, when deciding to enable the gate. Declarations alone do not select this
355
+ scheduler: a usable history file or an explicit CPU budget is also required.
356
+
357
+ The scheduler records how long each head waited. Runtime requests and phase reports
358
+ carry a private collection index and attempt token, so the plugin sums waits onto
359
+ the exact execution's `cpu.wait`, including repeated requests, retries and duplicate
360
+ selections. A wait before execution belongs to the first attempt. If the worker dies
361
+ before sending any phase report, its unreported crash retains that wait. The worker
362
+ also records the part inside the test span
363
+ as `runtime_wait`. All open fixture clocks pause during that wait, and test estimates
364
+ subtract it along with shared setup. CPU `elapsed` and `work` exclude the wait and
365
+ its process-tree CPU work, so the measured rate covers execution. Older records
366
+ without `runtime_wait` default to zero; pre-start waits are not subtracted again.
367
+ Event handlers are registered before each worker starts, even when another plugin
368
+ selects the scheduler: without an admission gate a request is granted immediately.
369
+ The cgroup quota behind `auto` is read from the process's own group, walking up to
370
+ the mount for the tightest limit, on cgroup v2 (`cpu.max`) and v1 (`cpu.cfs_quota_us`).
371
+ Which hierarchy holds the `cpu` controller decides: on a hybrid system a v1 line
372
+ naming it in `/proc/self/cgroup` wins over the `0::` v2 line, whose hierarchy has no CPU
373
+ controller there. Throttling is read from the same group's `cpu.stat` (`throttled_usec`
374
+ on v2, `throttled_time` in nanoseconds on v1).
375
+
376
+ ## Memory admission
377
+
378
+ Memory goes through the same gate as CPU, as a second `Admission` per domain counted
379
+ in bytes, and a test is granted only when its reservation fits both: `_raise` checks
380
+ every gate with `fits` first and reserves on all of them or none, and a refused worker
381
+ waits in every line with its respective need. Reservations, releases, forced
382
+ admissions, withdrawals and stall resolution act on both gates in step, so their
383
+ `busy` and `idle` states never disagree. The rules that let CPU admission exceed its
384
+ limit are safe for memory for the same reasons they are safe for CPU: backfilling
385
+ never exceeds the limit (it lends out slots pledged to the head), a request above the
386
+ limit is clamped so the test runs alone rather than never, and a forced admission
387
+ happens only when nothing runs in the domain, so nothing else's memory is at risk.
388
+ Pressure feedback moves only the CPU limit.
389
+
390
+ Demand comes from history, never from declarations. `Estimates.from_run` keeps, per
391
+ test, the largest `peak - base` any attempt showed (the worst case is what an
392
+ out-of-memory kill depends on), and per shared fixture key the memory the attempt that
393
+ paid its set-up kept resident (`after - base`, split evenly when one attempt paid for
394
+ several). `Costs.charge` adds a `memory` to each `Charge`: the test's rise plus what
395
+ the fixtures alive around it keep, projected through the lane's fixture state exactly
396
+ as CPU holds are, since workers report nothing about memory at run time. A fixture with
397
+ recorded memory belongs to its tests' families even when its set-up was too quick to
398
+ matter for time. `_need_memory` is the twin of `_need`; an idle worker reserves what
399
+ its fixtures keep, and a worker whose next test could never fit next to what the
400
+ other workers' fixtures keep is parked like one blocked by CPU holds. A fixture
401
+ reached at run time adds its recorded memory to the queued charges when its
402
+ request is granted.
403
+
404
+ The budget is `--timing-memory`: a size, or `auto`, which takes 80% of the smallest
405
+ memory the domain's workers report (physical memory capped by the cgroup limit,
406
+ `memory.max` on v2 and `memory.limit_in_bytes` on v1, read like the CPU quota). The
407
+ share is deliberate: estimates are rises above each worker's footprint, and the
408
+ footprints, the controller and the rest of the host are not in them. Without a
409
+ recorded run every estimate is zero and the gate admits everything; the summary says
410
+ so. The planner ignores memory: with tests kept apart by the gate, balancing lanes by
411
+ memory would add little, and the lane plan can still move.
412
+
413
+ ## Measuring CPU work
414
+
415
+ `ProcessTreeClock` uses `os.times()` for worker CPU time and, on POSIX, reaped child
416
+ CPU time. Where available, it adds live descendants through
417
+ `/proc/<pid>/task/*/children` and `/proc/<pid>/stat` on Linux, through `libproc` on
418
+ macOS (`proc_listchildpids` to find them, `proc_pidinfo` for their task times, in
419
+ Mach time units converted with `mach_timebase_info`), or psutil otherwise. psutil is
420
+ the last resort because its `children` scans the whole process table, about ten
421
+ milliseconds on macOS, and the clock is read several times per test: with psutil
422
+ installed, 3,000 trivial tests went from under a second to over a minute.
423
+ Readings account for a waited-for child moving from the live total into the reaped
424
+ total. Process discovery is a snapshot, so exits during traversal or descendants
425
+ that outlive or detach from their parents can leave gaps. Every record reports its
426
+ coverage: `tree`, `reaped` (worker plus waited-for children), or `self` (worker only,
427
+ including Windows without psutil).
428
+
429
+ The meter takes a reading at the start of setup and when the teardown report is made,
430
+ and the fixture timer takes readings around shared set-ups, so a record carries the
431
+ test's total `work` and the `setup_work` inside set-ups charged to it. Incomplete
432
+ tests without a teardown report may have no CPU record. Each record also carries
433
+ the host's PSI `some` share and whether the cgroup was throttled during
434
+ the test; `Estimates.from_run` treats such attempts as contended and prefers a clean
435
+ attempt of the same test when one exists, keeping the contended ones in the file.
436
+
437
+ ## Measuring memory
438
+
439
+ Memory is recorded so that a later run can keep tests that need a lot of it from
440
+ running at the same time; nothing schedules on it yet. `ResidentMemory` reads the
441
+ resident set size of the process running the tests and, where they can be listed, of
442
+ its live descendants: `/proc/self/statm` and the `/proc` walk shared with CPU
443
+ measurement on Linux, the `libproc` reader shared with the CPU clock on macOS,
444
+ `GetProcessMemoryInfo` on Windows for the worker alone. psutil fills in what the
445
+ platform readers cannot do, today descendants on Windows, and is never preferred
446
+ over a native reader. Descendants are summed, so pages they share count more than
447
+ once; the total errs on the large side.
448
+
449
+ A high-water mark such as `ru_maxrss` never comes down, so it cannot say what one
450
+ test needed: only the test that first raised the worker's peak would show anything.
451
+ `MemorySampler` therefore polls from a daemon thread while a window is open and
452
+ keeps the highest total. The worker's own size is read every 20 ms; it costs about a
453
+ microsecond on macOS and ten on Linux. Descendants are listed at most every 100 ms,
454
+ and while none are found the interval doubles up to a second, since listing costs an
455
+ order of magnitude more (and far more with psutil). Opening or closing a window never
456
+ lists descendants by itself, so a fast test costs two readings of its own process, a
457
+ few microseconds. The meter opens the window at `pytest_runtest_setup` and closes it
458
+ when the teardown report is made, so shared fixture set-ups are charged to the test
459
+ that paid for them, as their time is. The window goes on the teardown report as
460
+ `timing_memory` and into the JSON as `memory`: `base`, `peak` and `after` in bytes,
461
+ and the coverage (`tree` or `self`). A platform with no reading records nothing.
462
+ The sampler sleeps between windows and is closed at `pytest_unconfigure`.
463
+
464
+ A test shorter than the sampling interval is seen only at its edges: its record is
465
+ what was resident before and after it, and a buffer allocated and freed inside it
466
+ is missed. Faulting in enough memory to matter takes longer than one interval.
467
+
468
+ `peak - base` is the attempt's rise: what it needed on top of the worker's footprint.
469
+ `after - base` is what stayed resident, which for the first test of a session fixture
470
+ is roughly the fixture. Both are biased by allocator behaviour: a heap that already
471
+ grew for an earlier test can serve a later one without raising RSS, so a rise can
472
+ undercount a test whose allocations reuse freed heap, and a freed buffer the allocator
473
+ keeps can leave `after` high. Large buffers and subprocess memory, the usual causes of
474
+ an out-of-memory kill, are mapped and unmapped directly and measure well.
475
+
476
+ ## Feedback
477
+
478
+ Admission's pressure feedback moves a domain's limit, never its budget, on evidence
479
+ of contention: increasing cgroup quota throttling, or a PSI `some` share at or above
480
+ 25% while the run's own measured CPU rate is below three quarters of the limit. That
481
+ rate sums the latest measured rate of each local worker currently running a test;
482
+ idle, departed and remote workers contribute nothing. High PSI alone does not lower
483
+ the limit when the run's own measured throughput is high enough.
484
+ Three contended samples in a row lower the limit by one slot; eight clean
485
+ samples in a row raise it by one, up to the budget; a ten-second cooldown separates
486
+ moves; samples are taken at most once a second, on the controller's host only, and
487
+ only when the host has the signals. Running tests are never touched: a lower limit
488
+ defers the next admissions. Without PSI or cgroup statistics the limit stays at the
489
+ budget.
490
+
491
+ ## Evaluation
492
+
493
+ `tests/test_schedule.py` covers estimates, fixture lifetimes, planning, rebalancing
494
+ and gated dispatch with fake workers, plus pytester runs with real workers.
495
+ `tests/test_admission.py` exercises reservations, the waiting line, backfilling and
496
+ pressure feedback. `tests/test_cpu.py` runs subprocess workloads, cold fixture
497
+ set-ups, held slots, crashes and `-x`; `tests/test_runtime_admission.py` covers dynamic
498
+ fixtures, runtime waits, cancellation, retries and shutdown recovery.
499
+
500
+ `benchmarks/cpu_bench.py` generates suites of tiny tests, single-CPU tests, internally
501
+ parallel tests with and without deadlines, mixed weights, an expensive session
502
+ fixture and background load. It compares configurations using whole-process wall
503
+ time (including pytest startup and reporting), slowest heavy test, failures, fixture
504
+ set-ups, throttling and pressure. For configurations producing a timing report, its
505
+ JSON also includes pytest session time; otherwise that field uses process wall
506
+ time. Results depend on the machine and load; the script can also run under a
507
+ container CPU quota.