pytest-timing 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/ARCHITECTURE.md +91 -7
  2. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/CHANGELOG.md +28 -0
  3. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/PKG-INFO +52 -7
  4. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/README.md +51 -6
  5. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/pyproject.toml +1 -1
  6. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/__init__.py +1 -1
  7. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/collector.py +13 -1
  8. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/demand.py +28 -3
  9. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/model.py +54 -0
  10. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/plugin.py +96 -7
  11. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/schedule.py +71 -12
  12. pytest_timing-0.3.0/src/pytest_timing/telemetry.py +834 -0
  13. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/xdist_scheduler.py +201 -58
  14. pytest_timing-0.3.0/tests/test_memory.py +216 -0
  15. pytest_timing-0.3.0/tests/test_memory_gate.py +272 -0
  16. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_schedule.py +39 -11
  17. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/uv.lock +1 -1
  18. pytest_timing-0.2.0/src/pytest_timing/telemetry.py +0 -346
  19. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/.github/workflows/ci.yml +0 -0
  20. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/.github/workflows/release.yml +0 -0
  21. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/.gitignore +0 -0
  22. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/LICENSE +0 -0
  23. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/benchmarks/cpu_bench.py +0 -0
  24. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/docs/report.png +0 -0
  25. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/examples/test_demo.py +0 -0
  26. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/admission.py +0 -0
  27. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/cli.py +0 -0
  28. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/fixtures.py +0 -0
  29. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/outputs.py +0 -0
  30. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/render/__init__.py +0 -0
  31. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/render/ascii.py +0 -0
  32. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/render/html.py +0 -0
  33. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/render/trace.py +0 -0
  34. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/static/__init__.py +0 -0
  35. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/static/report.html +0 -0
  36. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/src/pytest_timing/xdist_compat.py +0 -0
  37. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/conftest.py +0 -0
  38. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/fixtures/legacy_complete_false.json +0 -0
  39. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/fixtures/legacy_complete_true.json +0 -0
  40. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/fixtures/legacy_no_flag.json +0 -0
  41. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_admission.py +0 -0
  42. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_ascii.py +0 -0
  43. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_benchmarks.py +0 -0
  44. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_cli.py +0 -0
  45. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_collector.py +0 -0
  46. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_cpu.py +0 -0
  47. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_fixtures.py +0 -0
  48. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_html.py +0 -0
  49. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_plugin.py +0 -0
  50. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_runtime_admission.py +0 -0
  51. {pytest_timing-0.2.0 → pytest_timing-0.3.0}/tests/test_trace.py +0 -0
@@ -14,9 +14,9 @@ schedules a run from the previous one. For usage, see the [README](README.md).
14
14
  | `cli.py` | `pytest-timing render` and `pytest-timing merge`. |
15
15
  | `fixtures.py` | Times shared fixture set-up where tests run and lists dependencies on reports. |
16
16
  | `schedule.py` | Duration estimates, the fixture-aware cost model and the planner. Free of xdist. |
17
- | `demand.py` | Declared CPU demand: the `timing_cpu` marker, the fixture decorator, the declarations a worker sends, and the meter that records each test's CPU work. |
17
+ | `demand.py` | Declared CPU demand: the `timing_cpu` marker, the fixture decorator, the declarations a worker sends, and the meter that records each test's CPU work and resident memory. |
18
18
  | `admission.py` | One host's CPU budget, reservations, fair waiting line and pressure feedback. Free of xdist and platform reads. |
19
- | `telemetry.py` | Platform reads: affinity and cgroup quota, CPU time of a process tree, pressure and throttling. |
19
+ | `telemetry.py` | Platform reads: affinity and cgroup quota, CPU time and resident memory of a process tree, pressure and throttling. |
20
20
  | `xdist_scheduler.py` | The xdist scheduler that drives workers with the planner and the admission gate, in `load` and `worksteal` mode. |
21
21
  | `xdist_compat.py` | Everything that reads pytest-xdist internals (versions, worker detection, why the controller stopped, the custom worker event and controller command). |
22
22
 
@@ -85,10 +85,13 @@ optimistic there.
85
85
 
86
86
  `Estimates.from_run` subtracts shared set-up and runtime admission waits from each
87
87
  attempt's duration to estimate the test's *own* time. Crashed attempts and nonpositive
88
- results are excluded. For each node id it uses the longest clean attempt, or the
89
- longest contended attempt when no clean one exists. Tests without a usable estimate
90
- get the mean of those estimates, or zero if none exist. Fixture dependencies are
91
- combined across attempts, and each fixture key's set-up cost is the longest recorded.
88
+ results are excluded. For each node id it ranks the attempts clean before contended
89
+ and passing before failing, and takes the median of the best rank present: a failed
90
+ attempt usually stops early, and the longest attempt grows with the number of
91
+ attempts, so a history merged from several runs would drift upward. Tests without a
92
+ usable estimate get the mean of those estimates, or zero if none exist. Fixture
93
+ dependencies are combined across attempts, and each fixture key's set-up cost is the
94
+ median of the recorded ones.
92
95
 
93
96
  A missing or unreadable history file leaves xdist's scheduler in place unless an
94
97
  explicit CPU budget was requested. With that budget the custom scheduler still runs,
@@ -370,11 +373,53 @@ naming it in `/proc/self/cgroup` wins over the `0::` v2 line, whose hierarchy ha
370
373
  controller there. Throttling is read from the same group's `cpu.stat` (`throttled_usec`
371
374
  on v2, `throttled_time` in nanoseconds on v1).
372
375
 
376
+ ## Memory admission
377
+
378
+ Memory goes through the same gate as CPU, as a second `Admission` per domain counted
379
+ in bytes, and a test is granted only when its reservation fits both: `_raise` checks
380
+ every gate with `fits` first and reserves on all of them or none, and a refused worker
381
+ waits in every line with its respective need. Reservations, releases, forced
382
+ admissions, withdrawals and stall resolution act on both gates in step, so their
383
+ `busy` and `idle` states never disagree. The rules that let CPU admission exceed its
384
+ limit are safe for memory for the same reasons they are safe for CPU: backfilling
385
+ never exceeds the limit (it lends out slots pledged to the head), a request above the
386
+ limit is clamped so the test runs alone rather than never, and a forced admission
387
+ happens only when nothing runs in the domain, so nothing else's memory is at risk.
388
+ Pressure feedback moves only the CPU limit.
389
+
390
+ Demand comes from history, never from declarations. `Estimates.from_run` keeps, per
391
+ test, the largest `peak - base` any attempt showed (the worst case is what an
392
+ out-of-memory kill depends on), and per shared fixture key the memory the attempt that
393
+ paid its set-up kept resident (`after - base`, split evenly when one attempt paid for
394
+ several). `Costs.charge` adds a `memory` to each `Charge`: the test's rise plus what
395
+ the fixtures alive around it keep, projected through the lane's fixture state exactly
396
+ as CPU holds are, since workers report nothing about memory at run time. A fixture with
397
+ recorded memory belongs to its tests' families even when its set-up was too quick to
398
+ matter for time. `_need_memory` is the twin of `_need`; an idle worker reserves what
399
+ its fixtures keep, and a worker whose next test could never fit next to what the
400
+ other workers' fixtures keep is parked like one blocked by CPU holds. A fixture
401
+ reached at run time adds its recorded memory to the queued charges when its
402
+ request is granted.
403
+
404
+ The budget is `--timing-memory`: a size, or `auto`, which takes 80% of the smallest
405
+ memory the domain's workers report (physical memory capped by the cgroup limit,
406
+ `memory.max` on v2 and `memory.limit_in_bytes` on v1, read like the CPU quota). The
407
+ share is deliberate: estimates are rises above each worker's footprint, and the
408
+ footprints, the controller and the rest of the host are not in them. Without a
409
+ recorded run every estimate is zero and the gate admits everything; the summary says
410
+ so. The planner ignores memory: with tests kept apart by the gate, balancing lanes by
411
+ memory would add little, and the lane plan can still move.
412
+
373
413
  ## Measuring CPU work
374
414
 
375
415
  `ProcessTreeClock` uses `os.times()` for worker CPU time and, on POSIX, reaped child
376
416
  CPU time. Where available, it adds live descendants through
377
- `/proc/<pid>/task/*/children` and `/proc/<pid>/stat` on Linux, or psutil otherwise.
417
+ `/proc/<pid>/task/*/children` and `/proc/<pid>/stat` on Linux, through `libproc` on
418
+ macOS (`proc_listchildpids` to find them, `proc_pidinfo` for their task times, in
419
+ Mach time units converted with `mach_timebase_info`), or psutil otherwise. psutil is
420
+ the last resort because its `children` scans the whole process table, about ten
421
+ milliseconds on macOS, and the clock is read several times per test: with psutil
422
+ installed, 3,000 trivial tests went from under a second to over a minute.
378
423
  Readings account for a waited-for child moving from the live total into the reaped
379
424
  total. Process discovery is a snapshot, so exits during traversal or descendants
380
425
  that outlive or detach from their parents can leave gaps. Every record reports its
@@ -389,6 +434,45 @@ the host's PSI `some` share and whether the cgroup was throttled during
389
434
  the test; `Estimates.from_run` treats such attempts as contended and prefers a clean
390
435
  attempt of the same test when one exists, keeping the contended ones in the file.
391
436
 
437
+ ## Measuring memory
438
+
439
+ Memory is recorded so that a later run can keep tests that need a lot of it from
440
+ running at the same time; nothing schedules on it yet. `ResidentMemory` reads the
441
+ resident set size of the process running the tests and, where they can be listed, of
442
+ its live descendants: `/proc/self/statm` and the `/proc` walk shared with CPU
443
+ measurement on Linux, the `libproc` reader shared with the CPU clock on macOS,
444
+ `GetProcessMemoryInfo` on Windows for the worker alone. psutil fills in what the
445
+ platform readers cannot do, today descendants on Windows, and is never preferred
446
+ over a native reader. Descendants are summed, so pages they share count more than
447
+ once; the total errs on the large side.
448
+
449
+ A high-water mark such as `ru_maxrss` never comes down, so it cannot say what one
450
+ test needed: only the test that first raised the worker's peak would show anything.
451
+ `MemorySampler` therefore polls from a daemon thread while a window is open and
452
+ keeps the highest total. The worker's own size is read every 20 ms; it costs about a
453
+ microsecond on macOS and ten on Linux. Descendants are listed at most every 100 ms,
454
+ and while none are found the interval doubles up to a second, since listing costs an
455
+ order of magnitude more (and far more with psutil). Opening or closing a window never
456
+ lists descendants by itself, so a fast test costs two readings of its own process, a
457
+ few microseconds. The meter opens the window at `pytest_runtest_setup` and closes it
458
+ when the teardown report is made, so shared fixture set-ups are charged to the test
459
+ that paid for them, as their time is. The window goes on the teardown report as
460
+ `timing_memory` and into the JSON as `memory`: `base`, `peak` and `after` in bytes,
461
+ and the coverage (`tree` or `self`). A platform with no reading records nothing.
462
+ The sampler sleeps between windows and is closed at `pytest_unconfigure`.
463
+
464
+ A test shorter than the sampling interval is seen only at its edges: its record is
465
+ what was resident before and after it, and a buffer allocated and freed inside it
466
+ is missed. Faulting in enough memory to matter takes longer than one interval.
467
+
468
+ `peak - base` is the attempt's rise: what it needed on top of the worker's footprint.
469
+ `after - base` is what stayed resident, which for the first test of a session fixture
470
+ is roughly the fixture. Both are biased by allocator behaviour: a heap that already
471
+ grew for an earlier test can serve a later one without raising RSS, so a rise can
472
+ undercount a test whose allocations reuse freed heap, and a freed buffer the allocator
473
+ keeps can leave `after` high. Large buffers and subprocess memory, the usual causes of
474
+ an out-of-memory kill, are mapped and unmapped directly and measure well.
475
+
392
476
  ## Feedback
393
477
 
394
478
  Admission's pressure feedback moves a domain's limit, never its budget, on evidence
@@ -1,5 +1,33 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.3.0
4
+
5
+ - Estimate a test's cost from the median of its best attempts instead of its
6
+ longest one. The longest attempt grows with the number of attempts, so a history
7
+ merged from several runs drifted upward and a flaky test's worst run was taken as
8
+ its cost. Attempts are ranked clean before contended and passing before failing,
9
+ and fixture set-up costs use the median too.
10
+
11
+ - Memory-aware admission under pytest-xdist. `--timing-memory SIZE|auto` (ini
12
+ `timing_memory`, env `PYTEST_TIMING_MEMORY`) sets a memory budget per host, and the
13
+ memory each test needed in the run at `--timing-schedule` keeps tests apart whose
14
+ recorded needs would not fit in it together. Nothing is declared: the first run
15
+ records, the next one gates. Memory and CPU budgets are checked together, so a
16
+ test starts only when both fit. A shared fixture keeps what its first test left
17
+ resident reserved while it is alive. The run's JSON gains `memory` with the budget
18
+ and admission summary.
19
+ - On macOS, CPU time and memory of live subprocesses are read through `libproc`
20
+ instead of psutil. psutil's child listing scans the whole process table, and the
21
+ CPU clock is read several times per test: with psutil installed, `--timing` on
22
+ 3,000 trivial tests took over a minute instead of under a second. psutil is now
23
+ used only where no native reader covers subprocesses, which is Windows.
24
+ - Record each test's resident memory. A sampler thread in the process running the
25
+ tests polls the resident set size of the worker (and, on Linux, macOS or with
26
+ psutil, its live subprocesses) while a test runs. Test JSON records now include `memory` when
27
+ the platform provides a reading: `base` before the test's set-up, `peak` during
28
+ it and `after` at its teardown, in bytes, with the measurement coverage. Nothing
29
+ schedules on it yet.
30
+
3
31
  ## 0.2.0
4
32
 
5
33
  - Raise the minimum supported pytest-xdist version to 3.7. Running without xdist
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pytest-timing
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Record test timings under pytest-xdist and render cargo-style timing reports (ASCII Gantt and HTML).
5
5
  Project-URL: Homepage, https://github.com/messense/pytest-timing
6
6
  Author-email: messense <messense@icloud.com>
@@ -78,10 +78,11 @@ pip install "pytest-timing[xdist]" # with pytest-xdist
78
78
  ```
79
79
 
80
80
  Python 3.10+ and pytest 7.3+. Distributed runs need pytest-xdist 3.7+.
81
- CPU records include live subprocesses on Linux through `/proc`, or through
82
- [psutil](https://pypi.org/project/psutil/) when installed. Without either, POSIX
83
- records include children the worker has waited for; Windows records cover only the
84
- worker itself. Each CPU record identifies its measurement coverage.
81
+ CPU and memory records include live subprocesses on Linux through `/proc`, on macOS
82
+ through `libproc`, and on Windows through [psutil](https://pypi.org/project/psutil/)
83
+ when installed. Without psutil, Windows CPU records cover only the worker itself and
84
+ memory records only the worker; other POSIX platforms count children the worker has
85
+ waited for. Each record identifies its measurement coverage.
85
86
 
86
87
  ## Options
87
88
 
@@ -94,6 +95,7 @@ worker itself. Each CPU record identifies its measurement coverage.
94
95
  | `--timing-html-file PATH`, `--timing-json-file PATH`, `--timing-trace-file PATH` | Same, to an explicit path. |
95
96
  | `--timing-schedule PATH` | Under xdist, plan the workers' queues from the JSON run at `PATH`. |
96
97
  | `--timing-cpus N\|auto` | Under xdist, admit tests against a budget of `N` CPU slots per host (`auto` detects it). |
98
+ | `--timing-memory SIZE\|auto` | Under xdist, keep tests whose recorded memory would not fit in `SIZE` together from running at once (`auto` takes 80% of the host's memory). |
97
99
  | `--timing-top=N` | Rows in the slowest-tests section (default 10, `0` hides it). |
98
100
  | `--timing-min=SECONDS` | Hide tests shorter than this from the slowest-tests section. |
99
101
  | `--timing-ascii-style=unicode\|ascii` | Chart glyphs. |
@@ -107,7 +109,8 @@ and schedule paths are resolved from pytest's root directory.
107
109
 
108
110
  The ini key `timing` is a boolean; `timing_html`, `timing_json` and `timing_trace`
109
111
  accept paths or `true` for the default filenames. Other ini keys are
110
- `timing_schedule`, `timing_cpus`, `timing_top`, `timing_min` and `timing_ascii_style`.
112
+ `timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
113
+ `timing_ascii_style`.
111
114
  Environment variables use the uppercase key prefixed by `PYTEST_`, for example
112
115
  `PYTEST_TIMING=1`, `PYTEST_TIMING_HTML=path` or `PYTEST_TIMING_CPUS=auto`.
113
116
  `--timing-width` is command-line only.
@@ -120,7 +123,13 @@ environment, which wins over ini. Timing is enabled if any source requests it:
120
123
 
121
124
  **JSON** is the plugin's own record of the run: tests with their worker, phases and
122
125
  shared fixtures, plus CPU telemetry when available (elapsed time, measured CPU time,
123
- statically declared demand and time waiting for CPU slots). It also records worker
126
+ statically declared demand and time waiting for CPU slots) and the resident memory
127
+ of the worker's process tree around each test, in bytes: `base` before its set-up,
128
+ `peak` during it, and `after` at its teardown. The difference between `peak` and
129
+ `base` is what the test needed on top of the worker's existing footprint; `after`
130
+ minus `base` is what stayed, such as a shared fixture it set up. With
131
+ `--timing-memory`, the run's `memory` describes the budget and how admission went.
132
+ It also records worker
124
133
  lifecycles, detected host CPU environments, and how the session ended (`finished`,
125
134
  `collect_only`, `interrupted`, `aborted`, `internal_error`, with pytest's reason). It is
126
135
  the input to the CLI below and to `--timing-schedule`.
@@ -280,6 +289,42 @@ domains with their own budgets. With `-x` or `--maxfail`, xdist shuts every work
280
289
  down; a test a worker was holding without having been admitted is withdrawn and
281
290
  never starts.
282
291
 
292
+ ## Keep memory-hungry tests apart
293
+
294
+ Two tests that each need several gigabytes are fine on their own and fatal together:
295
+ the kernel kills a worker, or the whole job. Nothing needs declaring for this. Every
296
+ run records how much resident memory each test needed on top of its worker's footprint
297
+ (`memory` in the JSON), and the next run keeps tests apart when their recorded needs
298
+ would not fit in the budget at the same time:
299
+
300
+ ```
301
+ pytest -n 4 --timing-json --timing-schedule pytest-timing.json --timing-memory 12G
302
+ ```
303
+
304
+ `--timing-memory` takes a size with a binary suffix (`512M`, `12G`) or `auto`, which
305
+ takes 80% of physical memory, capped by the cgroup memory limit. It reads the memory
306
+ each test needed from the run at `--timing-schedule`, so the first run only records;
307
+ the summary says how many tests had recorded memory. A test with no record weighs
308
+ nothing until it has run once. A shared fixture keeps what its first test left resident
309
+ (a loaded model, say) reserved for as long as it is alive, on that worker.
310
+
311
+ What to expect:
312
+
313
+ - The gate is the same fair waiting line as for CPU: a worker whose next test does
314
+ not fit waits with its fixtures alive, and the oldest request goes first. Memory
315
+ and CPU budgets are checked together; a test starts only when both fit.
316
+ - Estimates are the largest rise any recorded attempt showed. They are relative to
317
+ the worker's footprint, so `auto` leaves a fifth of the host for the workers
318
+ themselves, the controller and everything else; set a smaller budget on a shared
319
+ machine.
320
+ - A test recorded above the budget runs alone, and the summary counts it. Allocator
321
+ behaviour can make an estimate low: a test that reuses heap an earlier test freed
322
+ shows a smaller rise than it needs. Large buffers and subprocesses, the usual
323
+ cause of an out-of-memory kill, measure well.
324
+ - Memory does not enter the planner: lanes are still balanced by duration and fixture
325
+ cost, and the gate only delays starts. Under `--dist load` and `worksteal`, like
326
+ CPU admission.
327
+
283
328
  ## Re-render or merge saved runs
284
329
 
285
330
  ```
@@ -52,10 +52,11 @@ pip install "pytest-timing[xdist]" # with pytest-xdist
52
52
  ```
53
53
 
54
54
  Python 3.10+ and pytest 7.3+. Distributed runs need pytest-xdist 3.7+.
55
- CPU records include live subprocesses on Linux through `/proc`, or through
56
- [psutil](https://pypi.org/project/psutil/) when installed. Without either, POSIX
57
- records include children the worker has waited for; Windows records cover only the
58
- worker itself. Each CPU record identifies its measurement coverage.
55
+ CPU and memory records include live subprocesses on Linux through `/proc`, on macOS
56
+ through `libproc`, and on Windows through [psutil](https://pypi.org/project/psutil/)
57
+ when installed. Without psutil, Windows CPU records cover only the worker itself and
58
+ memory records only the worker; other POSIX platforms count children the worker has
59
+ waited for. Each record identifies its measurement coverage.
59
60
 
60
61
  ## Options
61
62
 
@@ -68,6 +69,7 @@ worker itself. Each CPU record identifies its measurement coverage.
68
69
  | `--timing-html-file PATH`, `--timing-json-file PATH`, `--timing-trace-file PATH` | Same, to an explicit path. |
69
70
  | `--timing-schedule PATH` | Under xdist, plan the workers' queues from the JSON run at `PATH`. |
70
71
  | `--timing-cpus N\|auto` | Under xdist, admit tests against a budget of `N` CPU slots per host (`auto` detects it). |
72
+ | `--timing-memory SIZE\|auto` | Under xdist, keep tests whose recorded memory would not fit in `SIZE` together from running at once (`auto` takes 80% of the host's memory). |
71
73
  | `--timing-top=N` | Rows in the slowest-tests section (default 10, `0` hides it). |
72
74
  | `--timing-min=SECONDS` | Hide tests shorter than this from the slowest-tests section. |
73
75
  | `--timing-ascii-style=unicode\|ascii` | Chart glyphs. |
@@ -81,7 +83,8 @@ and schedule paths are resolved from pytest's root directory.
81
83
 
82
84
  The ini key `timing` is a boolean; `timing_html`, `timing_json` and `timing_trace`
83
85
  accept paths or `true` for the default filenames. Other ini keys are
84
- `timing_schedule`, `timing_cpus`, `timing_top`, `timing_min` and `timing_ascii_style`.
86
+ `timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
87
+ `timing_ascii_style`.
85
88
  Environment variables use the uppercase key prefixed by `PYTEST_`, for example
86
89
  `PYTEST_TIMING=1`, `PYTEST_TIMING_HTML=path` or `PYTEST_TIMING_CPUS=auto`.
87
90
  `--timing-width` is command-line only.
@@ -94,7 +97,13 @@ environment, which wins over ini. Timing is enabled if any source requests it:
94
97
 
95
98
  **JSON** is the plugin's own record of the run: tests with their worker, phases and
96
99
  shared fixtures, plus CPU telemetry when available (elapsed time, measured CPU time,
97
- statically declared demand and time waiting for CPU slots). It also records worker
100
+ statically declared demand and time waiting for CPU slots) and the resident memory
101
+ of the worker's process tree around each test, in bytes: `base` before its set-up,
102
+ `peak` during it, and `after` at its teardown. The difference between `peak` and
103
+ `base` is what the test needed on top of the worker's existing footprint; `after`
104
+ minus `base` is what stayed, such as a shared fixture it set up. With
105
+ `--timing-memory`, the run's `memory` describes the budget and how admission went.
106
+ It also records worker
98
107
  lifecycles, detected host CPU environments, and how the session ended (`finished`,
99
108
  `collect_only`, `interrupted`, `aborted`, `internal_error`, with pytest's reason). It is
100
109
  the input to the CLI below and to `--timing-schedule`.
@@ -254,6 +263,42 @@ domains with their own budgets. With `-x` or `--maxfail`, xdist shuts every work
254
263
  down; a test a worker was holding without having been admitted is withdrawn and
255
264
  never starts.
256
265
 
266
+ ## Keep memory-hungry tests apart
267
+
268
+ Two tests that each need several gigabytes are fine on their own and fatal together:
269
+ the kernel kills a worker, or the whole job. Nothing needs declaring for this. Every
270
+ run records how much resident memory each test needed on top of its worker's footprint
271
+ (`memory` in the JSON), and the next run keeps tests apart when their recorded needs
272
+ would not fit in the budget at the same time:
273
+
274
+ ```
275
+ pytest -n 4 --timing-json --timing-schedule pytest-timing.json --timing-memory 12G
276
+ ```
277
+
278
+ `--timing-memory` takes a size with a binary suffix (`512M`, `12G`) or `auto`, which
279
+ takes 80% of physical memory, capped by the cgroup memory limit. It reads the memory
280
+ each test needed from the run at `--timing-schedule`, so the first run only records;
281
+ the summary says how many tests had recorded memory. A test with no record weighs
282
+ nothing until it has run once. A shared fixture keeps what its first test left resident
283
+ (a loaded model, say) reserved for as long as it is alive, on that worker.
284
+
285
+ What to expect:
286
+
287
+ - The gate is the same fair waiting line as for CPU: a worker whose next test does
288
+ not fit waits with its fixtures alive, and the oldest request goes first. Memory
289
+ and CPU budgets are checked together; a test starts only when both fit.
290
+ - Estimates are the largest rise any recorded attempt showed. They are relative to
291
+ the worker's footprint, so `auto` leaves a fifth of the host for the workers
292
+ themselves, the controller and everything else; set a smaller budget on a shared
293
+ machine.
294
+ - A test recorded above the budget runs alone, and the summary counts it. Allocator
295
+ behaviour can make an estimate low: a test that reuses heap an earlier test freed
296
+ shows a smaller rise than it needs. Large buffers and subprocesses, the usual
297
+ cause of an out-of-memory kill, measure well.
298
+ - Memory does not enter the planner: lanes are still balanced by duration and fixture
299
+ cost, and the gate only delays starts. Under `--dist load` and `worksteal`, like
300
+ CPU admission.
301
+
257
302
  ## Re-render or merge saved runs
258
303
 
259
304
  ```
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pytest-timing"
7
- version = "0.2.0"
7
+ version = "0.3.0"
8
8
  description = "Record test timings under pytest-xdist and render cargo-style timing reports (ASCII Gantt and HTML)."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -4,6 +4,6 @@ from __future__ import annotations
4
4
 
5
5
  from pytest_timing.demand import cpu
6
6
 
7
- __version__ = "0.2.0"
7
+ __version__ = "0.3.0"
8
8
 
9
9
  __all__ = ["__version__", "cpu"]
@@ -17,7 +17,16 @@ from __future__ import annotations
17
17
  from dataclasses import dataclass
18
18
  from typing import Any
19
19
 
20
- from pytest_timing.model import PHASES, CpuRecord, Phase, Run, RunInfo, TestSpan, Worker
20
+ from pytest_timing.model import (
21
+ PHASES,
22
+ CpuRecord,
23
+ MemoryRecord,
24
+ Phase,
25
+ Run,
26
+ RunInfo,
27
+ TestSpan,
28
+ Worker,
29
+ )
21
30
 
22
31
  CRASH_WHEN = "???" # xdist synthesises a report with this ``when`` for crashed items
23
32
  RETRIED = "rerun"
@@ -38,6 +47,7 @@ class PhaseReport:
38
47
  received: float = 0.0 # epoch seconds when the controller saw the report
39
48
  fixtures: dict[str, float | None] | None = None # shared fixtures, setup and call
40
49
  cpu: dict[str, Any] | None = None # the worker's CPU record, on the teardown report
50
+ memory: dict[str, Any] | None = None # the worker's memory record, likewise
41
51
  execution: tuple[int, int] | None = None # collection index, attempt on that worker
42
52
 
43
53
 
@@ -131,6 +141,8 @@ class Collector:
131
141
  span.fixtures.update(report.fixtures)
132
142
  if report.cpu:
133
143
  span.cpu = CpuRecord.from_dict(report.cpu)
144
+ if report.memory:
145
+ span.memory = MemoryRecord.from_dict(report.memory)
134
146
  span.start = min(span.start, start)
135
147
  span.stop = max(span.stop, stop)
136
148
  self._apply_outcome(span, report)
@@ -1,4 +1,4 @@
1
- """CPU declarations and per-test measurement in the process running the tests.
1
+ """CPU declarations and per-test CPU and memory measurement where the tests run.
2
2
 
3
3
  Declarations precede xdist collection reports. Dynamic fixture setup requests
4
4
  reserve slots before execution; holds follow actual setup/finalization events.
@@ -16,11 +16,13 @@ import pytest
16
16
 
17
17
  from pytest_timing import xdist_compat
18
18
  from pytest_timing.fixtures import FixtureTimer, _key_for, defined_at, fixture_key, fixturedefs
19
- from pytest_timing.telemetry import Pressure, ProcessTreeClock, host_cpu
19
+ from pytest_timing.telemetry import MemorySampler, Pressure, ProcessTreeClock, host_cpu
20
20
 
21
21
  MARKER = "timing_cpu"
22
22
  FIXTURE_ATTR = "_pytest_timing_cpu"
23
23
  REPORT_ATTR = "timing_cpu"
24
+ MEMORY_ATTR = "timing_memory"
25
+ """Report attribute carrying the attempt's resident-memory window, on teardown."""
24
26
  EXECUTION_ATTR = "timing_execution"
25
27
  """Private report identity: collection index and attempt, including retries."""
26
28
  EVENT = "timing_cpu"
@@ -237,7 +239,7 @@ class Declarations:
237
239
 
238
240
 
239
241
  class CpuMeter:
240
- """Measures each test's CPU work and attaches it to the teardown report.
242
+ """Measures each test's CPU work and memory and attaches them to the teardown report.
241
243
 
242
244
  Runs wherever tests run: in every xdist worker, or in the main process without
243
245
  xdist. After collection it also sends the :class:`Declarations` to the controller
@@ -255,6 +257,9 @@ class CpuMeter:
255
257
  set-up's slots (``timing_request``) before the set-up starts and blocks until
256
258
  they are granted (``timing_grant``), so such a set-up goes through the same
257
259
  gate as a declared one.
260
+
261
+ ``memory`` samples resident memory from set-up to the teardown report; the
262
+ window (``timing_memory``) goes on that report next to the CPU record.
258
263
  """
259
264
 
260
265
  def __init__(
@@ -263,11 +268,13 @@ class CpuMeter:
263
268
  send: Callable[[str, dict[str, Any]], None] | None = None,
264
269
  clock: ProcessTreeClock | None = None,
265
270
  pressure: Pressure | None = None,
271
+ memory: MemorySampler | None = None,
266
272
  ) -> None:
267
273
  self.timer = timer
268
274
  self.send = send
269
275
  self.clock = clock or ProcessTreeClock()
270
276
  self.pressure = pressure or Pressure()
277
+ self.memory = memory
271
278
  self.declarations: Declarations | None = None
272
279
  self.cancelled: set[int] = set() # item indices withdrawn by the controller
273
280
  self._live = False # can the controller reach this worker while a test runs
@@ -427,6 +434,8 @@ class CpuMeter:
427
434
  self._started = time.perf_counter()
428
435
  self._work = self.clock.seconds()
429
436
  self._throttled = self.pressure.throttled()
437
+ if self.memory is not None:
438
+ self.memory.begin()
430
439
 
431
440
  @pytest.hookimpl(hookwrapper=True)
432
441
  def pytest_runtest_makereport(self, item: pytest.Item, call: Any) -> Generator[None, Any, None]:
@@ -436,6 +445,18 @@ class CpuMeter:
436
445
  setattr(outcome.get_result(), EXECUTION_ATTR, (index, self._attempts.get(index, 0)))
437
446
  if call.when != "teardown" or not self._started:
438
447
  return
448
+ window = self.memory.end() if self.memory is not None else None
449
+ if window is not None:
450
+ setattr(
451
+ outcome.get_result(),
452
+ MEMORY_ATTR,
453
+ {
454
+ "base": window.base,
455
+ "peak": window.peak,
456
+ "after": window.after,
457
+ "coverage": window.coverage,
458
+ },
459
+ )
439
460
  elapsed = max(0.0, time.perf_counter() - self._started - self.timer.wait)
440
461
  work = self.clock.seconds() - self._work - self.timer.wait_work
441
462
  throttled_now = self.pressure.throttled()
@@ -456,3 +477,7 @@ class CpuMeter:
456
477
  }
457
478
  setattr(outcome.get_result(), REPORT_ATTR, record)
458
479
  self._started = 0.0
480
+
481
+ def pytest_unconfigure(self) -> None:
482
+ if self.memory is not None:
483
+ self.memory.close()
@@ -18,6 +18,9 @@ Semantics that every renderer must agree on live here, once:
18
18
  time, measured CPU work of the worker and its descendants, declared demand, how much
19
19
  of the process tree the measurement covered, and how long the controller held the
20
20
  test back for CPU slots. ``RunInfo.cpu`` describes the hosts and the budget.
21
+ * ``TestSpan.memory`` is the resident memory of the worker's process tree around the
22
+ attempt (:class:`MemoryRecord`): before it, at its peak, and after it, in bytes.
23
+ ``RunInfo.memory`` describes the memory budget and its admission, when one was set.
21
24
  """
22
25
 
23
26
  from __future__ import annotations
@@ -201,6 +204,49 @@ CONTENDED_PRESSURE = 0.25
201
204
  """PSI share above which an attempt is not a clean observation of its duration."""
202
205
 
203
206
 
207
+ @dataclass(slots=True)
208
+ class MemoryRecord:
209
+ """Resident memory around one attempt, in bytes; see :mod:`pytest_timing.telemetry`.
210
+
211
+ ``base`` is what the worker's process tree had resident when set-up began,
212
+ ``peak`` the highest reading until the teardown report, ``after`` the reading at
213
+ that report. ``coverage`` says whether descendants were counted (``tree``) or
214
+ only the worker (``self``). A worker's heap rarely shrinks, so ``rise`` is
215
+ what the attempt needed on top of what was already there, and ``retained`` what
216
+ stayed resident afterwards: a shared fixture it set up, or a leak.
217
+ """
218
+
219
+ base: int = 0
220
+ peak: int = 0
221
+ after: int = 0
222
+ coverage: str = "none"
223
+
224
+ @property
225
+ def rise(self) -> int:
226
+ return max(0, self.peak - self.base)
227
+
228
+ @property
229
+ def retained(self) -> int:
230
+ return max(0, self.after - self.base)
231
+
232
+ def to_dict(self) -> dict[str, Any]:
233
+ return {
234
+ "base": self.base,
235
+ "peak": self.peak,
236
+ "after": self.after,
237
+ "coverage": self.coverage,
238
+ }
239
+
240
+ @classmethod
241
+ def from_dict(cls, data: dict[str, Any]) -> MemoryRecord:
242
+ return cls(
243
+ base=int(data.get("base", 0)),
244
+ peak=int(data.get("peak", 0)),
245
+ after=int(data.get("after", 0)),
246
+ coverage=str(data.get("coverage", "none")),
247
+ )
248
+
249
+
204
250
  @dataclass(slots=True)
205
251
  class TestSpan:
206
252
  """One attempt at running one test occurrence on one worker.
@@ -221,6 +267,7 @@ class TestSpan:
221
267
  occurrence: int = 0
222
268
  fixtures: dict[str, float | None] = field(default_factory=dict)
223
269
  cpu: CpuRecord | None = None
270
+ memory: MemoryRecord | None = None
224
271
 
225
272
  @property
226
273
  def duration(self) -> float:
@@ -258,12 +305,15 @@ class TestSpan:
258
305
  doc["fixtures"] = {key: _opt_round(seconds) for key, seconds in self.fixtures.items()}
259
306
  if self.cpu is not None:
260
307
  doc["cpu"] = self.cpu.to_dict()
308
+ if self.memory is not None:
309
+ doc["memory"] = self.memory.to_dict()
261
310
  return doc
262
311
 
263
312
  @classmethod
264
313
  def from_dict(cls, data: dict[str, Any]) -> TestSpan:
265
314
  fixtures = data.get("fixtures") or {}
266
315
  cpu = data.get("cpu")
316
+ memory = data.get("memory")
267
317
  return cls(
268
318
  nodeid=str(data["nodeid"]),
269
319
  worker=str(data["worker"]),
@@ -281,6 +331,7 @@ class TestSpan:
281
331
  for key, seconds in dict(fixtures).items()
282
332
  },
283
333
  cpu=CpuRecord.from_dict(dict(cpu)) if isinstance(cpu, dict) else None,
334
+ memory=MemoryRecord.from_dict(dict(memory)) if isinstance(memory, dict) else None,
284
335
  )
285
336
 
286
337
 
@@ -351,6 +402,7 @@ class RunInfo:
351
402
  dist: str | None = None
352
403
  numprocesses: int | None = None
353
404
  cpu: dict[str, Any] | None = None # hosts, budget and admission summary
405
+ memory: dict[str, Any] | None = None # the memory budget and its admission summary
354
406
 
355
407
  @property
356
408
  def wall(self) -> float:
@@ -381,6 +433,7 @@ class RunInfo:
381
433
  "dist": self.dist,
382
434
  "numprocesses": self.numprocesses,
383
435
  "cpu": self.cpu,
436
+ "memory": self.memory,
384
437
  }
385
438
 
386
439
  @classmethod
@@ -407,6 +460,7 @@ class RunInfo:
407
460
  dist=None if data.get("dist") is None else str(data["dist"]),
408
461
  numprocesses=None if numprocesses is None else int(numprocesses),
409
462
  cpu=dict(data["cpu"]) if isinstance(data.get("cpu"), dict) else None,
463
+ memory=dict(data["memory"]) if isinstance(data.get("memory"), dict) else None,
410
464
  )
411
465
 
412
466