pytest-timing 0.2.0__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/ARCHITECTURE.md +109 -7
  2. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/CHANGELOG.md +49 -0
  3. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/PKG-INFO +68 -9
  4. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/README.md +67 -8
  5. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/pyproject.toml +1 -1
  6. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/__init__.py +1 -1
  7. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/admission.py +2 -1
  8. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/collector.py +32 -6
  9. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/demand.py +28 -3
  10. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/model.py +70 -0
  11. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/plugin.py +108 -10
  12. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/render/ascii.py +14 -2
  13. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/schedule.py +109 -13
  14. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/static/report.html +16 -3
  15. pytest_timing-0.3.1/src/pytest_timing/telemetry.py +834 -0
  16. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/xdist_scheduler.py +244 -66
  17. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_ascii.py +13 -0
  18. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_collector.py +13 -1
  19. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_html.py +15 -0
  20. pytest_timing-0.3.1/tests/test_memory.py +216 -0
  21. pytest_timing-0.3.1/tests/test_memory_gate.py +376 -0
  22. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_runtime_admission.py +5 -1
  23. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_schedule.py +39 -11
  24. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/uv.lock +1 -1
  25. pytest_timing-0.2.0/src/pytest_timing/telemetry.py +0 -346
  26. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/.github/workflows/ci.yml +0 -0
  27. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/.github/workflows/release.yml +0 -0
  28. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/.gitignore +0 -0
  29. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/LICENSE +0 -0
  30. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/benchmarks/cpu_bench.py +0 -0
  31. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/docs/report.png +0 -0
  32. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/examples/test_demo.py +0 -0
  33. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/cli.py +0 -0
  34. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/fixtures.py +0 -0
  35. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/outputs.py +0 -0
  36. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/render/__init__.py +0 -0
  37. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/render/html.py +0 -0
  38. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/render/trace.py +0 -0
  39. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/static/__init__.py +0 -0
  40. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/src/pytest_timing/xdist_compat.py +0 -0
  41. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/conftest.py +0 -0
  42. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/fixtures/legacy_complete_false.json +0 -0
  43. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/fixtures/legacy_complete_true.json +0 -0
  44. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/fixtures/legacy_no_flag.json +0 -0
  45. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_admission.py +0 -0
  46. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_benchmarks.py +0 -0
  47. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_cli.py +0 -0
  48. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_cpu.py +0 -0
  49. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_fixtures.py +0 -0
  50. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_plugin.py +0 -0
  51. {pytest_timing-0.2.0 → pytest_timing-0.3.1}/tests/test_trace.py +0 -0
@@ -14,9 +14,9 @@ schedules a run from the previous one. For usage, see the [README](README.md).
14
14
  | `cli.py` | `pytest-timing render` and `pytest-timing merge`. |
15
15
  | `fixtures.py` | Times shared fixture set-up where tests run and lists dependencies on reports. |
16
16
  | `schedule.py` | Duration estimates, the fixture-aware cost model and the planner. Free of xdist. |
17
- | `demand.py` | Declared CPU demand: the `timing_cpu` marker, the fixture decorator, the declarations a worker sends, and the meter that records each test's CPU work. |
17
+ | `demand.py` | Declared CPU demand: the `timing_cpu` marker, the fixture decorator, the declarations a worker sends, and the meter that records each test's CPU work and resident memory. |
18
18
  | `admission.py` | One host's CPU budget, reservations, fair waiting line and pressure feedback. Free of xdist and platform reads. |
19
- | `telemetry.py` | Platform reads: affinity and cgroup quota, CPU time of a process tree, pressure and throttling. |
19
+ | `telemetry.py` | Platform reads: affinity and cgroup quota, CPU time and resident memory of a process tree, pressure and throttling. |
20
20
  | `xdist_scheduler.py` | The xdist scheduler that drives workers with the planner and the admission gate, in `load` and `worksteal` mode. |
21
21
  | `xdist_compat.py` | Everything that reads pytest-xdist internals (versions, worker detection, why the controller stopped, the custom worker event and controller command). |
22
22
 
@@ -85,10 +85,13 @@ optimistic there.
85
85
 
86
86
  `Estimates.from_run` subtracts shared set-up and runtime admission waits from each
87
87
  attempt's duration to estimate the test's *own* time. Crashed attempts and nonpositive
88
- results are excluded. For each node id it uses the longest clean attempt, or the
89
- longest contended attempt when no clean one exists. Tests without a usable estimate
90
- get the mean of those estimates, or zero if none exist. Fixture dependencies are
91
- combined across attempts, and each fixture key's set-up cost is the longest recorded.
88
+ results are excluded. For each node id it ranks the attempts clean before contended
89
+ and passing before failing, and takes the median of the best rank present: a failed
90
+ attempt usually stops early, and the longest attempt grows with the number of
91
+ attempts, so a history merged from several runs would drift upward. Tests without a
92
+ usable estimate get the mean of those estimates, or zero if none exist. Fixture
93
+ dependencies are combined across attempts, and each fixture key's set-up cost is the
94
+ median of the recorded ones.
92
95
 
93
96
  A missing or unreadable history file leaves xdist's scheduler in place unless an
94
97
  explicit CPU budget was requested. With that budget the custom scheduler still runs,
@@ -370,11 +373,71 @@ naming it in `/proc/self/cgroup` wins over the `0::` v2 line, whose hierarchy ha
370
373
  controller there. Throttling is read from the same group's `cpu.stat` (`throttled_usec`
371
374
  on v2, `throttled_time` in nanoseconds on v1).
372
375
 
376
+ ## Memory admission
377
+
378
+ Memory goes through the same gate as CPU, as a second `Admission` per domain counted
379
+ in bytes, and a test is granted only when its reservation fits both: `_raise` checks
380
+ every gate with `fits` first and reserves on all of them or none, and a refused worker
381
+ waits in every line with its respective need. Reservations, releases, forced
382
+ admissions, withdrawals and stall resolution act on both gates in step, so their
383
+ `busy` and `idle` states never disagree. The rules that let CPU admission exceed its
384
+ limit are safe for memory for the same reasons they are safe for CPU: backfilling
385
+ never exceeds the limit (it lends out slots pledged to the head), a request above the
386
+ limit is clamped so the test runs alone rather than never, and a forced admission
387
+ happens only when nothing runs in the domain, so nothing else's memory is at risk.
388
+ Pressure feedback moves only the CPU limit.
389
+
390
+ Demand comes from history, never from declarations. `Estimates.from_run` keeps, per
391
+ test, the largest need any attempt showed (the worst case is what an out-of-memory
392
+ kill depends on), and per shared fixture key the memory the attempt that paid its
393
+ set-up kept resident (`after - base`, split evenly when one attempt paid for
394
+ several). A payer's rise spans its set-ups, so its own need is `peak - after`: what
395
+ it used beyond what stayed, which is the fixtures'. Session- and package-scoped
396
+ fixtures are the exception: what they keep is every worker's baseline
397
+ (`is_baseline`), never attributed, charged or projected. Every worker sets them up
398
+ once and never lets go, so gating on them cannot spare the host; it can only hold
399
+ tests back, or park a worker until stall resolution shuts it down, which is what a
400
+ large session fixture did before this rule. The first attempt on each worker is a
401
+ warm-up window (lazy imports, caches, the allocator's first growth), so its rise and
402
+ residual count only for a test or fixture with no other attempt.
403
+
404
+ `Costs.charge` adds a `memory` to each `Charge`: the test's need plus what the
405
+ module- and class-scoped fixtures alive around it keep, projected through the lane's
406
+ fixture state exactly as CPU holds are, since workers report nothing about memory at
407
+ run time. A fixture with recorded memory belongs to its tests' families even when its
408
+ set-up was too quick to matter for time. `_need_memory` is the twin of `_need`; an
409
+ idle worker reserves what its fixtures keep, and a worker whose next test could never
410
+ fit next to what the other workers' fixtures keep is parked like one blocked by CPU
411
+ holds. A fixture reached at run time adds its recorded memory to the queued charges
412
+ when its request is granted.
413
+
414
+ The budget is `--timing-memory`: a size, or `auto`, which takes 80% of the smallest
415
+ memory the domain's workers report (physical memory capped by the cgroup limit,
416
+ `memory.max` on v2 and `memory.limit_in_bytes` on v1, read like the CPU quota). The
417
+ share is deliberate: estimates are rises above each worker's footprint, and the
418
+ footprints (session fixtures included), the controller and the rest of the host are
419
+ not in them. Without a recorded run every estimate is zero and the gate admits
420
+ everything; the summary says so. The planner ignores memory: with tests kept apart by
421
+ the gate, balancing lanes by memory would add little, and the lane plan can still move.
422
+
423
+ Every `Admission` has a `kind` (`cpu` or `memory`), and a refusal records which kinds
424
+ refused (`refused`), as does passing a test over for what other workers keep. When
425
+ the worker is admitted, the wait it records (`AdmissionWait.gates`) says which gates
426
+ held it, and the collector puts it on the test's `cpu.wait` or `memory.wait`
427
+ accordingly. A worker that leaves while parked, having run nothing it waited for,
428
+ has no test to carry its wait: `_release_runtime` records it in `parked`, and each
429
+ gate's summary reports the parked time and worker count beside the tests' waits.
430
+
373
431
  ## Measuring CPU work
374
432
 
375
433
  `ProcessTreeClock` uses `os.times()` for worker CPU time and, on POSIX, reaped child
376
434
  CPU time. Where available, it adds live descendants through
377
- `/proc/<pid>/task/*/children` and `/proc/<pid>/stat` on Linux, or psutil otherwise.
435
+ `/proc/<pid>/task/*/children` and `/proc/<pid>/stat` on Linux, through `libproc` on
436
+ macOS (`proc_listchildpids` to find them, `proc_pidinfo` for their task times, in
437
+ Mach time units converted with `mach_timebase_info`), or psutil otherwise. psutil is
438
+ the last resort because its `children` scans the whole process table, about ten
439
+ milliseconds on macOS, and the clock is read several times per test: with psutil
440
+ installed, 3,000 trivial tests went from under a second to over a minute.
378
441
  Readings account for a waited-for child moving from the live total into the reaped
379
442
  total. Process discovery is a snapshot, so exits during traversal or descendants
380
443
  that outlive or detach from their parents can leave gaps. Every record reports its
@@ -389,6 +452,45 @@ the host's PSI `some` share and whether the cgroup was throttled during
389
452
  the test; `Estimates.from_run` treats such attempts as contended and prefers a clean
390
453
  attempt of the same test when one exists, keeping the contended ones in the file.
391
454
 
455
+ ## Measuring memory
456
+
457
+ Memory is recorded so that a later run can keep tests that need a lot of it from
458
+ running at the same time; nothing schedules on it yet. `ResidentMemory` reads the
459
+ resident set size of the process running the tests and, where they can be listed, of
460
+ its live descendants: `/proc/self/statm` and the `/proc` walk shared with CPU
461
+ measurement on Linux, the `libproc` reader shared with the CPU clock on macOS,
462
+ `GetProcessMemoryInfo` on Windows for the worker alone. psutil fills in what the
463
+ platform readers cannot do, today descendants on Windows, and is never preferred
464
+ over a native reader. Descendants are summed, so pages they share count more than
465
+ once; the total errs on the large side.
466
+
467
+ A high-water mark such as `ru_maxrss` never comes down, so it cannot say what one
468
+ test needed: only the test that first raised the worker's peak would show anything.
469
+ `MemorySampler` therefore polls from a daemon thread while a window is open and
470
+ keeps the highest total. The worker's own size is read every 20 ms; it costs about a
471
+ microsecond on macOS and ten on Linux. Descendants are listed at most every 100 ms,
472
+ and while none are found the interval doubles up to a second, since listing costs an
473
+ order of magnitude more (and far more with psutil). Opening or closing a window never
474
+ lists descendants by itself, so a fast test costs two readings of its own process, a
475
+ few microseconds. The meter opens the window at `pytest_runtest_setup` and closes it
476
+ when the teardown report is made, so shared fixture set-ups are charged to the test
477
+ that paid for them, as their time is. The window goes on the teardown report as
478
+ `timing_memory` and into the JSON as `memory`: `base`, `peak` and `after` in bytes,
479
+ and the coverage (`tree` or `self`). A platform with no reading records nothing.
480
+ The sampler sleeps between windows and is closed at `pytest_unconfigure`.
481
+
482
+ A test shorter than the sampling interval is seen only at its edges: its record is
483
+ what was resident before and after it, and a buffer allocated and freed inside it
484
+ is missed. Faulting in enough memory to matter takes longer than one interval.
485
+
486
+ `peak - base` is the attempt's rise: what it needed on top of the worker's footprint.
487
+ `after - base` is what stayed resident, which for the first test of a session fixture
488
+ is roughly the fixture. Both are biased by allocator behaviour: a heap that already
489
+ grew for an earlier test can serve a later one without raising RSS, so a rise can
490
+ undercount a test whose allocations reuse freed heap, and a freed buffer the allocator
491
+ keeps can leave `after` high. Large buffers and subprocess memory, the usual causes of
492
+ an out-of-memory kill, are mapped and unmapped directly and measure well.
493
+
392
494
  ## Feedback
393
495
 
394
496
  Admission's pressure feedback moves a domain's limit, never its budget, on evidence
@@ -1,5 +1,54 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.3.1
4
+
5
+ - Memory admission no longer charges what session- and package-scoped fixtures keep
6
+ resident. Every worker sets them up once and never lets go, so counting them once
7
+ per lane, and again against every other lane, could only hold tests back or park
8
+ a worker until it was shut down: on a suite with a large session fixture the gate
9
+ left two of eight workers nearly idle and made the run almost three times longer
10
+ while the host had tens of gigabytes free. Their memory is now the worker's
11
+ baseline, which the budget's headroom covers (#7).
12
+ - Memory estimates are what a test needs of its own. An attempt that set up shared
13
+ fixtures was charged its whole rise and, again, what the fixtures kept, twice the
14
+ residual; its own need is now the rise beyond what stayed. The first attempt on
15
+ each worker, whose window also covers the worker's warm-up, counts only for a
16
+ test or fixture with no other attempt (#7).
17
+ - Waits say which gate held the test. A test the memory gate held back carries the
18
+ seconds in `memory.wait`, beside `cpu.wait` for CPU slots, and the summary line
19
+ reports memory waits even when a CPU budget is set. Time a worker sat parked and
20
+ then left without running what it waited for is reported per gate as well, instead
21
+ of vanishing. The HTML report's table gains a "Held" column and its hover details a
22
+ "held" line, and the terminal's slowest-tests rows say how long each was held (#7).
23
+
24
+ ## 0.3.0
25
+
26
+ - Estimate a test's cost from the median of its best attempts instead of its
27
+ longest one. The longest attempt grows with the number of attempts, so a history
28
+ merged from several runs drifted upward and a flaky test's worst run was taken as
29
+ its cost. Attempts are ranked clean before contended and passing before failing,
30
+ and fixture set-up costs use the median too.
31
+
32
+ - Memory-aware admission under pytest-xdist. `--timing-memory SIZE|auto` (ini
33
+ `timing_memory`, env `PYTEST_TIMING_MEMORY`) sets a memory budget per host, and the
34
+ memory each test needed in the run at `--timing-schedule` keeps tests apart whose
35
+ recorded needs would not fit in it together. Nothing is declared: the first run
36
+ records, the next one gates. Memory and CPU budgets are checked together, so a
37
+ test starts only when both fit. A shared fixture keeps what its first test left
38
+ resident reserved while it is alive. The run's JSON gains `memory` with the budget
39
+ and admission summary.
40
+ - On macOS, CPU time and memory of live subprocesses are read through `libproc`
41
+ instead of psutil. psutil's child listing scans the whole process table, and the
42
+ CPU clock is read several times per test: with psutil installed, `--timing` on
43
+ 3,000 trivial tests took over a minute instead of under a second. psutil is now
44
+ used only where no native reader covers subprocesses, which is Windows.
45
+ - Record each test's resident memory. A sampler thread in the process running the
46
+ tests polls the resident set size of the worker (and, on Linux, macOS or with
47
+ psutil, its live subprocesses) while a test runs. Test JSON records now include `memory` when
48
+ the platform provides a reading: `base` before the test's set-up, `peak` during
49
+ it and `after` at its teardown, in bytes, with the measurement coverage. Nothing
50
+ schedules on it yet.
51
+
3
52
  ## 0.2.0
4
53
 
5
54
  - Raise the minimum supported pytest-xdist version to 3.7. Running without xdist
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pytest-timing
3
- Version: 0.2.0
3
+ Version: 0.3.1
4
4
  Summary: Record test timings under pytest-xdist and render cargo-style timing reports (ASCII Gantt and HTML).
5
5
  Project-URL: Homepage, https://github.com/messense/pytest-timing
6
6
  Author-email: messense <messense@icloud.com>
@@ -78,10 +78,11 @@ pip install "pytest-timing[xdist]" # with pytest-xdist
78
78
  ```
79
79
 
80
80
  Python 3.10+ and pytest 7.3+. Distributed runs need pytest-xdist 3.7+.
81
- CPU records include live subprocesses on Linux through `/proc`, or through
82
- [psutil](https://pypi.org/project/psutil/) when installed. Without either, POSIX
83
- records include children the worker has waited for; Windows records cover only the
84
- worker itself. Each CPU record identifies its measurement coverage.
81
+ CPU and memory records include live subprocesses on Linux through `/proc`, on macOS
82
+ through `libproc`, and on Windows through [psutil](https://pypi.org/project/psutil/)
83
+ when installed. Without psutil, Windows CPU records cover only the worker itself and
84
+ memory records only the worker; other POSIX platforms count children the worker has
85
+ waited for. Each record identifies its measurement coverage.
85
86
 
86
87
  ## Options
87
88
 
@@ -94,6 +95,7 @@ worker itself. Each CPU record identifies its measurement coverage.
94
95
  | `--timing-html-file PATH`, `--timing-json-file PATH`, `--timing-trace-file PATH` | Same, to an explicit path. |
95
96
  | `--timing-schedule PATH` | Under xdist, plan the workers' queues from the JSON run at `PATH`. |
96
97
  | `--timing-cpus N\|auto` | Under xdist, admit tests against a budget of `N` CPU slots per host (`auto` detects it). |
98
+ | `--timing-memory SIZE\|auto` | Under xdist, keep tests whose recorded memory would not fit in `SIZE` together from running at once (`auto` takes 80% of the host's memory). |
97
99
  | `--timing-top=N` | Rows in the slowest-tests section (default 10, `0` hides it). |
98
100
  | `--timing-min=SECONDS` | Hide tests shorter than this from the slowest-tests section. |
99
101
  | `--timing-ascii-style=unicode\|ascii` | Chart glyphs. |
@@ -107,7 +109,8 @@ and schedule paths are resolved from pytest's root directory.
107
109
 
108
110
  The ini key `timing` is a boolean; `timing_html`, `timing_json` and `timing_trace`
109
111
  accept paths or `true` for the default filenames. Other ini keys are
110
- `timing_schedule`, `timing_cpus`, `timing_top`, `timing_min` and `timing_ascii_style`.
112
+ `timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
113
+ `timing_ascii_style`.
111
114
  Environment variables use the uppercase key prefixed by `PYTEST_`, for example
112
115
  `PYTEST_TIMING=1`, `PYTEST_TIMING_HTML=path` or `PYTEST_TIMING_CPUS=auto`.
113
116
  `--timing-width` is command-line only.
@@ -120,7 +123,16 @@ environment, which wins over ini. Timing is enabled if any source requests it:
120
123
 
121
124
  **JSON** is the plugin's own record of the run: tests with their worker, phases and
122
125
  shared fixtures, plus CPU telemetry when available (elapsed time, measured CPU time,
123
- statically declared demand and time waiting for CPU slots). It also records worker
126
+ statically declared demand and time waiting for CPU slots) and the resident memory
127
+ of the worker's process tree around each test, in bytes: `base` before its set-up,
128
+ `peak` during it, and `after` at its teardown. The difference between `peak` and
129
+ `base` is what the test needed on top of the worker's existing footprint; `after`
130
+ minus `base` is what stayed, such as a shared fixture it set up. With
131
+ `--timing-memory`, the run's `memory` describes the budget and how admission went,
132
+ and a test the memory gate held back carries the seconds it waited in `memory.wait`,
133
+ as `cpu.wait` does for the CPU gate; the HTML report's table and hover details and
134
+ the terminal's slowest-tests rows show them.
135
+ It also records worker
124
136
  lifecycles, detected host CPU environments, and how the session ended (`finished`,
125
137
  `collect_only`, `interrupted`, `aborted`, `internal_error`, with pytest's reason). It is
126
138
  the input to the CLI below and to `--timing-schedule`.
@@ -259,8 +271,8 @@ What to expect:
259
271
  forces the oldest blocked request if needed. Such forced reservations can exceed
260
272
  the limit and are counted in the summary.
261
273
  - The summary reports the budget, the number of tests over one slot, and how long
262
- tests waited for slots in total. Each test's JSON record carries its own wait.
263
- `cpu.runtime_wait` records admission waits inside a running test. These waits
274
+ tests waited for slots in total. Each test's JSON record carries its own wait in
275
+ `cpu.wait`. `cpu.runtime_wait` records admission waits inside a running test. These waits
264
276
  are excluded from duration estimates, fixture setup costs and measured CPU rate.
265
277
  A runtime request that gets no grant within five minutes cancels the fixture
266
278
  setup. It waits to recover the test's original reservation before failing, so
@@ -280,6 +292,53 @@ domains with their own budgets. With `-x` or `--maxfail`, xdist shuts every work
280
292
  down; a test a worker was holding without having been admitted is withdrawn and
281
293
  never starts.
282
294
 
295
+ ## Keep memory-hungry tests apart
296
+
297
+ Two tests that each need several gigabytes are fine on their own and fatal together:
298
+ the kernel kills a worker, or the whole job. Nothing needs declaring for this. Every
299
+ run records how much resident memory each test needed on top of its worker's footprint
300
+ (`memory` in the JSON), and the next run keeps tests apart when their recorded needs
301
+ would not fit in the budget at the same time:
302
+
303
+ ```
304
+ pytest -n 4 --timing-json --timing-schedule pytest-timing.json --timing-memory 12G
305
+ ```
306
+
307
+ `--timing-memory` takes a size with a binary suffix (`512M`, `12G`) or `auto`, which
308
+ takes 80% of physical memory, capped by the cgroup memory limit. It reads the memory
309
+ each test needed from the run at `--timing-schedule`, so the first run only records;
310
+ the summary says how many tests had recorded memory. A test with no record weighs
311
+ nothing until it has run once. A module- or class-scoped fixture keeps what its first
312
+ test left resident (a loaded model, say) reserved for as long as it is alive, on that
313
+ worker. What session- and package-scoped fixtures keep is treated as part of every
314
+ worker's footprint instead: each worker sets them up once and never lets go, so
315
+ gating on them could only delay the run, never spare the host.
316
+
317
+ What to expect:
318
+
319
+ - The gate is the same fair waiting line as for CPU: a worker whose next test does
320
+ not fit waits with its fixtures alive, and the oldest request goes first. Memory
321
+ and CPU budgets are checked together; a test starts only when both fit.
322
+ - Estimates are the largest need any recorded attempt showed. For an attempt that
323
+ set up shared fixtures, what stayed resident afterwards is the fixtures' (or the
324
+ worker's footprint), and the test's own need is what it used beyond that. The
325
+ first test on each worker also pays for the worker's warm-up, so its record is
326
+ used only for a test or fixture with no other attempt. Estimates are relative to
327
+ the worker's footprint, so `auto` leaves a fifth of the host for the workers
328
+ themselves, their session fixtures, the controller and everything else; set a
329
+ smaller budget on a shared machine, and a larger share of headroom when session
330
+ fixtures are big.
331
+ - A test recorded above the budget runs alone, and the summary counts it. Allocator
332
+ behaviour can make an estimate low: a test that reuses heap an earlier test freed
333
+ shows a smaller rise than it needs. Large buffers and subprocesses, the usual
334
+ cause of an out-of-memory kill, measure well.
335
+ - Memory does not enter the planner: lanes are still balanced by duration and fixture
336
+ cost, and the gate only delays starts. Under `--dist load` and `worksteal`, like
337
+ CPU admission.
338
+ - The summary says how many tests waited for memory and for how long, and how long
339
+ workers sat parked at the gate without running what they waited for. Each held
340
+ test's JSON record carries its wait in `memory.wait`.
341
+
283
342
  ## Re-render or merge saved runs
284
343
 
285
344
  ```
@@ -52,10 +52,11 @@ pip install "pytest-timing[xdist]" # with pytest-xdist
52
52
  ```
53
53
 
54
54
  Python 3.10+ and pytest 7.3+. Distributed runs need pytest-xdist 3.7+.
55
- CPU records include live subprocesses on Linux through `/proc`, or through
56
- [psutil](https://pypi.org/project/psutil/) when installed. Without either, POSIX
57
- records include children the worker has waited for; Windows records cover only the
58
- worker itself. Each CPU record identifies its measurement coverage.
55
+ CPU and memory records include live subprocesses on Linux through `/proc`, on macOS
56
+ through `libproc`, and on Windows through [psutil](https://pypi.org/project/psutil/)
57
+ when installed. Without psutil, Windows CPU records cover only the worker itself and
58
+ memory records only the worker; other POSIX platforms count children the worker has
59
+ waited for. Each record identifies its measurement coverage.
59
60
 
60
61
  ## Options
61
62
 
@@ -68,6 +69,7 @@ worker itself. Each CPU record identifies its measurement coverage.
68
69
  | `--timing-html-file PATH`, `--timing-json-file PATH`, `--timing-trace-file PATH` | Same, to an explicit path. |
69
70
  | `--timing-schedule PATH` | Under xdist, plan the workers' queues from the JSON run at `PATH`. |
70
71
  | `--timing-cpus N\|auto` | Under xdist, admit tests against a budget of `N` CPU slots per host (`auto` detects it). |
72
+ | `--timing-memory SIZE\|auto` | Under xdist, keep tests whose recorded memory would not fit in `SIZE` together from running at once (`auto` takes 80% of the host's memory). |
71
73
  | `--timing-top=N` | Rows in the slowest-tests section (default 10, `0` hides it). |
72
74
  | `--timing-min=SECONDS` | Hide tests shorter than this from the slowest-tests section. |
73
75
  | `--timing-ascii-style=unicode\|ascii` | Chart glyphs. |
@@ -81,7 +83,8 @@ and schedule paths are resolved from pytest's root directory.
81
83
 
82
84
  The ini key `timing` is a boolean; `timing_html`, `timing_json` and `timing_trace`
83
85
  accept paths or `true` for the default filenames. Other ini keys are
84
- `timing_schedule`, `timing_cpus`, `timing_top`, `timing_min` and `timing_ascii_style`.
86
+ `timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
87
+ `timing_ascii_style`.
85
88
  Environment variables use the uppercase key prefixed by `PYTEST_`, for example
86
89
  `PYTEST_TIMING=1`, `PYTEST_TIMING_HTML=path` or `PYTEST_TIMING_CPUS=auto`.
87
90
  `--timing-width` is command-line only.
@@ -94,7 +97,16 @@ environment, which wins over ini. Timing is enabled if any source requests it:
94
97
 
95
98
  **JSON** is the plugin's own record of the run: tests with their worker, phases and
96
99
  shared fixtures, plus CPU telemetry when available (elapsed time, measured CPU time,
97
- statically declared demand and time waiting for CPU slots). It also records worker
100
+ statically declared demand and time waiting for CPU slots) and the resident memory
101
+ of the worker's process tree around each test, in bytes: `base` before its set-up,
102
+ `peak` during it, and `after` at its teardown. The difference between `peak` and
103
+ `base` is what the test needed on top of the worker's existing footprint; `after`
104
+ minus `base` is what stayed, such as a shared fixture it set up. With
105
+ `--timing-memory`, the run's `memory` describes the budget and how admission went,
106
+ and a test the memory gate held back carries the seconds it waited in `memory.wait`,
107
+ as `cpu.wait` does for the CPU gate; the HTML report's table and hover details and
108
+ the terminal's slowest-tests rows show them.
109
+ It also records worker
98
110
  lifecycles, detected host CPU environments, and how the session ended (`finished`,
99
111
  `collect_only`, `interrupted`, `aborted`, `internal_error`, with pytest's reason). It is
100
112
  the input to the CLI below and to `--timing-schedule`.
@@ -233,8 +245,8 @@ What to expect:
233
245
  forces the oldest blocked request if needed. Such forced reservations can exceed
234
246
  the limit and are counted in the summary.
235
247
  - The summary reports the budget, the number of tests over one slot, and how long
236
- tests waited for slots in total. Each test's JSON record carries its own wait.
237
- `cpu.runtime_wait` records admission waits inside a running test. These waits
248
+ tests waited for slots in total. Each test's JSON record carries its own wait in
249
+ `cpu.wait`. `cpu.runtime_wait` records admission waits inside a running test. These waits
238
250
  are excluded from duration estimates, fixture setup costs and measured CPU rate.
239
251
  A runtime request that gets no grant within five minutes cancels the fixture
240
252
  setup. It waits to recover the test's original reservation before failing, so
@@ -254,6 +266,53 @@ domains with their own budgets. With `-x` or `--maxfail`, xdist shuts every work
254
266
  down; a test a worker was holding without having been admitted is withdrawn and
255
267
  never starts.
256
268
 
269
+ ## Keep memory-hungry tests apart
270
+
271
+ Two tests that each need several gigabytes are fine on their own and fatal together:
272
+ the kernel kills a worker, or the whole job. Nothing needs declaring for this. Every
273
+ run records how much resident memory each test needed on top of its worker's footprint
274
+ (`memory` in the JSON), and the next run keeps tests apart when their recorded needs
275
+ would not fit in the budget at the same time:
276
+
277
+ ```
278
+ pytest -n 4 --timing-json --timing-schedule pytest-timing.json --timing-memory 12G
279
+ ```
280
+
281
+ `--timing-memory` takes a size with a binary suffix (`512M`, `12G`) or `auto`, which
282
+ takes 80% of physical memory, capped by the cgroup memory limit. It reads the memory
283
+ each test needed from the run at `--timing-schedule`, so the first run only records;
284
+ the summary says how many tests had recorded memory. A test with no record weighs
285
+ nothing until it has run once. A module- or class-scoped fixture keeps what its first
286
+ test left resident (a loaded model, say) reserved for as long as it is alive, on that
287
+ worker. What session- and package-scoped fixtures keep is treated as part of every
288
+ worker's footprint instead: each worker sets them up once and never lets go, so
289
+ gating on them could only delay the run, never spare the host.
290
+
291
+ What to expect:
292
+
293
+ - The gate is the same fair waiting line as for CPU: a worker whose next test does
294
+ not fit waits with its fixtures alive, and the oldest request goes first. Memory
295
+ and CPU budgets are checked together; a test starts only when both fit.
296
+ - Estimates are the largest need any recorded attempt showed. For an attempt that
297
+ set up shared fixtures, what stayed resident afterwards is the fixtures' (or the
298
+ worker's footprint), and the test's own need is what it used beyond that. The
299
+ first test on each worker also pays for the worker's warm-up, so its record is
300
+ used only for a test or fixture with no other attempt. Estimates are relative to
301
+ the worker's footprint, so `auto` leaves a fifth of the host for the workers
302
+ themselves, their session fixtures, the controller and everything else; set a
303
+ smaller budget on a shared machine, and a larger share of headroom when session
304
+ fixtures are big.
305
+ - A test recorded above the budget runs alone, and the summary counts it. Allocator
306
+ behaviour can make an estimate low: a test that reuses heap an earlier test freed
307
+ shows a smaller rise than it needs. Large buffers and subprocesses, the usual
308
+ cause of an out-of-memory kill, measure well.
309
+ - Memory does not enter the planner: lanes are still balanced by duration and fixture
310
+ cost, and the gate only delays starts. Under `--dist load` and `worksteal`, like
311
+ CPU admission.
312
+ - The summary says how many tests waited for memory and for how long, and how long
313
+ workers sat parked at the gate without running what they waited for. Each held
314
+ test's JSON record carries its wait in `memory.wait`.
315
+
257
316
  ## Re-render or merge saved runs
258
317
 
259
318
  ```
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pytest-timing"
7
- version = "0.2.0"
7
+ version = "0.3.1"
8
8
  description = "Record test timings under pytest-xdist and render cargo-style timing reports (ASCII Gantt and HTML)."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -4,6 +4,6 @@ from __future__ import annotations
4
4
 
5
5
  from pytest_timing.demand import cpu
6
6
 
7
- __version__ = "0.2.0"
7
+ __version__ = "0.3.1"
8
8
 
9
9
  __all__ = ["__version__", "cpu"]
@@ -40,8 +40,9 @@ class Admission:
40
40
  RECOVER_AFTER = 8
41
41
  COOLDOWN = 10.0
42
42
 
43
- def __init__(self, budget: int, name: str = "local") -> None:
43
+ def __init__(self, budget: int, name: str = "local", kind: str = "cpu") -> None:
44
44
  self.name = name
45
+ self.kind = kind # what the slots are: ``cpu`` or ``memory`` (bytes)
45
46
  self.budget = max(1, int(budget))
46
47
  self.limit = self.budget
47
48
  self.reserved: dict[Hashable, Reservation] = {}
@@ -14,10 +14,20 @@ correct span; they do not change that public occurrence/attempt numbering.
14
14
 
15
15
  from __future__ import annotations
16
16
 
17
+ from collections.abc import Iterable
17
18
  from dataclasses import dataclass
18
19
  from typing import Any
19
20
 
20
- from pytest_timing.model import PHASES, CpuRecord, Phase, Run, RunInfo, TestSpan, Worker
21
+ from pytest_timing.model import (
22
+ PHASES,
23
+ CpuRecord,
24
+ MemoryRecord,
25
+ Phase,
26
+ Run,
27
+ RunInfo,
28
+ TestSpan,
29
+ Worker,
30
+ )
21
31
 
22
32
  CRASH_WHEN = "???" # xdist synthesises a report with this ``when`` for crashed items
23
33
  RETRIED = "rerun"
@@ -38,6 +48,7 @@ class PhaseReport:
38
48
  received: float = 0.0 # epoch seconds when the controller saw the report
39
49
  fixtures: dict[str, float | None] | None = None # shared fixtures, setup and call
40
50
  cpu: dict[str, Any] | None = None # the worker's CPU record, on the teardown report
51
+ memory: dict[str, Any] | None = None # the worker's memory record, likewise
41
52
  execution: tuple[int, int] | None = None # collection index, attempt on that worker
42
53
 
43
54
 
@@ -73,12 +84,19 @@ class Collector:
73
84
  worker.items = items
74
85
 
75
86
  def add_wait(
76
- self, worker_id: str, nodeid: str, index: int, attempt: int, seconds: float
87
+ self,
88
+ worker_id: str,
89
+ nodeid: str,
90
+ index: int,
91
+ attempt: int,
92
+ seconds: float,
93
+ gates: Iterable[str] = (),
77
94
  ) -> None:
78
95
  """Add an admission delay to the exact execution that waited.
79
96
 
80
97
  A nodeid can have several selections and retries. Looking up its last span
81
- at session finish would put every delay on the final one instead.
98
+ at session finish would put every delay on the final one instead. ``gates``
99
+ names what held the test (``cpu``, ``memory``); unnamed, the CPU gate did.
82
100
  """
83
101
  span = self._executions.get((worker_id, index, attempt))
84
102
  if span is None:
@@ -88,9 +106,15 @@ class Collector:
88
106
  span = self._last.get((worker_id, nodeid))
89
107
  if span is None or span.outcome != "crashed" or span.phases or span.attempt != attempt:
90
108
  return
91
- if span.cpu is None:
92
- span.cpu = CpuRecord(elapsed=span.duration)
93
- span.cpu.wait += seconds
109
+ gates = set(gates)
110
+ if "cpu" in gates or not gates:
111
+ if span.cpu is None:
112
+ span.cpu = CpuRecord(elapsed=span.duration)
113
+ span.cpu.wait += seconds
114
+ if "memory" in gates:
115
+ if span.memory is None:
116
+ span.memory = MemoryRecord() # no reading, but the wait is a fact
117
+ span.memory.wait += seconds
94
118
 
95
119
  def worker_down(self, worker_id: str, epoch: float, error: str | None) -> None:
96
120
  worker = self._worker(worker_id)
@@ -131,6 +155,8 @@ class Collector:
131
155
  span.fixtures.update(report.fixtures)
132
156
  if report.cpu:
133
157
  span.cpu = CpuRecord.from_dict(report.cpu)
158
+ if report.memory:
159
+ span.memory = MemoryRecord.from_dict(report.memory)
134
160
  span.start = min(span.start, start)
135
161
  span.stop = max(span.stop, stop)
136
162
  self._apply_outcome(span, report)
@@ -1,4 +1,4 @@
1
- """CPU declarations and per-test measurement in the process running the tests.
1
+ """CPU declarations and per-test CPU and memory measurement where the tests run.
2
2
 
3
3
  Declarations precede xdist collection reports. Dynamic fixture setup requests
4
4
  reserve slots before execution; holds follow actual setup/finalization events.
@@ -16,11 +16,13 @@ import pytest
16
16
 
17
17
  from pytest_timing import xdist_compat
18
18
  from pytest_timing.fixtures import FixtureTimer, _key_for, defined_at, fixture_key, fixturedefs
19
- from pytest_timing.telemetry import Pressure, ProcessTreeClock, host_cpu
19
+ from pytest_timing.telemetry import MemorySampler, Pressure, ProcessTreeClock, host_cpu
20
20
 
21
21
  MARKER = "timing_cpu"
22
22
  FIXTURE_ATTR = "_pytest_timing_cpu"
23
23
  REPORT_ATTR = "timing_cpu"
24
+ MEMORY_ATTR = "timing_memory"
25
+ """Report attribute carrying the attempt's resident-memory window, on teardown."""
24
26
  EXECUTION_ATTR = "timing_execution"
25
27
  """Private report identity: collection index and attempt, including retries."""
26
28
  EVENT = "timing_cpu"
@@ -237,7 +239,7 @@ class Declarations:
237
239
 
238
240
 
239
241
  class CpuMeter:
240
- """Measures each test's CPU work and attaches it to the teardown report.
242
+ """Measures each test's CPU work and memory and attaches them to the teardown report.
241
243
 
242
244
  Runs wherever tests run: in every xdist worker, or in the main process without
243
245
  xdist. After collection it also sends the :class:`Declarations` to the controller
@@ -255,6 +257,9 @@ class CpuMeter:
255
257
  set-up's slots (``timing_request``) before the set-up starts and blocks until
256
258
  they are granted (``timing_grant``), so such a set-up goes through the same
257
259
  gate as a declared one.
260
+
261
+ ``memory`` samples resident memory from set-up to the teardown report; the
262
+ window (``timing_memory``) goes on that report next to the CPU record.
258
263
  """
259
264
 
260
265
  def __init__(
@@ -263,11 +268,13 @@ class CpuMeter:
263
268
  send: Callable[[str, dict[str, Any]], None] | None = None,
264
269
  clock: ProcessTreeClock | None = None,
265
270
  pressure: Pressure | None = None,
271
+ memory: MemorySampler | None = None,
266
272
  ) -> None:
267
273
  self.timer = timer
268
274
  self.send = send
269
275
  self.clock = clock or ProcessTreeClock()
270
276
  self.pressure = pressure or Pressure()
277
+ self.memory = memory
271
278
  self.declarations: Declarations | None = None
272
279
  self.cancelled: set[int] = set() # item indices withdrawn by the controller
273
280
  self._live = False # can the controller reach this worker while a test runs
@@ -427,6 +434,8 @@ class CpuMeter:
427
434
  self._started = time.perf_counter()
428
435
  self._work = self.clock.seconds()
429
436
  self._throttled = self.pressure.throttled()
437
+ if self.memory is not None:
438
+ self.memory.begin()
430
439
 
431
440
  @pytest.hookimpl(hookwrapper=True)
432
441
  def pytest_runtest_makereport(self, item: pytest.Item, call: Any) -> Generator[None, Any, None]:
@@ -436,6 +445,18 @@ class CpuMeter:
436
445
  setattr(outcome.get_result(), EXECUTION_ATTR, (index, self._attempts.get(index, 0)))
437
446
  if call.when != "teardown" or not self._started:
438
447
  return
448
+ window = self.memory.end() if self.memory is not None else None
449
+ if window is not None:
450
+ setattr(
451
+ outcome.get_result(),
452
+ MEMORY_ATTR,
453
+ {
454
+ "base": window.base,
455
+ "peak": window.peak,
456
+ "after": window.after,
457
+ "coverage": window.coverage,
458
+ },
459
+ )
439
460
  elapsed = max(0.0, time.perf_counter() - self._started - self.timer.wait)
440
461
  work = self.clock.seconds() - self._work - self.timer.wait_work
441
462
  throttled_now = self.pressure.throttled()
@@ -456,3 +477,7 @@ class CpuMeter:
456
477
  }
457
478
  setattr(outcome.get_result(), REPORT_ATTR, record)
458
479
  self._started = 0.0
480
+
481
+ def pytest_unconfigure(self) -> None:
482
+ if self.memory is not None:
483
+ self.memory.close()