pytest-timing 0.3.1__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/ARCHITECTURE.md +21 -1
  2. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/CHANGELOG.md +33 -0
  3. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/PKG-INFO +76 -3
  4. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/README.md +75 -2
  5. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/pyproject.toml +1 -1
  6. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/__init__.py +1 -1
  7. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/cli.py +122 -3
  8. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/collector.py +7 -0
  9. pytest_timing-0.4.0/src/pytest_timing/compare.py +200 -0
  10. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/demand.py +20 -10
  11. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/model.py +24 -0
  12. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/outputs.py +4 -3
  13. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/plugin.py +26 -4
  14. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/schedule.py +54 -4
  15. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/static/report.html +416 -73
  16. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/xdist_scheduler.py +58 -33
  17. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/conftest.py +11 -2
  18. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_cli.py +20 -0
  19. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_collector.py +29 -0
  20. pytest_timing-0.4.0/tests/test_compare.py +207 -0
  21. pytest_timing-0.4.0/tests/test_html.py +461 -0
  22. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_memory_gate.py +28 -0
  23. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_plugin.py +42 -0
  24. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_runtime_admission.py +62 -2
  25. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_schedule.py +128 -0
  26. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/uv.lock +1 -1
  27. pytest_timing-0.3.1/tests/test_html.py +0 -101
  28. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/.github/workflows/ci.yml +0 -0
  29. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/.github/workflows/release.yml +0 -0
  30. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/.gitignore +0 -0
  31. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/LICENSE +0 -0
  32. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/benchmarks/cpu_bench.py +0 -0
  33. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/docs/report.png +0 -0
  34. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/examples/test_demo.py +0 -0
  35. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/admission.py +0 -0
  36. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/fixtures.py +0 -0
  37. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/render/__init__.py +0 -0
  38. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/render/ascii.py +0 -0
  39. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/render/html.py +0 -0
  40. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/render/trace.py +0 -0
  41. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/static/__init__.py +0 -0
  42. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/telemetry.py +0 -0
  43. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/xdist_compat.py +0 -0
  44. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/fixtures/legacy_complete_false.json +0 -0
  45. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/fixtures/legacy_complete_true.json +0 -0
  46. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/fixtures/legacy_no_flag.json +0 -0
  47. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_admission.py +0 -0
  48. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_ascii.py +0 -0
  49. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_benchmarks.py +0 -0
  50. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_cpu.py +0 -0
  51. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_fixtures.py +0 -0
  52. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_memory.py +0 -0
  53. {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_trace.py +0 -0
@@ -11,9 +11,10 @@ schedules a run from the previous one. For usage, see the [README](README.md).
11
11
  | `collector.py` | Folds phase reports and worker events into a `Run`. Knows nothing about pytest objects. |
12
12
  | `model.py` | The recorded run: `Run`, `Worker`, `TestSpan`, `Phase`, and the JSON representation. |
13
13
  | `outputs.py` | The ASCII, HTML and trace renderers, in one table shared by the plugin and the CLI. |
14
- | `cli.py` | `pytest-timing render` and `pytest-timing merge`. |
14
+ | `cli.py` | `pytest-timing render`, `merge` and `compare`. |
15
15
  | `fixtures.py` | Times shared fixture set-up where tests run and lists dependencies on reports. |
16
16
  | `schedule.py` | Duration estimates, the fixture-aware cost model and the planner. Free of xdist. |
17
+ | `compare.py` | Saved-run comparison, final-attempt aggregation and per-test CI budgets. Free of pytest hooks and output I/O. |
17
18
  | `demand.py` | Declared CPU demand: the `timing_cpu` marker, the fixture decorator, the declarations a worker sends, and the meter that records each test's CPU work and resident memory. |
18
19
  | `admission.py` | One host's CPU budget, reservations, fair waiting line and pressure feedback. Free of xdist and platform reads. |
19
20
  | `telemetry.py` | Platform reads: affinity and cgroup quota, CPU time and resident memory of a process tree, pressure and throttling. |
@@ -97,6 +98,14 @@ A missing or unreadable history file leaves xdist's scheduler in place unless an
97
98
  explicit CPU budget was requested. With that budget the custom scheduler still runs,
98
99
  using equal 1 ms test estimates. It supports only `load` and `worksteal`.
99
100
 
101
+ ## Saved-run comparisons
102
+
103
+ Saved-run comparisons use a separate policy from scheduling estimates: compare
104
+ the median of final attempts, preserving missing metrics and rejecting incomplete
105
+ budget checks. Scheduling may choose uncontended historical observations instead.
106
+ Keeping these policies separate avoids treating a failed or missing observation as
107
+ a successful CI check merely because it was usable for queue planning.
108
+
100
109
  ## Cost model
101
110
 
102
111
  A shared fixture makes the cost of a test depend on the worker. Every test has a
@@ -118,10 +127,16 @@ CPU work, peak demand and fixture transition. `Costs.project` folds these into a
118
127
 
119
128
  The scheduler keeps one `Charge` per dispatched item. It records the holds already
120
129
  included in the peak, so actual live holds supplement it without double counting.
130
+ It also records each fixture's contribution to CPU holds and retained memory.
131
+ Runtime admission previews and commits the same reconciliation: a fixture already
132
+ covered by history adds only a missing contribution, never its full cost again.
121
133
  Charged duration, CPU work and the pre-dispatch fixture checkpoint remain historical
122
134
  facts even when a runtime request changes the reservation. Stealing refunds the same
123
135
  record and restores that checkpoint.
124
136
  Ordinary transfers and steals compare finish times with the same CPU-work floor.
137
+ After placement, a stable module/class grouping is also tried within each worker's
138
+ assignment. It replaces exact-family ordering only when `Costs.project` predicts
139
+ less lane time and no more CPU work, so parameter locality can still win.
125
140
 
126
141
  ## Planning
127
142
 
@@ -364,6 +379,11 @@ as `runtime_wait`. All open fixture clocks pause during that wait, and test esti
364
379
  subtract it along with shared setup. CPU `elapsed` and `work` exclude the wait and
365
380
  its process-tree CPU work, so the measured rate covers execution. Older records
366
381
  without `runtime_wait` default to zero; pre-start waits are not subtracted again.
382
+ The controller also attaches each wait's end epoch and monotonic duration as an
383
+ optional `TestSpan.admission_waits` interval on the run axis. Model rebasing moves
384
+ these intervals with phases and test spans. The HTML report displays them once
385
+ even when both gates refused the same interval. Older totals without intervals stay
386
+ unlocated; parked workers without an executed item are only in gate summaries.
367
387
  Event handlers are registered before each worker starts, even when another plugin
368
388
  selects the scheduler: without an admission gate a request is granted immediately.
369
389
  The cgroup quota behind `auto` is read from the process's own group, walking up to
@@ -1,5 +1,38 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.4.0
4
+
5
+ - Keep slowest-test labels visible at the end of the timeline and wrap long labels
6
+ below their bars when neither side has room.
7
+ - Add `pytest-timing compare` with per-test duration, phase, CPU and memory changes,
8
+ optional CI regression budgets, machine-readable JSON and explicit unavailable checks.
9
+ - Add shared fixture costs by worker and linked time-range selection in HTML reports.
10
+ Zooming and scrolling resample the visible interval for finer detail.
11
+ - Record admission wait intervals on their test attempts, preserve them through
12
+ merging and show them in lanes, a wait graph and range statistics.
13
+ - Add `--timing-capture=light` to skip optional CPU and memory measurements while
14
+ retaining durations and resource admission; the default remains full capture.
15
+ - Bound HTML chart samples and ticks independently of zoom, retaining peak values
16
+ over each bucket; cache merged-test totals and top-ten hover details.
17
+ - Try module/class-contiguous ordering across overlapping fixture families without
18
+ changing worker assignments or increasing modeled CPU work.
19
+ - Trace-only output no longer builds an unused JSON report document.
20
+ - Runtime fixture admission no longer reserves a dynamic fixture's held CPU slots
21
+ or retained memory twice when it was already included in scheduling history.
22
+ - The HTML CPU graph spreads recorded CPU time over the full test span, including
23
+ runtime admission waits, so the graph preserves the recorded total. Test details
24
+ continue to show average CPU use during execution, with those waits excluded.
25
+
26
+ ## 0.3.2
27
+
28
+ - The HTML report shows CPU and memory usage. The table gains "CPU time", "CPUs"
29
+ (the CPUs a test kept busy on average, beside the slots it declared) and "Memory"
30
+ (what it needed on top of its worker's footprint) columns, the hover details say
31
+ the same with what the measurement covered, the header totals the run and names
32
+ each host's budget, and two graphs show CPUs busy and resident memory over time,
33
+ estimated from the per-test records. All of it appears only where there was a
34
+ reading.
35
+
3
36
  ## 0.3.1
4
37
 
5
38
  - Memory admission no longer charges what session- and package-scoped fixtures keep
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pytest-timing
3
- Version: 0.3.1
3
+ Version: 0.4.0
4
4
  Summary: Record test timings under pytest-xdist and render cargo-style timing reports (ASCII Gantt and HTML).
5
5
  Project-URL: Homepage, https://github.com/messense/pytest-timing
6
6
  Author-email: messense <messense@icloud.com>
@@ -89,6 +89,7 @@ waited for. Each record identifies its measurement coverage.
89
89
  | Option | Effect |
90
90
  |---|---|
91
91
  | `--timing` | Record timings and print the ASCII chart in the terminal summary. |
92
+ | `--timing-capture=full\|light` | Full metrics (default), or timings without CPU/memory measurement. Implies timing. |
92
93
  | `--timing-html` | Write a self-contained HTML report to `pytest-timing.html`. |
93
94
  | `--timing-json` | Write the recorded run to `pytest-timing.json`. |
94
95
  | `--timing-trace` | Write a Chrome trace file for [Perfetto](https://ui.perfetto.dev) to `pytest-timing.trace.json`. |
@@ -109,7 +110,7 @@ and schedule paths are resolved from pytest's root directory.
109
110
 
110
111
  The ini key `timing` is a boolean; `timing_html`, `timing_json` and `timing_trace`
111
112
  accept paths or `true` for the default filenames. Other ini keys are
112
- `timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
113
+ `timing_capture`, `timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
113
114
  `timing_ascii_style`.
114
115
  Environment variables use the uppercase key prefixed by `PYTEST_`, for example
115
116
  `PYTEST_TIMING=1`, `PYTEST_TIMING_HTML=path` or `PYTEST_TIMING_CPUS=auto`.
@@ -119,6 +120,16 @@ For output paths, scheduling and formatting, the command line wins over the
119
120
  environment, which wins over ini. Timing is enabled if any source requests it:
120
121
  `PYTEST_TIMING=0` does not disable `timing=true` in ini or an enabled output.
121
122
 
123
+ Use `--timing-capture=light` (or `PYTEST_TIMING_CAPTURE=light`, or
124
+ `timing_capture = light` in ini) for timing-only collection. It keeps phase and
125
+ shared-fixture durations, worker lifecycle and admission waits, but does not create
126
+ a process-tree CPU reader or memory sampler. CPU work/rate, pressure and resident
127
+ memory measurements are unavailable; wait-only records have unknown measurement
128
+ coverage, rather than a measured zero. Declared CPU admission and memory admission
129
+ from an existing full report still apply. Capture stays `full` unless explicitly
130
+ changed; use full capture when refreshing resource estimates for future scheduling.
131
+ Capture mode names are case-insensitive in CLI, environment and ini settings.
132
+
122
133
  ## Outputs
123
134
 
124
135
  **JSON** is the plugin's own record of the run: tests with their worker, phases and
@@ -138,7 +149,17 @@ lifecycles, detected host CPU environments, and how the session ended (`finished
138
149
  the input to the CLI below and to `--timing-schedule`.
139
150
 
140
151
  **HTML** is a single file with no external dependencies, so it can be attached to a CI
141
- job as an artifact and opened anywhere.
152
+ job as an artifact and opened anywhere. Where CPU and memory were measured it shows
153
+ them: each test's CPU time, the CPUs it kept busy on average (beside the slots it
154
+ declared) and the memory it needed on top of its worker's footprint, in the table and
155
+ the hover details; the run's totals and budgets in the header; and CPU and memory
156
+ usage over time. The two graphs are drawn from the per-test records, not sampled: a
157
+ test's CPU time is spread evenly over its run, and a worker counts at its running
158
+ test's peak, so they show where the load was rather than its exact shape.
159
+ Each curve uses at most 4096 display samples, even at deep zoom. A sample keeps
160
+ the maximum over its time bucket; hover reports that bucket's time range. This
161
+ bounds display memory without dropping short peaks. Merged tiny-test bars cache
162
+ their total duration and ten longest tests for repeated hover.
142
163
 
143
164
  **Trace** is a Chrome Trace Event file. Open it in
144
165
  [Perfetto UI](https://ui.perfetto.dev) with "Open trace file", or in `chrome://tracing`.
@@ -349,6 +370,58 @@ pytest-timing merge shard1.json shard2.json -o all.json
349
370
  `render` produces any of the outputs from a saved JSON run. `merge` places several runs
350
371
  (for example CI shards) on one shared time axis using their absolute start times.
351
372
 
373
+ ## Compare runs and enforce CI budgets
374
+
375
+ ```
376
+ pytest-timing compare before.json after.json
377
+ pytest-timing compare before.json after.json --budget duration=10% --budget memory=64MiB --json comparison.json
378
+ ```
379
+
380
+ Comparisons match tests by node ID and show total, setup, call and teardown time,
381
+ CPU time, and peak memory rise above the worker's baseline. For each run, only the
382
+ final retry of each worker/occurrence is used, then repeated observations are
383
+ combined with a median. The output includes retry counts and added/removed tests.
384
+ Missing measurements stay unavailable, including partial coverage across repeats.
385
+
386
+ A budget sets the maximum increase **per common test**. Percentages are relative
387
+ to the baseline; a zero baseline permits no increase under a percentage budget.
388
+ Time limits use seconds or `ms`; memory limits use bytes, `KiB`, `MiB` or `GiB`.
389
+ Each metric accepts one budget. Added and removed tests are listed but not budgeted.
390
+ Exit codes are `0` for success (or a comparison without budgets), `1` for exceeded
391
+ budgets, and `2` for invalid input or unavailable checks. Budget checks need complete
392
+ runs, at least one common test, successful final attempts and every requested metric.
393
+ Differences in Python, pytest, worker count or distribution mode produce warnings;
394
+ use comparable environments and repeat noisy workloads before choosing thresholds.
395
+
396
+ ## Explore the HTML report
397
+
398
+ The scale control zooms all timelines together. Scrolling any timeline moves the
399
+ others to the same time. Curves are resampled from the visible window, so a narrow
400
+ peak gains detail as you zoom in, with at most 4,096 samples per curve. Hover values
401
+ are the maximum within the displayed bucket.
402
+
403
+ Enter start and end times in seconds and select **Select range** to fit that interval
404
+ across lanes, concurrency, admission waits, CPU and memory. The test and fixture
405
+ tables follow the selection and existing filters. **Clear range** returns to the
406
+ whole run. The range summary covers all tests, independently of text/outcome/fixture
407
+ filters; CPU totals integrate the original estimated curve, never its peak buckets.
408
+ CPU time is spread evenly over each test span, while the memory curve places the
409
+ recorded peak over that span; neither is a sampled execution trace.
410
+
411
+ The shared fixture table groups recorded setup costs by fixture key and worker,
412
+ showing repetition across tests and workers. Select a fixture to see its associated
413
+ tests and use **Clear fixture filter** to restore them. Costs belong to overlapping
414
+ tests: fixture setup timestamps are not recorded, so costs cannot be clipped to a
415
+ selected interval. A `None` fixture value means reuse and contributes no setup.
416
+
417
+ New reports include controller-observed `admission_waits` intervals on the exact
418
+ test attempt that waited. Lane strips and the wait graph display those intervals;
419
+ one interval refused by both CPU and memory counts once in the graph, summary and
420
+ Held column (its hover still lists both gates).
421
+ Older JSON files remain readable, but their wait totals cannot be placed on a
422
+ timeline. Workers parked without ever running a test remain in the run-level gate
423
+ summary rather than being assigned to an unrelated test.
424
+
352
425
  ## Overhead
353
426
 
354
427
  Timing collection also measures fixtures and CPU work in each worker. Its overhead
@@ -63,6 +63,7 @@ waited for. Each record identifies its measurement coverage.
63
63
  | Option | Effect |
64
64
  |---|---|
65
65
  | `--timing` | Record timings and print the ASCII chart in the terminal summary. |
66
+ | `--timing-capture=full\|light` | Full metrics (default), or timings without CPU/memory measurement. Implies timing. |
66
67
  | `--timing-html` | Write a self-contained HTML report to `pytest-timing.html`. |
67
68
  | `--timing-json` | Write the recorded run to `pytest-timing.json`. |
68
69
  | `--timing-trace` | Write a Chrome trace file for [Perfetto](https://ui.perfetto.dev) to `pytest-timing.trace.json`. |
@@ -83,7 +84,7 @@ and schedule paths are resolved from pytest's root directory.
83
84
 
84
85
  The ini key `timing` is a boolean; `timing_html`, `timing_json` and `timing_trace`
85
86
  accept paths or `true` for the default filenames. Other ini keys are
86
- `timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
87
+ `timing_capture`, `timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
87
88
  `timing_ascii_style`.
88
89
  Environment variables use the uppercase key prefixed by `PYTEST_`, for example
89
90
  `PYTEST_TIMING=1`, `PYTEST_TIMING_HTML=path` or `PYTEST_TIMING_CPUS=auto`.
@@ -93,6 +94,16 @@ For output paths, scheduling and formatting, the command line wins over the
93
94
  environment, which wins over ini. Timing is enabled if any source requests it:
94
95
  `PYTEST_TIMING=0` does not disable `timing=true` in ini or an enabled output.
95
96
 
97
+ Use `--timing-capture=light` (or `PYTEST_TIMING_CAPTURE=light`, or
98
+ `timing_capture = light` in ini) for timing-only collection. It keeps phase and
99
+ shared-fixture durations, worker lifecycle and admission waits, but does not create
100
+ a process-tree CPU reader or memory sampler. CPU work/rate, pressure and resident
101
+ memory measurements are unavailable; wait-only records have unknown measurement
102
+ coverage, rather than a measured zero. Declared CPU admission and memory admission
103
+ from an existing full report still apply. Capture stays `full` unless explicitly
104
+ changed; use full capture when refreshing resource estimates for future scheduling.
105
+ Capture mode names are case-insensitive in CLI, environment and ini settings.
106
+
96
107
  ## Outputs
97
108
 
98
109
  **JSON** is the plugin's own record of the run: tests with their worker, phases and
@@ -112,7 +123,17 @@ lifecycles, detected host CPU environments, and how the session ended (`finished
112
123
  the input to the CLI below and to `--timing-schedule`.
113
124
 
114
125
  **HTML** is a single file with no external dependencies, so it can be attached to a CI
115
- job as an artifact and opened anywhere.
126
+ job as an artifact and opened anywhere. Where CPU and memory were measured it shows
127
+ them: each test's CPU time, the CPUs it kept busy on average (beside the slots it
128
+ declared) and the memory it needed on top of its worker's footprint, in the table and
129
+ the hover details; the run's totals and budgets in the header; and CPU and memory
130
+ usage over time. The two graphs are drawn from the per-test records, not sampled: a
131
+ test's CPU time is spread evenly over its run, and a worker counts at its running
132
+ test's peak, so they show where the load was rather than its exact shape.
133
+ Each curve uses at most 4096 display samples, even at deep zoom. A sample keeps
134
+ the maximum over its time bucket; hover reports that bucket's time range. This
135
+ bounds display memory without dropping short peaks. Merged tiny-test bars cache
136
+ their total duration and ten longest tests for repeated hover.
116
137
 
117
138
  **Trace** is a Chrome Trace Event file. Open it in
118
139
  [Perfetto UI](https://ui.perfetto.dev) with "Open trace file", or in `chrome://tracing`.
@@ -323,6 +344,58 @@ pytest-timing merge shard1.json shard2.json -o all.json
323
344
  `render` produces any of the outputs from a saved JSON run. `merge` places several runs
324
345
  (for example CI shards) on one shared time axis using their absolute start times.
325
346
 
347
+ ## Compare runs and enforce CI budgets
348
+
349
+ ```
350
+ pytest-timing compare before.json after.json
351
+ pytest-timing compare before.json after.json --budget duration=10% --budget memory=64MiB --json comparison.json
352
+ ```
353
+
354
+ Comparisons match tests by node ID and show total, setup, call and teardown time,
355
+ CPU time, and peak memory rise above the worker's baseline. For each run, only the
356
+ final retry of each worker/occurrence is used, then repeated observations are
357
+ combined with a median. The output includes retry counts and added/removed tests.
358
+ Missing measurements stay unavailable, including partial coverage across repeats.
359
+
360
+ A budget sets the maximum increase **per common test**. Percentages are relative
361
+ to the baseline; a zero baseline permits no increase under a percentage budget.
362
+ Time limits use seconds or `ms`; memory limits use bytes, `KiB`, `MiB` or `GiB`.
363
+ Each metric accepts one budget. Added and removed tests are listed but not budgeted.
364
+ Exit codes are `0` for success (or a comparison without budgets), `1` for exceeded
365
+ budgets, and `2` for invalid input or unavailable checks. Budget checks need complete
366
+ runs, at least one common test, successful final attempts and every requested metric.
367
+ Differences in Python, pytest, worker count or distribution mode produce warnings;
368
+ use comparable environments and repeat noisy workloads before choosing thresholds.
369
+
370
+ ## Explore the HTML report
371
+
372
+ The scale control zooms all timelines together. Scrolling any timeline moves the
373
+ others to the same time. Curves are resampled from the visible window, so a narrow
374
+ peak gains detail as you zoom in, with at most 4,096 samples per curve. Hover values
375
+ are the maximum within the displayed bucket.
376
+
377
+ Enter start and end times in seconds and select **Select range** to fit that interval
378
+ across lanes, concurrency, admission waits, CPU and memory. The test and fixture
379
+ tables follow the selection and existing filters. **Clear range** returns to the
380
+ whole run. The range summary covers all tests, independently of text/outcome/fixture
381
+ filters; CPU totals integrate the original estimated curve, never its peak buckets.
382
+ CPU time is spread evenly over each test span, while the memory curve places the
383
+ recorded peak over that span; neither is a sampled execution trace.
384
+
385
+ The shared fixture table groups recorded setup costs by fixture key and worker,
386
+ showing repetition across tests and workers. Select a fixture to see its associated
387
+ tests and use **Clear fixture filter** to restore them. Costs belong to overlapping
388
+ tests: fixture setup timestamps are not recorded, so costs cannot be clipped to a
389
+ selected interval. A `None` fixture value means reuse and contributes no setup.
390
+
391
+ New reports include controller-observed `admission_waits` intervals on the exact
392
+ test attempt that waited. Lane strips and the wait graph display those intervals;
393
+ one interval refused by both CPU and memory counts once in the graph, summary and
394
+ Held column (its hover still lists both gates).
395
+ Older JSON files remain readable, but their wait totals cannot be placed on a
396
+ timeline. Workers parked without ever running a test remain in the run-level gate
397
+ summary rather than being assigned to an unrelated test.
398
+
326
399
  ## Overhead
327
400
 
328
401
  Timing collection also measures fixtures and CPU work in each worker. Its overhead
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pytest-timing"
7
- version = "0.3.1"
7
+ version = "0.4.0"
8
8
  description = "Record test timings under pytest-xdist and render cargo-style timing reports (ASCII Gantt and HTML)."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -4,6 +4,6 @@ from __future__ import annotations
4
4
 
5
5
  from pytest_timing.demand import cpu
6
6
 
7
- __version__ = "0.3.1"
7
+ __version__ = "0.4.0"
8
8
 
9
9
  __all__ = ["__version__", "cpu"]
@@ -1,18 +1,106 @@
1
- """``pytest-timing`` command line: re-render or merge saved JSON runs."""
1
+ """``pytest-timing`` command line: render, merge or compare saved JSON runs."""
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
5
  import argparse
6
+ import json
7
+ import math
6
8
  import sys
7
9
  from pathlib import Path
10
+ from typing import Any
8
11
 
12
+ from pytest_timing.compare import compare_runs, format_comparison, parse_budget
9
13
  from pytest_timing.model import Run, RunInfo, TestSpan, Worker
10
14
  from pytest_timing.outputs import OUTPUTS, write_output
11
15
  from pytest_timing.render.ascii import render_ascii
12
16
 
13
17
 
18
+ def _number(value: Any, where: str) -> None:
19
+ if not isinstance(value, (int, float, str)):
20
+ raise ValueError(f"{where} must be a number")
21
+ try:
22
+ finite = math.isfinite(float(value))
23
+ except OverflowError:
24
+ finite = False
25
+ if not finite:
26
+ raise ValueError(f"{where} must be finite")
27
+
28
+
29
+ def _record(
30
+ value: Any, where: str, *, required: str = "", numbers: str = "", optional: str = ""
31
+ ) -> dict[str, Any]:
32
+ if not isinstance(value, dict):
33
+ raise ValueError(f"{where} must be an object")
34
+ for key in required.split():
35
+ if key not in value:
36
+ raise ValueError(f"{where}.{key} is required")
37
+ for key in numbers.split():
38
+ if key in value:
39
+ _number(value[key], f"{where}.{key}")
40
+ for key in optional.split():
41
+ if value.get(key) is not None:
42
+ _number(value[key], f"{where}.{key}")
43
+ return value
44
+
45
+
46
+ def _array(value: Any, where: str) -> list[Any]:
47
+ if not isinstance(value, list):
48
+ raise ValueError(f"{where} must be an array")
49
+ return value
50
+
51
+
14
52
  def _load(path: str) -> Run:
15
- return Run.from_json(Path(path).read_text("utf-8"))
53
+ # Check external input before model conversion. Programming errors in the
54
+ # model, comparison or formatter must propagate rather than becoming exit 2.
55
+ doc = _record(
56
+ json.loads(Path(path).read_text("utf-8")), "report", required="run", numbers="schema"
57
+ )
58
+ info = _record(
59
+ doc["run"],
60
+ "run",
61
+ required="start stop",
62
+ numbers="start stop",
63
+ optional="numprocesses exit_status",
64
+ )
65
+ if info.get("termination") is not None and not isinstance(info["termination"], str):
66
+ raise ValueError("run.termination must be a string")
67
+ _array(info.get("argv", []), "run.argv")
68
+ for i, value in enumerate(_array(doc.get("workers", []), "workers")):
69
+ _record(value, f"workers[{i}]", required="id", optional="start ready collected down items")
70
+ for i, value in enumerate(_array(doc.get("tests", []), "tests")):
71
+ where = f"tests[{i}]"
72
+ test = _record(
73
+ value,
74
+ where,
75
+ required="nodeid worker outcome start stop",
76
+ numbers="start stop attempt occurrence",
77
+ )
78
+ for name, phase in _record(test.get("phases", {}), f"{where}.phases").items():
79
+ at = f"{where}.phases.{name}"
80
+ if len(_array(phase, at)) != 3:
81
+ raise ValueError(f"{at} must contain start, stop and duration")
82
+ for number in phase:
83
+ _number(number, at)
84
+ for key, seconds in _record(test.get("fixtures") or {}, f"{where}.fixtures").items():
85
+ if seconds is not None:
86
+ _number(seconds, f"{where}.fixtures.{key}")
87
+ if test.get("cpu") is not None:
88
+ _record(
89
+ test["cpu"],
90
+ f"{where}.cpu",
91
+ numbers="elapsed work setup_work demand wait runtime_wait",
92
+ optional="pressure",
93
+ )
94
+ if test.get("memory") is not None:
95
+ _record(test["memory"], f"{where}.memory", numbers="base peak after wait")
96
+ for j, value in enumerate(
97
+ _array(test.get("admission_waits", []), f"{where}.admission_waits")
98
+ ):
99
+ at = f"{where}.admission_waits[{j}]"
100
+ wait = _record(value, at, required="start stop", numbers="start stop")
101
+ if not all(isinstance(g, str) for g in _array(wait.get("gates", []), f"{at}.gates")):
102
+ raise ValueError(f"{at}.gates must contain strings")
103
+ return Run.from_dict(doc)
16
104
 
17
105
 
18
106
  def cmd_render(args: argparse.Namespace) -> int:
@@ -22,7 +110,8 @@ def cmd_render(args: argparse.Namespace) -> int:
22
110
  for kind, output in OUTPUTS.items():
23
111
  path = getattr(args, kind)
24
112
  if path:
25
- doc = doc or run.to_dict()
113
+ if doc is None and output.needs_doc:
114
+ doc = run.to_dict()
26
115
  print(write_output(output, Path(path), run, doc))
27
116
  wrote = True
28
117
  if args.ascii or not wrote:
@@ -39,6 +128,20 @@ def cmd_render(args: argparse.Namespace) -> int:
39
128
  return 0
40
129
 
41
130
 
131
+ def cmd_compare(args: argparse.Namespace) -> int:
132
+ try:
133
+ result = compare_runs(_load(args.before), _load(args.after), args.budget)
134
+ if args.json:
135
+ Path(args.json).write_text(
136
+ json.dumps(result, indent=2, allow_nan=False), encoding="utf-8"
137
+ )
138
+ print(format_comparison(result, args.top))
139
+ return int(result["exit_code"])
140
+ except (OSError, ValueError) as exc:
141
+ print(f"pytest-timing compare: {exc}", file=sys.stderr)
142
+ return 2
143
+
144
+
42
145
  def _uniquify(candidate: str, used: set[str]) -> str:
43
146
  """``candidate``, or ``candidate#2``, ``#3``... until it is not in ``used``."""
44
147
  name, n = candidate, 1
@@ -137,6 +240,22 @@ def build_parser() -> argparse.ArgumentParser:
137
240
  merge.add_argument("runs", nargs="+", help="pytest-timing JSON files")
138
241
  merge.add_argument("-o", "--output", required=True, metavar="PATH")
139
242
  merge.set_defaults(func=cmd_merge)
243
+ compare = sub.add_parser(
244
+ "compare", help="compare two saved runs and check per-test regression budgets"
245
+ )
246
+ compare.add_argument("before", help="baseline JSON run")
247
+ compare.add_argument("after", help="current JSON run")
248
+ compare.add_argument(
249
+ "--budget",
250
+ action="append",
251
+ type=parse_budget,
252
+ default=[],
253
+ metavar="METRIC=LIMIT",
254
+ help="maximum increase per common test, e.g. duration=10%%, cpu=50ms, memory=64MiB",
255
+ )
256
+ compare.add_argument("--json", metavar="PATH", help="write all differences and budget results")
257
+ compare.add_argument("--top", type=int, default=20, help="test details to print (default: 20)")
258
+ compare.set_defaults(func=cmd_compare)
140
259
  return parser
141
260
 
142
261
 
@@ -26,6 +26,7 @@ from pytest_timing.model import (
26
26
  Run,
27
27
  RunInfo,
28
28
  TestSpan,
29
+ WaitInterval,
29
30
  Worker,
30
31
  )
31
32
 
@@ -91,6 +92,7 @@ class Collector:
91
92
  attempt: int,
92
93
  seconds: float,
93
94
  gates: Iterable[str] = (),
95
+ ended: float | None = None,
94
96
  ) -> None:
95
97
  """Add an admission delay to the exact execution that waited.
96
98
 
@@ -107,6 +109,11 @@ class Collector:
107
109
  if span is None or span.outcome != "crashed" or span.phases or span.attempt != attempt:
108
110
  return
109
111
  gates = set(gates)
112
+ if ended is not None:
113
+ stop = self._rel(ended)
114
+ span.admission_waits.append(
115
+ WaitInterval(stop - seconds, stop, tuple(sorted(gates or {"cpu"})))
116
+ )
110
117
  if "cpu" in gates or not gates:
111
118
  if span.cpu is None:
112
119
  span.cpu = CpuRecord(elapsed=span.duration)