pg-workload 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. {pg_workload-0.3.0/src/pg_workload.egg-info → pg_workload-0.4.0}/PKG-INFO +42 -3
  2. {pg_workload-0.3.0 → pg_workload-0.4.0}/README.md +40 -1
  3. {pg_workload-0.3.0 → pg_workload-0.4.0}/pyproject.toml +2 -2
  4. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/__init__.py +3 -2
  5. pg_workload-0.4.0/src/pg_workload/bundled/profiles/bloat_vacuum/README.md +32 -0
  6. pg_workload-0.4.0/src/pg_workload/bundled/profiles/bloat_vacuum/generator.py +55 -0
  7. pg_workload-0.4.0/src/pg_workload/bundled/profiles/bloat_vacuum/profile.yml +37 -0
  8. pg_workload-0.4.0/src/pg_workload/bundled/profiles/bloat_vacuum/sql/01_churn.sql +27 -0
  9. pg_workload-0.4.0/src/pg_workload/bundled/profiles/bloat_vacuum/sql/accounts-schema.sql +17 -0
  10. pg_workload-0.4.0/src/pg_workload/bundled/profiles/emulate_errors/README.md +28 -0
  11. pg_workload-0.4.0/src/pg_workload/bundled/profiles/imdb/README.md +32 -0
  12. pg_workload-0.4.0/src/pg_workload/bundled/profiles/jsonb_docs/README.md +34 -0
  13. pg_workload-0.4.0/src/pg_workload/bundled/profiles/jsonb_docs/generator.py +56 -0
  14. pg_workload-0.4.0/src/pg_workload/bundled/profiles/jsonb_docs/profile.yml +38 -0
  15. pg_workload-0.4.0/src/pg_workload/bundled/profiles/jsonb_docs/sql/01_write.sql +20 -0
  16. pg_workload-0.4.0/src/pg_workload/bundled/profiles/jsonb_docs/sql/02_read.sql +18 -0
  17. pg_workload-0.4.0/src/pg_workload/bundled/profiles/jsonb_docs/sql/docs-schema.sql +15 -0
  18. pg_workload-0.4.0/src/pg_workload/bundled/profiles/many_objects/README.md +29 -0
  19. pg_workload-0.4.0/src/pg_workload/bundled/profiles/pagila/README.md +32 -0
  20. pg_workload-0.4.0/src/pg_workload/bundled/profiles/partition_aging/README.md +33 -0
  21. pg_workload-0.4.0/src/pg_workload/bundled/profiles/partition_aging/generator.py +64 -0
  22. pg_workload-0.4.0/src/pg_workload/bundled/profiles/partition_aging/profile.yml +77 -0
  23. pg_workload-0.4.0/src/pg_workload/bundled/profiles/partition_aging/sql/01_ingest.sql +6 -0
  24. pg_workload-0.4.0/src/pg_workload/bundled/profiles/partition_aging/sql/02_read_recent.sql +11 -0
  25. pg_workload-0.4.0/src/pg_workload/bundled/profiles/partition_aging/sql/03_read_history.sql +6 -0
  26. pg_workload-0.4.0/src/pg_workload/bundled/profiles/partition_aging/sql/events-schema.sql +12 -0
  27. pg_workload-0.4.0/src/pg_workload/bundled/profiles/pss_overflow/README.md +33 -0
  28. pg_workload-0.4.0/src/pg_workload/bundled/profiles/queue_skip_locked/README.md +34 -0
  29. pg_workload-0.4.0/src/pg_workload/bundled/profiles/queue_skip_locked/generator.py +59 -0
  30. pg_workload-0.4.0/src/pg_workload/bundled/profiles/queue_skip_locked/profile.yml +45 -0
  31. pg_workload-0.4.0/src/pg_workload/bundled/profiles/queue_skip_locked/sql/01_enqueue.sql +5 -0
  32. pg_workload-0.4.0/src/pg_workload/bundled/profiles/queue_skip_locked/sql/02_dequeue.sql +16 -0
  33. pg_workload-0.4.0/src/pg_workload/bundled/profiles/queue_skip_locked/sql/03_settle.sql +14 -0
  34. pg_workload-0.4.0/src/pg_workload/bundled/profiles/queue_skip_locked/sql/queue-schema.sql +18 -0
  35. pg_workload-0.4.0/src/pg_workload/bundled/profiles/simple_stock/README.md +30 -0
  36. pg_workload-0.4.0/src/pg_workload/bundled/profiles/simple_stock_spec_symbols/README.md +30 -0
  37. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/schema/pg_workload-v1.schema.json +1 -0
  38. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/orchestration.py +1 -0
  39. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/pg_client.py +9 -6
  40. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/profiles.py +7 -0
  41. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/runner.py +31 -0
  42. {pg_workload-0.3.0 → pg_workload-0.4.0/src/pg_workload.egg-info}/PKG-INFO +42 -3
  43. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload.egg-info/SOURCES.txt +32 -0
  44. {pg_workload-0.3.0 → pg_workload-0.4.0}/tests/test_docker_postgres18.py +51 -1
  45. {pg_workload-0.3.0 → pg_workload-0.4.0}/tests/test_workload.py +164 -0
  46. {pg_workload-0.3.0 → pg_workload-0.4.0}/LICENSE +0 -0
  47. {pg_workload-0.3.0 → pg_workload-0.4.0}/MANIFEST.in +0 -0
  48. {pg_workload-0.3.0 → pg_workload-0.4.0}/THIRD_PARTY_NOTICES.md +0 -0
  49. {pg_workload-0.3.0 → pg_workload-0.4.0}/setup.cfg +0 -0
  50. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/__main__.py +0 -0
  51. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/assets.py +0 -0
  52. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/emulate_errors/profile.yml +0 -0
  53. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/emulate_errors/sql/01_workload.sql +0 -0
  54. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/emulate_errors/sql/schema.sql +0 -0
  55. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/imdb/generator.py +0 -0
  56. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/imdb/profile.yml +0 -0
  57. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/imdb/sql/01_company_catalog.sql +0 -0
  58. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/imdb/sql/02_people_by_keyword.sql +0 -0
  59. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/imdb/sql/03_keyword_trends.sql +0 -0
  60. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/imdb/sql/04_genre_cast.sql +0 -0
  61. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/imdb/sql/05_join_stress.sql +0 -0
  62. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/imdb/sql/indexes.sql +0 -0
  63. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/imdb/sql/schema.sql +0 -0
  64. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/many_objects/profile.yml +0 -0
  65. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/many_objects/sql/01_workload.sql +0 -0
  66. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/many_objects/sql/prepare_schema_01.sql +0 -0
  67. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/many_objects/sql/prepare_schema_02.sql +0 -0
  68. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pagila/generator.py +0 -0
  69. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pagila/profile.yml +0 -0
  70. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pagila/sql/01_select.sql +0 -0
  71. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pagila/sql/02_insert.sql +0 -0
  72. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pagila/sql/03_update.sql +0 -0
  73. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pagila/sql/04_delete.sql +0 -0
  74. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pagila/sql/pagila-schema.sql +0 -0
  75. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pss_overflow/generator.py +0 -0
  76. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pss_overflow/profile.yml +0 -0
  77. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pss_overflow/sql/01_workload.sql +0 -0
  78. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/pss_overflow/sql/pss_overflow-schema.sql +0 -0
  79. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/simple_stock/generator.py +0 -0
  80. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/simple_stock/profile.yml +0 -0
  81. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/simple_stock/sql/01_workload.sql +0 -0
  82. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/simple_stock/sql/indexes.sql +0 -0
  83. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/simple_stock/sql/stock-schema.sql +0 -0
  84. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/simple_stock_spec_symbols/generator.py +0 -0
  85. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/simple_stock_spec_symbols/profile.yml +0 -0
  86. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/simple_stock_spec_symbols/sql/01_workload.sql +0 -0
  87. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/simple_stock_spec_symbols/sql/indexes.sql +0 -0
  88. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/bundled/profiles/simple_stock_spec_symbols/sql/stock-schema.sql +0 -0
  89. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/cli.py +0 -0
  90. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/common.py +0 -0
  91. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/config.py +0 -0
  92. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/control.py +0 -0
  93. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/logging.py +0 -0
  94. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/resources.py +0 -0
  95. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/scheduler.py +0 -0
  96. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload/state.py +0 -0
  97. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload.egg-info/dependency_links.txt +0 -0
  98. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload.egg-info/entry_points.txt +0 -0
  99. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload.egg-info/requires.txt +0 -0
  100. {pg_workload-0.3.0 → pg_workload-0.4.0}/src/pg_workload.egg-info/top_level.txt +0 -0
@@ -1,7 +1,7 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pg-workload
3
- Version: 0.3.0
4
- Summary: PostgreSQL backend activity emulator for diverse plans, logs, and runtime statistics.
3
+ Version: 0.4.0
4
+ Summary: PostgreSQL workload emulator for diagnostic and observability testing
5
5
  Author-email: O2eg <oleg.ispu@yandex.ru>
6
6
  License-Expression: MIT
7
7
  Project-URL: Homepage, https://github.com/O2eg/pg_workload
@@ -57,6 +57,24 @@ The distribution name and command are `pg-workload`; the import package is `pg_w
57
57
  wheel contains immutable profile templates and a public `pg_workload/v1` JSON Schema. Runtime
58
58
  state, logs, credentials, and generated table contents are never packaged.
59
59
 
60
+ ## Related projects
61
+
62
+ pg_workload is conceptually different from a benchmark tool such as
63
+ [pg_perf_bench](https://github.com/O2eg/pg_perf_bench): a benchmark measures a system
64
+ (maximum TPS across a controlled load sweep, with a recreated database and captured
65
+ environment evidence for every point), while pg_workload does not measure anything — it
66
+ emulates diverse background activity so that an otherwise empty database starts to "live":
67
+ queries, plans, locks, WAL, autovacuum, errors, and logs. That living database is then
68
+ observed from the other side: diagnostic reports, log parsing, dashboards, and monitoring
69
+ pipelines are tested against it.
70
+
71
+ | Project | How it is used together with pg_workload |
72
+ |---|---|
73
+ | [pg_stand](https://github.com/O2eg/pg_stand) | Deploys the disposable Docker stands (PostgreSQL 10–18, replication topologies, diagnostic extensions preloaded) that pg_workload loads — the canonical `pg-stand -> pg-workload -> pg-diag` target. |
74
+ | [pg_diag](https://github.com/O2eg/pg_diag) | Captures JSON/HTML diagnostic reports of the activity pg_workload generates. Bundled profile READMEs document which pg_diag sections each profile exercises, and `prepare-db` installs the matching default extensions (`pg_stat_statements`, `pg_buffercache`). |
75
+ | [pg_play](https://github.com/O2eg/pg_play) | Orchestrates pg_workload through the versioned machine contract `pg_play/component/v1` (plan hashes, desired-state scheduler control) — see [docs/pg_play-integration.md](docs/pg_play-integration.md). |
76
+ | [pg_configurator](https://github.com/O2eg/pg_configurator) | Generates versioned PostgreSQL configuration candidates; apply a candidate to the stand and rerun the same pg_workload profiles to observe how behavior changes with the settings. |
77
+
60
78
  ## Installation
61
79
 
62
80
  ### Install from PyPI
@@ -211,14 +229,25 @@ Use `--bin-dir /path/to/postgresql/bin` when client binaries are outside
211
229
  | `many_objects` | Metadata-heavy schemas and partitions | scale-aware SQL object creation |
212
230
  | `emulate_errors` | Intentional SQL errors | SQL seed rows only |
213
231
  | `pss_overflow` | `pg_stat_statements` churn | scaled Python generator; extension/preload required |
232
+ | `queue_skip_locked` | Producer/consumer task queue with `FOR UPDATE SKIP LOCKED` | synthetic Python generator |
233
+ | `jsonb_docs` | JSONB document store with GIN containment and `jsonb_set` updates | synthetic Python generator |
234
+ | `partition_aging` | Daily-partitioned time-series with partition pruning and aging DDL | synthetic Python generator |
235
+ | `bloat_vacuum` | Dead-tuple churn from non-HOT updates with scheduled `VACUUM` | synthetic Python generator |
214
236
 
215
237
  No profile requires a dump, CSV file, or network download. The Pagila schema is redistributed
216
238
  under its upstream license; see `THIRD_PARTY_NOTICES.md`. Its rows are generated locally. The
217
239
  bundled `imdb` profile is an original, compact movie-domain model and does not contain Join Order
218
240
  Benchmark SQL or IMDB source data.
219
241
 
242
+ Every profile directory contains a `README.md` with the scenario description, its jobs, the
243
+ pg_diag sections it is meant to exercise, and scale/observation guidance.
244
+
220
245
  ## Profile contract
221
246
 
247
+ See [docs/profile-authoring-cookbook.md](docs/profile-authoring-cookbook.md) for a complete
248
+ guide to writing custom profiles, including generator conventions and worked recipes for
249
+ queue, JSONB, partitioning, and bloat scenarios.
250
+
222
251
  Profiles live at `data/<profile>/profile.yml`. Version 1 is identified by the stable string
223
252
  `pg_workload/v1`:
224
253
 
@@ -228,6 +257,7 @@ name: my_workload
228
257
  schema: my_workload
229
258
  description: Small example workload.
230
259
  requires_write: false
260
+ min_pg_version: 12 # optional major-version floor for the target server
231
261
 
232
262
  prepare:
233
263
  steps:
@@ -265,6 +295,7 @@ conditions that JSON Schema alone cannot safely enforce:
265
295
  - job names are unique and types are `pgbench` or `psql`;
266
296
  - a pgbench job defines exactly one of `duration` or `transactions`;
267
297
  - threads do not exceed clients;
298
+ - `min_pg_version`, when present, is an integer major version of at least 10;
268
299
  - SQL files, generators, script globs, and logs stay inside their profile boundary even when
269
300
  symlinks are present;
270
301
  - generator steps reference an existing Python file;
@@ -272,7 +303,10 @@ conditions that JSON Schema alone cannot safely enforce:
272
303
 
273
304
  `validate` checks the complete contract and all local paths without connecting to PostgreSQL.
274
305
  `install` executes `prepare.steps` in declared order. A normal data profile therefore creates its
275
- schema, runs its generator, and only then creates indexes and statistics.
306
+ schema, runs its generator, and only then creates indexes and statistics. When a profile declares
307
+ `min_pg_version`, `install` also queries the target `server_version_num` and refuses to install
308
+ on an older server — this fails fast instead of breaking later on a missing feature
309
+ (for example, `jsonb_docs` requires PostgreSQL 12 for SQL/JSON path queries).
276
310
 
277
311
  ```bash
278
312
  pg-workload validate
@@ -320,6 +354,10 @@ pg-workload run --profile simple_stock --pgbench-transactions 100
320
354
  | `simple_stock_spec_symbols` | the same cardinalities with hostile and Unicode identifiers/values |
321
355
  | `imdb` | 10,000 companies; 100,000 people; 100,000 titles; 1,300,000 fact rows |
322
356
  | `pagila` | 600 customers; 1,000 films; 4,500 inventory; 16,000 rentals; 16,500 payments |
357
+ | `queue_skip_locked` | 20,000 pending + 5,000 done queue tasks |
358
+ | `jsonb_docs` | 50,000 JSONB documents |
359
+ | `partition_aging` | 33 daily partitions; 200,000 events skewed toward recent days |
360
+ | `bloat_vacuum` | 100,000 accounts, pre-bloated by one non-HOT update round |
323
361
 
324
362
  For example, `--scale 2` approximately doubles scalable tables. Very small values retain a
325
363
  profile-specific minimum so foreign-key structure and query selectivity remain meaningful in CI.
@@ -397,6 +435,7 @@ After `init`, the working directory is:
397
435
  project/
398
436
  data/<profile>/
399
437
  profile.yml
438
+ README.md # scenario, jobs, pg_diag watch list, scale guidance
400
439
  generator.py # only when the profile needs table data
401
440
  sql/
402
441
  log/ # runtime, ignored
@@ -18,6 +18,24 @@ The distribution name and command are `pg-workload`; the import package is `pg_w
18
18
  wheel contains immutable profile templates and a public `pg_workload/v1` JSON Schema. Runtime
19
19
  state, logs, credentials, and generated table contents are never packaged.
20
20
 
21
+ ## Related projects
22
+
23
+ pg_workload is conceptually different from a benchmark tool such as
24
+ [pg_perf_bench](https://github.com/O2eg/pg_perf_bench): a benchmark measures a system
25
+ (maximum TPS across a controlled load sweep, with a recreated database and captured
26
+ environment evidence for every point), while pg_workload does not measure anything — it
27
+ emulates diverse background activity so that an otherwise empty database starts to "live":
28
+ queries, plans, locks, WAL, autovacuum, errors, and logs. That living database is then
29
+ observed from the other side: diagnostic reports, log parsing, dashboards, and monitoring
30
+ pipelines are tested against it.
31
+
32
+ | Project | How it is used together with pg_workload |
33
+ |---|---|
34
+ | [pg_stand](https://github.com/O2eg/pg_stand) | Deploys the disposable Docker stands (PostgreSQL 10–18, replication topologies, diagnostic extensions preloaded) that pg_workload loads — the canonical `pg-stand -> pg-workload -> pg-diag` target. |
35
+ | [pg_diag](https://github.com/O2eg/pg_diag) | Captures JSON/HTML diagnostic reports of the activity pg_workload generates. Bundled profile READMEs document which pg_diag sections each profile exercises, and `prepare-db` installs the matching default extensions (`pg_stat_statements`, `pg_buffercache`). |
36
+ | [pg_play](https://github.com/O2eg/pg_play) | Orchestrates pg_workload through the versioned machine contract `pg_play/component/v1` (plan hashes, desired-state scheduler control) — see [docs/pg_play-integration.md](docs/pg_play-integration.md). |
37
+ | [pg_configurator](https://github.com/O2eg/pg_configurator) | Generates versioned PostgreSQL configuration candidates; apply a candidate to the stand and rerun the same pg_workload profiles to observe how behavior changes with the settings. |
38
+
21
39
  ## Installation
22
40
 
23
41
  ### Install from PyPI
@@ -172,14 +190,25 @@ Use `--bin-dir /path/to/postgresql/bin` when client binaries are outside
172
190
  | `many_objects` | Metadata-heavy schemas and partitions | scale-aware SQL object creation |
173
191
  | `emulate_errors` | Intentional SQL errors | SQL seed rows only |
174
192
  | `pss_overflow` | `pg_stat_statements` churn | scaled Python generator; extension/preload required |
193
+ | `queue_skip_locked` | Producer/consumer task queue with `FOR UPDATE SKIP LOCKED` | synthetic Python generator |
194
+ | `jsonb_docs` | JSONB document store with GIN containment and `jsonb_set` updates | synthetic Python generator |
195
+ | `partition_aging` | Daily-partitioned time-series with partition pruning and aging DDL | synthetic Python generator |
196
+ | `bloat_vacuum` | Dead-tuple churn from non-HOT updates with scheduled `VACUUM` | synthetic Python generator |
175
197
 
176
198
  No profile requires a dump, CSV file, or network download. The Pagila schema is redistributed
177
199
  under its upstream license; see `THIRD_PARTY_NOTICES.md`. Its rows are generated locally. The
178
200
  bundled `imdb` profile is an original, compact movie-domain model and does not contain Join Order
179
201
  Benchmark SQL or IMDB source data.
180
202
 
203
+ Every profile directory contains a `README.md` with the scenario description, its jobs, the
204
+ pg_diag sections it is meant to exercise, and scale/observation guidance.
205
+
181
206
  ## Profile contract
182
207
 
208
+ See [docs/profile-authoring-cookbook.md](docs/profile-authoring-cookbook.md) for a complete
209
+ guide to writing custom profiles, including generator conventions and worked recipes for
210
+ queue, JSONB, partitioning, and bloat scenarios.
211
+
183
212
  Profiles live at `data/<profile>/profile.yml`. Version 1 is identified by the stable string
184
213
  `pg_workload/v1`:
185
214
 
@@ -189,6 +218,7 @@ name: my_workload
189
218
  schema: my_workload
190
219
  description: Small example workload.
191
220
  requires_write: false
221
+ min_pg_version: 12 # optional major-version floor for the target server
192
222
 
193
223
  prepare:
194
224
  steps:
@@ -226,6 +256,7 @@ conditions that JSON Schema alone cannot safely enforce:
226
256
  - job names are unique and types are `pgbench` or `psql`;
227
257
  - a pgbench job defines exactly one of `duration` or `transactions`;
228
258
  - threads do not exceed clients;
259
+ - `min_pg_version`, when present, is an integer major version of at least 10;
229
260
  - SQL files, generators, script globs, and logs stay inside their profile boundary even when
230
261
  symlinks are present;
231
262
  - generator steps reference an existing Python file;
@@ -233,7 +264,10 @@ conditions that JSON Schema alone cannot safely enforce:
233
264
 
234
265
  `validate` checks the complete contract and all local paths without connecting to PostgreSQL.
235
266
  `install` executes `prepare.steps` in declared order. A normal data profile therefore creates its
236
- schema, runs its generator, and only then creates indexes and statistics.
267
+ schema, runs its generator, and only then creates indexes and statistics. When a profile declares
268
+ `min_pg_version`, `install` also queries the target `server_version_num` and refuses to install
269
+ on an older server — this fails fast instead of breaking later on a missing feature
270
+ (for example, `jsonb_docs` requires PostgreSQL 12 for SQL/JSON path queries).
237
271
 
238
272
  ```bash
239
273
  pg-workload validate
@@ -281,6 +315,10 @@ pg-workload run --profile simple_stock --pgbench-transactions 100
281
315
  | `simple_stock_spec_symbols` | the same cardinalities with hostile and Unicode identifiers/values |
282
316
  | `imdb` | 10,000 companies; 100,000 people; 100,000 titles; 1,300,000 fact rows |
283
317
  | `pagila` | 600 customers; 1,000 films; 4,500 inventory; 16,000 rentals; 16,500 payments |
318
+ | `queue_skip_locked` | 20,000 pending + 5,000 done queue tasks |
319
+ | `jsonb_docs` | 50,000 JSONB documents |
320
+ | `partition_aging` | 33 daily partitions; 200,000 events skewed toward recent days |
321
+ | `bloat_vacuum` | 100,000 accounts, pre-bloated by one non-HOT update round |
284
322
 
285
323
  For example, `--scale 2` approximately doubles scalable tables. Very small values retain a
286
324
  profile-specific minimum so foreign-key structure and query selectivity remain meaningful in CI.
@@ -358,6 +396,7 @@ After `init`, the working directory is:
358
396
  project/
359
397
  data/<profile>/
360
398
  profile.yml
399
+ README.md # scenario, jobs, pg_diag watch list, scale guidance
361
400
  generator.py # only when the profile needs table data
362
401
  sql/
363
402
  log/ # runtime, ignored
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "pg-workload"
7
- version = "0.3.0"
8
- description = "PostgreSQL backend activity emulator for diverse plans, logs, and runtime statistics."
7
+ version = "0.4.0"
8
+ description = "PostgreSQL workload emulator for diagnostic and observability testing"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
11
11
  license = "MIT"
@@ -1,6 +1,6 @@
1
1
  """PostgreSQL workload generation and scheduling tools."""
2
2
 
3
- __version__ = "0.3.0"
3
+ __version__ = "0.4.0"
4
4
 
5
5
  from pg_workload.assets import bundled_profiles_root, initialize_project
6
6
  from pg_workload.cli import build_parser
@@ -15,7 +15,7 @@ from pg_workload.profiles import (
15
15
  validate_profile,
16
16
  )
17
17
  from pg_workload.resources import CpuSample, ResourceGuard, assert_resources_available
18
- from pg_workload.runner import check_profile_requirements, install_profiles, prepare_database
18
+ from pg_workload.runner import check_min_pg_version, check_profile_requirements, install_profiles, prepare_database
19
19
  from pg_workload.scheduler import job_recover_on_failure
20
20
  from pg_workload.state import effective_schedule, load_state, update_job_state
21
21
 
@@ -32,6 +32,7 @@ __all__ = [
32
32
  "build_parser",
33
33
  "build_runtime_config",
34
34
  "bundled_profiles_root",
35
+ "check_min_pg_version",
35
36
  "check_profile_requirements",
36
37
  "effective_schedule",
37
38
  "file_lock",
@@ -0,0 +1,32 @@
1
+ # bloat_vacuum
2
+
3
+ Dead-tuple generation on purpose: updates touch indexed columns (defeating HOT updates),
4
+ so dead tuples accumulate in the heap and indexes. A scheduled manual `VACUUM` and a bloat
5
+ reporting job make the whole cycle observable. Use it to watch autovacuum queueing, vacuum
6
+ progress, and table/index growth versus live row count.
7
+
8
+ ## Objects
9
+
10
+ - `bloat_vacuum.accounts` — 100,000 rows at scale 1, pre-bloated by one non-HOT update
11
+ round during install
12
+ - Secondary indexes on `balance`, `status`, `updated_at` — the columns the churn job updates
13
+
14
+ ## Jobs
15
+
16
+ | Job | Type | Interval | What it does |
17
+ |---|---|---|---|
18
+ | `churn` | pgbench | 60 s | non-HOT updates of random rows plus a small insert/delete tail |
19
+ | `vacuum_worker` | psql | 300 s | `VACUUM (ANALYZE) bloat_vacuum.accounts` — explicit vacuum activity |
20
+ | `bloat_report` | psql | 600 s | logs `n_dead_tup`, vacuum counts, and timestamps from `pg_stat_user_tables` |
21
+
22
+ ## What to watch in pg_diag
23
+
24
+ - `storage_vacuum` — dead tuples, autovacuum queue, last vacuum/analyze timestamps
25
+ - `maintenance_progress` — vacuum phases while `vacuum_worker` runs
26
+ - `object_workload` — table and index size growth versus `n_live_tup`
27
+
28
+ ## Recommended scale and observation window
29
+
30
+ - Scale 0.01–0.1; bloat accumulates per cycle, so longer runs show more.
31
+ - 10–15 minutes to see several churn cycles, at least one manual vacuum, and a bloat
32
+ report entry. Compare `data/bloat_vacuum/log/bloat_report.log` over time.
@@ -0,0 +1,55 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import math
5
+ import os
6
+ import subprocess
7
+
8
+
9
+ def scaled(base: int, scale: float, minimum: int) -> int:
10
+ return max(minimum, round(base * scale))
11
+
12
+
13
+ def main() -> None:
14
+ parser = argparse.ArgumentParser(description="Generate synthetic bloat_vacuum data")
15
+ parser.add_argument("--scale", type=float, required=True)
16
+ args = parser.parse_args()
17
+ if not math.isfinite(args.scale) or args.scale <= 0:
18
+ parser.error("--scale must be a finite number greater than zero")
19
+
20
+ account_count = scaled(100_000, args.scale, 1_000)
21
+
22
+ sql = f"""
23
+ SELECT setseed(0.90731255);
24
+
25
+ INSERT INTO bloat_vacuum.accounts (owner_name, balance, status, updated_at, filler)
26
+ SELECT
27
+ 'account owner ' || g,
28
+ round((power(random(), 1.7) * 50000)::numeric, 2),
29
+ (ARRAY['active', 'active', 'active', 'suspended', 'closed'])[1 + floor(random() * 5)::integer],
30
+ now() - (random() * interval '30 days'),
31
+ md5(g::text) || repeat('x', 40)
32
+ FROM generate_series(1, {account_count}) AS g;
33
+
34
+ -- Pre-bloat: one non-HOT update round over a tenth of the table, so the
35
+ -- first report already shows dead tuples.
36
+ UPDATE bloat_vacuum.accounts
37
+ SET balance = balance + 1,
38
+ updated_at = clock_timestamp(),
39
+ filler = md5(random()::text) || repeat('x', 40)
40
+ WHERE id % 10 = 0;
41
+
42
+ ANALYZE bloat_vacuum.accounts;
43
+ """
44
+ subprocess.run(
45
+ [os.environ.get("PG_WORKLOAD_PSQL", "psql"), "-X", "-q", "-v", "ON_ERROR_STOP=1"],
46
+ input=sql,
47
+ text=True,
48
+ check=True,
49
+ )
50
+
51
+ print(f"Generated bloat_vacuum: accounts={account_count}")
52
+
53
+
54
+ if __name__ == "__main__":
55
+ main()
@@ -0,0 +1,37 @@
1
+ api_version: pg_workload/v1
2
+ name: bloat_vacuum
3
+ schema: bloat_vacuum
4
+ description: Dead-tuple churn from non-HOT updates with scheduled VACUUM and bloat reporting.
5
+ requires_write: true
6
+ prepare:
7
+ steps:
8
+ - type: sql
9
+ path: sql/accounts-schema.sql
10
+ - type: generator
11
+ path: generator.py
12
+ jobs:
13
+ - name: churn
14
+ type: pgbench
15
+ interval: 60
16
+ log: churn.log
17
+ no_vacuum: true
18
+ clients: 2
19
+ threads: 2
20
+ transactions: 5
21
+ scripts:
22
+ - path: sql/01_churn.sql
23
+ - name: vacuum_worker
24
+ type: psql
25
+ interval: 300
26
+ log: vacuum_worker.log
27
+ command: "VACUUM (ANALYZE) bloat_vacuum.accounts;"
28
+ - name: bloat_report
29
+ type: psql
30
+ interval: 600
31
+ log: bloat_report.log
32
+ command: >-
33
+ SELECT relname, n_live_tup, n_dead_tup,
34
+ round(100.0 * n_dead_tup / NULLIF(n_live_tup + n_dead_tup, 0), 2) AS dead_pct,
35
+ vacuum_count, autovacuum_count, last_vacuum, last_autovacuum
36
+ FROM pg_stat_user_tables
37
+ WHERE schemaname = 'bloat_vacuum';
@@ -0,0 +1,27 @@
1
+ -- Non-HOT updates on indexed columns: dead tuples grow in heap and indexes.
2
+ UPDATE bloat_vacuum.accounts a
3
+ SET balance = a.balance + (random() * 100 - 50)::numeric(14, 2),
4
+ updated_at = clock_timestamp(),
5
+ filler = md5(random()::text) || repeat('x', 40)
6
+ WHERE a.id IN (
7
+ SELECT 1 + floor(random() * (SELECT max(id) FROM bloat_vacuum.accounts))::bigint
8
+ FROM generate_series(1, 10)
9
+ );
10
+
11
+ -- A small insert/delete tail keeps autovacuum interesting.
12
+ INSERT INTO bloat_vacuum.accounts (owner_name, balance, status, filler)
13
+ SELECT
14
+ 'churn user ' || g,
15
+ round((random() * 10000)::numeric, 2),
16
+ 'active',
17
+ md5(random()::text) || repeat('x', 40)
18
+ FROM generate_series(1, 3) AS g;
19
+
20
+ DELETE FROM bloat_vacuum.accounts
21
+ WHERE id IN (
22
+ SELECT id
23
+ FROM bloat_vacuum.accounts
24
+ WHERE owner_name LIKE 'churn user %'
25
+ ORDER BY random()
26
+ LIMIT 3
27
+ );
@@ -0,0 +1,17 @@
1
+ DROP SCHEMA IF EXISTS bloat_vacuum CASCADE;
2
+ CREATE SCHEMA bloat_vacuum;
3
+
4
+ -- Several secondary indexes on columns the churn job updates: every update
5
+ -- is non-HOT and leaves dead tuples in both the heap and the indexes.
6
+ CREATE TABLE bloat_vacuum.accounts (
7
+ id bigint GENERATED ALWAYS AS IDENTITY PRIMARY KEY,
8
+ owner_name text NOT NULL,
9
+ balance numeric(14, 2) NOT NULL,
10
+ status text NOT NULL,
11
+ updated_at timestamptz NOT NULL DEFAULT now(),
12
+ filler text NOT NULL
13
+ );
14
+
15
+ CREATE INDEX accounts_balance_idx ON bloat_vacuum.accounts (balance);
16
+ CREATE INDEX accounts_status_idx ON bloat_vacuum.accounts (status);
17
+ CREATE INDEX accounts_updated_at_idx ON bloat_vacuum.accounts (updated_at);
@@ -0,0 +1,28 @@
1
+ # emulate_errors
2
+
3
+ Intentional SQL errors: constraint violations, bad syntax, division by zero, and similar
4
+ failures. The job runs with `allow_failure: true`, so errors are expected and logged. Use
5
+ it to fill PostgreSQL error logs and transaction rollback statistics
6
+ (`pg_stat_database.xact_rollback`) without breaking the workload runner.
7
+
8
+ ## Objects
9
+
10
+ - `emulate_errors.accounts`, `emulate_errors.transactions` — small fixed seed
11
+ (`--scale` is ignored by design)
12
+
13
+ ## Jobs
14
+
15
+ | Job | Type | Interval | What it does |
16
+ |---|---|---|---|
17
+ | `errors` | pgbench | 60 s | statements that fail in various ways, with `allow_failure` |
18
+
19
+ ## What to watch in pg_diag
20
+
21
+ - PostgreSQL CSV logs — ERROR/FATAL entries with statements (pg_stand stands collect them)
22
+ - `snapshot_delta_workload` — rollback-heavy transaction statistics
23
+ - `sql_workload` — failing statements still visible in `pg_stat_statements`
24
+
25
+ ## Recommended scale and observation window
26
+
27
+ - Any scale (ignored); the seed is intentionally tiny.
28
+ - 5 minutes is enough to produce a representative error stream.
@@ -0,0 +1,32 @@
1
+ # imdb
2
+
3
+ Analytical read-only workload over a synthetic movie-domain model: multi-way joins,
4
+ aggregations, selective lookups, and skewed popularity distributions. An original compact
5
+ schema — not the Join Order Benchmark and no IMDB source data (both excluded for licensing
6
+ reasons). Use it to exercise the planner and to produce interesting plans for
7
+ `auto_explain` and `pg_stat_statements`.
8
+
9
+ ## Objects (at scale 1)
10
+
11
+ - 10,000 companies; 100,000 people; 100,000 titles; ~1,300,000 fact rows
12
+ (`cast_info`, `movie_keyword`, `movie_company`, `movie_info`)
13
+ - Secondary indexes created after the data load
14
+
15
+ ## Jobs
16
+
17
+ | Job | Type | Interval | What it does |
18
+ |---|---|---|---|
19
+ | `analytical_selects` | pgbench | 60 s | five join/aggregation scripts of varying selectivity |
20
+
21
+ ## What to watch in pg_diag
22
+
23
+ - `sql_workload` — top statements by total/mean time with real join plans
24
+ - `object_workload` — sequential vs index scan split on fact tables
25
+ - `buffer_cache` — relation caching behaviour under repeated analytical reads
26
+ - `snapshot_charts_db` — read I/O rates during the observation window
27
+
28
+ ## Recommended scale and observation window
29
+
30
+ - Scale 1 for meaningful planning; scale 0.1 still keeps join selectivity sensible.
31
+ - 5–10 minutes of scheduled runs; analytical queries dominate `pg_stat_statements`
32
+ quickly because the job interval is 60 s.
@@ -0,0 +1,34 @@
1
+ # jsonb_docs
2
+
3
+ Document-store workload over a `jsonb` column: GIN-indexed containment queries, jsonpath
4
+ extraction, and in-place `jsonb_set` updates of wide documents. Use it to exercise GIN
5
+ indexes, TOAST, and the WAL volume produced by JSONB updates.
6
+
7
+ ## Objects
8
+
9
+ - `jsonb_docs.documents` — 50,000 documents at scale 1, skewed `customer_id` and tags
10
+ - `documents_doc_gin` — GIN index with `jsonb_path_ops` (containment)
11
+ - `documents_status_idx` — expression index on `(doc ->> 'status')`
12
+
13
+ ## Jobs
14
+
15
+ | Job | Type | Interval | What it does |
16
+ |---|---|---|---|
17
+ | `writers` | pgbench | 60 s | insert documents; `jsonb_set` updates on random rows |
18
+ | `readers` | pgbench | 60 s | `@>` containment, `jsonb_path_query`, status aggregation |
19
+ | `analyze_docs` | psql | 1800 s | `ANALYZE jsonb_docs.documents` |
20
+
21
+ ## Requirements
22
+
23
+ - PostgreSQL 12+ (`min_pg_version: 12`) — the reader script uses SQL/JSON path queries
24
+
25
+ ## What to watch in pg_diag
26
+
27
+ - `indexes` — GIN vs expression index usage and size
28
+ - `wal_io_checkpoints` — WAL volume from full-row JSONB updates
29
+ - `storage_vacuum` — TOAST storage and dead tuples from updates
30
+
31
+ ## Recommended scale and observation window
32
+
33
+ - Scale 0.01–0.1 for CI; scale 1 makes GIN vs seq-scan choices visible.
34
+ - 5–10 minutes of scheduled runs before a snapshots window.
@@ -0,0 +1,56 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import math
5
+ import os
6
+ import subprocess
7
+
8
+
9
+ def scaled(base: int, scale: float, minimum: int) -> int:
10
+ return max(minimum, round(base * scale))
11
+
12
+
13
+ def main() -> None:
14
+ parser = argparse.ArgumentParser(description="Generate synthetic jsonb_docs data")
15
+ parser.add_argument("--scale", type=float, required=True)
16
+ args = parser.parse_args()
17
+ if not math.isfinite(args.scale) or args.scale <= 0:
18
+ parser.error("--scale must be a finite number greater than zero")
19
+
20
+ doc_count = scaled(50_000, args.scale, 1_000)
21
+
22
+ sql = f"""
23
+ SELECT setseed(0.73051984);
24
+
25
+ INSERT INTO jsonb_docs.documents (doc_type, doc, created_at)
26
+ SELECT
27
+ (ARRAY['order', 'invoice', 'ticket', 'profile'])[1 + floor(random() * 4)::integer],
28
+ jsonb_build_object(
29
+ 'status', (ARRAY['new', 'open', 'closed'])[1 + floor(random() * 3)::integer],
30
+ 'customer_id', 1 + floor(power(random(), 2) * 10000)::integer,
31
+ 'amount', round((power(random(), 1.5) * 1000)::numeric, 2),
32
+ 'priority', floor(random() * 5)::integer,
33
+ 'tags', (
34
+ SELECT jsonb_agg('tag-' || t)
35
+ FROM (SELECT DISTINCT floor(random() * 50)::integer AS t
36
+ FROM generate_series(1, 1 + floor(random() * 4)::integer)) AS s
37
+ ),
38
+ 'comment', 'Synthetic document ' || g
39
+ ),
40
+ now() - (random() * interval '7 days')
41
+ FROM generate_series(1, {doc_count}) AS g;
42
+
43
+ ANALYZE jsonb_docs.documents;
44
+ """
45
+ subprocess.run(
46
+ [os.environ.get("PG_WORKLOAD_PSQL", "psql"), "-X", "-q", "-v", "ON_ERROR_STOP=1"],
47
+ input=sql,
48
+ text=True,
49
+ check=True,
50
+ )
51
+
52
+ print(f"Generated jsonb_docs: documents={doc_count}")
53
+
54
+
55
+ if __name__ == "__main__":
56
+ main()
@@ -0,0 +1,38 @@
1
+ api_version: pg_workload/v1
2
+ name: jsonb_docs
3
+ schema: jsonb_docs
4
+ description: Document-store workload over jsonb with GIN containment and jsonb_set updates.
5
+ requires_write: true
6
+ min_pg_version: 12
7
+ prepare:
8
+ steps:
9
+ - type: sql
10
+ path: sql/docs-schema.sql
11
+ - type: generator
12
+ path: generator.py
13
+ jobs:
14
+ - name: writers
15
+ type: pgbench
16
+ interval: 60
17
+ log: writers.log
18
+ no_vacuum: true
19
+ clients: 2
20
+ threads: 2
21
+ transactions: 5
22
+ scripts:
23
+ - path: sql/01_write.sql
24
+ - name: readers
25
+ type: pgbench
26
+ interval: 60
27
+ log: readers.log
28
+ no_vacuum: true
29
+ clients: 2
30
+ threads: 2
31
+ transactions: 5
32
+ scripts:
33
+ - path: sql/02_read.sql
34
+ - name: analyze_docs
35
+ type: psql
36
+ interval: 1800
37
+ log: analyze_docs.log
38
+ command: "ANALYZE jsonb_docs.documents;"
@@ -0,0 +1,20 @@
1
+ INSERT INTO jsonb_docs.documents (doc_type, doc)
2
+ SELECT
3
+ (ARRAY['order', 'invoice', 'ticket', 'profile'])[1 + floor(random() * 4)::integer],
4
+ jsonb_build_object(
5
+ 'status', 'new',
6
+ 'customer_id', 1 + floor(power(random(), 2) * 10000)::integer,
7
+ 'amount', round((random() * 1000)::numeric, 2),
8
+ 'priority', floor(random() * 5)::integer,
9
+ 'tags', jsonb_build_array('tag-' || floor(random() * 50)::integer)
10
+ )
11
+ FROM generate_series(1, 10) AS g;
12
+
13
+ -- Non-HOT-friendly update of a wide jsonb column on random rows.
14
+ UPDATE jsonb_docs.documents d
15
+ SET doc = jsonb_set(d.doc, '{status}', to_jsonb('closed'::text), false),
16
+ updated_at = clock_timestamp()
17
+ WHERE d.id IN (
18
+ SELECT 1 + floor(random() * (SELECT max(id) FROM jsonb_docs.documents))::bigint
19
+ FROM generate_series(1, 5)
20
+ );
@@ -0,0 +1,18 @@
1
+ \set customer_id random(1, 10000)
2
+
3
+ -- GIN-indexed containment lookup.
4
+ SELECT count(*)
5
+ FROM jsonb_docs.documents
6
+ WHERE doc @> jsonb_build_object('customer_id', :customer_id);
7
+
8
+ -- Expression-index filter plus a jsonpath extraction.
9
+ SELECT d.id, jsonb_path_query(d.doc, '$.tags[*]') AS tag
10
+ FROM jsonb_docs.documents d
11
+ WHERE d.doc ->> 'status' = 'open'
12
+ LIMIT 50;
13
+
14
+ -- Aggregation over extracted keys.
15
+ SELECT doc ->> 'status' AS status, count(*) AS docs
16
+ FROM jsonb_docs.documents
17
+ GROUP BY 1
18
+ ORDER BY 2 DESC;
@@ -0,0 +1,15 @@
1
+ DROP SCHEMA IF EXISTS jsonb_docs CASCADE;
2
+ CREATE SCHEMA jsonb_docs;
3
+
4
+ CREATE TABLE jsonb_docs.documents (
5
+ id bigint GENERATED ALWAYS AS IDENTITY PRIMARY KEY,
6
+ doc_type text NOT NULL,
7
+ doc jsonb NOT NULL,
8
+ created_at timestamptz NOT NULL DEFAULT now(),
9
+ updated_at timestamptz NOT NULL DEFAULT now()
10
+ );
11
+
12
+ -- Containment (@>) lookups.
13
+ CREATE INDEX documents_doc_gin ON jsonb_docs.documents USING gin (doc jsonb_path_ops);
14
+ -- Equality on the most-filtered key.
15
+ CREATE INDEX documents_status_idx ON jsonb_docs.documents ((doc ->> 'status'));