django-data-shape 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (98) hide show
  1. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/CHANGELOG.md +69 -1
  2. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/PKG-INFO +46 -5
  3. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/README.md +42 -4
  4. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/__init__.py +6 -0
  5. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/build.py +101 -15
  6. django_data_shape-0.4.0/django_data_shape/fixtures/__init__.py +19 -0
  7. django_data_shape-0.4.0/django_data_shape/fixtures/scale_fixture.py +76 -0
  8. django_data_shape-0.4.0/django_data_shape/fixtures/shape_fixture.py +94 -0
  9. django_data_shape-0.4.0/django_data_shape/fixtures/skip_unless_postgres.py +36 -0
  10. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/require_postgres.py +9 -2
  11. django_data_shape-0.4.0/django_data_shape/scale_protocol.py +48 -0
  12. django_data_shape-0.4.0/django_data_shape/scaled_shape.py +102 -0
  13. django_data_shape-0.4.0/django_data_shape/scaled_world.py +77 -0
  14. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/version.py +1 -1
  15. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/docs/index.md +18 -2
  16. django_data_shape-0.4.0/docs/pytest.md +332 -0
  17. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/docs/reference.md +14 -0
  18. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/docs/relations.md +6 -0
  19. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/mkdocs.yml +1 -0
  20. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/pyproject.toml +15 -1
  21. django_data_shape-0.4.0/tests/scale_protocol_consumers.py +54 -0
  22. django_data_shape-0.4.0/tests/scale_protocol_impostors.py +32 -0
  23. django_data_shape-0.4.0/tests/test_build_portable.py +125 -0
  24. django_data_shape-0.4.0/tests/test_fixtures.py +139 -0
  25. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_require_postgres.py +21 -0
  26. django_data_shape-0.4.0/tests/test_scale_protocol.py +46 -0
  27. django_data_shape-0.4.0/tests/test_scaled_shape.py +139 -0
  28. django_data_shape-0.4.0/tests/test_scaled_world.py +132 -0
  29. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/testapp/models.py +13 -0
  30. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/uv.lock +7 -1
  31. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/CODEOWNERS +0 -0
  32. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/SECURITY.md +0 -0
  33. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/dependabot.yml +0 -0
  34. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/workflows/release.yml +0 -0
  35. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/workflows/tests.yml +0 -0
  36. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/workflows/upstream-drift.yml +0 -0
  37. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.gitignore +0 -0
  38. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.pre-commit-config.yaml +0 -0
  39. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/CLAUDE.md +0 -0
  40. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/LICENSE +0 -0
  41. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/Makefile +0 -0
  42. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/build_result.py +0 -0
  43. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/__init__.py +0 -0
  44. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/bounded.py +0 -0
  45. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/constant.py +0 -0
  46. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/distribution.py +0 -0
  47. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/sequential.py +0 -0
  48. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/skew.py +0 -0
  49. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/uniform.py +0 -0
  50. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/zipf.py +0 -0
  51. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/fan_out.py +0 -0
  52. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/fan_out_plan.py +0 -0
  53. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/generate_rows.py +0 -0
  54. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/infer_key_strategy.py +0 -0
  55. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/invalid_shape.py +0 -0
  56. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/keys/__init__.py +0 -0
  57. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/keys/key_function.py +0 -0
  58. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/keys/key_strategy.py +0 -0
  59. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/keys/sequential_keys.py +0 -0
  60. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/keys/uuid_keys.py +0 -0
  61. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/order_tables.py +0 -0
  62. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/py.typed +0 -0
  63. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/resolve_fan_out.py +0 -0
  64. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/shape.py +0 -0
  65. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/shape_not_empty.py +0 -0
  66. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/table.py +0 -0
  67. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/table_result.py +0 -0
  68. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/unsupported_backend.py +0 -0
  69. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/utils.py +0 -0
  70. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/docs/keys.md +0 -0
  71. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/scripts/release-publish.sh +0 -0
  72. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/__init__.py +0 -0
  73. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/conftest.py +0 -0
  74. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/conftest_settings.py +0 -0
  75. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/__init__.py +0 -0
  76. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/test_constant.py +0 -0
  77. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/test_sequential.py +0 -0
  78. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/test_skew.py +0 -0
  79. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/test_uniform.py +0 -0
  80. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/test_zipf.py +0 -0
  81. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/keys/__init__.py +0 -0
  82. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/keys/test_key_function.py +0 -0
  83. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/keys/test_sequential_keys.py +0 -0
  84. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/keys/test_uuid_keys.py +0 -0
  85. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_build.py +0 -0
  86. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_build_graph.py +0 -0
  87. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_build_keys.py +0 -0
  88. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_documentation.py +0 -0
  89. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_fan_out.py +0 -0
  90. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_generate_rows.py +0 -0
  91. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_infer_key_strategy.py +0 -0
  92. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_order_tables.py +0 -0
  93. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_resolve_fan_out.py +0 -0
  94. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_shape.py +0 -0
  95. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_table.py +0 -0
  96. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_utils.py +0 -0
  97. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_version.py +0 -0
  98. {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/testapp/__init__.py +0 -0
@@ -7,6 +7,73 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.4.0] — 2026-09-02
11
+
12
+ ### Added
13
+ - **The pytest surface**, in `django_data_shape.fixtures`. `shape_fixture(shape)`
14
+ returns a session-scoped fixture that builds a shape once for a whole run;
15
+ bind it to a name in `conftest.py` and request that name from a test. It
16
+ composes with pytest-django rather than replacing it, asking for
17
+ `django_db_setup` and `django_db_blocker` by name so the coupling is to two
18
+ fixture names and not to pytest-django's internals. Session scope is
19
+ load-bearing: pytest creates higher-scoped fixtures first, so the rows are
20
+ committed before the transaction that wraps a test is opened, and every test
21
+ sees them while everything a test writes is rolled back with it.
22
+ - **The scale protocol**, which is what a growth assertion asks a world for:
23
+ make the world be at factor F, then let the caller run its block.
24
+ `scaled_world(shape, factor)` is a context manager that builds it and undoes
25
+ it; `scale_fixture(shape)` is the same thing as a fixture; and `ScaleProtocol`
26
+ is the structural type both satisfy, so a consumer asserting that a query
27
+ count is `O(1)` rather than `O(N)` depends on the shape of the call rather
28
+ than on this package. What the context manager yields is a plain row count,
29
+ not a `BuildResult`, for the same reason: a seam a stranger cannot implement
30
+ is not a seam.
31
+ - `scaled_shape(shape, factor)`, the declaration transform underneath it. **A
32
+ factor varies the declaration rather than subsetting one larger build**, because
33
+ a subset is not a smaller database but the same database with a filter -- the
34
+ statistics still describe every row, and the block under test would have to
35
+ cooperate by restricting itself, which puts the harness inside the thing being
36
+ measured. Every table scales, parents included, so the average fan-out is the
37
+ same at every factor and two worlds differ in size alone. The scaled tables go
38
+ through `Table`'s own constructor, so a declaration that only holds at its
39
+ original size is refused at the factor that breaks it, naming the factor.
40
+ - `skip_unless_postgres(connection, operation)`, the pytest twin of the backend
41
+ refusal. Both fixtures skip with the refusal's own message as the reason where
42
+ a shaped database cannot exist, so a suite that also runs on SQLite reports
43
+ what it did not check rather than passing over a database nobody shaped.
44
+ - A `pytest` extra. It is an extra rather than a dependency because the rest of
45
+ the package has nothing to do with pytest, which is also why these fixtures are
46
+ not re-exported from the top-level `__init__`: importing them is what requires
47
+ pytest, not importing the package.
48
+ - **`build(shape, require_statistics=False)` loads rows on any backend.** It asks
49
+ for rows and cardinality rather than for a database the planner can reason
50
+ about, and it is written as a requirement being dropped rather than as work
51
+ being skipped: on PostgreSQL it changes nothing at all, since `COPY` and
52
+ `ANALYZE` are both free and leaving them out would manufacture the unanalyzed
53
+ table this package exists to condemn. Elsewhere the rows are inserted in chunks
54
+ and no statistics are gathered. SQLite's own `ANALYZE` is deliberately not run,
55
+ because running it would claim the plan realism this package says it will not
56
+ claim. The driver check is unaffected: psycopg 2 is still refused on a
57
+ PostgreSQL connection, because the vendor picks the route and not the caller.
58
+ - **The growth harness works on every backend Django supports** and no longer
59
+ skips. A query count is an ORM property and means the same anywhere, so a
60
+ growth assertion is honest off PostgreSQL where a plan assertion is not, and
61
+ the scale harness was the only thing standing between a consumer on SQLite and
62
+ the milestone's headline seam. `shape_fixture` still skips: it exists to build
63
+ a world a planner will believe.
64
+
65
+ ### Fixed
66
+ - `ScaleProtocol` rejected the implementations its own docstring offered. A
67
+ structural type matches parameter names too, so a hand-rolled callable taking
68
+ `n` rather than `factor` did not satisfy it -- exactly the five-line callable
69
+ the documentation tells a consumer to write. The factor is positional-only now,
70
+ and two files of consumers and impostors are type-checked by the suite, because
71
+ a type-level claim with no type-level test is what let this ship.
72
+ - `ShapeNotEmpty` names the likely cause and not only the remedy. The first
73
+ consumer met it by composing both fixtures over one model, where the rows are
74
+ real, correct and written by a fixture the failing test never mentions -- so
75
+ "empty the table first" read as advice about somebody else's data.
76
+
10
77
  ## [0.3.0] — 2026-09-01
11
78
 
12
79
  ### Fixed
@@ -147,7 +214,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
147
214
  into `COPY FROM STDIN`, which psycopg 2 cannot do without materialising them
148
215
  first.
149
216
 
150
- [Unreleased]: https://github.com/Artui/django-data-shape/compare/v0.3.0...HEAD
217
+ [Unreleased]: https://github.com/Artui/django-data-shape/compare/v0.4.0...HEAD
218
+ [0.4.0]: https://github.com/Artui/django-data-shape/compare/v0.3.0...v0.4.0
151
219
  [0.3.0]: https://github.com/Artui/django-data-shape/compare/v0.2.0...v0.3.0
152
220
  [0.2.0]: https://github.com/Artui/django-data-shape/compare/v0.1.1...v0.2.0
153
221
  [0.1.1]: https://github.com/Artui/django-data-shape/compare/v0.1.0...v0.1.1
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: django-data-shape
3
- Version: 0.3.0
3
+ Version: 0.4.0
4
4
  Summary: A realistically shaped test database from Django models: declare cardinality, skew and fan-out, load by COPY, and make the query planner believe it.
5
5
  Project-URL: Homepage, https://github.com/Artui/django-data-shape
6
6
  Project-URL: Repository, https://github.com/Artui/django-data-shape
@@ -36,6 +36,9 @@ Requires-Python: >=3.10
36
36
  Requires-Dist: django>=4.2
37
37
  Provides-Extra: postgres
38
38
  Requires-Dist: psycopg[binary]>=3.2.0; extra == 'postgres'
39
+ Provides-Extra: pytest
40
+ Requires-Dist: pytest-django>=4.9.0; extra == 'pytest'
41
+ Requires-Dist: pytest>=8.0.0; extra == 'pytest'
39
42
  Description-Content-Type: text/markdown
40
43
 
41
44
  # django-data-shape
@@ -99,7 +102,10 @@ build(shape)
99
102
 
100
103
  `build()` generates the rows, loads them with `COPY`, moves the identity sequence
101
104
  past the keys it assigned, and runs `ANALYZE` so the planner can see the shape.
102
- It raises on any backend that is not PostgreSQL rather than degrading quietly.
105
+ It raises on any backend that is not PostgreSQL rather than degrading quietly --
106
+ unless you say `require_statistics=False`, which asks for rows and cardinality
107
+ instead of a database the planner can reason about, and is what the growth
108
+ harness below is built on.
103
109
 
104
110
  ## Relations
105
111
 
@@ -126,6 +132,37 @@ The parents can be rows this package built or rows your own code did -- their
126
132
  real keys are read, not assumed, so the ORM can own the small tables while this
127
133
  owns the large ones.
128
134
 
135
+ ## From pytest
136
+
137
+ ```python
138
+ # conftest.py
139
+ from django_data_shape import Constant, Shape, Table
140
+ from django_data_shape.fixtures import scale_fixture, shape_fixture
141
+
142
+ orders = shape_fixture(Shape(Table(Order, rows=100_000, status=Constant("complete"))))
143
+ world = scale_fixture(Shape(Table(Order, rows=100, status=Constant("complete"))))
144
+ ```
145
+
146
+ `orders` is one world built once for the whole session, composed with
147
+ pytest-django rather than replacing it. `world` is the **scale protocol**: make
148
+ the world be at factor F, then let the caller run its block, which is what a
149
+ query count asserted to be `O(1)` rather than `O(N)` needs.
150
+
151
+ ```python
152
+ def test_the_dashboard_does_not_grow(world, django_assert_num_queries):
153
+ for factor in (1, 10):
154
+ with world(factor):
155
+ with django_assert_num_queries(3):
156
+ dashboard()
157
+ ```
158
+
159
+ A factor varies the declaration rather than subsetting one larger build, and
160
+ `pip install 'django-data-shape[pytest]'` is what these two need. **The growth
161
+ harness works on any backend Django supports**, because a query count is an ORM
162
+ property and means the same everywhere; the session world **skips with a stated
163
+ reason** where a shaped database cannot exist, because a plan over it is the
164
+ thing it exists to make honest.
165
+
129
166
  ## What it expects, and what it refuses
130
167
 
131
168
  A declaration that cannot describe a database raises before a row is generated,
@@ -133,7 +170,10 @@ naming the field. In particular:
133
170
 
134
171
  - **PostgreSQL and psycopg 3.** Rows stream into `COPY FROM STDIN`, which
135
172
  psycopg 2 cannot do without materialising them first. Both are refused by name
136
- rather than degraded around.
173
+ rather than degraded around. PostgreSQL is required for the statistics half
174
+ only: `build(shape, require_statistics=False)` loads rows on any backend and
175
+ claims nothing about a plan. psycopg 2 is refused either way, because the
176
+ vendor picks the route and not the caller.
137
177
  - **A key type it can assign.** Integer keys count from one and UUID keys are
138
178
  derived from the seed; anything else is refused rather than guessed, and
139
179
  `keys=KeyFunction(...)` declares one.
@@ -145,8 +185,9 @@ naming the field. In particular:
145
185
 
146
186
  ## Status
147
187
 
148
- Early. Single tables and the model graph. Derived fields, collections copied
149
- along a join, per-group invariants and template-database reuse come next.
188
+ Early. Single tables, the model graph and the pytest surface. Derived fields,
189
+ collections copied along a join, per-group invariants and template-database reuse
190
+ come next.
150
191
 
151
192
  Full documentation: <https://artui.github.io/django-data-shape/>
152
193
 
@@ -59,7 +59,10 @@ build(shape)
59
59
 
60
60
  `build()` generates the rows, loads them with `COPY`, moves the identity sequence
61
61
  past the keys it assigned, and runs `ANALYZE` so the planner can see the shape.
62
- It raises on any backend that is not PostgreSQL rather than degrading quietly.
62
+ It raises on any backend that is not PostgreSQL rather than degrading quietly --
63
+ unless you say `require_statistics=False`, which asks for rows and cardinality
64
+ instead of a database the planner can reason about, and is what the growth
65
+ harness below is built on.
63
66
 
64
67
  ## Relations
65
68
 
@@ -86,6 +89,37 @@ The parents can be rows this package built or rows your own code did -- their
86
89
  real keys are read, not assumed, so the ORM can own the small tables while this
87
90
  owns the large ones.
88
91
 
92
+ ## From pytest
93
+
94
+ ```python
95
+ # conftest.py
96
+ from django_data_shape import Constant, Shape, Table
97
+ from django_data_shape.fixtures import scale_fixture, shape_fixture
98
+
99
+ orders = shape_fixture(Shape(Table(Order, rows=100_000, status=Constant("complete"))))
100
+ world = scale_fixture(Shape(Table(Order, rows=100, status=Constant("complete"))))
101
+ ```
102
+
103
+ `orders` is one world built once for the whole session, composed with
104
+ pytest-django rather than replacing it. `world` is the **scale protocol**: make
105
+ the world be at factor F, then let the caller run its block, which is what a
106
+ query count asserted to be `O(1)` rather than `O(N)` needs.
107
+
108
+ ```python
109
+ def test_the_dashboard_does_not_grow(world, django_assert_num_queries):
110
+ for factor in (1, 10):
111
+ with world(factor):
112
+ with django_assert_num_queries(3):
113
+ dashboard()
114
+ ```
115
+
116
+ A factor varies the declaration rather than subsetting one larger build, and
117
+ `pip install 'django-data-shape[pytest]'` is what these two need. **The growth
118
+ harness works on any backend Django supports**, because a query count is an ORM
119
+ property and means the same everywhere; the session world **skips with a stated
120
+ reason** where a shaped database cannot exist, because a plan over it is the
121
+ thing it exists to make honest.
122
+
89
123
  ## What it expects, and what it refuses
90
124
 
91
125
  A declaration that cannot describe a database raises before a row is generated,
@@ -93,7 +127,10 @@ naming the field. In particular:
93
127
 
94
128
  - **PostgreSQL and psycopg 3.** Rows stream into `COPY FROM STDIN`, which
95
129
  psycopg 2 cannot do without materialising them first. Both are refused by name
96
- rather than degraded around.
130
+ rather than degraded around. PostgreSQL is required for the statistics half
131
+ only: `build(shape, require_statistics=False)` loads rows on any backend and
132
+ claims nothing about a plan. psycopg 2 is refused either way, because the
133
+ vendor picks the route and not the caller.
97
134
  - **A key type it can assign.** Integer keys count from one and UUID keys are
98
135
  derived from the seed; anything else is refused rather than guessed, and
99
136
  `keys=KeyFunction(...)` declares one.
@@ -105,8 +142,9 @@ naming the field. In particular:
105
142
 
106
143
  ## Status
107
144
 
108
- Early. Single tables and the model graph. Derived fields, collections copied
109
- along a join, per-group invariants and template-database reuse come next.
145
+ Early. Single tables, the model graph and the pytest surface. Derived fields,
146
+ collections copied along a join, per-group invariants and template-database reuse
147
+ come next.
110
148
 
111
149
  Full documentation: <https://artui.github.io/django-data-shape/>
112
150
 
@@ -15,6 +15,9 @@ from django_data_shape.keys.key_function import KeyFunction
15
15
  from django_data_shape.keys.key_strategy import KeyStrategy
16
16
  from django_data_shape.keys.sequential_keys import SequentialKeys
17
17
  from django_data_shape.keys.uuid_keys import UuidKeys
18
+ from django_data_shape.scale_protocol import ScaleProtocol
19
+ from django_data_shape.scaled_shape import scaled_shape
20
+ from django_data_shape.scaled_world import scaled_world
18
21
  from django_data_shape.shape import Shape
19
22
  from django_data_shape.shape_not_empty import ShapeNotEmpty
20
23
  from django_data_shape.table import Table
@@ -31,6 +34,7 @@ __all__ = [
31
34
  "InvalidShape",
32
35
  "KeyFunction",
33
36
  "KeyStrategy",
37
+ "ScaleProtocol",
34
38
  "SequentialKeys",
35
39
  "UuidKeys",
36
40
  "Sequential",
@@ -44,4 +48,6 @@ __all__ = [
44
48
  "UnsupportedBackend",
45
49
  "__version__",
46
50
  "build",
51
+ "scaled_shape",
52
+ "scaled_world",
47
53
  ]
@@ -2,6 +2,8 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ from collections.abc import Iterator
6
+ from itertools import islice
5
7
  from typing import Any, cast
6
8
 
7
9
  from django.core.management.color import no_style
@@ -20,8 +22,14 @@ from django_data_shape.shape_not_empty import ShapeNotEmpty
20
22
  from django_data_shape.table import Table
21
23
  from django_data_shape.table_result import TableResult
22
24
 
25
+ # Big enough that the per-statement overhead disappears and small enough that
26
+ # the peak list is a rounding error next to the rows themselves.
27
+ _INSERT_CHUNK = 1000
23
28
 
24
- def build(shape: Shape, using: str = DEFAULT_DB_ALIAS) -> BuildResult:
29
+
30
+ def build(
31
+ shape: Shape, using: str = DEFAULT_DB_ALIAS, *, require_statistics: bool = True
32
+ ) -> BuildResult:
25
33
  """Generate, load, reset sequences and analyze every table in ``shape``.
26
34
 
27
35
  The order of those steps is the whole function, and it is not
@@ -36,9 +44,24 @@ def build(shape: Shape, using: str = DEFAULT_DB_ALIAS) -> BuildResult:
36
44
  statistics work proper -- per-column targets, and caching a built database
37
45
  as a template -- comes later. A loader that leaves its table unanalyzed
38
46
  ships the exact state this package exists to condemn.
47
+
48
+ ``require_statistics=False`` asks for rows and cardinality rather than for a
49
+ database the planner can reason about, and it is the only way to build on a
50
+ backend without ``COPY`` and column statistics. It is written as a
51
+ requirement being dropped rather than as work being skipped, because that is
52
+ what it does: **on PostgreSQL it changes nothing at all** -- the load is
53
+ still ``COPY`` and ``ANALYZE`` still runs, since both are free and leaving
54
+ them out would manufacture the unanalyzed table this package exists to
55
+ condemn. Elsewhere the rows are inserted instead and no statistics are
56
+ gathered, so cardinality is real and nothing about a plan is claimed.
57
+
58
+ The distinction is the one this package draws everywhere: generation and
59
+ cardinality are backend-neutral, planner realism is not. A query *count* is
60
+ an ORM property and means the same on any backend, which is why a growth
61
+ assertion can be honest here while a plan assertion still cannot.
39
62
  """
40
63
  connection = connections[using]
41
- require_postgres(connection, "Building a shape")
64
+ require_postgres(connection, "Building a shape", statistics=require_statistics)
42
65
 
43
66
  results: list[TableResult] = []
44
67
  # One transaction around every table. Without it a shape whose second table
@@ -82,6 +105,13 @@ def _require_empty(connection: Any, table: Table) -> None:
82
105
  the same table collides on the primary key. That surfaced as a bare
83
106
  UniqueViolation naming an index, which tells the reader nothing about what
84
107
  they did or what to do instead.
108
+
109
+ The message names the likely cause and not only the remedy, because the
110
+ first consumer met this from a direction the remedy does not fit: a
111
+ session-scoped ``shape_fixture`` and a scaled world pointed at one model.
112
+ The rows are then real, correct, and put there by a fixture the failing test
113
+ never mentions -- so "empty the table first" reads as advice about somebody
114
+ else's data.
85
115
  """
86
116
  with connection.cursor() as cursor:
87
117
  cursor.execute(f"SELECT EXISTS (SELECT 1 FROM {connection.ops.quote_name(table.db_table)})")
@@ -89,7 +119,10 @@ def _require_empty(connection: Any, table: Table) -> None:
89
119
  if row[0]:
90
120
  raise ShapeNotEmpty(
91
121
  f"{table.db_table} already holds rows, and this package assigns primary keys from 1, "
92
- "so building over them would collide. Empty the table first."
122
+ "so building over them would collide. If nothing in the test wrote them, the usual "
123
+ "cause is a world that was already there: a session-scoped shape_fixture over this "
124
+ "model holds its rows for the whole run, and a scaled world cannot build over them. "
125
+ "Give the two different models, or empty this table first."
93
126
  )
94
127
 
95
128
 
@@ -98,12 +131,7 @@ def _require_empty(connection: Any, table: Table) -> None:
98
131
  # class, so annotating the real wrapper type would mean asserting the checker
99
132
  # out of the way on every line that uses it.
100
133
  def _load(connection: Any, table: Table, seed: int, plans: dict[str, FanOutPlan]) -> int:
101
- """Stream generated rows into the table with ``COPY FROM STDIN``.
102
-
103
- ``bulk_create`` is the obvious alternative and is roughly an order of
104
- magnitude too slow at the row counts that make a plan meaningful. ``COPY``
105
- is also why the generator yields tuples rather than model instances: there
106
- is no instance to build, and no ``save`` to run.
134
+ """Stream generated rows into the table, by the fastest route the backend has.
107
135
 
108
136
  Each declared value is passed through its field's ``get_db_prep_save``
109
137
  first, and that is not a formality. Skipping it is how a naive datetime got
@@ -111,7 +139,9 @@ def _load(connection: Any, table: Table, seed: int, plans: dict[str, FanOutPlan]
111
139
  non-UTC ``TIME_ZONE`` -- silently, on the exact column ``Sequential`` exists
112
140
  to make realistic -- and how a ``JSONField`` failed to load at all. The
113
141
  generator stays backend-neutral because the preparation happens here rather
114
- than inside it.
142
+ than inside it, and that is also why the preparation sits **above** the
143
+ branch below: a value is prepared by its field for its connection, which is
144
+ the same work whichever statement carries it.
115
145
 
116
146
  Returns the number of rows the database actually took, not the number
117
147
  declared. They are the same today and stop being so once deduplicated
@@ -120,12 +150,31 @@ def _load(connection: Any, table: Table, seed: int, plans: dict[str, FanOutPlan]
120
150
  quote = connection.ops.quote_name
121
151
  pk_field = table.model._meta.pk
122
152
  columns = [quote(pk_field.column)] + [quote(field.column) for _, field in table.columns()]
123
- statement = f"COPY {quote(table.db_table)} ({', '.join(columns)}) FROM STDIN"
124
153
  # The primary key is prepared like every other value. It did not need to be
125
154
  # while keys were always integers; a UUID key does, and a strategy the
126
155
  # caller wrote could return anything its column accepts.
127
156
  prepare = [pk_field.get_db_prep_save] + [field.get_db_prep_save for _, field in table.columns()]
157
+ rows = (
158
+ tuple(prep(value, connection) for prep, value in zip(prepare, row, strict=True))
159
+ for row in generate_rows(table, seed, plans)
160
+ )
161
+
162
+ if connection.vendor == "postgresql":
163
+ return _copy(connection, table.db_table, columns, rows)
164
+ return _insert(connection, table.db_table, columns, rows)
128
165
 
166
+
167
+ def _copy(
168
+ connection: Any, db_table: str, columns: list[str], rows: Iterator[tuple[Any, ...]]
169
+ ) -> int:
170
+ """``COPY FROM STDIN``, which is the reason this package can be worth using.
171
+
172
+ ``bulk_create`` is the obvious alternative and is roughly an order of
173
+ magnitude too slow at the row counts that make a plan meaningful. ``COPY``
174
+ is also why the generator yields tuples rather than model instances: there
175
+ is no instance to build, and no ``save`` to run.
176
+ """
177
+ statement = f"COPY {connection.ops.quote_name(db_table)} ({', '.join(columns)}) FROM STDIN"
129
178
  with connection.cursor() as cursor:
130
179
  # ``copy`` is not in Django's WRAP_ERROR_ATTRS, so without this a
131
180
  # Postgres error escapes as a raw psycopg exception: the caller cannot
@@ -133,13 +182,41 @@ def _load(connection: Any, table: Table, seed: int, plans: dict[str, FanOutPlan]
133
182
  # block never learns it needs a rollback, so the next query inside it
134
183
  # fails with "current transaction is aborted" instead of a Django error.
135
184
  with connection.wrap_database_errors, cursor.copy(statement) as copy:
136
- for row in generate_rows(table, seed, plans):
137
- copy.write_row(
138
- tuple(prep(value, connection) for prep, value in zip(prepare, row, strict=True))
139
- )
185
+ for row in rows:
186
+ copy.write_row(row)
140
187
  return int(cursor.rowcount)
141
188
 
142
189
 
190
+ def _insert(
191
+ connection: Any, db_table: str, columns: list[str], rows: Iterator[tuple[Any, ...]]
192
+ ) -> int:
193
+ """The portable route, for a backend that has no ``COPY``.
194
+
195
+ Reached only when the caller said it does not require planner statistics,
196
+ which is the honest shape of the trade: this writes real rows in real
197
+ cardinality and buys nothing at all for a plan. It is slower than ``COPY``
198
+ and that is the wrong thing to worry about at the sizes it is for --
199
+ measured on SQLite, the insert costs about 1.6 ms per thousand rows against
200
+ 8 ms to generate them, so the load is not what a growth assertion pays for.
201
+
202
+ Chunked rather than handed the whole iterator, because ``executemany``
203
+ materialises what it is given: streaming into ``COPY`` is the property this
204
+ package is built on, and a portable path that quietly held a million tuples
205
+ in memory would be a different bargain from the one above.
206
+ """
207
+ placeholders = ", ".join(["%s"] * len(columns))
208
+ statement = (
209
+ f"INSERT INTO {connection.ops.quote_name(db_table)} "
210
+ f"({', '.join(columns)}) VALUES ({placeholders})"
211
+ )
212
+ loaded = 0
213
+ with connection.cursor() as cursor:
214
+ while chunk := list(islice(rows, _INSERT_CHUNK)):
215
+ cursor.executemany(statement, chunk)
216
+ loaded += len(chunk)
217
+ return loaded
218
+
219
+
143
220
  def _reset_sequence(connection: Any, table: Table) -> None:
144
221
  """Move the identity sequence past the keys this package just assigned.
145
222
 
@@ -161,6 +238,15 @@ def _analyze(connection: Any, table: Table) -> None:
161
238
  selectivity and commits to it, which is how a two-million-row table gets
162
239
  bitmap-scanned through an index for a value matching 98% of it. Measured at
163
240
  81 ms on that table, because ``ANALYZE`` samples rather than scans.
241
+
242
+ Nothing is gathered on another backend, and SQLite is the case worth being
243
+ explicit about: it has an ``ANALYZE`` of its own and running it is one line.
244
+ It is deliberately not run. Plan realism on SQLite is out of this package's
245
+ scope -- support the generation, refuse the pretence -- and a table with
246
+ ``sqlite_stat1`` behind it would be this package claiming, in the only way a
247
+ library can, that the plan over it means something.
164
248
  """
249
+ if connection.vendor != "postgresql":
250
+ return
165
251
  with connection.cursor() as cursor:
166
252
  cursor.execute(f"ANALYZE {connection.ops.quote_name(table.db_table)}")
@@ -0,0 +1,19 @@
1
+ """The pytest surface, deliberately outside the top-level re-exports.
2
+
3
+ Everything in here imports ``pytest``, and ``pytest`` is an optional extra
4
+ rather than a dependency. Re-exporting these from ``django_data_shape/__init__``
5
+ would make ``import django_data_shape`` fail for a project that runs its suite
6
+ with Django's own runner, so the import boundary is drawn where the dependency
7
+ boundary already is: ``from django_data_shape.fixtures import shape_fixture``
8
+ says out loud that pytest is required for this half and not for the other.
9
+ """
10
+
11
+ from django_data_shape.fixtures.scale_fixture import scale_fixture
12
+ from django_data_shape.fixtures.shape_fixture import shape_fixture
13
+ from django_data_shape.fixtures.skip_unless_postgres import skip_unless_postgres
14
+
15
+ __all__ = [
16
+ "scale_fixture",
17
+ "shape_fixture",
18
+ "skip_unless_postgres",
19
+ ]
@@ -0,0 +1,76 @@
1
+ """One shape, offered at whatever size the test asks for."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from functools import partial
6
+
7
+ import pytest
8
+ from django.db import DEFAULT_DB_ALIAS
9
+
10
+ from django_data_shape.scale_protocol import ScaleProtocol
11
+ from django_data_shape.scaled_world import scaled_world
12
+ from django_data_shape.shape import Shape
13
+
14
+
15
+ # ``object`` for the same reason as in shape_fixture: the type of a fixture is
16
+ # pytest's own and it changed shape between pytest 8.0 and 8.4.
17
+ def scale_fixture(shape: Shape, *, using: str = DEFAULT_DB_ALIAS) -> object:
18
+ """A pytest fixture yielding a :class:`~django_data_shape.scale_protocol.ScaleProtocol`.
19
+
20
+ The pytest face of the scale protocol. Bind it in ``conftest.py``::
21
+
22
+ from django_data_shape import Constant, Shape, Table
23
+ from django_data_shape.fixtures import scale_fixture
24
+
25
+ world = scale_fixture(Shape(Table(Order, rows=100, status=Constant("new"))))
26
+
27
+ and a growth assertion has somewhere to ask for a bigger world::
28
+
29
+ def test_the_dashboard_query_is_constant(world, django_assert_num_queries):
30
+ for factor in (1, 10):
31
+ with world(factor):
32
+ with django_assert_num_queries(3):
33
+ dashboard()
34
+
35
+ The declared row counts are the world at factor 1, so the base declaration
36
+ should be the smallest world that still means something -- a hundred rows
37
+ against a thousand is the regime this is for, and it is milliseconds per
38
+ factor. Size, in the two-million-row sense that makes a query *plan*
39
+ realistic, is a different assertion with a different cost and does not vary
40
+ a factor at all.
41
+
42
+ Function-scoped, and it requests pytest-django's ``db`` fixture, which does
43
+ two things worth knowing. A test using this needs no ``django_db`` marker of
44
+ its own. And each world is then built inside the transaction that wraps the
45
+ test, so tearing it down is a savepoint rollback: cheap, exact, and leaving
46
+ the test's own transaction usable afterwards. A test that marks itself
47
+ ``transaction=True`` still works -- the marker wins over the fixture, and
48
+ the rollback is then an ordinary one.
49
+
50
+ **Not over a model a session world already holds.** Each world here is built
51
+ from empty and undone again, so a table that
52
+ :func:`~django_data_shape.fixtures.shape_fixture.shape_fixture` filled for
53
+ the session is one this cannot build into at all -- the rows are still there,
54
+ and the build is refused. The two compose over a graph by taking different
55
+ models, not by taking turns over one.
56
+
57
+ **It works on any backend Django supports**, because what a growth
58
+ assertion measures -- the number of queries a block emits -- is an ORM
59
+ property rather than a planner one. Where the backend has ``COPY`` and
60
+ column statistics the world is built with them; where it does not, the rows
61
+ are inserted and no statistics are gathered, so the cardinality is real and
62
+ nothing about a plan is claimed. That is the one place this package builds
63
+ outside PostgreSQL, and it is allowed precisely because the assertion it
64
+ serves does not need the planner. A plan assertion still skips: see
65
+ :func:`~django_data_shape.fixtures.skip_unless_postgres.skip_unless_postgres`.
66
+ """
67
+
68
+ @pytest.fixture
69
+ def scaled_worlds(db: None) -> ScaleProtocol:
70
+ # Binding, not wrapping. The fixture's whole job is to attach one shape
71
+ # and one connection to the protocol; everything about what a world is
72
+ # and how it is undone belongs to scaled_world, where a consumer not
73
+ # using pytest can reach it too.
74
+ return partial(scaled_world, shape, using=using)
75
+
76
+ return scaled_worlds
@@ -0,0 +1,94 @@
1
+ """One shape, built once for a whole test session."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ import pytest
8
+ from django.db import DEFAULT_DB_ALIAS, connections
9
+
10
+ from django_data_shape.build import build
11
+ from django_data_shape.build_result import BuildResult
12
+ from django_data_shape.fixtures.skip_unless_postgres import skip_unless_postgres
13
+ from django_data_shape.shape import Shape
14
+
15
+
16
+ # The return type is ``object`` on purpose. What ``pytest.fixture`` hands back
17
+ # is pytest's own -- a plain function on pytest 8.0 and a
18
+ # ``FixtureFunctionDefinition`` from 8.4 onwards -- so naming it here would pin
19
+ # this package's floor to a pytest version in exchange for nothing: the caller
20
+ # binds the value to a name and never calls it.
21
+ def shape_fixture(shape: Shape, *, using: str = DEFAULT_DB_ALIAS) -> object:
22
+ """A session-scoped pytest fixture that builds ``shape`` once.
23
+
24
+ Bind it to a name in ``conftest.py`` and request that name from a test::
25
+
26
+ from django_data_shape import Constant, Shape, Table
27
+ from django_data_shape.fixtures import shape_fixture
28
+
29
+ orders = shape_fixture(Shape(Table(Order, rows=100_000, status=Constant("new"))))
30
+
31
+ ::
32
+
33
+ import pytest
34
+
35
+ @pytest.mark.django_db
36
+ def test_the_dashboard_query(orders):
37
+ assert orders.rows == 100_000
38
+
39
+ It **composes with pytest-django rather than replacing it**. The fixture
40
+ requests ``django_db_setup``, which is the seam a project overrides to
41
+ decide how its test database is made, so whatever a project has done there
42
+ -- a template database, ``--reuse-db``, a different creation strategy -- has
43
+ already happened before a row is generated. It then writes through
44
+ ``django_db_blocker.unblock()``, the mechanism pytest-django documents for
45
+ populating a database once. Neither of those is imported: they are asked for
46
+ by name, so this package depends on two fixture names and not on
47
+ pytest-django's internals.
48
+
49
+ **Session scope is load-bearing, not a performance choice.** pytest creates
50
+ higher-scoped fixtures before lower-scoped ones, so a session-scoped build
51
+ always runs before the function-scoped ``db`` fixture opens the transaction
52
+ that wraps a test -- which is what makes the rows committed and visible to
53
+ every later test. A function-scoped build would be ordered against ``db`` by
54
+ the accident of argument order, and on the losing side of that order it
55
+ would be rolled back with the test that happened to build it.
56
+
57
+ Yields the :class:`~django_data_shape.build_result.BuildResult`, so a test
58
+ can assert on the size of the world it was handed.
59
+
60
+ **One caveat, and it is worth stating plainly**: a test marked
61
+ ``django_db(transaction=True)`` truncates every table at teardown, and takes
62
+ the session's rows with it. Nothing rebuilds them, so a later test that
63
+ reads this fixture is measuring an empty database. Keep transactional tests
64
+ off the tables a shape owns, mark them ``serialized_rollback=True``, or
65
+ build per test with
66
+ :func:`~django_data_shape.scaled_world.scaled_world` at factor 1 -- which
67
+ undoes itself and therefore does not care.
68
+
69
+ **One world per table.** A session world holds its rows for the whole run,
70
+ so :func:`~django_data_shape.fixtures.scale_fixture.scale_fixture` over the
71
+ same model cannot build: the second build meets a table that is not empty
72
+ and is refused. Give the two different models -- the session world the tables
73
+ a plan assertion needs to be big, the scale harness the tables a growth
74
+ assertion counts. It is the first thing a consumer composing both hits, and
75
+ the refusal now names it.
76
+
77
+ On a connection that cannot carry a shaped database the fixture skips with
78
+ the reason rather than raising, so a suite that also runs on SQLite reports
79
+ what it did not check instead of erroring or, worse, passing.
80
+ """
81
+
82
+ # The blocker is typed loosely because naming its class would mean importing
83
+ # pytest_django, and the whole point of asking for it by name is that this
84
+ # package does not. It is a pytest-django fixture; what is used of it is
85
+ # ``unblock()``.
86
+ @pytest.fixture(scope="session")
87
+ def built_shape(django_db_setup: None, django_db_blocker: Any) -> BuildResult:
88
+ # ``vendor`` is a class attribute, so the gate is read before anything
89
+ # is unblocked and a skip costs no connection at all.
90
+ skip_unless_postgres(connections[using], "Building a shape for the test session")
91
+ with django_db_blocker.unblock():
92
+ return build(shape, using=using)
93
+
94
+ return built_shape