django-data-shape 0.3.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/CHANGELOG.md +69 -1
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/PKG-INFO +46 -5
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/README.md +42 -4
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/__init__.py +6 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/build.py +101 -15
- django_data_shape-0.4.0/django_data_shape/fixtures/__init__.py +19 -0
- django_data_shape-0.4.0/django_data_shape/fixtures/scale_fixture.py +76 -0
- django_data_shape-0.4.0/django_data_shape/fixtures/shape_fixture.py +94 -0
- django_data_shape-0.4.0/django_data_shape/fixtures/skip_unless_postgres.py +36 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/require_postgres.py +9 -2
- django_data_shape-0.4.0/django_data_shape/scale_protocol.py +48 -0
- django_data_shape-0.4.0/django_data_shape/scaled_shape.py +102 -0
- django_data_shape-0.4.0/django_data_shape/scaled_world.py +77 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/version.py +1 -1
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/docs/index.md +18 -2
- django_data_shape-0.4.0/docs/pytest.md +332 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/docs/reference.md +14 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/docs/relations.md +6 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/mkdocs.yml +1 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/pyproject.toml +15 -1
- django_data_shape-0.4.0/tests/scale_protocol_consumers.py +54 -0
- django_data_shape-0.4.0/tests/scale_protocol_impostors.py +32 -0
- django_data_shape-0.4.0/tests/test_build_portable.py +125 -0
- django_data_shape-0.4.0/tests/test_fixtures.py +139 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_require_postgres.py +21 -0
- django_data_shape-0.4.0/tests/test_scale_protocol.py +46 -0
- django_data_shape-0.4.0/tests/test_scaled_shape.py +139 -0
- django_data_shape-0.4.0/tests/test_scaled_world.py +132 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/testapp/models.py +13 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/uv.lock +7 -1
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/CODEOWNERS +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/SECURITY.md +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/dependabot.yml +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/workflows/release.yml +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/workflows/tests.yml +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.github/workflows/upstream-drift.yml +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.gitignore +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/.pre-commit-config.yaml +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/CLAUDE.md +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/LICENSE +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/Makefile +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/build_result.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/__init__.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/bounded.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/constant.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/distribution.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/sequential.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/skew.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/uniform.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/distributions/zipf.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/fan_out.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/fan_out_plan.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/generate_rows.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/infer_key_strategy.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/invalid_shape.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/keys/__init__.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/keys/key_function.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/keys/key_strategy.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/keys/sequential_keys.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/keys/uuid_keys.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/order_tables.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/py.typed +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/resolve_fan_out.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/shape.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/shape_not_empty.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/table.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/table_result.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/unsupported_backend.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/django_data_shape/utils.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/docs/keys.md +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/scripts/release-publish.sh +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/__init__.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/conftest.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/conftest_settings.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/__init__.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/test_constant.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/test_sequential.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/test_skew.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/test_uniform.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/distributions/test_zipf.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/keys/__init__.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/keys/test_key_function.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/keys/test_sequential_keys.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/keys/test_uuid_keys.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_build.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_build_graph.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_build_keys.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_documentation.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_fan_out.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_generate_rows.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_infer_key_strategy.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_order_tables.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_resolve_fan_out.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_shape.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_table.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_utils.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/test_version.py +0 -0
- {django_data_shape-0.3.0 → django_data_shape-0.4.0}/tests/testapp/__init__.py +0 -0
|
@@ -7,6 +7,73 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.4.0] — 2026-09-02
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
- **The pytest surface**, in `django_data_shape.fixtures`. `shape_fixture(shape)`
|
|
14
|
+
returns a session-scoped fixture that builds a shape once for a whole run;
|
|
15
|
+
bind it to a name in `conftest.py` and request that name from a test. It
|
|
16
|
+
composes with pytest-django rather than replacing it, asking for
|
|
17
|
+
`django_db_setup` and `django_db_blocker` by name so the coupling is to two
|
|
18
|
+
fixture names and not to pytest-django's internals. Session scope is
|
|
19
|
+
load-bearing: pytest creates higher-scoped fixtures first, so the rows are
|
|
20
|
+
committed before the transaction that wraps a test is opened, and every test
|
|
21
|
+
sees them while everything a test writes is rolled back with it.
|
|
22
|
+
- **The scale protocol**, which is what a growth assertion asks a world for:
|
|
23
|
+
make the world be at factor F, then let the caller run its block.
|
|
24
|
+
`scaled_world(shape, factor)` is a context manager that builds it and undoes
|
|
25
|
+
it; `scale_fixture(shape)` is the same thing as a fixture; and `ScaleProtocol`
|
|
26
|
+
is the structural type both satisfy, so a consumer asserting that a query
|
|
27
|
+
count is `O(1)` rather than `O(N)` depends on the shape of the call rather
|
|
28
|
+
than on this package. What the context manager yields is a plain row count,
|
|
29
|
+
not a `BuildResult`, for the same reason: a seam a stranger cannot implement
|
|
30
|
+
is not a seam.
|
|
31
|
+
- `scaled_shape(shape, factor)`, the declaration transform underneath it. **A
|
|
32
|
+
factor varies the declaration rather than subsetting one larger build**, because
|
|
33
|
+
a subset is not a smaller database but the same database with a filter -- the
|
|
34
|
+
statistics still describe every row, and the block under test would have to
|
|
35
|
+
cooperate by restricting itself, which puts the harness inside the thing being
|
|
36
|
+
measured. Every table scales, parents included, so the average fan-out is the
|
|
37
|
+
same at every factor and two worlds differ in size alone. The scaled tables go
|
|
38
|
+
through `Table`'s own constructor, so a declaration that only holds at its
|
|
39
|
+
original size is refused at the factor that breaks it, naming the factor.
|
|
40
|
+
- `skip_unless_postgres(connection, operation)`, the pytest twin of the backend
|
|
41
|
+
refusal. Both fixtures skip with the refusal's own message as the reason where
|
|
42
|
+
a shaped database cannot exist, so a suite that also runs on SQLite reports
|
|
43
|
+
what it did not check rather than passing over a database nobody shaped.
|
|
44
|
+
- A `pytest` extra. It is an extra rather than a dependency because the rest of
|
|
45
|
+
the package has nothing to do with pytest, which is also why these fixtures are
|
|
46
|
+
not re-exported from the top-level `__init__`: importing them is what requires
|
|
47
|
+
pytest, not importing the package.
|
|
48
|
+
- **`build(shape, require_statistics=False)` loads rows on any backend.** It asks
|
|
49
|
+
for rows and cardinality rather than for a database the planner can reason
|
|
50
|
+
about, and it is written as a requirement being dropped rather than as work
|
|
51
|
+
being skipped: on PostgreSQL it changes nothing at all, since `COPY` and
|
|
52
|
+
`ANALYZE` are both free and leaving them out would manufacture the unanalyzed
|
|
53
|
+
table this package exists to condemn. Elsewhere the rows are inserted in chunks
|
|
54
|
+
and no statistics are gathered. SQLite's own `ANALYZE` is deliberately not run,
|
|
55
|
+
because running it would claim the plan realism this package says it will not
|
|
56
|
+
claim. The driver check is unaffected: psycopg 2 is still refused on a
|
|
57
|
+
PostgreSQL connection, because the vendor picks the route and not the caller.
|
|
58
|
+
- **The growth harness works on every backend Django supports** and no longer
|
|
59
|
+
skips. A query count is an ORM property and means the same anywhere, so a
|
|
60
|
+
growth assertion is honest off PostgreSQL where a plan assertion is not, and
|
|
61
|
+
the scale harness was the only thing standing between a consumer on SQLite and
|
|
62
|
+
the milestone's headline seam. `shape_fixture` still skips: it exists to build
|
|
63
|
+
a world a planner will believe.
|
|
64
|
+
|
|
65
|
+
### Fixed
|
|
66
|
+
- `ScaleProtocol` rejected the implementations its own docstring offered. A
|
|
67
|
+
structural type matches parameter names too, so a hand-rolled callable taking
|
|
68
|
+
`n` rather than `factor` did not satisfy it -- exactly the five-line callable
|
|
69
|
+
the documentation tells a consumer to write. The factor is positional-only now,
|
|
70
|
+
and two files of consumers and impostors are type-checked by the suite, because
|
|
71
|
+
a type-level claim with no type-level test is what let this ship.
|
|
72
|
+
- `ShapeNotEmpty` names the likely cause and not only the remedy. The first
|
|
73
|
+
consumer met it by composing both fixtures over one model, where the rows are
|
|
74
|
+
real, correct and written by a fixture the failing test never mentions -- so
|
|
75
|
+
"empty the table first" read as advice about somebody else's data.
|
|
76
|
+
|
|
10
77
|
## [0.3.0] — 2026-09-01
|
|
11
78
|
|
|
12
79
|
### Fixed
|
|
@@ -147,7 +214,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
147
214
|
into `COPY FROM STDIN`, which psycopg 2 cannot do without materialising them
|
|
148
215
|
first.
|
|
149
216
|
|
|
150
|
-
[Unreleased]: https://github.com/Artui/django-data-shape/compare/v0.
|
|
217
|
+
[Unreleased]: https://github.com/Artui/django-data-shape/compare/v0.4.0...HEAD
|
|
218
|
+
[0.4.0]: https://github.com/Artui/django-data-shape/compare/v0.3.0...v0.4.0
|
|
151
219
|
[0.3.0]: https://github.com/Artui/django-data-shape/compare/v0.2.0...v0.3.0
|
|
152
220
|
[0.2.0]: https://github.com/Artui/django-data-shape/compare/v0.1.1...v0.2.0
|
|
153
221
|
[0.1.1]: https://github.com/Artui/django-data-shape/compare/v0.1.0...v0.1.1
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: django-data-shape
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: A realistically shaped test database from Django models: declare cardinality, skew and fan-out, load by COPY, and make the query planner believe it.
|
|
5
5
|
Project-URL: Homepage, https://github.com/Artui/django-data-shape
|
|
6
6
|
Project-URL: Repository, https://github.com/Artui/django-data-shape
|
|
@@ -36,6 +36,9 @@ Requires-Python: >=3.10
|
|
|
36
36
|
Requires-Dist: django>=4.2
|
|
37
37
|
Provides-Extra: postgres
|
|
38
38
|
Requires-Dist: psycopg[binary]>=3.2.0; extra == 'postgres'
|
|
39
|
+
Provides-Extra: pytest
|
|
40
|
+
Requires-Dist: pytest-django>=4.9.0; extra == 'pytest'
|
|
41
|
+
Requires-Dist: pytest>=8.0.0; extra == 'pytest'
|
|
39
42
|
Description-Content-Type: text/markdown
|
|
40
43
|
|
|
41
44
|
# django-data-shape
|
|
@@ -99,7 +102,10 @@ build(shape)
|
|
|
99
102
|
|
|
100
103
|
`build()` generates the rows, loads them with `COPY`, moves the identity sequence
|
|
101
104
|
past the keys it assigned, and runs `ANALYZE` so the planner can see the shape.
|
|
102
|
-
It raises on any backend that is not PostgreSQL rather than degrading quietly
|
|
105
|
+
It raises on any backend that is not PostgreSQL rather than degrading quietly --
|
|
106
|
+
unless you say `require_statistics=False`, which asks for rows and cardinality
|
|
107
|
+
instead of a database the planner can reason about, and is what the growth
|
|
108
|
+
harness below is built on.
|
|
103
109
|
|
|
104
110
|
## Relations
|
|
105
111
|
|
|
@@ -126,6 +132,37 @@ The parents can be rows this package built or rows your own code did -- their
|
|
|
126
132
|
real keys are read, not assumed, so the ORM can own the small tables while this
|
|
127
133
|
owns the large ones.
|
|
128
134
|
|
|
135
|
+
## From pytest
|
|
136
|
+
|
|
137
|
+
```python
|
|
138
|
+
# conftest.py
|
|
139
|
+
from django_data_shape import Constant, Shape, Table
|
|
140
|
+
from django_data_shape.fixtures import scale_fixture, shape_fixture
|
|
141
|
+
|
|
142
|
+
orders = shape_fixture(Shape(Table(Order, rows=100_000, status=Constant("complete"))))
|
|
143
|
+
world = scale_fixture(Shape(Table(Order, rows=100, status=Constant("complete"))))
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
`orders` is one world built once for the whole session, composed with
|
|
147
|
+
pytest-django rather than replacing it. `world` is the **scale protocol**: make
|
|
148
|
+
the world be at factor F, then let the caller run its block, which is what a
|
|
149
|
+
query count asserted to be `O(1)` rather than `O(N)` needs.
|
|
150
|
+
|
|
151
|
+
```python
|
|
152
|
+
def test_the_dashboard_does_not_grow(world, django_assert_num_queries):
|
|
153
|
+
for factor in (1, 10):
|
|
154
|
+
with world(factor):
|
|
155
|
+
with django_assert_num_queries(3):
|
|
156
|
+
dashboard()
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
A factor varies the declaration rather than subsetting one larger build, and
|
|
160
|
+
`pip install 'django-data-shape[pytest]'` is what these two need. **The growth
|
|
161
|
+
harness works on any backend Django supports**, because a query count is an ORM
|
|
162
|
+
property and means the same everywhere; the session world **skips with a stated
|
|
163
|
+
reason** where a shaped database cannot exist, because a plan over it is the
|
|
164
|
+
thing it exists to make honest.
|
|
165
|
+
|
|
129
166
|
## What it expects, and what it refuses
|
|
130
167
|
|
|
131
168
|
A declaration that cannot describe a database raises before a row is generated,
|
|
@@ -133,7 +170,10 @@ naming the field. In particular:
|
|
|
133
170
|
|
|
134
171
|
- **PostgreSQL and psycopg 3.** Rows stream into `COPY FROM STDIN`, which
|
|
135
172
|
psycopg 2 cannot do without materialising them first. Both are refused by name
|
|
136
|
-
rather than degraded around.
|
|
173
|
+
rather than degraded around. PostgreSQL is required for the statistics half
|
|
174
|
+
only: `build(shape, require_statistics=False)` loads rows on any backend and
|
|
175
|
+
claims nothing about a plan. psycopg 2 is refused either way, because the
|
|
176
|
+
vendor picks the route and not the caller.
|
|
137
177
|
- **A key type it can assign.** Integer keys count from one and UUID keys are
|
|
138
178
|
derived from the seed; anything else is refused rather than guessed, and
|
|
139
179
|
`keys=KeyFunction(...)` declares one.
|
|
@@ -145,8 +185,9 @@ naming the field. In particular:
|
|
|
145
185
|
|
|
146
186
|
## Status
|
|
147
187
|
|
|
148
|
-
Early. Single tables and the
|
|
149
|
-
along a join, per-group invariants and template-database reuse
|
|
188
|
+
Early. Single tables, the model graph and the pytest surface. Derived fields,
|
|
189
|
+
collections copied along a join, per-group invariants and template-database reuse
|
|
190
|
+
come next.
|
|
150
191
|
|
|
151
192
|
Full documentation: <https://artui.github.io/django-data-shape/>
|
|
152
193
|
|
|
@@ -59,7 +59,10 @@ build(shape)
|
|
|
59
59
|
|
|
60
60
|
`build()` generates the rows, loads them with `COPY`, moves the identity sequence
|
|
61
61
|
past the keys it assigned, and runs `ANALYZE` so the planner can see the shape.
|
|
62
|
-
It raises on any backend that is not PostgreSQL rather than degrading quietly
|
|
62
|
+
It raises on any backend that is not PostgreSQL rather than degrading quietly --
|
|
63
|
+
unless you say `require_statistics=False`, which asks for rows and cardinality
|
|
64
|
+
instead of a database the planner can reason about, and is what the growth
|
|
65
|
+
harness below is built on.
|
|
63
66
|
|
|
64
67
|
## Relations
|
|
65
68
|
|
|
@@ -86,6 +89,37 @@ The parents can be rows this package built or rows your own code did -- their
|
|
|
86
89
|
real keys are read, not assumed, so the ORM can own the small tables while this
|
|
87
90
|
owns the large ones.
|
|
88
91
|
|
|
92
|
+
## From pytest
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
# conftest.py
|
|
96
|
+
from django_data_shape import Constant, Shape, Table
|
|
97
|
+
from django_data_shape.fixtures import scale_fixture, shape_fixture
|
|
98
|
+
|
|
99
|
+
orders = shape_fixture(Shape(Table(Order, rows=100_000, status=Constant("complete"))))
|
|
100
|
+
world = scale_fixture(Shape(Table(Order, rows=100, status=Constant("complete"))))
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
`orders` is one world built once for the whole session, composed with
|
|
104
|
+
pytest-django rather than replacing it. `world` is the **scale protocol**: make
|
|
105
|
+
the world be at factor F, then let the caller run its block, which is what a
|
|
106
|
+
query count asserted to be `O(1)` rather than `O(N)` needs.
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
def test_the_dashboard_does_not_grow(world, django_assert_num_queries):
|
|
110
|
+
for factor in (1, 10):
|
|
111
|
+
with world(factor):
|
|
112
|
+
with django_assert_num_queries(3):
|
|
113
|
+
dashboard()
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
A factor varies the declaration rather than subsetting one larger build, and
|
|
117
|
+
`pip install 'django-data-shape[pytest]'` is what these two need. **The growth
|
|
118
|
+
harness works on any backend Django supports**, because a query count is an ORM
|
|
119
|
+
property and means the same everywhere; the session world **skips with a stated
|
|
120
|
+
reason** where a shaped database cannot exist, because a plan over it is the
|
|
121
|
+
thing it exists to make honest.
|
|
122
|
+
|
|
89
123
|
## What it expects, and what it refuses
|
|
90
124
|
|
|
91
125
|
A declaration that cannot describe a database raises before a row is generated,
|
|
@@ -93,7 +127,10 @@ naming the field. In particular:
|
|
|
93
127
|
|
|
94
128
|
- **PostgreSQL and psycopg 3.** Rows stream into `COPY FROM STDIN`, which
|
|
95
129
|
psycopg 2 cannot do without materialising them first. Both are refused by name
|
|
96
|
-
rather than degraded around.
|
|
130
|
+
rather than degraded around. PostgreSQL is required for the statistics half
|
|
131
|
+
only: `build(shape, require_statistics=False)` loads rows on any backend and
|
|
132
|
+
claims nothing about a plan. psycopg 2 is refused either way, because the
|
|
133
|
+
vendor picks the route and not the caller.
|
|
97
134
|
- **A key type it can assign.** Integer keys count from one and UUID keys are
|
|
98
135
|
derived from the seed; anything else is refused rather than guessed, and
|
|
99
136
|
`keys=KeyFunction(...)` declares one.
|
|
@@ -105,8 +142,9 @@ naming the field. In particular:
|
|
|
105
142
|
|
|
106
143
|
## Status
|
|
107
144
|
|
|
108
|
-
Early. Single tables and the
|
|
109
|
-
along a join, per-group invariants and template-database reuse
|
|
145
|
+
Early. Single tables, the model graph and the pytest surface. Derived fields,
|
|
146
|
+
collections copied along a join, per-group invariants and template-database reuse
|
|
147
|
+
come next.
|
|
110
148
|
|
|
111
149
|
Full documentation: <https://artui.github.io/django-data-shape/>
|
|
112
150
|
|
|
@@ -15,6 +15,9 @@ from django_data_shape.keys.key_function import KeyFunction
|
|
|
15
15
|
from django_data_shape.keys.key_strategy import KeyStrategy
|
|
16
16
|
from django_data_shape.keys.sequential_keys import SequentialKeys
|
|
17
17
|
from django_data_shape.keys.uuid_keys import UuidKeys
|
|
18
|
+
from django_data_shape.scale_protocol import ScaleProtocol
|
|
19
|
+
from django_data_shape.scaled_shape import scaled_shape
|
|
20
|
+
from django_data_shape.scaled_world import scaled_world
|
|
18
21
|
from django_data_shape.shape import Shape
|
|
19
22
|
from django_data_shape.shape_not_empty import ShapeNotEmpty
|
|
20
23
|
from django_data_shape.table import Table
|
|
@@ -31,6 +34,7 @@ __all__ = [
|
|
|
31
34
|
"InvalidShape",
|
|
32
35
|
"KeyFunction",
|
|
33
36
|
"KeyStrategy",
|
|
37
|
+
"ScaleProtocol",
|
|
34
38
|
"SequentialKeys",
|
|
35
39
|
"UuidKeys",
|
|
36
40
|
"Sequential",
|
|
@@ -44,4 +48,6 @@ __all__ = [
|
|
|
44
48
|
"UnsupportedBackend",
|
|
45
49
|
"__version__",
|
|
46
50
|
"build",
|
|
51
|
+
"scaled_shape",
|
|
52
|
+
"scaled_world",
|
|
47
53
|
]
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
from collections.abc import Iterator
|
|
6
|
+
from itertools import islice
|
|
5
7
|
from typing import Any, cast
|
|
6
8
|
|
|
7
9
|
from django.core.management.color import no_style
|
|
@@ -20,8 +22,14 @@ from django_data_shape.shape_not_empty import ShapeNotEmpty
|
|
|
20
22
|
from django_data_shape.table import Table
|
|
21
23
|
from django_data_shape.table_result import TableResult
|
|
22
24
|
|
|
25
|
+
# Big enough that the per-statement overhead disappears and small enough that
|
|
26
|
+
# the peak list is a rounding error next to the rows themselves.
|
|
27
|
+
_INSERT_CHUNK = 1000
|
|
23
28
|
|
|
24
|
-
|
|
29
|
+
|
|
30
|
+
def build(
|
|
31
|
+
shape: Shape, using: str = DEFAULT_DB_ALIAS, *, require_statistics: bool = True
|
|
32
|
+
) -> BuildResult:
|
|
25
33
|
"""Generate, load, reset sequences and analyze every table in ``shape``.
|
|
26
34
|
|
|
27
35
|
The order of those steps is the whole function, and it is not
|
|
@@ -36,9 +44,24 @@ def build(shape: Shape, using: str = DEFAULT_DB_ALIAS) -> BuildResult:
|
|
|
36
44
|
statistics work proper -- per-column targets, and caching a built database
|
|
37
45
|
as a template -- comes later. A loader that leaves its table unanalyzed
|
|
38
46
|
ships the exact state this package exists to condemn.
|
|
47
|
+
|
|
48
|
+
``require_statistics=False`` asks for rows and cardinality rather than for a
|
|
49
|
+
database the planner can reason about, and it is the only way to build on a
|
|
50
|
+
backend without ``COPY`` and column statistics. It is written as a
|
|
51
|
+
requirement being dropped rather than as work being skipped, because that is
|
|
52
|
+
what it does: **on PostgreSQL it changes nothing at all** -- the load is
|
|
53
|
+
still ``COPY`` and ``ANALYZE`` still runs, since both are free and leaving
|
|
54
|
+
them out would manufacture the unanalyzed table this package exists to
|
|
55
|
+
condemn. Elsewhere the rows are inserted instead and no statistics are
|
|
56
|
+
gathered, so cardinality is real and nothing about a plan is claimed.
|
|
57
|
+
|
|
58
|
+
The distinction is the one this package draws everywhere: generation and
|
|
59
|
+
cardinality are backend-neutral, planner realism is not. A query *count* is
|
|
60
|
+
an ORM property and means the same on any backend, which is why a growth
|
|
61
|
+
assertion can be honest here while a plan assertion still cannot.
|
|
39
62
|
"""
|
|
40
63
|
connection = connections[using]
|
|
41
|
-
require_postgres(connection, "Building a shape")
|
|
64
|
+
require_postgres(connection, "Building a shape", statistics=require_statistics)
|
|
42
65
|
|
|
43
66
|
results: list[TableResult] = []
|
|
44
67
|
# One transaction around every table. Without it a shape whose second table
|
|
@@ -82,6 +105,13 @@ def _require_empty(connection: Any, table: Table) -> None:
|
|
|
82
105
|
the same table collides on the primary key. That surfaced as a bare
|
|
83
106
|
UniqueViolation naming an index, which tells the reader nothing about what
|
|
84
107
|
they did or what to do instead.
|
|
108
|
+
|
|
109
|
+
The message names the likely cause and not only the remedy, because the
|
|
110
|
+
first consumer met this from a direction the remedy does not fit: a
|
|
111
|
+
session-scoped ``shape_fixture`` and a scaled world pointed at one model.
|
|
112
|
+
The rows are then real, correct, and put there by a fixture the failing test
|
|
113
|
+
never mentions -- so "empty the table first" reads as advice about somebody
|
|
114
|
+
else's data.
|
|
85
115
|
"""
|
|
86
116
|
with connection.cursor() as cursor:
|
|
87
117
|
cursor.execute(f"SELECT EXISTS (SELECT 1 FROM {connection.ops.quote_name(table.db_table)})")
|
|
@@ -89,7 +119,10 @@ def _require_empty(connection: Any, table: Table) -> None:
|
|
|
89
119
|
if row[0]:
|
|
90
120
|
raise ShapeNotEmpty(
|
|
91
121
|
f"{table.db_table} already holds rows, and this package assigns primary keys from 1, "
|
|
92
|
-
"so building over them would collide.
|
|
122
|
+
"so building over them would collide. If nothing in the test wrote them, the usual "
|
|
123
|
+
"cause is a world that was already there: a session-scoped shape_fixture over this "
|
|
124
|
+
"model holds its rows for the whole run, and a scaled world cannot build over them. "
|
|
125
|
+
"Give the two different models, or empty this table first."
|
|
93
126
|
)
|
|
94
127
|
|
|
95
128
|
|
|
@@ -98,12 +131,7 @@ def _require_empty(connection: Any, table: Table) -> None:
|
|
|
98
131
|
# class, so annotating the real wrapper type would mean asserting the checker
|
|
99
132
|
# out of the way on every line that uses it.
|
|
100
133
|
def _load(connection: Any, table: Table, seed: int, plans: dict[str, FanOutPlan]) -> int:
|
|
101
|
-
"""Stream generated rows into the table
|
|
102
|
-
|
|
103
|
-
``bulk_create`` is the obvious alternative and is roughly an order of
|
|
104
|
-
magnitude too slow at the row counts that make a plan meaningful. ``COPY``
|
|
105
|
-
is also why the generator yields tuples rather than model instances: there
|
|
106
|
-
is no instance to build, and no ``save`` to run.
|
|
134
|
+
"""Stream generated rows into the table, by the fastest route the backend has.
|
|
107
135
|
|
|
108
136
|
Each declared value is passed through its field's ``get_db_prep_save``
|
|
109
137
|
first, and that is not a formality. Skipping it is how a naive datetime got
|
|
@@ -111,7 +139,9 @@ def _load(connection: Any, table: Table, seed: int, plans: dict[str, FanOutPlan]
|
|
|
111
139
|
non-UTC ``TIME_ZONE`` -- silently, on the exact column ``Sequential`` exists
|
|
112
140
|
to make realistic -- and how a ``JSONField`` failed to load at all. The
|
|
113
141
|
generator stays backend-neutral because the preparation happens here rather
|
|
114
|
-
than inside it
|
|
142
|
+
than inside it, and that is also why the preparation sits **above** the
|
|
143
|
+
branch below: a value is prepared by its field for its connection, which is
|
|
144
|
+
the same work whichever statement carries it.
|
|
115
145
|
|
|
116
146
|
Returns the number of rows the database actually took, not the number
|
|
117
147
|
declared. They are the same today and stop being so once deduplicated
|
|
@@ -120,12 +150,31 @@ def _load(connection: Any, table: Table, seed: int, plans: dict[str, FanOutPlan]
|
|
|
120
150
|
quote = connection.ops.quote_name
|
|
121
151
|
pk_field = table.model._meta.pk
|
|
122
152
|
columns = [quote(pk_field.column)] + [quote(field.column) for _, field in table.columns()]
|
|
123
|
-
statement = f"COPY {quote(table.db_table)} ({', '.join(columns)}) FROM STDIN"
|
|
124
153
|
# The primary key is prepared like every other value. It did not need to be
|
|
125
154
|
# while keys were always integers; a UUID key does, and a strategy the
|
|
126
155
|
# caller wrote could return anything its column accepts.
|
|
127
156
|
prepare = [pk_field.get_db_prep_save] + [field.get_db_prep_save for _, field in table.columns()]
|
|
157
|
+
rows = (
|
|
158
|
+
tuple(prep(value, connection) for prep, value in zip(prepare, row, strict=True))
|
|
159
|
+
for row in generate_rows(table, seed, plans)
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
if connection.vendor == "postgresql":
|
|
163
|
+
return _copy(connection, table.db_table, columns, rows)
|
|
164
|
+
return _insert(connection, table.db_table, columns, rows)
|
|
128
165
|
|
|
166
|
+
|
|
167
|
+
def _copy(
|
|
168
|
+
connection: Any, db_table: str, columns: list[str], rows: Iterator[tuple[Any, ...]]
|
|
169
|
+
) -> int:
|
|
170
|
+
"""``COPY FROM STDIN``, which is the reason this package can be worth using.
|
|
171
|
+
|
|
172
|
+
``bulk_create`` is the obvious alternative and is roughly an order of
|
|
173
|
+
magnitude too slow at the row counts that make a plan meaningful. ``COPY``
|
|
174
|
+
is also why the generator yields tuples rather than model instances: there
|
|
175
|
+
is no instance to build, and no ``save`` to run.
|
|
176
|
+
"""
|
|
177
|
+
statement = f"COPY {connection.ops.quote_name(db_table)} ({', '.join(columns)}) FROM STDIN"
|
|
129
178
|
with connection.cursor() as cursor:
|
|
130
179
|
# ``copy`` is not in Django's WRAP_ERROR_ATTRS, so without this a
|
|
131
180
|
# Postgres error escapes as a raw psycopg exception: the caller cannot
|
|
@@ -133,13 +182,41 @@ def _load(connection: Any, table: Table, seed: int, plans: dict[str, FanOutPlan]
|
|
|
133
182
|
# block never learns it needs a rollback, so the next query inside it
|
|
134
183
|
# fails with "current transaction is aborted" instead of a Django error.
|
|
135
184
|
with connection.wrap_database_errors, cursor.copy(statement) as copy:
|
|
136
|
-
for row in
|
|
137
|
-
copy.write_row(
|
|
138
|
-
tuple(prep(value, connection) for prep, value in zip(prepare, row, strict=True))
|
|
139
|
-
)
|
|
185
|
+
for row in rows:
|
|
186
|
+
copy.write_row(row)
|
|
140
187
|
return int(cursor.rowcount)
|
|
141
188
|
|
|
142
189
|
|
|
190
|
+
def _insert(
|
|
191
|
+
connection: Any, db_table: str, columns: list[str], rows: Iterator[tuple[Any, ...]]
|
|
192
|
+
) -> int:
|
|
193
|
+
"""The portable route, for a backend that has no ``COPY``.
|
|
194
|
+
|
|
195
|
+
Reached only when the caller said it does not require planner statistics,
|
|
196
|
+
which is the honest shape of the trade: this writes real rows in real
|
|
197
|
+
cardinality and buys nothing at all for a plan. It is slower than ``COPY``
|
|
198
|
+
and that is the wrong thing to worry about at the sizes it is for --
|
|
199
|
+
measured on SQLite, the insert costs about 1.6 ms per thousand rows against
|
|
200
|
+
8 ms to generate them, so the load is not what a growth assertion pays for.
|
|
201
|
+
|
|
202
|
+
Chunked rather than handed the whole iterator, because ``executemany``
|
|
203
|
+
materialises what it is given: streaming into ``COPY`` is the property this
|
|
204
|
+
package is built on, and a portable path that quietly held a million tuples
|
|
205
|
+
in memory would be a different bargain from the one above.
|
|
206
|
+
"""
|
|
207
|
+
placeholders = ", ".join(["%s"] * len(columns))
|
|
208
|
+
statement = (
|
|
209
|
+
f"INSERT INTO {connection.ops.quote_name(db_table)} "
|
|
210
|
+
f"({', '.join(columns)}) VALUES ({placeholders})"
|
|
211
|
+
)
|
|
212
|
+
loaded = 0
|
|
213
|
+
with connection.cursor() as cursor:
|
|
214
|
+
while chunk := list(islice(rows, _INSERT_CHUNK)):
|
|
215
|
+
cursor.executemany(statement, chunk)
|
|
216
|
+
loaded += len(chunk)
|
|
217
|
+
return loaded
|
|
218
|
+
|
|
219
|
+
|
|
143
220
|
def _reset_sequence(connection: Any, table: Table) -> None:
|
|
144
221
|
"""Move the identity sequence past the keys this package just assigned.
|
|
145
222
|
|
|
@@ -161,6 +238,15 @@ def _analyze(connection: Any, table: Table) -> None:
|
|
|
161
238
|
selectivity and commits to it, which is how a two-million-row table gets
|
|
162
239
|
bitmap-scanned through an index for a value matching 98% of it. Measured at
|
|
163
240
|
81 ms on that table, because ``ANALYZE`` samples rather than scans.
|
|
241
|
+
|
|
242
|
+
Nothing is gathered on another backend, and SQLite is the case worth being
|
|
243
|
+
explicit about: it has an ``ANALYZE`` of its own and running it is one line.
|
|
244
|
+
It is deliberately not run. Plan realism on SQLite is out of this package's
|
|
245
|
+
scope -- support the generation, refuse the pretence -- and a table with
|
|
246
|
+
``sqlite_stat1`` behind it would be this package claiming, in the only way a
|
|
247
|
+
library can, that the plan over it means something.
|
|
164
248
|
"""
|
|
249
|
+
if connection.vendor != "postgresql":
|
|
250
|
+
return
|
|
165
251
|
with connection.cursor() as cursor:
|
|
166
252
|
cursor.execute(f"ANALYZE {connection.ops.quote_name(table.db_table)}")
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""The pytest surface, deliberately outside the top-level re-exports.
|
|
2
|
+
|
|
3
|
+
Everything in here imports ``pytest``, and ``pytest`` is an optional extra
|
|
4
|
+
rather than a dependency. Re-exporting these from ``django_data_shape/__init__``
|
|
5
|
+
would make ``import django_data_shape`` fail for a project that runs its suite
|
|
6
|
+
with Django's own runner, so the import boundary is drawn where the dependency
|
|
7
|
+
boundary already is: ``from django_data_shape.fixtures import shape_fixture``
|
|
8
|
+
says out loud that pytest is required for this half and not for the other.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from django_data_shape.fixtures.scale_fixture import scale_fixture
|
|
12
|
+
from django_data_shape.fixtures.shape_fixture import shape_fixture
|
|
13
|
+
from django_data_shape.fixtures.skip_unless_postgres import skip_unless_postgres
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"scale_fixture",
|
|
17
|
+
"shape_fixture",
|
|
18
|
+
"skip_unless_postgres",
|
|
19
|
+
]
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""One shape, offered at whatever size the test asks for."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from functools import partial
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
from django.db import DEFAULT_DB_ALIAS
|
|
9
|
+
|
|
10
|
+
from django_data_shape.scale_protocol import ScaleProtocol
|
|
11
|
+
from django_data_shape.scaled_world import scaled_world
|
|
12
|
+
from django_data_shape.shape import Shape
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
# ``object`` for the same reason as in shape_fixture: the type of a fixture is
|
|
16
|
+
# pytest's own and it changed shape between pytest 8.0 and 8.4.
|
|
17
|
+
def scale_fixture(shape: Shape, *, using: str = DEFAULT_DB_ALIAS) -> object:
|
|
18
|
+
"""A pytest fixture yielding a :class:`~django_data_shape.scale_protocol.ScaleProtocol`.
|
|
19
|
+
|
|
20
|
+
The pytest face of the scale protocol. Bind it in ``conftest.py``::
|
|
21
|
+
|
|
22
|
+
from django_data_shape import Constant, Shape, Table
|
|
23
|
+
from django_data_shape.fixtures import scale_fixture
|
|
24
|
+
|
|
25
|
+
world = scale_fixture(Shape(Table(Order, rows=100, status=Constant("new"))))
|
|
26
|
+
|
|
27
|
+
and a growth assertion has somewhere to ask for a bigger world::
|
|
28
|
+
|
|
29
|
+
def test_the_dashboard_query_is_constant(world, django_assert_num_queries):
|
|
30
|
+
for factor in (1, 10):
|
|
31
|
+
with world(factor):
|
|
32
|
+
with django_assert_num_queries(3):
|
|
33
|
+
dashboard()
|
|
34
|
+
|
|
35
|
+
The declared row counts are the world at factor 1, so the base declaration
|
|
36
|
+
should be the smallest world that still means something -- a hundred rows
|
|
37
|
+
against a thousand is the regime this is for, and it is milliseconds per
|
|
38
|
+
factor. Size, in the two-million-row sense that makes a query *plan*
|
|
39
|
+
realistic, is a different assertion with a different cost and does not vary
|
|
40
|
+
a factor at all.
|
|
41
|
+
|
|
42
|
+
Function-scoped, and it requests pytest-django's ``db`` fixture, which does
|
|
43
|
+
two things worth knowing. A test using this needs no ``django_db`` marker of
|
|
44
|
+
its own. And each world is then built inside the transaction that wraps the
|
|
45
|
+
test, so tearing it down is a savepoint rollback: cheap, exact, and leaving
|
|
46
|
+
the test's own transaction usable afterwards. A test that marks itself
|
|
47
|
+
``transaction=True`` still works -- the marker wins over the fixture, and
|
|
48
|
+
the rollback is then an ordinary one.
|
|
49
|
+
|
|
50
|
+
**Not over a model a session world already holds.** Each world here is built
|
|
51
|
+
from empty and undone again, so a table that
|
|
52
|
+
:func:`~django_data_shape.fixtures.shape_fixture.shape_fixture` filled for
|
|
53
|
+
the session is one this cannot build into at all -- the rows are still there,
|
|
54
|
+
and the build is refused. The two compose over a graph by taking different
|
|
55
|
+
models, not by taking turns over one.
|
|
56
|
+
|
|
57
|
+
**It works on any backend Django supports**, because what a growth
|
|
58
|
+
assertion measures -- the number of queries a block emits -- is an ORM
|
|
59
|
+
property rather than a planner one. Where the backend has ``COPY`` and
|
|
60
|
+
column statistics the world is built with them; where it does not, the rows
|
|
61
|
+
are inserted and no statistics are gathered, so the cardinality is real and
|
|
62
|
+
nothing about a plan is claimed. That is the one place this package builds
|
|
63
|
+
outside PostgreSQL, and it is allowed precisely because the assertion it
|
|
64
|
+
serves does not need the planner. A plan assertion still skips: see
|
|
65
|
+
:func:`~django_data_shape.fixtures.skip_unless_postgres.skip_unless_postgres`.
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
@pytest.fixture
|
|
69
|
+
def scaled_worlds(db: None) -> ScaleProtocol:
|
|
70
|
+
# Binding, not wrapping. The fixture's whole job is to attach one shape
|
|
71
|
+
# and one connection to the protocol; everything about what a world is
|
|
72
|
+
# and how it is undone belongs to scaled_world, where a consumer not
|
|
73
|
+
# using pytest can reach it too.
|
|
74
|
+
return partial(scaled_world, shape, using=using)
|
|
75
|
+
|
|
76
|
+
return scaled_worlds
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""One shape, built once for a whole test session."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
from django.db import DEFAULT_DB_ALIAS, connections
|
|
9
|
+
|
|
10
|
+
from django_data_shape.build import build
|
|
11
|
+
from django_data_shape.build_result import BuildResult
|
|
12
|
+
from django_data_shape.fixtures.skip_unless_postgres import skip_unless_postgres
|
|
13
|
+
from django_data_shape.shape import Shape
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
# The return type is ``object`` on purpose. What ``pytest.fixture`` hands back
|
|
17
|
+
# is pytest's own -- a plain function on pytest 8.0 and a
|
|
18
|
+
# ``FixtureFunctionDefinition`` from 8.4 onwards -- so naming it here would pin
|
|
19
|
+
# this package's floor to a pytest version in exchange for nothing: the caller
|
|
20
|
+
# binds the value to a name and never calls it.
|
|
21
|
+
def shape_fixture(shape: Shape, *, using: str = DEFAULT_DB_ALIAS) -> object:
|
|
22
|
+
"""A session-scoped pytest fixture that builds ``shape`` once.
|
|
23
|
+
|
|
24
|
+
Bind it to a name in ``conftest.py`` and request that name from a test::
|
|
25
|
+
|
|
26
|
+
from django_data_shape import Constant, Shape, Table
|
|
27
|
+
from django_data_shape.fixtures import shape_fixture
|
|
28
|
+
|
|
29
|
+
orders = shape_fixture(Shape(Table(Order, rows=100_000, status=Constant("new"))))
|
|
30
|
+
|
|
31
|
+
::
|
|
32
|
+
|
|
33
|
+
import pytest
|
|
34
|
+
|
|
35
|
+
@pytest.mark.django_db
|
|
36
|
+
def test_the_dashboard_query(orders):
|
|
37
|
+
assert orders.rows == 100_000
|
|
38
|
+
|
|
39
|
+
It **composes with pytest-django rather than replacing it**. The fixture
|
|
40
|
+
requests ``django_db_setup``, which is the seam a project overrides to
|
|
41
|
+
decide how its test database is made, so whatever a project has done there
|
|
42
|
+
-- a template database, ``--reuse-db``, a different creation strategy -- has
|
|
43
|
+
already happened before a row is generated. It then writes through
|
|
44
|
+
``django_db_blocker.unblock()``, the mechanism pytest-django documents for
|
|
45
|
+
populating a database once. Neither of those is imported: they are asked for
|
|
46
|
+
by name, so this package depends on two fixture names and not on
|
|
47
|
+
pytest-django's internals.
|
|
48
|
+
|
|
49
|
+
**Session scope is load-bearing, not a performance choice.** pytest creates
|
|
50
|
+
higher-scoped fixtures before lower-scoped ones, so a session-scoped build
|
|
51
|
+
always runs before the function-scoped ``db`` fixture opens the transaction
|
|
52
|
+
that wraps a test -- which is what makes the rows committed and visible to
|
|
53
|
+
every later test. A function-scoped build would be ordered against ``db`` by
|
|
54
|
+
the accident of argument order, and on the losing side of that order it
|
|
55
|
+
would be rolled back with the test that happened to build it.
|
|
56
|
+
|
|
57
|
+
Yields the :class:`~django_data_shape.build_result.BuildResult`, so a test
|
|
58
|
+
can assert on the size of the world it was handed.
|
|
59
|
+
|
|
60
|
+
**One caveat, and it is worth stating plainly**: a test marked
|
|
61
|
+
``django_db(transaction=True)`` truncates every table at teardown, and takes
|
|
62
|
+
the session's rows with it. Nothing rebuilds them, so a later test that
|
|
63
|
+
reads this fixture is measuring an empty database. Keep transactional tests
|
|
64
|
+
off the tables a shape owns, mark them ``serialized_rollback=True``, or
|
|
65
|
+
build per test with
|
|
66
|
+
:func:`~django_data_shape.scaled_world.scaled_world` at factor 1 -- which
|
|
67
|
+
undoes itself and therefore does not care.
|
|
68
|
+
|
|
69
|
+
**One world per table.** A session world holds its rows for the whole run,
|
|
70
|
+
so :func:`~django_data_shape.fixtures.scale_fixture.scale_fixture` over the
|
|
71
|
+
same model cannot build: the second build meets a table that is not empty
|
|
72
|
+
and is refused. Give the two different models -- the session world the tables
|
|
73
|
+
a plan assertion needs to be big, the scale harness the tables a growth
|
|
74
|
+
assertion counts. It is the first thing a consumer composing both hits, and
|
|
75
|
+
the refusal now names it.
|
|
76
|
+
|
|
77
|
+
On a connection that cannot carry a shaped database the fixture skips with
|
|
78
|
+
the reason rather than raising, so a suite that also runs on SQLite reports
|
|
79
|
+
what it did not check instead of erroring or, worse, passing.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
# The blocker is typed loosely because naming its class would mean importing
|
|
83
|
+
# pytest_django, and the whole point of asking for it by name is that this
|
|
84
|
+
# package does not. It is a pytest-django fixture; what is used of it is
|
|
85
|
+
# ``unblock()``.
|
|
86
|
+
@pytest.fixture(scope="session")
|
|
87
|
+
def built_shape(django_db_setup: None, django_db_blocker: Any) -> BuildResult:
|
|
88
|
+
# ``vendor`` is a class attribute, so the gate is read before anything
|
|
89
|
+
# is unblocked and a skip costs no connection at all.
|
|
90
|
+
skip_unless_postgres(connections[using], "Building a shape for the test session")
|
|
91
|
+
with django_db_blocker.unblock():
|
|
92
|
+
return build(shape, using=using)
|
|
93
|
+
|
|
94
|
+
return built_shape
|