django-data-shape 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- django_data_shape-0.2.0/CHANGELOG.md +114 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/PKG-INFO +21 -5
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/README.md +20 -4
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/__init__.py +6 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/build.py +30 -5
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/__init__.py +3 -1
- django_data_shape-0.2.0/django_data_shape/distributions/bounded.py +25 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/constant.py +4 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/skew.py +4 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/uniform.py +16 -2
- django_data_shape-0.2.0/django_data_shape/distributions/zipf.py +37 -0
- django_data_shape-0.2.0/django_data_shape/fan_out.py +82 -0
- django_data_shape-0.2.0/django_data_shape/fan_out_plan.py +75 -0
- django_data_shape-0.2.0/django_data_shape/generate_rows.py +60 -0
- django_data_shape-0.2.0/django_data_shape/order_tables.py +65 -0
- django_data_shape-0.2.0/django_data_shape/resolve_fan_out.py +121 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/shape.py +12 -2
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/table.py +100 -16
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/version.py +1 -1
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/docs/index.md +16 -9
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/docs/reference.md +6 -0
- django_data_shape-0.2.0/docs/relations.md +118 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/mkdocs.yml +1 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/pyproject.toml +1 -1
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/distributions/test_constant.py +6 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/distributions/test_skew.py +4 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/distributions/test_uniform.py +15 -0
- django_data_shape-0.2.0/tests/distributions/test_zipf.py +40 -0
- django_data_shape-0.2.0/tests/test_build_graph.py +175 -0
- django_data_shape-0.2.0/tests/test_fan_out.py +42 -0
- django_data_shape-0.2.0/tests/test_order_tables.py +46 -0
- django_data_shape-0.2.0/tests/test_resolve_fan_out.py +141 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_shape.py +8 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_table.py +60 -2
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/testapp/models.py +36 -0
- django_data_shape-0.1.0/CHANGELOG.md +0 -53
- django_data_shape-0.1.0/django_data_shape/generate_rows.py +0 -43
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/CODEOWNERS +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/SECURITY.md +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/dependabot.yml +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/workflows/release.yml +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/workflows/tests.yml +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/workflows/upstream-drift.yml +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.gitignore +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.pre-commit-config.yaml +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/CLAUDE.md +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/LICENSE +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/Makefile +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/build_result.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/distribution.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/sequential.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/invalid_shape.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/py.typed +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/require_postgres.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/shape_not_empty.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/table_result.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/unsupported_backend.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/utils.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/scripts/release-publish.sh +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/__init__.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/conftest.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/conftest_settings.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/distributions/__init__.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/distributions/test_sequential.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_build.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_generate_rows.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_require_postgres.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_utils.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_version.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/testapp/__init__.py +0 -0
- {django_data_shape-0.1.0 → django_data_shape-0.2.0}/uv.lock +0 -0
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented in this file.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
6
|
+
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
|
+
|
|
8
|
+
## [Unreleased]
|
|
9
|
+
|
|
10
|
+
## [0.2.0] — 2026-09-01
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
- `FanOut`, which declares how a foreign key's children spread across their
|
|
14
|
+
parents: a size distribution, a `childless` share for parents with no children
|
|
15
|
+
at all, a `null` share for nullable columns, and `placement`.
|
|
16
|
+
- `Zipf`, the heavy-tailed weight distribution fan-out is realistically drawn
|
|
17
|
+
from. A table where every parent has ten children is not merely tidy -- it is
|
|
18
|
+
the one shape in which the planner is never wrong, because its `n_distinct`
|
|
19
|
+
average is the truth.
|
|
20
|
+
- Tables load in dependency order, and a cycle of fan-outs is refused by name.
|
|
21
|
+
|
|
22
|
+
### The two representation decisions
|
|
23
|
+
- **Fan-out reads the parent's real keys rather than assuming the dense `1..N`
|
|
24
|
+
range this package assigns.** The case that matters is the hybrid: a project
|
|
25
|
+
builds its fifty companies with the ORM, where the row count is small and the
|
|
26
|
+
ORM is the right tool, and asks this package only for the two million orders.
|
|
27
|
+
Referential integrity then holds by construction, because every key emitted
|
|
28
|
+
came out of the parent table.
|
|
29
|
+
- **A fan-out is a partition of the child key range, not a per-child draw.**
|
|
30
|
+
Parent `j` owns rows `[start, end)`. A per-child draw cannot be inverted, and
|
|
31
|
+
"which children belong to parent T" is what a mirrored collection needs. The
|
|
32
|
+
childless tail and `placement` both fall out of the partition for free.
|
|
33
|
+
|
|
34
|
+
### Notes
|
|
35
|
+
- `placement` defaults to `arrival`. Emitting children parent by parent gives a
|
|
36
|
+
perfectly clustered table that no production system has and that flatters
|
|
37
|
+
every index scan over the foreign key.
|
|
38
|
+
- A self-referential fan-out is refused: it would read keys from a table still
|
|
39
|
+
empty at load time. Self-referential trees are their own feature.
|
|
40
|
+
- A relation needs a `FanOut` and a plain column refuses one, in both
|
|
41
|
+
directions.
|
|
42
|
+
|
|
43
|
+
## [0.1.1] — 2026-09-01
|
|
44
|
+
|
|
45
|
+
### Added
|
|
46
|
+
- `Bounded`, an optional second protocol for distributions that can say how many
|
|
47
|
+
distinct values they produce. `Constant` and `Skew` implement it. It is
|
|
48
|
+
separate from `Distribution` on purpose: adding the method there would make it
|
|
49
|
+
required, so a custom distribution written against the single-method protocol
|
|
50
|
+
would stop satisfying it.
|
|
51
|
+
- A declaration that provably cannot be loaded is now refused at declaration
|
|
52
|
+
time. A `Constant` on a unique column with more than one row, or a `Skew` with
|
|
53
|
+
fewer values than rows, is arithmetic rather than a subtle problem, and it used
|
|
54
|
+
to be discovered by the database partway through a load that had already
|
|
55
|
+
written most of a table. Only single-column uniqueness is checked; multi-column
|
|
56
|
+
constraints are satisfiable through combinations across independently declared
|
|
57
|
+
columns, which is an analysis rather than a comparison.
|
|
58
|
+
|
|
59
|
+
### Fixed
|
|
60
|
+
- `Uniform` with `places` raised `decimal.InvalidOperation` past 28 significant
|
|
61
|
+
digits -- Python's default context precision -- from inside the `COPY` loop, on
|
|
62
|
+
a column such as `numeric(30, 2)` that would have accepted the value. The
|
|
63
|
+
precision needed is now derived from the declared bounds.
|
|
64
|
+
- `Table` and `Shape` attributes are read-only. Every rule they enforce runs once
|
|
65
|
+
in `__init__`, so while the attributes were writable a declaration could be
|
|
66
|
+
edited afterwards into one that would have been refused, with nothing
|
|
67
|
+
re-checking it.
|
|
68
|
+
|
|
69
|
+
## [0.1.0] — 2026-09-01
|
|
70
|
+
|
|
71
|
+
### Added
|
|
72
|
+
- The shape vocabulary: `Shape`, `Table`, and the `Skew`, `Uniform`, `Sequential`
|
|
73
|
+
and `Constant` distributions, behind a single-method `Distribution` protocol.
|
|
74
|
+
- `build()`, which generates rows, loads them with `COPY FROM STDIN`, moves the
|
|
75
|
+
identity sequence past the keys it assigned, and runs `ANALYZE`. The order is
|
|
76
|
+
owned by the library: loading into a table analyzed while empty leaves the
|
|
77
|
+
planner applying old statistics to a new row count, which is a worse lie than
|
|
78
|
+
having no statistics at all.
|
|
79
|
+
- `InvalidShape`, raised at declaration time, and `UnsupportedBackend`, raised
|
|
80
|
+
for any connection that is not PostgreSQL. Generation is backend-neutral;
|
|
81
|
+
`COPY` and planner statistics are not, and degrading quietly would produce the
|
|
82
|
+
false confidence this package exists to remove.
|
|
83
|
+
- Primary keys are assigned as a dense `1..N` range rather than declared. That is
|
|
84
|
+
what will let a foreign key be satisfied without a lookup once relations land,
|
|
85
|
+
and what makes a self-referential tree acyclic by construction.
|
|
86
|
+
- A field carrying a Django `default=` is filled with the value `save()` would
|
|
87
|
+
have written, because defaults are applied by `save()` and nothing here calls
|
|
88
|
+
it. A callable default is refused instead of guessed: `uuid4` varies per row
|
|
89
|
+
and `dict` does not, and nothing on the field distinguishes them.
|
|
90
|
+
|
|
91
|
+
- `ShapeNotEmpty`, raised before anything is written when a target table already
|
|
92
|
+
holds rows. Keys start at 1 on every build, so the collision was previously a
|
|
93
|
+
unique-violation naming an index, which says nothing about what to do instead.
|
|
94
|
+
- The whole build runs in one transaction, so a shape whose second table fails
|
|
95
|
+
leaves nothing behind and can be re-run after a fix.
|
|
96
|
+
- Every declared value passes through its field's `get_db_prep_save`. Without it
|
|
97
|
+
a naive datetime was stored verbatim rather than localised -- hours from where
|
|
98
|
+
`save()` puts it under a non-UTC `TIME_ZONE` -- and a `JSONField` could not be
|
|
99
|
+
written at all.
|
|
100
|
+
|
|
101
|
+
### Notes
|
|
102
|
+
- Relations are refused in both directions: declaring one raises, and so does
|
|
103
|
+
omitting one that cannot be null. An optional foreign key may be omitted and
|
|
104
|
+
loads entirely `NULL`, which is documented rather than left to be discovered.
|
|
105
|
+
- Only integer primary keys are supported. Any other kind is refused, rather than
|
|
106
|
+
a dense `1..N` integer range being written into a character column.
|
|
107
|
+
- psycopg 3 is required and psycopg 2 is refused by name. Rows stream straight
|
|
108
|
+
into `COPY FROM STDIN`, which psycopg 2 cannot do without materialising them
|
|
109
|
+
first.
|
|
110
|
+
|
|
111
|
+
[Unreleased]: https://github.com/Artui/django-data-shape/compare/v0.2.0...HEAD
|
|
112
|
+
[0.2.0]: https://github.com/Artui/django-data-shape/compare/v0.1.1...v0.2.0
|
|
113
|
+
[0.1.1]: https://github.com/Artui/django-data-shape/compare/v0.1.0...v0.1.1
|
|
114
|
+
[0.1.0]: https://github.com/Artui/django-data-shape/compare/v0.0.0...v0.1.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: django-data-shape
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: A realistically shaped test database from Django models: declare cardinality, skew and fan-out, load by COPY, and make the query planner believe it.
|
|
5
5
|
Project-URL: Homepage, https://github.com/Artui/django-data-shape
|
|
6
6
|
Project-URL: Repository, https://github.com/Artui/django-data-shape
|
|
@@ -96,12 +96,28 @@ build(shape)
|
|
|
96
96
|
past the keys it assigned, and runs `ANALYZE` so the planner can see the shape.
|
|
97
97
|
It raises on any backend that is not PostgreSQL rather than degrading quietly.
|
|
98
98
|
|
|
99
|
+
## Relations
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
Table(
|
|
103
|
+
Order,
|
|
104
|
+
rows=2_000_000,
|
|
105
|
+
# A distribution, not a number: giving every parent ten children is the one
|
|
106
|
+
# shape in which the planner is never wrong, because its n_distinct average
|
|
107
|
+
# is then the truth.
|
|
108
|
+
company=FanOut(Zipf(1.2), childless=0.35),
|
|
109
|
+
...
|
|
110
|
+
)
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
The parents can be rows this package built or rows your own code did -- their
|
|
114
|
+
real keys are read, not assumed, so the ORM can own the small tables while this
|
|
115
|
+
owns the large ones.
|
|
116
|
+
|
|
99
117
|
## Status
|
|
100
118
|
|
|
101
|
-
Early.
|
|
102
|
-
|
|
103
|
-
releases after it; declaring a relation raises today rather than generating ids
|
|
104
|
-
that point at nothing.
|
|
119
|
+
Early. Single tables and the model graph. Derived fields, collections copied
|
|
120
|
+
along a join, per-group invariants and template-database reuse come next.
|
|
105
121
|
|
|
106
122
|
Full documentation: <https://artui.github.io/django-data-shape/>
|
|
107
123
|
|
|
@@ -56,12 +56,28 @@ build(shape)
|
|
|
56
56
|
past the keys it assigned, and runs `ANALYZE` so the planner can see the shape.
|
|
57
57
|
It raises on any backend that is not PostgreSQL rather than degrading quietly.
|
|
58
58
|
|
|
59
|
+
## Relations
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
Table(
|
|
63
|
+
Order,
|
|
64
|
+
rows=2_000_000,
|
|
65
|
+
# A distribution, not a number: giving every parent ten children is the one
|
|
66
|
+
# shape in which the planner is never wrong, because its n_distinct average
|
|
67
|
+
# is then the truth.
|
|
68
|
+
company=FanOut(Zipf(1.2), childless=0.35),
|
|
69
|
+
...
|
|
70
|
+
)
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
The parents can be rows this package built or rows your own code did -- their
|
|
74
|
+
real keys are read, not assumed, so the ORM can own the small tables while this
|
|
75
|
+
owns the large ones.
|
|
76
|
+
|
|
59
77
|
## Status
|
|
60
78
|
|
|
61
|
-
Early.
|
|
62
|
-
|
|
63
|
-
releases after it; declaring a relation raises today rather than generating ids
|
|
64
|
-
that point at nothing.
|
|
79
|
+
Early. Single tables and the model graph. Derived fields, collections copied
|
|
80
|
+
along a join, per-group invariants and template-database reuse come next.
|
|
65
81
|
|
|
66
82
|
Full documentation: <https://artui.github.io/django-data-shape/>
|
|
67
83
|
|
|
@@ -2,11 +2,14 @@
|
|
|
2
2
|
|
|
3
3
|
from django_data_shape.build import build
|
|
4
4
|
from django_data_shape.build_result import BuildResult
|
|
5
|
+
from django_data_shape.distributions.bounded import Bounded
|
|
5
6
|
from django_data_shape.distributions.constant import Constant
|
|
6
7
|
from django_data_shape.distributions.distribution import Distribution
|
|
7
8
|
from django_data_shape.distributions.sequential import Sequential
|
|
8
9
|
from django_data_shape.distributions.skew import Skew
|
|
9
10
|
from django_data_shape.distributions.uniform import Uniform
|
|
11
|
+
from django_data_shape.distributions.zipf import Zipf
|
|
12
|
+
from django_data_shape.fan_out import FanOut
|
|
10
13
|
from django_data_shape.invalid_shape import InvalidShape
|
|
11
14
|
from django_data_shape.shape import Shape
|
|
12
15
|
from django_data_shape.shape_not_empty import ShapeNotEmpty
|
|
@@ -16,9 +19,11 @@ from django_data_shape.unsupported_backend import UnsupportedBackend
|
|
|
16
19
|
from django_data_shape.version import __version__
|
|
17
20
|
|
|
18
21
|
__all__ = [
|
|
22
|
+
"Bounded",
|
|
19
23
|
"BuildResult",
|
|
20
24
|
"Constant",
|
|
21
25
|
"Distribution",
|
|
26
|
+
"FanOut",
|
|
22
27
|
"InvalidShape",
|
|
23
28
|
"Sequential",
|
|
24
29
|
"Shape",
|
|
@@ -27,6 +32,7 @@ __all__ = [
|
|
|
27
32
|
"Table",
|
|
28
33
|
"TableResult",
|
|
29
34
|
"Uniform",
|
|
35
|
+
"Zipf",
|
|
30
36
|
"UnsupportedBackend",
|
|
31
37
|
"__version__",
|
|
32
38
|
"build",
|
|
@@ -2,14 +2,19 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
-
from typing import Any
|
|
5
|
+
from typing import Any, cast
|
|
6
6
|
|
|
7
7
|
from django.core.management.color import no_style
|
|
8
8
|
from django.db import DEFAULT_DB_ALIAS, connections, transaction
|
|
9
|
+
from django.db.models import Model
|
|
9
10
|
|
|
10
11
|
from django_data_shape.build_result import BuildResult
|
|
12
|
+
from django_data_shape.fan_out import FanOut
|
|
13
|
+
from django_data_shape.fan_out_plan import FanOutPlan
|
|
11
14
|
from django_data_shape.generate_rows import generate_rows
|
|
15
|
+
from django_data_shape.order_tables import order_tables
|
|
12
16
|
from django_data_shape.require_postgres import require_postgres
|
|
17
|
+
from django_data_shape.resolve_fan_out import resolve_fan_out
|
|
13
18
|
from django_data_shape.shape import Shape
|
|
14
19
|
from django_data_shape.shape_not_empty import ShapeNotEmpty
|
|
15
20
|
from django_data_shape.table import Table
|
|
@@ -41,15 +46,35 @@ def build(shape: Shape, using: str = DEFAULT_DB_ALIAS) -> BuildResult:
|
|
|
41
46
|
# -- fix the shape, run it again -- fails on a duplicate key rather than on
|
|
42
47
|
# the original problem.
|
|
43
48
|
with transaction.atomic(using=using):
|
|
44
|
-
|
|
49
|
+
# Parents first. Not because the database insists -- Django's foreign
|
|
50
|
+
# keys are deferred, so any order commits -- but because a fan-out reads
|
|
51
|
+
# its parent's real keys, and a table with no rows yet has none.
|
|
52
|
+
for table in order_tables(shape.tables):
|
|
45
53
|
_require_empty(connection, table)
|
|
46
|
-
|
|
54
|
+
plans = _resolve(connection, table, shape.seed)
|
|
55
|
+
loaded = _load(connection, table, shape.seed, plans)
|
|
47
56
|
_reset_sequence(connection, table)
|
|
48
57
|
_analyze(connection, table)
|
|
49
58
|
results.append(TableResult(table=table.db_table, rows=loaded))
|
|
50
59
|
return BuildResult(tables=tuple(results))
|
|
51
60
|
|
|
52
61
|
|
|
62
|
+
def _resolve(connection: Any, table: Table, seed: int) -> dict[str, FanOutPlan]:
|
|
63
|
+
"""Partition each declared relation over the parent keys that exist."""
|
|
64
|
+
plans: dict[str, FanOutPlan] = {}
|
|
65
|
+
for name, field in table.relations():
|
|
66
|
+
plans[name] = resolve_fan_out(
|
|
67
|
+
cast("FanOut", table.fields[name]),
|
|
68
|
+
cast("type[Model]", field.related_model),
|
|
69
|
+
table.rows,
|
|
70
|
+
seed,
|
|
71
|
+
table.db_table,
|
|
72
|
+
name,
|
|
73
|
+
connection,
|
|
74
|
+
)
|
|
75
|
+
return plans
|
|
76
|
+
|
|
77
|
+
|
|
53
78
|
def _require_empty(connection: Any, table: Table) -> None:
|
|
54
79
|
"""Refuse to build on top of rows that are already there.
|
|
55
80
|
|
|
@@ -72,7 +97,7 @@ def _require_empty(connection: Any, table: Table) -> None:
|
|
|
72
97
|
# API reached through Django's cursor wrapper, and it appears on no Django base
|
|
73
98
|
# class, so annotating the real wrapper type would mean asserting the checker
|
|
74
99
|
# out of the way on every line that uses it.
|
|
75
|
-
def _load(connection: Any, table: Table, seed: int) -> int:
|
|
100
|
+
def _load(connection: Any, table: Table, seed: int, plans: dict[str, FanOutPlan]) -> int:
|
|
76
101
|
"""Stream generated rows into the table with ``COPY FROM STDIN``.
|
|
77
102
|
|
|
78
103
|
``bulk_create`` is the obvious alternative and is roughly an order of
|
|
@@ -105,7 +130,7 @@ def _load(connection: Any, table: Table, seed: int) -> int:
|
|
|
105
130
|
# block never learns it needs a rollback, so the next query inside it
|
|
106
131
|
# fails with "current transaction is aborted" instead of a Django error.
|
|
107
132
|
with connection.wrap_database_errors, cursor.copy(statement) as copy:
|
|
108
|
-
for row in generate_rows(table, seed):
|
|
133
|
+
for row in generate_rows(table, seed, plans):
|
|
109
134
|
copy.write_row(
|
|
110
135
|
(
|
|
111
136
|
row[0],
|
{django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/__init__.py
RENAMED
|
@@ -1,9 +1,11 @@
|
|
|
1
1
|
"""The declared value distributions."""
|
|
2
2
|
|
|
3
|
+
from django_data_shape.distributions.bounded import Bounded
|
|
3
4
|
from django_data_shape.distributions.constant import Constant
|
|
4
5
|
from django_data_shape.distributions.distribution import Distribution
|
|
5
6
|
from django_data_shape.distributions.sequential import Sequential
|
|
6
7
|
from django_data_shape.distributions.skew import Skew
|
|
7
8
|
from django_data_shape.distributions.uniform import Uniform
|
|
9
|
+
from django_data_shape.distributions.zipf import Zipf
|
|
8
10
|
|
|
9
|
-
__all__ = ["Constant", "Distribution", "Sequential", "Skew", "Uniform"]
|
|
11
|
+
__all__ = ["Bounded", "Constant", "Distribution", "Sequential", "Skew", "Uniform", "Zipf"]
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Distributions that can only ever produce so many different values."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Protocol, runtime_checkable
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@runtime_checkable
|
|
9
|
+
class Bounded(Protocol):
|
|
10
|
+
"""A distribution with a known, finite number of distinct values.
|
|
11
|
+
|
|
12
|
+
Deliberately a **second** protocol rather than a method on
|
|
13
|
+
:class:`~django_data_shape.distributions.distribution.Distribution`. Adding
|
|
14
|
+
it there would make it required, and a custom distribution written against
|
|
15
|
+
the single-method protocol would stop satisfying it -- so the one thing this
|
|
16
|
+
exists to prevent, a declaration that cannot describe a database, would be
|
|
17
|
+
bought by breaking every declaration someone had already written.
|
|
18
|
+
|
|
19
|
+
Structural and runtime-checkable, so a distribution opts in by having the
|
|
20
|
+
method and nothing has to register anywhere. A distribution that cannot
|
|
21
|
+
answer -- one drawing from a continuous range, say -- simply does not
|
|
22
|
+
implement it, and is treated as unbounded rather than as suspicious.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
def distinct_values(self) -> int: ...
|
{django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/constant.py
RENAMED
|
@@ -22,5 +22,9 @@ class Constant:
|
|
|
22
22
|
def value(self, row: int, draw: float) -> object:
|
|
23
23
|
return self._value
|
|
24
24
|
|
|
25
|
+
def distinct_values(self) -> int:
|
|
26
|
+
"""One, by definition. See ``Bounded``."""
|
|
27
|
+
return 1
|
|
28
|
+
|
|
25
29
|
def __repr__(self) -> str:
|
|
26
30
|
return f"Constant({self._value!r})"
|
|
@@ -67,5 +67,9 @@ class Skew:
|
|
|
67
67
|
# is cheaper than renormalising on every draw.
|
|
68
68
|
return self._values[-1]
|
|
69
69
|
|
|
70
|
+
def distinct_values(self) -> int:
|
|
71
|
+
"""How many values were declared. See ``Bounded``."""
|
|
72
|
+
return len(self._values)
|
|
73
|
+
|
|
70
74
|
def __repr__(self) -> str:
|
|
71
75
|
return f"Skew({self._weights!r})"
|
{django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/uniform.py
RENAMED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import math
|
|
6
|
-
from decimal import Decimal
|
|
6
|
+
from decimal import Decimal, localcontext
|
|
7
7
|
|
|
8
8
|
from django_data_shape.invalid_shape import InvalidShape
|
|
9
9
|
|
|
@@ -38,12 +38,26 @@ class Uniform:
|
|
|
38
38
|
self._low = low
|
|
39
39
|
self._high = high
|
|
40
40
|
self._places = places
|
|
41
|
+
# Python's default decimal context carries 28 significant digits, and
|
|
42
|
+
# rounding past it raises InvalidOperation -- which used to surface from
|
|
43
|
+
# inside the COPY loop, on a numeric(30, 2) column that would have
|
|
44
|
+
# accepted the value perfectly well. The precision needed is the integer
|
|
45
|
+
# digits of the widest bound plus the decimal places, and the guard
|
|
46
|
+
# covers the rounding step itself.
|
|
47
|
+
magnitude = max(abs(low), abs(high), 1.0)
|
|
48
|
+
self._precision = max(28, int(math.log10(magnitude)) + (places or 0) + 5)
|
|
41
49
|
|
|
42
50
|
def value(self, row: int, draw: float) -> object:
|
|
43
51
|
raw = self._low + draw * (self._high - self._low)
|
|
44
52
|
if self._places is None:
|
|
45
53
|
return raw
|
|
46
|
-
|
|
54
|
+
with localcontext() as context:
|
|
55
|
+
context.prec = self._precision
|
|
56
|
+
# repr() rather than the float itself: Decimal(float) takes the full
|
|
57
|
+
# binary expansion, which rounds tie cases the other way --
|
|
58
|
+
# Decimal(2.675) rounds to 2.67 where Decimal(repr(2.675)) gives the
|
|
59
|
+
# 2.68 a reader expects from the literal they wrote.
|
|
60
|
+
return round(Decimal(repr(raw)), self._places)
|
|
47
61
|
|
|
48
62
|
def __repr__(self) -> str:
|
|
49
63
|
places = "" if self._places is None else f", places={self._places}"
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Heavy-tailed weights: a few very large, a long tail of very small."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
|
|
7
|
+
from django_data_shape.invalid_shape import InvalidShape
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class Zipf:
|
|
11
|
+
"""Positive weights following a power law of exponent ``s``.
|
|
12
|
+
|
|
13
|
+
The distribution fan-out is realistically drawn from, and the reason
|
|
14
|
+
declaring fan-out is worth doing at all. A customer table where every
|
|
15
|
+
customer has ten orders is not merely tidy, it is the one shape in which the
|
|
16
|
+
planner is never wrong: its ``n_distinct`` average *is* the truth, so a join
|
|
17
|
+
estimate cannot miss. Give the head a thousand orders and the tail one, and
|
|
18
|
+
the same estimate is out by orders of magnitude in both directions -- which
|
|
19
|
+
is what production looks like and what a test database has to reproduce.
|
|
20
|
+
|
|
21
|
+
Inverse transform of a Pareto: ``(1 - draw) ** (-1 / s)``. Larger ``s``
|
|
22
|
+
means a lighter tail; values near 1 are the classic Zipf regime.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
def __init__(self, s: float = 1.2) -> None:
|
|
26
|
+
if not math.isfinite(s) or s <= 0:
|
|
27
|
+
raise InvalidShape(f"Zipf needs a positive finite exponent, got s={s}.")
|
|
28
|
+
self._s = s
|
|
29
|
+
|
|
30
|
+
def value(self, row: int, draw: float) -> object:
|
|
31
|
+
# ``1 - draw`` rather than ``draw`` so the singularity sits at the top of
|
|
32
|
+
# the interval: draw is in [0, 1), so 1 - draw is in (0, 1] and never
|
|
33
|
+
# zero, which keeps the power finite for every possible draw.
|
|
34
|
+
return (1.0 - draw) ** (-1.0 / self._s)
|
|
35
|
+
|
|
36
|
+
def __repr__(self) -> str:
|
|
37
|
+
return f"Zipf({self._s!r})"
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""How a foreign key's children are spread across their parents."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from django_data_shape.distributions.distribution import Distribution
|
|
6
|
+
from django_data_shape.invalid_shape import InvalidShape
|
|
7
|
+
|
|
8
|
+
PLACEMENTS = ("arrival", "grouped")
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class FanOut:
|
|
12
|
+
"""A distribution over how many children each parent has.
|
|
13
|
+
|
|
14
|
+
Declared as a shape over an **already fixed** child row count, never as a
|
|
15
|
+
multiplier. Rows times fan-out makes cardinality emergent, and a table whose
|
|
16
|
+
size is emergent is a table nothing can assert about.
|
|
17
|
+
|
|
18
|
+
``sizes`` supplies a relative weight per parent -- ``Zipf()`` for the
|
|
19
|
+
realistic heavy tail, ``Uniform(1, 10)`` for something flatter. The weights
|
|
20
|
+
are normalised, so their scale is irrelevant and only their spread matters.
|
|
21
|
+
|
|
22
|
+
``childless`` is the share of parents with **no** children at all. It is
|
|
23
|
+
called out separately because it is the case hand-written fixtures always
|
|
24
|
+
omit and the one that changes what a join does: a parent nobody references
|
|
25
|
+
is the difference between an inner and an outer join returning the same
|
|
26
|
+
thing and returning different things.
|
|
27
|
+
|
|
28
|
+
``null`` is the share of children whose foreign key is NULL, which is only
|
|
29
|
+
meaningful on a nullable column. **It thins the partition uniformly after**
|
|
30
|
+
it is computed, so ``sizes`` describes the pre-null spread. Stated because
|
|
31
|
+
the alternative -- partitioning only the non-null children -- would make the
|
|
32
|
+
declared distribution mean something subtly different from what it says.
|
|
33
|
+
|
|
34
|
+
``placement`` decides where children sit physically, and it is not
|
|
35
|
+
cosmetic. Emitting them parent by parent gives a perfectly clustered table
|
|
36
|
+
that no production system has and that flatters every index scan; the
|
|
37
|
+
default interleaves them the way rows really arrive.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
def __init__(
|
|
41
|
+
self,
|
|
42
|
+
sizes: Distribution,
|
|
43
|
+
*,
|
|
44
|
+
childless: float = 0.0,
|
|
45
|
+
null: float = 0.0,
|
|
46
|
+
placement: str = "arrival",
|
|
47
|
+
) -> None:
|
|
48
|
+
for name, share in (("childless", childless), ("null", null)):
|
|
49
|
+
if not 0.0 <= share < 1.0:
|
|
50
|
+
raise InvalidShape(
|
|
51
|
+
f"FanOut {name} is a share of the whole and must be in [0, 1), got {share}."
|
|
52
|
+
)
|
|
53
|
+
if placement not in PLACEMENTS:
|
|
54
|
+
raise InvalidShape(
|
|
55
|
+
f"FanOut placement must be one of {', '.join(PLACEMENTS)}, got {placement!r}."
|
|
56
|
+
)
|
|
57
|
+
self._sizes = sizes
|
|
58
|
+
self._childless = childless
|
|
59
|
+
self._null = null
|
|
60
|
+
self._placement = placement
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def sizes(self) -> Distribution:
|
|
64
|
+
return self._sizes
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def childless(self) -> float:
|
|
68
|
+
return self._childless
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def null(self) -> float:
|
|
72
|
+
return self._null
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def placement(self) -> str:
|
|
76
|
+
return self._placement
|
|
77
|
+
|
|
78
|
+
def __repr__(self) -> str:
|
|
79
|
+
return (
|
|
80
|
+
f"FanOut({self._sizes!r}, childless={self._childless!r}, "
|
|
81
|
+
f"null={self._null!r}, placement={self._placement!r})"
|
|
82
|
+
)
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""A fan-out resolved against real parent keys."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import bisect
|
|
6
|
+
from math import gcd
|
|
7
|
+
|
|
8
|
+
from django_data_shape.utils import draw
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class FanOutPlan:
|
|
12
|
+
"""Which parent owns each child row, decided once and answered in O(1).
|
|
13
|
+
|
|
14
|
+
A **partition of the child key range**: parent ``j`` owns rows
|
|
15
|
+
``[starts[j], starts[j + 1])``. That representation is the whole point, and
|
|
16
|
+
it was chosen over the obvious alternative -- drawing a parent per child --
|
|
17
|
+
because a per-child draw **cannot be inverted**. Asking "which children
|
|
18
|
+
belong to parent T" is what a mirrored collection needs, and against a draw
|
|
19
|
+
the only answer is to index every row.
|
|
20
|
+
|
|
21
|
+
Two consequences fall out for free. A childless parent is one whose range is
|
|
22
|
+
empty, so the tail everybody forgets is representable rather than
|
|
23
|
+
approximated. And physical placement becomes a pure question of the order
|
|
24
|
+
rows are *emitted* in, entirely separate from which parent owns them -- the
|
|
25
|
+
same split this design keeps finding between one order and another.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
def __init__(
|
|
29
|
+
self,
|
|
30
|
+
keys: list[int],
|
|
31
|
+
starts: list[int],
|
|
32
|
+
rows: int,
|
|
33
|
+
null_stream: int,
|
|
34
|
+
null_share: float,
|
|
35
|
+
interleave: bool,
|
|
36
|
+
) -> None:
|
|
37
|
+
self._keys = keys
|
|
38
|
+
self._starts = starts
|
|
39
|
+
self._rows = rows
|
|
40
|
+
self._null_stream = null_stream
|
|
41
|
+
self._null_share = null_share
|
|
42
|
+
self._stride = _stride(rows) if interleave else 1
|
|
43
|
+
|
|
44
|
+
def key_for(self, row: int) -> int | None:
|
|
45
|
+
"""The parent key for one child row, or None where the key is null."""
|
|
46
|
+
if self._null_share and draw(self._null_stream, row) < self._null_share:
|
|
47
|
+
return None
|
|
48
|
+
slot = (row * self._stride) % self._rows
|
|
49
|
+
# bisect_right, not left: a parent with an empty range shares its start
|
|
50
|
+
# with the next one, and bisect_right steps past every duplicate to the
|
|
51
|
+
# last parent whose range actually begins at or below the slot. That is
|
|
52
|
+
# what makes a childless parent unreachable rather than special-cased.
|
|
53
|
+
return self._keys[bisect.bisect_right(self._starts, slot) - 1]
|
|
54
|
+
|
|
55
|
+
def sizes(self) -> list[int]:
|
|
56
|
+
"""How many children each parent ended up with, in parent-key order."""
|
|
57
|
+
bounds = [*self._starts, self._rows]
|
|
58
|
+
return [bounds[i + 1] - bounds[i] for i in range(len(self._keys))]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _stride(rows: int) -> int:
|
|
62
|
+
"""A multiplier that walks every slot exactly once, scattering as it goes.
|
|
63
|
+
|
|
64
|
+
``row * stride % rows`` is a bijection precisely when the two are coprime,
|
|
65
|
+
so children land in an order unrelated to their parent without buffering a
|
|
66
|
+
permutation or holding any state. Starting near the golden ratio of ``rows``
|
|
67
|
+
gives the low-discrepancy spread that makes consecutive children come from
|
|
68
|
+
unrelated parents, which is what "arrival order" means physically.
|
|
69
|
+
"""
|
|
70
|
+
if rows < 3:
|
|
71
|
+
return 1
|
|
72
|
+
candidate = max(2, int(rows * 0.6180339887498949))
|
|
73
|
+
while gcd(candidate, rows) != 1:
|
|
74
|
+
candidate += 1
|
|
75
|
+
return candidate
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Turning a declared table into the tuples COPY will consume."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Iterator, Mapping
|
|
6
|
+
from functools import partial
|
|
7
|
+
from typing import Any, cast
|
|
8
|
+
|
|
9
|
+
from django_data_shape.distributions.distribution import Distribution
|
|
10
|
+
from django_data_shape.fan_out_plan import FanOutPlan
|
|
11
|
+
from django_data_shape.table import Table
|
|
12
|
+
from django_data_shape.utils import draw, field_stream
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def generate_rows(
|
|
16
|
+
table: Table, seed: int, plans: Mapping[str, FanOutPlan] | None = None
|
|
17
|
+
) -> Iterator[tuple[Any, ...]]:
|
|
18
|
+
"""Yield one tuple per row: the primary key, then each declared column.
|
|
19
|
+
|
|
20
|
+
Rows, not model instances. The ORM is the wrong tool at these counts -- the
|
|
21
|
+
difference between a load measured in seconds and one measured in minutes --
|
|
22
|
+
and nothing here needs an instance, because no ``save`` will run and no
|
|
23
|
+
signal should fire.
|
|
24
|
+
|
|
25
|
+
Primary keys are a dense ``1..N`` because this package assigns them, which is
|
|
26
|
+
what lets a child's foreign key be satisfied without a lookup and what makes
|
|
27
|
+
a self-referential tree acyclic by construction. It also obliges the caller
|
|
28
|
+
to reset the sequence afterwards; see ``build``.
|
|
29
|
+
|
|
30
|
+
``plans`` carries the resolved fan-out for each relation column. Resolving
|
|
31
|
+
happens outside this function because it has to read the parent's real keys
|
|
32
|
+
out of the database, and keeping the query there leaves generation itself
|
|
33
|
+
backend-neutral and testable without a connection.
|
|
34
|
+
|
|
35
|
+
A generator rather than a list: a million tuples is real memory, psycopg
|
|
36
|
+
writes them one at a time anyway, and materialising the set would buy
|
|
37
|
+
nothing but peak RSS.
|
|
38
|
+
"""
|
|
39
|
+
plans = plans or {}
|
|
40
|
+
# Each column is reduced to one callable of the row index before the loop
|
|
41
|
+
# starts. At a million rows the loop body runs a million times per column, so
|
|
42
|
+
# the branch between a fan-out and a value distribution is worth deciding
|
|
43
|
+
# once rather than a million times.
|
|
44
|
+
emit: list[Callable[[int], object]] = []
|
|
45
|
+
for name, _field in table.columns():
|
|
46
|
+
plan = plans.get(name)
|
|
47
|
+
if plan is not None:
|
|
48
|
+
emit.append(plan.key_for)
|
|
49
|
+
continue
|
|
50
|
+
distribution = cast("Distribution", table.fields[name])
|
|
51
|
+
emit.append(
|
|
52
|
+
partial(_from_distribution, distribution, field_stream(seed, table.db_table, name))
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
for row in range(table.rows):
|
|
56
|
+
yield (row + 1, *(produce(row) for produce in emit))
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _from_distribution(distribution: Distribution, stream: int, row: int) -> object:
|
|
60
|
+
return distribution.value(row, draw(stream, row))
|