django-data-shape 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. django_data_shape-0.2.0/CHANGELOG.md +114 -0
  2. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/PKG-INFO +21 -5
  3. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/README.md +20 -4
  4. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/__init__.py +6 -0
  5. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/build.py +30 -5
  6. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/__init__.py +3 -1
  7. django_data_shape-0.2.0/django_data_shape/distributions/bounded.py +25 -0
  8. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/constant.py +4 -0
  9. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/skew.py +4 -0
  10. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/uniform.py +16 -2
  11. django_data_shape-0.2.0/django_data_shape/distributions/zipf.py +37 -0
  12. django_data_shape-0.2.0/django_data_shape/fan_out.py +82 -0
  13. django_data_shape-0.2.0/django_data_shape/fan_out_plan.py +75 -0
  14. django_data_shape-0.2.0/django_data_shape/generate_rows.py +60 -0
  15. django_data_shape-0.2.0/django_data_shape/order_tables.py +65 -0
  16. django_data_shape-0.2.0/django_data_shape/resolve_fan_out.py +121 -0
  17. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/shape.py +12 -2
  18. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/table.py +100 -16
  19. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/version.py +1 -1
  20. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/docs/index.md +16 -9
  21. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/docs/reference.md +6 -0
  22. django_data_shape-0.2.0/docs/relations.md +118 -0
  23. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/mkdocs.yml +1 -0
  24. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/pyproject.toml +1 -1
  25. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/distributions/test_constant.py +6 -0
  26. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/distributions/test_skew.py +4 -0
  27. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/distributions/test_uniform.py +15 -0
  28. django_data_shape-0.2.0/tests/distributions/test_zipf.py +40 -0
  29. django_data_shape-0.2.0/tests/test_build_graph.py +175 -0
  30. django_data_shape-0.2.0/tests/test_fan_out.py +42 -0
  31. django_data_shape-0.2.0/tests/test_order_tables.py +46 -0
  32. django_data_shape-0.2.0/tests/test_resolve_fan_out.py +141 -0
  33. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_shape.py +8 -0
  34. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_table.py +60 -2
  35. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/testapp/models.py +36 -0
  36. django_data_shape-0.1.0/CHANGELOG.md +0 -53
  37. django_data_shape-0.1.0/django_data_shape/generate_rows.py +0 -43
  38. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/CODEOWNERS +0 -0
  39. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/SECURITY.md +0 -0
  40. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/dependabot.yml +0 -0
  41. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/workflows/release.yml +0 -0
  42. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/workflows/tests.yml +0 -0
  43. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.github/workflows/upstream-drift.yml +0 -0
  44. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.gitignore +0 -0
  45. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/.pre-commit-config.yaml +0 -0
  46. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/CLAUDE.md +0 -0
  47. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/LICENSE +0 -0
  48. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/Makefile +0 -0
  49. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/build_result.py +0 -0
  50. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/distribution.py +0 -0
  51. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/distributions/sequential.py +0 -0
  52. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/invalid_shape.py +0 -0
  53. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/py.typed +0 -0
  54. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/require_postgres.py +0 -0
  55. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/shape_not_empty.py +0 -0
  56. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/table_result.py +0 -0
  57. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/unsupported_backend.py +0 -0
  58. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/django_data_shape/utils.py +0 -0
  59. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/scripts/release-publish.sh +0 -0
  60. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/__init__.py +0 -0
  61. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/conftest.py +0 -0
  62. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/conftest_settings.py +0 -0
  63. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/distributions/__init__.py +0 -0
  64. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/distributions/test_sequential.py +0 -0
  65. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_build.py +0 -0
  66. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_generate_rows.py +0 -0
  67. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_require_postgres.py +0 -0
  68. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_utils.py +0 -0
  69. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/test_version.py +0 -0
  70. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/tests/testapp/__init__.py +0 -0
  71. {django_data_shape-0.1.0 → django_data_shape-0.2.0}/uv.lock +0 -0
@@ -0,0 +1,114 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented in this file.
4
+
5
+ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
6
+ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
+
8
+ ## [Unreleased]
9
+
10
+ ## [0.2.0] — 2026-09-01
11
+
12
+ ### Added
13
+ - `FanOut`, which declares how a foreign key's children spread across their
14
+ parents: a size distribution, a `childless` share for parents with no children
15
+ at all, a `null` share for nullable columns, and `placement`.
16
+ - `Zipf`, the heavy-tailed weight distribution fan-out is realistically drawn
17
+ from. A table where every parent has ten children is not merely tidy -- it is
18
+ the one shape in which the planner is never wrong, because its `n_distinct`
19
+ average is the truth.
20
+ - Tables load in dependency order, and a cycle of fan-outs is refused by name.
21
+
22
+ ### The two representation decisions
23
+ - **Fan-out reads the parent's real keys rather than assuming the dense `1..N`
24
+ range this package assigns.** The case that matters is the hybrid: a project
25
+ builds its fifty companies with the ORM, where the row count is small and the
26
+ ORM is the right tool, and asks this package only for the two million orders.
27
+ Referential integrity then holds by construction, because every key emitted
28
+ came out of the parent table.
29
+ - **A fan-out is a partition of the child key range, not a per-child draw.**
30
+ Parent `j` owns rows `[start, end)`. A per-child draw cannot be inverted, and
31
+ "which children belong to parent T" is what a mirrored collection needs. The
32
+ childless tail and `placement` both fall out of the partition for free.
33
+
34
+ ### Notes
35
+ - `placement` defaults to `arrival`. Emitting children parent by parent gives a
36
+ perfectly clustered table that no production system has and that flatters
37
+ every index scan over the foreign key.
38
+ - A self-referential fan-out is refused: it would read keys from a table still
39
+ empty at load time. Self-referential trees are their own feature.
40
+ - A relation needs a `FanOut` and a plain column refuses one, in both
41
+ directions.
42
+
43
+ ## [0.1.1] — 2026-09-01
44
+
45
+ ### Added
46
+ - `Bounded`, an optional second protocol for distributions that can say how many
47
+ distinct values they produce. `Constant` and `Skew` implement it. It is
48
+ separate from `Distribution` on purpose: adding the method there would make it
49
+ required, so a custom distribution written against the single-method protocol
50
+ would stop satisfying it.
51
+ - A declaration that provably cannot be loaded is now refused at declaration
52
+ time. A `Constant` on a unique column with more than one row, or a `Skew` with
53
+ fewer values than rows, is arithmetic rather than a subtle problem, and it used
54
+ to be discovered by the database partway through a load that had already
55
+ written most of a table. Only single-column uniqueness is checked; multi-column
56
+ constraints are satisfiable through combinations across independently declared
57
+ columns, which is an analysis rather than a comparison.
58
+
59
+ ### Fixed
60
+ - `Uniform` with `places` raised `decimal.InvalidOperation` past 28 significant
61
+ digits -- Python's default context precision -- from inside the `COPY` loop, on
62
+ a column such as `numeric(30, 2)` that would have accepted the value. The
63
+ precision needed is now derived from the declared bounds.
64
+ - `Table` and `Shape` attributes are read-only. Every rule they enforce runs once
65
+ in `__init__`, so while the attributes were writable a declaration could be
66
+ edited afterwards into one that would have been refused, with nothing
67
+ re-checking it.
68
+
69
+ ## [0.1.0] — 2026-09-01
70
+
71
+ ### Added
72
+ - The shape vocabulary: `Shape`, `Table`, and the `Skew`, `Uniform`, `Sequential`
73
+ and `Constant` distributions, behind a single-method `Distribution` protocol.
74
+ - `build()`, which generates rows, loads them with `COPY FROM STDIN`, moves the
75
+ identity sequence past the keys it assigned, and runs `ANALYZE`. The order is
76
+ owned by the library: loading into a table analyzed while empty leaves the
77
+ planner applying old statistics to a new row count, which is a worse lie than
78
+ having no statistics at all.
79
+ - `InvalidShape`, raised at declaration time, and `UnsupportedBackend`, raised
80
+ for any connection that is not PostgreSQL. Generation is backend-neutral;
81
+ `COPY` and planner statistics are not, and degrading quietly would produce the
82
+ false confidence this package exists to remove.
83
+ - Primary keys are assigned as a dense `1..N` range rather than declared. That is
84
+ what will let a foreign key be satisfied without a lookup once relations land,
85
+ and what makes a self-referential tree acyclic by construction.
86
+ - A field carrying a Django `default=` is filled with the value `save()` would
87
+ have written, because defaults are applied by `save()` and nothing here calls
88
+ it. A callable default is refused instead of guessed: `uuid4` varies per row
89
+ and `dict` does not, and nothing on the field distinguishes them.
90
+
91
+ - `ShapeNotEmpty`, raised before anything is written when a target table already
92
+ holds rows. Keys start at 1 on every build, so the collision was previously a
93
+ unique-violation naming an index, which says nothing about what to do instead.
94
+ - The whole build runs in one transaction, so a shape whose second table fails
95
+ leaves nothing behind and can be re-run after a fix.
96
+ - Every declared value passes through its field's `get_db_prep_save`. Without it
97
+ a naive datetime was stored verbatim rather than localised -- hours from where
98
+ `save()` puts it under a non-UTC `TIME_ZONE` -- and a `JSONField` could not be
99
+ written at all.
100
+
101
+ ### Notes
102
+ - Relations are refused in both directions: declaring one raises, and so does
103
+ omitting one that cannot be null. An optional foreign key may be omitted and
104
+ loads entirely `NULL`, which is documented rather than left to be discovered.
105
+ - Only integer primary keys are supported. Any other kind is refused, rather than
106
+ a dense `1..N` integer range being written into a character column.
107
+ - psycopg 3 is required and psycopg 2 is refused by name. Rows stream straight
108
+ into `COPY FROM STDIN`, which psycopg 2 cannot do without materialising them
109
+ first.
110
+
111
+ [Unreleased]: https://github.com/Artui/django-data-shape/compare/v0.2.0...HEAD
112
+ [0.2.0]: https://github.com/Artui/django-data-shape/compare/v0.1.1...v0.2.0
113
+ [0.1.1]: https://github.com/Artui/django-data-shape/compare/v0.1.0...v0.1.1
114
+ [0.1.0]: https://github.com/Artui/django-data-shape/compare/v0.0.0...v0.1.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: django-data-shape
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: A realistically shaped test database from Django models: declare cardinality, skew and fan-out, load by COPY, and make the query planner believe it.
5
5
  Project-URL: Homepage, https://github.com/Artui/django-data-shape
6
6
  Project-URL: Repository, https://github.com/Artui/django-data-shape
@@ -96,12 +96,28 @@ build(shape)
96
96
  past the keys it assigned, and runs `ANALYZE` so the planner can see the shape.
97
97
  It raises on any backend that is not PostgreSQL rather than degrading quietly.
98
98
 
99
+ ## Relations
100
+
101
+ ```python
102
+ Table(
103
+ Order,
104
+ rows=2_000_000,
105
+ # A distribution, not a number: giving every parent ten children is the one
106
+ # shape in which the planner is never wrong, because its n_distinct average
107
+ # is then the truth.
108
+ company=FanOut(Zipf(1.2), childless=0.35),
109
+ ...
110
+ )
111
+ ```
112
+
113
+ The parents can be rows this package built or rows your own code did -- their
114
+ real keys are read, not assumed, so the ORM can own the small tables while this
115
+ owns the large ones.
116
+
99
117
  ## Status
100
118
 
101
- Early. This release covers single tables. Foreign-key fan-out as a distribution,
102
- physical placement, per-group invariants and template-database reuse are the
103
- releases after it; declaring a relation raises today rather than generating ids
104
- that point at nothing.
119
+ Early. Single tables and the model graph. Derived fields, collections copied
120
+ along a join, per-group invariants and template-database reuse come next.
105
121
 
106
122
  Full documentation: <https://artui.github.io/django-data-shape/>
107
123
 
@@ -56,12 +56,28 @@ build(shape)
56
56
  past the keys it assigned, and runs `ANALYZE` so the planner can see the shape.
57
57
  It raises on any backend that is not PostgreSQL rather than degrading quietly.
58
58
 
59
+ ## Relations
60
+
61
+ ```python
62
+ Table(
63
+ Order,
64
+ rows=2_000_000,
65
+ # A distribution, not a number: giving every parent ten children is the one
66
+ # shape in which the planner is never wrong, because its n_distinct average
67
+ # is then the truth.
68
+ company=FanOut(Zipf(1.2), childless=0.35),
69
+ ...
70
+ )
71
+ ```
72
+
73
+ The parents can be rows this package built or rows your own code did -- their
74
+ real keys are read, not assumed, so the ORM can own the small tables while this
75
+ owns the large ones.
76
+
59
77
  ## Status
60
78
 
61
- Early. This release covers single tables. Foreign-key fan-out as a distribution,
62
- physical placement, per-group invariants and template-database reuse are the
63
- releases after it; declaring a relation raises today rather than generating ids
64
- that point at nothing.
79
+ Early. Single tables and the model graph. Derived fields, collections copied
80
+ along a join, per-group invariants and template-database reuse come next.
65
81
 
66
82
  Full documentation: <https://artui.github.io/django-data-shape/>
67
83
 
@@ -2,11 +2,14 @@
2
2
 
3
3
  from django_data_shape.build import build
4
4
  from django_data_shape.build_result import BuildResult
5
+ from django_data_shape.distributions.bounded import Bounded
5
6
  from django_data_shape.distributions.constant import Constant
6
7
  from django_data_shape.distributions.distribution import Distribution
7
8
  from django_data_shape.distributions.sequential import Sequential
8
9
  from django_data_shape.distributions.skew import Skew
9
10
  from django_data_shape.distributions.uniform import Uniform
11
+ from django_data_shape.distributions.zipf import Zipf
12
+ from django_data_shape.fan_out import FanOut
10
13
  from django_data_shape.invalid_shape import InvalidShape
11
14
  from django_data_shape.shape import Shape
12
15
  from django_data_shape.shape_not_empty import ShapeNotEmpty
@@ -16,9 +19,11 @@ from django_data_shape.unsupported_backend import UnsupportedBackend
16
19
  from django_data_shape.version import __version__
17
20
 
18
21
  __all__ = [
22
+ "Bounded",
19
23
  "BuildResult",
20
24
  "Constant",
21
25
  "Distribution",
26
+ "FanOut",
22
27
  "InvalidShape",
23
28
  "Sequential",
24
29
  "Shape",
@@ -27,6 +32,7 @@ __all__ = [
27
32
  "Table",
28
33
  "TableResult",
29
34
  "Uniform",
35
+ "Zipf",
30
36
  "UnsupportedBackend",
31
37
  "__version__",
32
38
  "build",
@@ -2,14 +2,19 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
- from typing import Any
5
+ from typing import Any, cast
6
6
 
7
7
  from django.core.management.color import no_style
8
8
  from django.db import DEFAULT_DB_ALIAS, connections, transaction
9
+ from django.db.models import Model
9
10
 
10
11
  from django_data_shape.build_result import BuildResult
12
+ from django_data_shape.fan_out import FanOut
13
+ from django_data_shape.fan_out_plan import FanOutPlan
11
14
  from django_data_shape.generate_rows import generate_rows
15
+ from django_data_shape.order_tables import order_tables
12
16
  from django_data_shape.require_postgres import require_postgres
17
+ from django_data_shape.resolve_fan_out import resolve_fan_out
13
18
  from django_data_shape.shape import Shape
14
19
  from django_data_shape.shape_not_empty import ShapeNotEmpty
15
20
  from django_data_shape.table import Table
@@ -41,15 +46,35 @@ def build(shape: Shape, using: str = DEFAULT_DB_ALIAS) -> BuildResult:
41
46
  # -- fix the shape, run it again -- fails on a duplicate key rather than on
42
47
  # the original problem.
43
48
  with transaction.atomic(using=using):
44
- for table in shape.tables:
49
+ # Parents first. Not because the database insists -- Django's foreign
50
+ # keys are deferred, so any order commits -- but because a fan-out reads
51
+ # its parent's real keys, and a table with no rows yet has none.
52
+ for table in order_tables(shape.tables):
45
53
  _require_empty(connection, table)
46
- loaded = _load(connection, table, shape.seed)
54
+ plans = _resolve(connection, table, shape.seed)
55
+ loaded = _load(connection, table, shape.seed, plans)
47
56
  _reset_sequence(connection, table)
48
57
  _analyze(connection, table)
49
58
  results.append(TableResult(table=table.db_table, rows=loaded))
50
59
  return BuildResult(tables=tuple(results))
51
60
 
52
61
 
62
+ def _resolve(connection: Any, table: Table, seed: int) -> dict[str, FanOutPlan]:
63
+ """Partition each declared relation over the parent keys that exist."""
64
+ plans: dict[str, FanOutPlan] = {}
65
+ for name, field in table.relations():
66
+ plans[name] = resolve_fan_out(
67
+ cast("FanOut", table.fields[name]),
68
+ cast("type[Model]", field.related_model),
69
+ table.rows,
70
+ seed,
71
+ table.db_table,
72
+ name,
73
+ connection,
74
+ )
75
+ return plans
76
+
77
+
53
78
  def _require_empty(connection: Any, table: Table) -> None:
54
79
  """Refuse to build on top of rows that are already there.
55
80
 
@@ -72,7 +97,7 @@ def _require_empty(connection: Any, table: Table) -> None:
72
97
  # API reached through Django's cursor wrapper, and it appears on no Django base
73
98
  # class, so annotating the real wrapper type would mean asserting the checker
74
99
  # out of the way on every line that uses it.
75
- def _load(connection: Any, table: Table, seed: int) -> int:
100
+ def _load(connection: Any, table: Table, seed: int, plans: dict[str, FanOutPlan]) -> int:
76
101
  """Stream generated rows into the table with ``COPY FROM STDIN``.
77
102
 
78
103
  ``bulk_create`` is the obvious alternative and is roughly an order of
@@ -105,7 +130,7 @@ def _load(connection: Any, table: Table, seed: int) -> int:
105
130
  # block never learns it needs a rollback, so the next query inside it
106
131
  # fails with "current transaction is aborted" instead of a Django error.
107
132
  with connection.wrap_database_errors, cursor.copy(statement) as copy:
108
- for row in generate_rows(table, seed):
133
+ for row in generate_rows(table, seed, plans):
109
134
  copy.write_row(
110
135
  (
111
136
  row[0],
@@ -1,9 +1,11 @@
1
1
  """The declared value distributions."""
2
2
 
3
+ from django_data_shape.distributions.bounded import Bounded
3
4
  from django_data_shape.distributions.constant import Constant
4
5
  from django_data_shape.distributions.distribution import Distribution
5
6
  from django_data_shape.distributions.sequential import Sequential
6
7
  from django_data_shape.distributions.skew import Skew
7
8
  from django_data_shape.distributions.uniform import Uniform
9
+ from django_data_shape.distributions.zipf import Zipf
8
10
 
9
- __all__ = ["Constant", "Distribution", "Sequential", "Skew", "Uniform"]
11
+ __all__ = ["Bounded", "Constant", "Distribution", "Sequential", "Skew", "Uniform", "Zipf"]
@@ -0,0 +1,25 @@
1
+ """Distributions that can only ever produce so many different values."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Protocol, runtime_checkable
6
+
7
+
8
+ @runtime_checkable
9
+ class Bounded(Protocol):
10
+ """A distribution with a known, finite number of distinct values.
11
+
12
+ Deliberately a **second** protocol rather than a method on
13
+ :class:`~django_data_shape.distributions.distribution.Distribution`. Adding
14
+ it there would make it required, and a custom distribution written against
15
+ the single-method protocol would stop satisfying it -- so the one thing this
16
+ exists to prevent, a declaration that cannot describe a database, would be
17
+ bought by breaking every declaration someone had already written.
18
+
19
+ Structural and runtime-checkable, so a distribution opts in by having the
20
+ method and nothing has to register anywhere. A distribution that cannot
21
+ answer -- one drawing from a continuous range, say -- simply does not
22
+ implement it, and is treated as unbounded rather than as suspicious.
23
+ """
24
+
25
+ def distinct_values(self) -> int: ...
@@ -22,5 +22,9 @@ class Constant:
22
22
  def value(self, row: int, draw: float) -> object:
23
23
  return self._value
24
24
 
25
+ def distinct_values(self) -> int:
26
+ """One, by definition. See ``Bounded``."""
27
+ return 1
28
+
25
29
  def __repr__(self) -> str:
26
30
  return f"Constant({self._value!r})"
@@ -67,5 +67,9 @@ class Skew:
67
67
  # is cheaper than renormalising on every draw.
68
68
  return self._values[-1]
69
69
 
70
+ def distinct_values(self) -> int:
71
+ """How many values were declared. See ``Bounded``."""
72
+ return len(self._values)
73
+
70
74
  def __repr__(self) -> str:
71
75
  return f"Skew({self._weights!r})"
@@ -3,7 +3,7 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import math
6
- from decimal import Decimal
6
+ from decimal import Decimal, localcontext
7
7
 
8
8
  from django_data_shape.invalid_shape import InvalidShape
9
9
 
@@ -38,12 +38,26 @@ class Uniform:
38
38
  self._low = low
39
39
  self._high = high
40
40
  self._places = places
41
+ # Python's default decimal context carries 28 significant digits, and
42
+ # rounding past it raises InvalidOperation -- which used to surface from
43
+ # inside the COPY loop, on a numeric(30, 2) column that would have
44
+ # accepted the value perfectly well. The precision needed is the integer
45
+ # digits of the widest bound plus the decimal places, and the guard
46
+ # covers the rounding step itself.
47
+ magnitude = max(abs(low), abs(high), 1.0)
48
+ self._precision = max(28, int(math.log10(magnitude)) + (places or 0) + 5)
41
49
 
42
50
  def value(self, row: int, draw: float) -> object:
43
51
  raw = self._low + draw * (self._high - self._low)
44
52
  if self._places is None:
45
53
  return raw
46
- return round(Decimal(repr(raw)), self._places)
54
+ with localcontext() as context:
55
+ context.prec = self._precision
56
+ # repr() rather than the float itself: Decimal(float) takes the full
57
+ # binary expansion, which rounds tie cases the other way --
58
+ # Decimal(2.675) rounds to 2.67 where Decimal(repr(2.675)) gives the
59
+ # 2.68 a reader expects from the literal they wrote.
60
+ return round(Decimal(repr(raw)), self._places)
47
61
 
48
62
  def __repr__(self) -> str:
49
63
  places = "" if self._places is None else f", places={self._places}"
@@ -0,0 +1,37 @@
1
+ """Heavy-tailed weights: a few very large, a long tail of very small."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+
7
+ from django_data_shape.invalid_shape import InvalidShape
8
+
9
+
10
+ class Zipf:
11
+ """Positive weights following a power law of exponent ``s``.
12
+
13
+ The distribution fan-out is realistically drawn from, and the reason
14
+ declaring fan-out is worth doing at all. A customer table where every
15
+ customer has ten orders is not merely tidy, it is the one shape in which the
16
+ planner is never wrong: its ``n_distinct`` average *is* the truth, so a join
17
+ estimate cannot miss. Give the head a thousand orders and the tail one, and
18
+ the same estimate is out by orders of magnitude in both directions -- which
19
+ is what production looks like and what a test database has to reproduce.
20
+
21
+ Inverse transform of a Pareto: ``(1 - draw) ** (-1 / s)``. Larger ``s``
22
+ means a lighter tail; values near 1 are the classic Zipf regime.
23
+ """
24
+
25
+ def __init__(self, s: float = 1.2) -> None:
26
+ if not math.isfinite(s) or s <= 0:
27
+ raise InvalidShape(f"Zipf needs a positive finite exponent, got s={s}.")
28
+ self._s = s
29
+
30
+ def value(self, row: int, draw: float) -> object:
31
+ # ``1 - draw`` rather than ``draw`` so the singularity sits at the top of
32
+ # the interval: draw is in [0, 1), so 1 - draw is in (0, 1] and never
33
+ # zero, which keeps the power finite for every possible draw.
34
+ return (1.0 - draw) ** (-1.0 / self._s)
35
+
36
+ def __repr__(self) -> str:
37
+ return f"Zipf({self._s!r})"
@@ -0,0 +1,82 @@
1
+ """How a foreign key's children are spread across their parents."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from django_data_shape.distributions.distribution import Distribution
6
+ from django_data_shape.invalid_shape import InvalidShape
7
+
8
+ PLACEMENTS = ("arrival", "grouped")
9
+
10
+
11
+ class FanOut:
12
+ """A distribution over how many children each parent has.
13
+
14
+ Declared as a shape over an **already fixed** child row count, never as a
15
+ multiplier. Rows times fan-out makes cardinality emergent, and a table whose
16
+ size is emergent is a table nothing can assert about.
17
+
18
+ ``sizes`` supplies a relative weight per parent -- ``Zipf()`` for the
19
+ realistic heavy tail, ``Uniform(1, 10)`` for something flatter. The weights
20
+ are normalised, so their scale is irrelevant and only their spread matters.
21
+
22
+ ``childless`` is the share of parents with **no** children at all. It is
23
+ called out separately because it is the case hand-written fixtures always
24
+ omit and the one that changes what a join does: a parent nobody references
25
+ is the difference between an inner and an outer join returning the same
26
+ thing and returning different things.
27
+
28
+ ``null`` is the share of children whose foreign key is NULL, which is only
29
+ meaningful on a nullable column. **It thins the partition uniformly after**
30
+ it is computed, so ``sizes`` describes the pre-null spread. Stated because
31
+ the alternative -- partitioning only the non-null children -- would make the
32
+ declared distribution mean something subtly different from what it says.
33
+
34
+ ``placement`` decides where children sit physically, and it is not
35
+ cosmetic. Emitting them parent by parent gives a perfectly clustered table
36
+ that no production system has and that flatters every index scan; the
37
+ default interleaves them the way rows really arrive.
38
+ """
39
+
40
+ def __init__(
41
+ self,
42
+ sizes: Distribution,
43
+ *,
44
+ childless: float = 0.0,
45
+ null: float = 0.0,
46
+ placement: str = "arrival",
47
+ ) -> None:
48
+ for name, share in (("childless", childless), ("null", null)):
49
+ if not 0.0 <= share < 1.0:
50
+ raise InvalidShape(
51
+ f"FanOut {name} is a share of the whole and must be in [0, 1), got {share}."
52
+ )
53
+ if placement not in PLACEMENTS:
54
+ raise InvalidShape(
55
+ f"FanOut placement must be one of {', '.join(PLACEMENTS)}, got {placement!r}."
56
+ )
57
+ self._sizes = sizes
58
+ self._childless = childless
59
+ self._null = null
60
+ self._placement = placement
61
+
62
+ @property
63
+ def sizes(self) -> Distribution:
64
+ return self._sizes
65
+
66
+ @property
67
+ def childless(self) -> float:
68
+ return self._childless
69
+
70
+ @property
71
+ def null(self) -> float:
72
+ return self._null
73
+
74
+ @property
75
+ def placement(self) -> str:
76
+ return self._placement
77
+
78
+ def __repr__(self) -> str:
79
+ return (
80
+ f"FanOut({self._sizes!r}, childless={self._childless!r}, "
81
+ f"null={self._null!r}, placement={self._placement!r})"
82
+ )
@@ -0,0 +1,75 @@
1
+ """A fan-out resolved against real parent keys."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import bisect
6
+ from math import gcd
7
+
8
+ from django_data_shape.utils import draw
9
+
10
+
11
+ class FanOutPlan:
12
+ """Which parent owns each child row, decided once and answered in O(1).
13
+
14
+ A **partition of the child key range**: parent ``j`` owns rows
15
+ ``[starts[j], starts[j + 1])``. That representation is the whole point, and
16
+ it was chosen over the obvious alternative -- drawing a parent per child --
17
+ because a per-child draw **cannot be inverted**. Asking "which children
18
+ belong to parent T" is what a mirrored collection needs, and against a draw
19
+ the only answer is to index every row.
20
+
21
+ Two consequences fall out for free. A childless parent is one whose range is
22
+ empty, so the tail everybody forgets is representable rather than
23
+ approximated. And physical placement becomes a pure question of the order
24
+ rows are *emitted* in, entirely separate from which parent owns them -- the
25
+ same split this design keeps finding between one order and another.
26
+ """
27
+
28
+ def __init__(
29
+ self,
30
+ keys: list[int],
31
+ starts: list[int],
32
+ rows: int,
33
+ null_stream: int,
34
+ null_share: float,
35
+ interleave: bool,
36
+ ) -> None:
37
+ self._keys = keys
38
+ self._starts = starts
39
+ self._rows = rows
40
+ self._null_stream = null_stream
41
+ self._null_share = null_share
42
+ self._stride = _stride(rows) if interleave else 1
43
+
44
+ def key_for(self, row: int) -> int | None:
45
+ """The parent key for one child row, or None where the key is null."""
46
+ if self._null_share and draw(self._null_stream, row) < self._null_share:
47
+ return None
48
+ slot = (row * self._stride) % self._rows
49
+ # bisect_right, not left: a parent with an empty range shares its start
50
+ # with the next one, and bisect_right steps past every duplicate to the
51
+ # last parent whose range actually begins at or below the slot. That is
52
+ # what makes a childless parent unreachable rather than special-cased.
53
+ return self._keys[bisect.bisect_right(self._starts, slot) - 1]
54
+
55
+ def sizes(self) -> list[int]:
56
+ """How many children each parent ended up with, in parent-key order."""
57
+ bounds = [*self._starts, self._rows]
58
+ return [bounds[i + 1] - bounds[i] for i in range(len(self._keys))]
59
+
60
+
61
+ def _stride(rows: int) -> int:
62
+ """A multiplier that walks every slot exactly once, scattering as it goes.
63
+
64
+ ``row * stride % rows`` is a bijection precisely when the two are coprime,
65
+ so children land in an order unrelated to their parent without buffering a
66
+ permutation or holding any state. Starting near the golden ratio of ``rows``
67
+ gives the low-discrepancy spread that makes consecutive children come from
68
+ unrelated parents, which is what "arrival order" means physically.
69
+ """
70
+ if rows < 3:
71
+ return 1
72
+ candidate = max(2, int(rows * 0.6180339887498949))
73
+ while gcd(candidate, rows) != 1:
74
+ candidate += 1
75
+ return candidate
@@ -0,0 +1,60 @@
1
+ """Turning a declared table into the tuples COPY will consume."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable, Iterator, Mapping
6
+ from functools import partial
7
+ from typing import Any, cast
8
+
9
+ from django_data_shape.distributions.distribution import Distribution
10
+ from django_data_shape.fan_out_plan import FanOutPlan
11
+ from django_data_shape.table import Table
12
+ from django_data_shape.utils import draw, field_stream
13
+
14
+
15
+ def generate_rows(
16
+ table: Table, seed: int, plans: Mapping[str, FanOutPlan] | None = None
17
+ ) -> Iterator[tuple[Any, ...]]:
18
+ """Yield one tuple per row: the primary key, then each declared column.
19
+
20
+ Rows, not model instances. The ORM is the wrong tool at these counts -- the
21
+ difference between a load measured in seconds and one measured in minutes --
22
+ and nothing here needs an instance, because no ``save`` will run and no
23
+ signal should fire.
24
+
25
+ Primary keys are a dense ``1..N`` because this package assigns them, which is
26
+ what lets a child's foreign key be satisfied without a lookup and what makes
27
+ a self-referential tree acyclic by construction. It also obliges the caller
28
+ to reset the sequence afterwards; see ``build``.
29
+
30
+ ``plans`` carries the resolved fan-out for each relation column. Resolving
31
+ happens outside this function because it has to read the parent's real keys
32
+ out of the database, and keeping the query there leaves generation itself
33
+ backend-neutral and testable without a connection.
34
+
35
+ A generator rather than a list: a million tuples is real memory, psycopg
36
+ writes them one at a time anyway, and materialising the set would buy
37
+ nothing but peak RSS.
38
+ """
39
+ plans = plans or {}
40
+ # Each column is reduced to one callable of the row index before the loop
41
+ # starts. At a million rows the loop body runs a million times per column, so
42
+ # the branch between a fan-out and a value distribution is worth deciding
43
+ # once rather than a million times.
44
+ emit: list[Callable[[int], object]] = []
45
+ for name, _field in table.columns():
46
+ plan = plans.get(name)
47
+ if plan is not None:
48
+ emit.append(plan.key_for)
49
+ continue
50
+ distribution = cast("Distribution", table.fields[name])
51
+ emit.append(
52
+ partial(_from_distribution, distribution, field_stream(seed, table.db_table, name))
53
+ )
54
+
55
+ for row in range(table.rows):
56
+ yield (row + 1, *(produce(row) for produce in emit))
57
+
58
+
59
+ def _from_distribution(distribution: Distribution, stream: int, row: int) -> object:
60
+ return distribution.value(row, draw(stream, row))