dataframely 2.3.1__tar.gz → 2.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. {dataframely-2.3.1 → dataframely-2.5.0}/.github/copilot-instructions.md +23 -0
  2. {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/build.yml +8 -8
  3. {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/ci.yml +7 -7
  4. {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/copilot-setup-steps.yml +3 -3
  5. {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/nightly.yml +2 -2
  6. {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/scorecard.yml +3 -3
  7. {dataframely-2.3.1 → dataframely-2.5.0}/PKG-INFO +1 -1
  8. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_base_schema.py +13 -0
  9. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_compat.py +4 -1
  10. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/array.py +12 -3
  11. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/binary.py +6 -4
  12. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/decimal.py +5 -1
  13. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/list.py +13 -5
  14. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/string.py +4 -2
  15. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/struct.py +7 -4
  16. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/schema.py +58 -19
  17. {dataframely-2.3.1 → dataframely-2.5.0}/pixi.lock +7506 -7616
  18. {dataframely-2.3.1 → dataframely-2.5.0}/pyproject.toml +1 -1
  19. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_sample.py +13 -0
  20. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_decimal.py +56 -1
  21. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_list.py +13 -0
  22. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_sqlalchemy_columns.py +9 -4
  23. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_str.py +17 -0
  24. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_base.py +16 -0
  25. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_sample.py +26 -17
  26. {dataframely-2.3.1 → dataframely-2.5.0}/.copier-answers.yml +0 -0
  27. {dataframely-2.3.1 → dataframely-2.5.0}/.envrc +0 -0
  28. {dataframely-2.3.1 → dataframely-2.5.0}/.gitattributes +0 -0
  29. {dataframely-2.3.1 → dataframely-2.5.0}/.github/CODEOWNERS +0 -0
  30. {dataframely-2.3.1 → dataframely-2.5.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  31. {dataframely-2.3.1 → dataframely-2.5.0}/.github/dependabot.yml +0 -0
  32. {dataframely-2.3.1 → dataframely-2.5.0}/.github/instructions/tests.instructions.md +0 -0
  33. {dataframely-2.3.1 → dataframely-2.5.0}/.github/release-drafter.yml +0 -0
  34. {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/chore.yml +0 -0
  35. {dataframely-2.3.1 → dataframely-2.5.0}/.gitignore +0 -0
  36. {dataframely-2.3.1 → dataframely-2.5.0}/.pre-commit-config.yaml +0 -0
  37. {dataframely-2.3.1 → dataframely-2.5.0}/.prettierignore +0 -0
  38. {dataframely-2.3.1 → dataframely-2.5.0}/.prettierrc +0 -0
  39. {dataframely-2.3.1 → dataframely-2.5.0}/.readthedocs.yml +0 -0
  40. {dataframely-2.3.1 → dataframely-2.5.0}/Cargo.lock +0 -0
  41. {dataframely-2.3.1 → dataframely-2.5.0}/Cargo.toml +0 -0
  42. {dataframely-2.3.1 → dataframely-2.5.0}/LICENSE +0 -0
  43. {dataframely-2.3.1 → dataframely-2.5.0}/README.md +0 -0
  44. {dataframely-2.3.1 → dataframely-2.5.0}/SECURITY.md +0 -0
  45. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/__init__.py +0 -0
  46. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_deprecation.py +0 -0
  47. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_filter.py +0 -0
  48. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_match_to_schema.py +0 -0
  49. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_native.pyi +0 -0
  50. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_plugin.py +0 -0
  51. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_polars.py +0 -0
  52. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_pydantic.py +0 -0
  53. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_rule.py +0 -0
  54. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_serialization.py +0 -0
  55. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/__init__.py +0 -0
  56. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/_base.py +0 -0
  57. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/_exc.py +0 -0
  58. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/constants.py +0 -0
  59. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/delta.py +0 -0
  60. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/parquet.py +0 -0
  61. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_typing.py +0 -0
  62. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/collection/__init__.py +0 -0
  63. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/collection/_base.py +0 -0
  64. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/collection/collection.py +0 -0
  65. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/collection/filter_result.py +0 -0
  66. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/__init__.py +0 -0
  67. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/_base.py +0 -0
  68. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/_mixins.py +0 -0
  69. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/_registry.py +0 -0
  70. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/_utils.py +0 -0
  71. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/any.py +0 -0
  72. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/bool.py +0 -0
  73. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/categorical.py +0 -0
  74. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/datetime.py +0 -0
  75. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/enum.py +0 -0
  76. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/float.py +0 -0
  77. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/integer.py +0 -0
  78. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/object.py +0 -0
  79. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/config.py +0 -0
  80. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/exc.py +0 -0
  81. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/filter_result.py +0 -0
  82. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/functional.py +0 -0
  83. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/py.typed +0 -0
  84. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/random.py +0 -0
  85. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/__init__.py +0 -0
  86. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/const.py +0 -0
  87. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/factory.py +0 -0
  88. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/mask.py +0 -0
  89. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/rules.py +0 -0
  90. {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/storage.py +0 -0
  91. {dataframely-2.3.1 → dataframely-2.5.0}/docker-compose.yml +0 -0
  92. {dataframely-2.3.1 → dataframely-2.5.0}/docs/_static/custom.css +0 -0
  93. {dataframely-2.3.1 → dataframely-2.5.0}/docs/_static/favicon.ico +0 -0
  94. {dataframely-2.3.1 → dataframely-2.5.0}/docs/_templates/autosummary/class.rst +0 -0
  95. {dataframely-2.3.1 → dataframely-2.5.0}/docs/_templates/autosummary/method.rst +0 -0
  96. {dataframely-2.3.1 → dataframely-2.5.0}/docs/_templates/classes/column.rst +0 -0
  97. {dataframely-2.3.1 → dataframely-2.5.0}/docs/_templates/classes/error.rst +0 -0
  98. {dataframely-2.3.1 → dataframely-2.5.0}/docs/_templates/classes/filter_result.rst +0 -0
  99. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/generation.rst +0 -0
  100. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/index.rst +0 -0
  101. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/io.rst +0 -0
  102. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/metadata.rst +0 -0
  103. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/operations.rst +0 -0
  104. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/validation.rst +0 -0
  105. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/columns/index.rst +0 -0
  106. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/errors/index.rst +0 -0
  107. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/filter_result/failure_info.rst +0 -0
  108. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/filter_result/index.rst +0 -0
  109. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/index.rst +0 -0
  110. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/misc/index.rst +0 -0
  111. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/conversion.rst +0 -0
  112. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/generation.rst +0 -0
  113. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/index.rst +0 -0
  114. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/io.rst +0 -0
  115. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/metadata.rst +0 -0
  116. {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/validation.rst +0 -0
  117. {dataframely-2.3.1 → dataframely-2.5.0}/docs/conf.py +0 -0
  118. {dataframely-2.3.1 → dataframely-2.5.0}/docs/css/custom.css +0 -0
  119. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/development.md +0 -0
  120. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/examples/index.md +0 -0
  121. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/examples/real-world.ipynb +0 -0
  122. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/faq.md +0 -0
  123. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/column-metadata.md +0 -0
  124. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/data-generation.md +0 -0
  125. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/index.md +0 -0
  126. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/lazy-validation.md +0 -0
  127. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/primary-keys.md +0 -0
  128. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/serialization.md +0 -0
  129. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/sql-generation.md +0 -0
  130. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/index.md +0 -0
  131. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/migration/index.md +0 -0
  132. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/migration/v1-v2.md +0 -0
  133. {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/quickstart.md +0 -0
  134. {dataframely-2.3.1 → dataframely-2.5.0}/docs/index.md +0 -0
  135. {dataframely-2.3.1 → dataframely-2.5.0}/pixi.toml +0 -0
  136. {dataframely-2.3.1 → dataframely-2.5.0}/rust-toolchain.toml +0 -0
  137. {dataframely-2.3.1 → dataframely-2.5.0}/src/lib.rs +0 -0
  138. {dataframely-2.3.1 → dataframely-2.5.0}/src/polars_plugin/mod.rs +0 -0
  139. {dataframely-2.3.1 → dataframely-2.5.0}/src/polars_plugin/rule_failure.rs +0 -0
  140. {dataframely-2.3.1 → dataframely-2.5.0}/src/polars_plugin/utils.rs +0 -0
  141. {dataframely-2.3.1 → dataframely-2.5.0}/src/polars_plugin/validation_error.rs +0 -0
  142. {dataframely-2.3.1 → dataframely-2.5.0}/src/regex/errdefs.rs +0 -0
  143. {dataframely-2.3.1 → dataframely-2.5.0}/src/regex/mod.rs +0 -0
  144. {dataframely-2.3.1 → dataframely-2.5.0}/src/regex/repr.rs +0 -0
  145. {dataframely-2.3.1 → dataframely-2.5.0}/tests/benches/conftest.py +0 -0
  146. {dataframely-2.3.1 → dataframely-2.5.0}/tests/benches/test_collection.py +0 -0
  147. {dataframely-2.3.1 → dataframely-2.5.0}/tests/benches/test_failure.py +0 -0
  148. {dataframely-2.3.1 → dataframely-2.5.0}/tests/benches/test_schema.py +0 -0
  149. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_base.py +0 -0
  150. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_cast.py +0 -0
  151. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_collection_future_annotations.py +0 -0
  152. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_create_empty.py +0 -0
  153. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_filter_one_to_n.py +0 -0
  154. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_filter_validate.py +0 -0
  155. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_ignore_in_filter.py +0 -0
  156. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_implementation.py +0 -0
  157. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_join.py +0 -0
  158. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_matches.py +0 -0
  159. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_optional_members.py +0 -0
  160. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_propagate_row_failures.py +0 -0
  161. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_repr.py +0 -0
  162. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_serialization.py +0 -0
  163. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_storage.py +0 -0
  164. {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_validate_input.py +0 -0
  165. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/__init__.py +0 -0
  166. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_any.py +0 -0
  167. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_array.py +0 -0
  168. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_binary.py +0 -0
  169. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_datetime.py +0 -0
  170. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_enum.py +0 -0
  171. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_float.py +0 -0
  172. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_integer.py +0 -0
  173. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_object.py +0 -0
  174. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_string.py +0 -0
  175. {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_struct.py +0 -0
  176. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/__init__.py +0 -0
  177. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_alias.py +0 -0
  178. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_base.py +0 -0
  179. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_check.py +0 -0
  180. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_default_dtypes.py +0 -0
  181. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_matches.py +0 -0
  182. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_metadata.py +0 -0
  183. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_polars_schema.py +0 -0
  184. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_pyarrow.py +0 -0
  185. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_rules.py +0 -0
  186. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_sample.py +0 -0
  187. {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_utils.py +0 -0
  188. {dataframely-2.3.1 → dataframely-2.5.0}/tests/conftest.py +0 -0
  189. {dataframely-2.3.1 → dataframely-2.5.0}/tests/core_validation/__init__.py +0 -0
  190. {dataframely-2.3.1 → dataframely-2.5.0}/tests/core_validation/test_match_to_schema.py +0 -0
  191. {dataframely-2.3.1 → dataframely-2.5.0}/tests/core_validation/test_rule_evaluation.py +0 -0
  192. {dataframely-2.3.1 → dataframely-2.5.0}/tests/failure_info/test_storage.py +0 -0
  193. {dataframely-2.3.1 → dataframely-2.5.0}/tests/functional/test_concat.py +0 -0
  194. {dataframely-2.3.1 → dataframely-2.5.0}/tests/functional/test_relationships.py +0 -0
  195. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_cast.py +0 -0
  196. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_create_empty.py +0 -0
  197. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_create_empty_if_none.py +0 -0
  198. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_filter.py +0 -0
  199. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_inheritance.py +0 -0
  200. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_matches.py +0 -0
  201. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_read_write_parquet.py +0 -0
  202. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_repr.py +0 -0
  203. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_rule_implementation.py +0 -0
  204. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_serialization.py +0 -0
  205. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_storage.py +0 -0
  206. {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_validate.py +0 -0
  207. {dataframely-2.3.1 → dataframely-2.5.0}/tests/storage/test_delta.py +0 -0
  208. {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_compat.py +0 -0
  209. {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_config.py +0 -0
  210. {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_deprecation.py +0 -0
  211. {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_factory.py +0 -0
  212. {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_native_regex.py +0 -0
  213. {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_pydantic.py +0 -0
  214. {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_random.py +0 -0
  215. {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_serialization.py +0 -0
  216. {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_typing.py +0 -0
@@ -206,6 +206,29 @@ validated_df: dy.DataFrame[MySchema] = MySchema.validate(df, cast=True)
206
206
  4. **Documentation**: Update docstrings
207
207
  5. **API changes**: Ensure backward compatibility or document migration path
208
208
 
209
+ ### Pull request titles (required)
210
+
211
+ Pull request titles must follow the Conventional Commits format: `<type>[!]: <Subject>`
212
+
213
+ Allowed `type` values:
214
+
215
+ - `feat`: A new feature
216
+ - `fix`: A bug fix
217
+ - `docs`: Documentation only changes
218
+ - `style`: Changes that do not affect the meaning of the code (white-space, formatting, missing semi-colons, etc)
219
+ - `refactor`: A code change that neither fixes a bug nor adds a feature
220
+ - `perf`: A code change that improves performance
221
+ - `test`: Adding missing tests or correcting existing tests
222
+ - `build`: Changes that affect the build system or external dependencies
223
+ - `ci`: Changes to our CI configuration files and scripts
224
+ - `chore`: Other changes that don't modify src or test files
225
+ - `revert`: Reverts a previous commit
226
+
227
+ Additional rules:
228
+
229
+ - Use `!` only for **breaking changes**
230
+ - `Subject` must start with an **uppercase** letter and must **not** end with `.` or a trailing space
231
+
209
232
  ## Performance Considerations
210
233
 
211
234
  - Validation uses native polars expressions for performance
@@ -13,11 +13,11 @@ jobs:
13
13
  permissions:
14
14
  contents: read
15
15
  steps:
16
- - uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
16
+ - uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
17
17
  with:
18
18
  fetch-depth: 0
19
19
  - name: Set up pixi
20
- uses: prefix-dev/setup-pixi@28eb668aafebd9dede9d97c4ba1cd9989a4d0004 # v0.9.2
20
+ uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
21
21
  with:
22
22
  environments: build
23
23
  - name: Set version
@@ -25,7 +25,7 @@ jobs:
25
25
  - name: Build project
26
26
  run: pixi run -e build build-sdist
27
27
  - name: Upload package
28
- uses: actions/upload-artifact@330a01c490aca151604b8cf639adc76d48f6c5d4 # v5.0.0
28
+ uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0
29
29
  with:
30
30
  name: sdist
31
31
  path: dist/*
@@ -48,16 +48,16 @@ jobs:
48
48
  - target-platform: win-64
49
49
  os: windows-latest
50
50
  steps:
51
- - uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
51
+ - uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
52
52
  with:
53
53
  fetch-depth: 0
54
54
  - name: Set up pixi
55
- uses: prefix-dev/setup-pixi@28eb668aafebd9dede9d97c4ba1cd9989a4d0004 # v0.9.2
55
+ uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
56
56
  with:
57
57
  environments: build
58
58
  - name: Set version
59
59
  run: pixi run -e build set-version
60
- - uses: actions/setup-python@e797f83bcb11b83ae66e0230d6156d7c80228e7c # v6.0.0
60
+ - uses: actions/setup-python@83679a892e2d95755f2dac6acb0bfd1e9ac5d548 # v6.1.0
61
61
  with:
62
62
  python-version: "3.10"
63
63
  - name: Build wheel
@@ -70,7 +70,7 @@ jobs:
70
70
  - name: Check package
71
71
  run: pixi run -e build check-wheel
72
72
  - name: Upload package
73
- uses: actions/upload-artifact@330a01c490aca151604b8cf639adc76d48f6c5d4 # v5.0.0
73
+ uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0
74
74
  with:
75
75
  name: wheel-${{ matrix.target-platform }}
76
76
  path: dist/*
@@ -84,7 +84,7 @@ jobs:
84
84
  id-token: write
85
85
  environment: pypi
86
86
  steps:
87
- - uses: actions/download-artifact@018cc2cf5baa6db3ef3c5f8a56943fffe632ef53 # v6.0.0
87
+ - uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7.0.0
88
88
  with:
89
89
  path: dist
90
90
  merge-multiple: true
@@ -19,18 +19,18 @@ jobs:
19
19
  runs-on: ubuntu-latest
20
20
  steps:
21
21
  - name: Checkout branch
22
- uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
22
+ uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
23
23
  with:
24
24
  # needed for 'pre-commit-mirrors-insert-license'
25
25
  fetch-depth: 0
26
26
  - name: Set up pixi
27
- uses: prefix-dev/setup-pixi@28eb668aafebd9dede9d97c4ba1cd9989a4d0004 # v0.9.2
27
+ uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
28
28
  with:
29
29
  environments: default lint
30
30
  - name: Install Rust
31
31
  run: rustup show
32
32
  - name: Cache Rust dependencies
33
- uses: Swatinem/rust-cache@f13886b937689c021905a6b90929199931d60db1 # v2.8.1
33
+ uses: Swatinem/rust-cache@779680da715d629ac1d338a641029a2f4372abb5 # v2.8.2
34
34
  - name: pre-commit
35
35
  run: pixi run pre-commit-run --color=always --show-diff-on-failure
36
36
 
@@ -56,9 +56,9 @@ jobs:
56
56
  with_optionals: true
57
57
  steps:
58
58
  - name: Checkout branch
59
- uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
59
+ uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
60
60
  - name: Set up pixi
61
- uses: prefix-dev/setup-pixi@28eb668aafebd9dede9d97c4ba1cd9989a4d0004 # v0.9.2
61
+ uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
62
62
  with:
63
63
  environments: ${{ matrix.environment }}
64
64
  # FIXME: Remove when `s3_server` fixture does not start a process anymore
@@ -66,13 +66,13 @@ jobs:
66
66
  - name: Install Rust
67
67
  run: rustup show
68
68
  - name: Cache Rust dependencies
69
- uses: Swatinem/rust-cache@f13886b937689c021905a6b90929199931d60db1 # v2.8.1
69
+ uses: Swatinem/rust-cache@779680da715d629ac1d338a641029a2f4372abb5 # v2.8.2
70
70
  - name: Install repository
71
71
  run: pixi run -e ${{ matrix.environment }} postinstall
72
72
  - name: Run pytest
73
73
  run: pixi run -e ${{ matrix.environment }} test-coverage --color=yes ${{ matrix.with_optionals && '-m with_optionals' || '-m "not with_optionals"'}} --cov=dataframely --cov-report=xml
74
74
  - name: Upload codecov
75
- uses: codecov/codecov-action@5a1091511ad55cbe89839c7260b706298ca349f7 # v5.5.1
75
+ uses: codecov/codecov-action@671740ac38dd9b0130fbe1cec585b89eea48d3de # v5.5.2
76
76
  with:
77
77
  files: ./coverage.xml
78
78
  token: ${{ secrets.CODECOV_TOKEN }}
@@ -13,14 +13,14 @@ jobs:
13
13
  id-token: write
14
14
  steps:
15
15
  - name: Checkout branch
16
- uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
16
+ uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
17
17
  - name: Set up pixi
18
- uses: prefix-dev/setup-pixi@28eb668aafebd9dede9d97c4ba1cd9989a4d0004 # v0.9.2
18
+ uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
19
19
  with:
20
20
  environments: default
21
21
  - name: Install Rust
22
22
  run: rustup show
23
23
  - name: Cache Rust dependencies
24
- uses: Swatinem/rust-cache@f13886b937689c021905a6b90929199931d60db1 # v2.8.1
24
+ uses: Swatinem/rust-cache@779680da715d629ac1d338a641029a2f4372abb5 # v2.8.2
25
25
  - name: Install repository
26
26
  run: pixi run postinstall
@@ -23,9 +23,9 @@ jobs:
23
23
  os: [ubuntu-latest, windows-latest]
24
24
  steps:
25
25
  - name: Checkout branch
26
- uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
26
+ uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
27
27
  - name: Set up pixi
28
- uses: prefix-dev/setup-pixi@28eb668aafebd9dede9d97c4ba1cd9989a4d0004 # v0.9.2
28
+ uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
29
29
  with:
30
30
  environments: nightly
31
31
  - name: Install polars nightly
@@ -35,7 +35,7 @@ jobs:
35
35
 
36
36
  steps:
37
37
  - name: "Checkout code"
38
- uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
38
+ uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
39
39
  with:
40
40
  persist-credentials: false
41
41
 
@@ -65,7 +65,7 @@ jobs:
65
65
  # Upload the results as artifacts (optional). Commenting out will disable uploads of run results in SARIF
66
66
  # format to the repository Actions tab.
67
67
  - name: "Upload artifact"
68
- uses: actions/upload-artifact@330a01c490aca151604b8cf639adc76d48f6c5d4 # v5.0.0
68
+ uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0
69
69
  with:
70
70
  name: SARIF file
71
71
  path: results.sarif
@@ -74,6 +74,6 @@ jobs:
74
74
  # Upload the results to GitHub's code scanning dashboard (optional).
75
75
  # Commenting out will disable upload of results to your repo's Code Scanning dashboard
76
76
  - name: "Upload to code-scanning"
77
- uses: github/codeql-action/upload-sarif@0499de31b99561a6d14a36a5f662c2a54f91beee # v3.29.5
77
+ uses: github/codeql-action/upload-sarif@5d4e8d1aca955e8d8589aabd499c5cae939e33c7 # v3.29.5
78
78
  with:
79
79
  sarif_file: results.sarif
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dataframely
3
- Version: 2.3.1
3
+ Version: 2.5.0
4
4
  Classifier: Programming Language :: Python :: 3
5
5
  Classifier: Programming Language :: Python :: 3.10
6
6
  Classifier: Programming Language :: Python :: 3.11
@@ -162,6 +162,19 @@ class SchemaMeta(ABCMeta):
162
162
  f"Did you forget to add parentheses?"
163
163
  )
164
164
 
165
+ # Check for pl.DataType instance or type (e.g., pl.String() or pl.String instead of dy.String())
166
+ if isinstance(value, pl.DataType) or (
167
+ isinstance(value, type) and issubclass(value, pl.DataType)
168
+ ):
169
+ value_type = "instance" if isinstance(value, pl.DataType) else "type"
170
+ example = (
171
+ "pl.String()" if isinstance(value, pl.DataType) else "pl.String"
172
+ )
173
+ raise TypeError(
174
+ f"Schema member '{attr}' is a polars DataType {value_type}. "
175
+ f"Use dataframely column types (e.g., dy.String()) instead of polars types (e.g., {example})."
176
+ )
177
+
165
178
  return cls
166
179
 
167
180
  if not TYPE_CHECKING:
@@ -1,4 +1,4 @@
1
- # Copyright (c) QuantCo 2025-2025
1
+ # Copyright (c) QuantCo 2025-2026
2
2
  # SPDX-License-Identifier: BSD-3-Clause
3
3
 
4
4
 
@@ -29,6 +29,7 @@ except ImportError:
29
29
  try:
30
30
  import sqlalchemy as sa
31
31
  import sqlalchemy.dialects.mssql as sa_mssql
32
+ import sqlalchemy.dialects.postgresql as sa_postgresql
32
33
  from sqlalchemy import Dialect
33
34
  from sqlalchemy.dialects.mssql.pyodbc import MSDialect_pyodbc
34
35
  from sqlalchemy.dialects.postgresql.psycopg2 import PGDialect_psycopg2
@@ -36,6 +37,7 @@ try:
36
37
  except ImportError:
37
38
  sa = _DummyModule("sqlalchemy") # type: ignore
38
39
  sa_mssql = _DummyModule("sqlalchemy") # type: ignore
40
+ sa_postgresql = _DummyModule("sqlalchemy") # type: ignore
39
41
 
40
42
  class sa_TypeEngine: # type: ignore # noqa: N801
41
43
  pass
@@ -81,6 +83,7 @@ __all__ = [
81
83
  "pydantic_core_schema",
82
84
  "pydantic",
83
85
  "sa_mssql",
86
+ "sa_postgresql",
84
87
  "sa_TypeEngine",
85
88
  "sa",
86
89
  ]
@@ -1,4 +1,4 @@
1
- # Copyright (c) QuantCo 2025-2025
1
+ # Copyright (c) QuantCo 2025-2026
2
2
  # SPDX-License-Identifier: BSD-3-Clause
3
3
 
4
4
  from __future__ import annotations
@@ -97,8 +97,17 @@ class Array(Column):
97
97
  }
98
98
 
99
99
  def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
100
- # NOTE: We might want to add support for PostgreSQL's ARRAY type or use JSON in the future.
101
- raise NotImplementedError("SQL column cannot have 'Array' type.")
100
+ match dialect.name:
101
+ case "postgresql":
102
+ # Note that the length of the array in each dimension is not supported in SQLAlchemy
103
+ # That is because PostgreSQL does not enforce the length anyway
104
+ return sa.ARRAY(
105
+ self.inner.sqlalchemy_dtype(dialect), dimensions=len(self.shape)
106
+ )
107
+ case _:
108
+ raise NotImplementedError(
109
+ f"SQL column cannot have 'Array' type for dialect '{dialect}'."
110
+ )
102
111
 
103
112
  def _pyarrow_field_of_shape(self, shape: Sequence[int]) -> pa.Field:
104
113
  if shape:
@@ -1,4 +1,4 @@
1
- # Copyright (c) QuantCo 2025-2025
1
+ # Copyright (c) QuantCo 2025-2026
2
2
  # SPDX-License-Identifier: BSD-3-Clause
3
3
 
4
4
  from __future__ import annotations
@@ -21,9 +21,11 @@ class Binary(Column):
21
21
  return pl.Binary()
22
22
 
23
23
  def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
24
- if dialect.name == "mssql":
25
- return sa.VARBINARY()
26
- return sa.LargeBinary()
24
+ match dialect.name:
25
+ case "mssql":
26
+ return sa.VARBINARY()
27
+ case _:
28
+ return sa.LargeBinary()
27
29
 
28
30
  @property
29
31
  def pyarrow_dtype(self) -> pa.DataType:
@@ -98,7 +98,11 @@ class Decimal(OrdinalMixin[decimal.Decimal], Column):
98
98
  return pl.Decimal(self.precision, self.scale)
99
99
 
100
100
  def validate_dtype(self, dtype: PolarsDataType) -> bool:
101
- return dtype.is_decimal()
101
+ return (
102
+ isinstance(dtype, pl.Decimal)
103
+ and dtype.scale == self.scale
104
+ and (self.precision is None or dtype.precision == self.precision)
105
+ )
102
106
 
103
107
  def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
104
108
  if self.scale and not self.precision:
@@ -1,4 +1,4 @@
1
- # Copyright (c) QuantCo 2025-2025
1
+ # Copyright (c) QuantCo 2025-2026
2
2
  # SPDX-License-Identifier: BSD-3-Clause
3
3
 
4
4
  from __future__ import annotations
@@ -120,8 +120,13 @@ class List(Column):
120
120
  }
121
121
 
122
122
  def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
123
- # NOTE: We might want to add support for PostgreSQL's ARRAY type or use JSON in the future.
124
- raise NotImplementedError("SQL column cannot have 'List' type.")
123
+ match dialect.name:
124
+ case "postgresql":
125
+ return sa.ARRAY(self.inner.sqlalchemy_dtype(dialect))
126
+ case _:
127
+ raise NotImplementedError(
128
+ f"SQL column cannot have 'List' type for dialect '{dialect}'."
129
+ )
125
130
 
126
131
  @property
127
132
  def pyarrow_dtype(self) -> pa.DataType:
@@ -131,9 +136,12 @@ class List(Column):
131
136
  def _sample_unchecked(self, generator: Generator, n: int) -> pl.Series:
132
137
  # First, sample the number of items per list element
133
138
  # NOTE: We default to 32 for the upper bound as we need some kind of reasonable
134
- # upper bound if none is set.
139
+ # upper bound if none is set. If min_length is greater than 32, we use
140
+ # min_length as the default upper bound instead.
141
+ min_len = self.min_length or 0
142
+ default_max = max(32, min_len)
135
143
  element_lengths = generator.sample_int(
136
- n, min=self.min_length or 0, max=(self.max_length or 32) + 1
144
+ n, min=min_len, max=(self.max_length or default_max) + 1
137
145
  )
138
146
 
139
147
  # Then, we can sample the inner elements in a flat series
@@ -14,6 +14,8 @@ from dataframely.random import Generator
14
14
  from ._base import Check, Column
15
15
  from ._registry import register
16
16
 
17
+ DEFAULT_SAMPLING_REGEX = r"[0-9a-zA-Z]"
18
+
17
19
 
18
20
  @register
19
21
  class String(Column):
@@ -126,9 +128,9 @@ class String(Column):
126
128
  str_max = f"{self.max_length}" if self.max_length is not None else ""
127
129
  # NOTE: We generate single-byte unicode characters here as validation uses
128
130
  # `len_bytes()`. Potentially we need to be more accurate at some point...
129
- regex = f"[\x01-\x7a]{{{str_min},{str_max}}}"
131
+ regex = f"{DEFAULT_SAMPLING_REGEX}{{{str_min},{str_max}}}"
130
132
  else:
131
- regex = r"[\x01-\x7a]*"
133
+ regex = rf"{DEFAULT_SAMPLING_REGEX}*"
132
134
 
133
135
  return generator.sample_string(
134
136
  n,
@@ -1,4 +1,4 @@
1
- # Copyright (c) QuantCo 2025-2025
1
+ # Copyright (c) QuantCo 2025-2026
2
2
  # SPDX-License-Identifier: BSD-3-Clause
3
3
 
4
4
  from __future__ import annotations
@@ -8,7 +8,7 @@ from typing import Any, cast
8
8
 
9
9
  import polars as pl
10
10
 
11
- from dataframely._compat import pa, sa, sa_TypeEngine
11
+ from dataframely._compat import pa, sa, sa_postgresql, sa_TypeEngine
12
12
  from dataframely._polars import PolarsDataType
13
13
  from dataframely.random import Generator
14
14
 
@@ -107,8 +107,11 @@ class Struct(Column):
107
107
  }
108
108
 
109
109
  def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
110
- # NOTE: We might want to add support for PostgreSQL's JSON in the future.
111
- raise NotImplementedError("SQL column cannot have 'Struct' type.")
110
+ match dialect.name:
111
+ case "postgresql":
112
+ return sa_postgresql.JSONB()
113
+ case _:
114
+ raise NotImplementedError("SQL column cannot have 'Struct' type.")
112
115
 
113
116
  @property
114
117
  def pyarrow_dtype(self) -> pa.DataType:
@@ -7,7 +7,7 @@ import json
7
7
  import sys
8
8
  import warnings
9
9
  from abc import ABC
10
- from collections.abc import Iterable, Mapping, Sequence
10
+ from collections.abc import Mapping, Sequence
11
11
  from json import JSONDecodeError
12
12
  from pathlib import Path
13
13
  from typing import IO, Any, Literal, overload
@@ -177,7 +177,7 @@ class Schema(BaseSchema, ABC):
177
177
  num_rows: int | None = None,
178
178
  *,
179
179
  overrides: (
180
- Mapping[str, Iterable[Any]] | Sequence[Mapping[str, Any]] | None
180
+ Mapping[str, Sequence[Any] | Any] | Sequence[Mapping[str, Any]] | None
181
181
  ) = None,
182
182
  generator: Generator | None = None,
183
183
  ) -> DataFrame[Self]:
@@ -234,26 +234,22 @@ class Schema(BaseSchema, ABC):
234
234
  g = generator or Generator()
235
235
 
236
236
  # Precondition: valid overrides. We put them into a data frame to remember which
237
- # values have been used in the algorithm below.
238
- if overrides:
237
+ # values have been used in the algorithm below. When the user passes a sequence
238
+ # of mappings, they do not require to have the same keys. Hence, we have to
239
+ # remember that the data frame has "holes".
240
+ missing_override_indices: dict[str, pl.Series] = {}
241
+ if overrides is not None:
239
242
  override_keys = (
240
- set(overrides) if isinstance(overrides, Mapping) else set(overrides[0])
243
+ set(overrides)
244
+ if isinstance(overrides, Mapping)
245
+ else (
246
+ set.union(*[set(o.keys()) for o in overrides])
247
+ if len(overrides) > 0
248
+ else set()
249
+ )
241
250
  )
242
- if isinstance(overrides, Sequence):
243
- # Check that overrides entries are consistent. Not necessary for mapping
244
- # overrides as polars checks the series lists upon data frame construction.
245
- inconsistent_override_keys = [
246
- index
247
- for index, current in enumerate(overrides)
248
- if set(current) != override_keys
249
- ]
250
- if len(inconsistent_override_keys) > 0:
251
- raise ValueError(
252
- "The `overrides` entries at the following indices "
253
- "do not provide the same keys as the first entry: "
254
- f"{inconsistent_override_keys}."
255
- )
256
251
 
252
+ # Check that all override keys refer to valid columns
257
253
  column_names = set(cls.column_names())
258
254
  if not override_keys.issubset(column_names):
259
255
  raise ValueError(
@@ -261,6 +257,19 @@ class Schema(BaseSchema, ABC):
261
257
  "which are not in the schema."
262
258
  )
263
259
 
260
+ # Remember the "holes" of the inputs if overrides are provided as a sequence
261
+ if isinstance(overrides, Sequence):
262
+ for key in override_keys:
263
+ indices = [
264
+ i for i, override in enumerate(overrides) if key not in override
265
+ ]
266
+ if len(indices) > 0:
267
+ missing_override_indices[key] = pl.Series(indices)
268
+
269
+ # NOTE: Even if the user-provided overrides have "holes", we can still just
270
+ # create the data frame. Polars will fill the missing values with nulls, we
271
+ # will replace them later during sampling. If we were to already replace
272
+ # them here, we would not be able to resample these values.
264
273
  values = pl.DataFrame(
265
274
  overrides,
266
275
  schema={
@@ -323,6 +332,7 @@ class Schema(BaseSchema, ABC):
323
332
  used_values=values.slice(0, 0),
324
333
  remaining_values=values,
325
334
  override_expressions=override_expressions,
335
+ missing_value_indices=missing_override_indices,
326
336
  )
327
337
 
328
338
  sampling_rounds = 1
@@ -360,6 +370,7 @@ class Schema(BaseSchema, ABC):
360
370
  used_values=used_values,
361
371
  remaining_values=remaining_values,
362
372
  override_expressions=override_expressions,
373
+ missing_value_indices=missing_override_indices,
363
374
  )
364
375
  sampling_rounds += 1
365
376
 
@@ -388,6 +399,7 @@ class Schema(BaseSchema, ABC):
388
399
  used_values: pl.DataFrame,
389
400
  remaining_values: pl.DataFrame,
390
401
  override_expressions: list[pl.Expr],
402
+ missing_value_indices: dict[str, pl.Series],
391
403
  ) -> tuple[pl.DataFrame, pl.DataFrame, pl.DataFrame]:
392
404
  """Private method to sample a data frame with the schema including subsequent
393
405
  filtering.
@@ -406,6 +418,33 @@ class Schema(BaseSchema, ABC):
406
418
  }
407
419
  )
408
420
 
421
+ # If we have missing value indices, we need to sample new values for the
422
+ # indices that overlap with indices in the remaining values and replace them
423
+ # in the sampled data frame.
424
+ for name, indices in missing_value_indices.items():
425
+ remapped_indices = (
426
+ indices.to_frame("idx")
427
+ .join(
428
+ remaining_values.select("__row_index__").with_row_index(
429
+ "__row_index_loop__"
430
+ ),
431
+ left_on="idx",
432
+ right_on="__row_index__",
433
+ )
434
+ .select("__row_index_loop__")
435
+ .to_series()
436
+ )
437
+ if (num := len(remapped_indices)) > 0:
438
+ sampled_values = cls.columns()[name].sample(generator, num)
439
+ sampled = sampled.with_columns(
440
+ sampled[name]
441
+ # NOTE: We need to sort here as `scatter` requires sorted indices.
442
+ # Due to concatenations in `remaining_values`, the indices can go
443
+ # out of order.
444
+ .scatter(remapped_indices.sort(), sampled_values)
445
+ .alias(name)
446
+ )
447
+
409
448
  combined_dataframe = pl.concat([previous_result, sampled])
410
449
  # Pre-process columns before filtering.
411
450
  combined_dataframe = combined_dataframe.with_columns(override_expressions)