dataframely 2.1.0__tar.gz → 2.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (216) hide show
  1. {dataframely-2.1.0 → dataframely-2.3.1}/.github/copilot-instructions.md +1 -1
  2. {dataframely-2.1.0 → dataframely-2.3.1}/PKG-INFO +2 -2
  3. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/__init__.py +2 -0
  4. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/collection/_base.py +17 -0
  5. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/collection/collection.py +98 -46
  6. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/enum.py +2 -2
  7. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/exc.py +7 -0
  8. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/filter_result.py +2 -2
  9. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/schema.py +23 -6
  10. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/testing/storage.py +103 -12
  11. {dataframely-2.1.0 → dataframely-2.3.1}/pixi.lock +6371 -5953
  12. {dataframely-2.1.0 → dataframely-2.3.1}/pixi.toml +1 -1
  13. {dataframely-2.1.0 → dataframely-2.3.1}/pyproject.toml +2 -2
  14. dataframely-2.3.1/tests/collection/test_propagate_row_failures.py +126 -0
  15. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_serialization.py +35 -3
  16. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_storage.py +57 -7
  17. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_serialization.py +4 -4
  18. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_storage.py +62 -1
  19. {dataframely-2.1.0 → dataframely-2.3.1}/.copier-answers.yml +0 -0
  20. {dataframely-2.1.0 → dataframely-2.3.1}/.envrc +0 -0
  21. {dataframely-2.1.0 → dataframely-2.3.1}/.gitattributes +0 -0
  22. {dataframely-2.1.0 → dataframely-2.3.1}/.github/CODEOWNERS +0 -0
  23. {dataframely-2.1.0 → dataframely-2.3.1}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  24. {dataframely-2.1.0 → dataframely-2.3.1}/.github/dependabot.yml +0 -0
  25. {dataframely-2.1.0 → dataframely-2.3.1}/.github/instructions/tests.instructions.md +0 -0
  26. {dataframely-2.1.0 → dataframely-2.3.1}/.github/release-drafter.yml +0 -0
  27. {dataframely-2.1.0 → dataframely-2.3.1}/.github/workflows/build.yml +0 -0
  28. {dataframely-2.1.0 → dataframely-2.3.1}/.github/workflows/chore.yml +0 -0
  29. {dataframely-2.1.0 → dataframely-2.3.1}/.github/workflows/ci.yml +0 -0
  30. {dataframely-2.1.0 → dataframely-2.3.1}/.github/workflows/copilot-setup-steps.yml +0 -0
  31. {dataframely-2.1.0 → dataframely-2.3.1}/.github/workflows/nightly.yml +0 -0
  32. {dataframely-2.1.0 → dataframely-2.3.1}/.github/workflows/scorecard.yml +0 -0
  33. {dataframely-2.1.0 → dataframely-2.3.1}/.gitignore +0 -0
  34. {dataframely-2.1.0 → dataframely-2.3.1}/.pre-commit-config.yaml +0 -0
  35. {dataframely-2.1.0 → dataframely-2.3.1}/.prettierignore +0 -0
  36. {dataframely-2.1.0 → dataframely-2.3.1}/.prettierrc +0 -0
  37. {dataframely-2.1.0 → dataframely-2.3.1}/.readthedocs.yml +0 -0
  38. {dataframely-2.1.0 → dataframely-2.3.1}/Cargo.lock +0 -0
  39. {dataframely-2.1.0 → dataframely-2.3.1}/Cargo.toml +0 -0
  40. {dataframely-2.1.0 → dataframely-2.3.1}/LICENSE +0 -0
  41. {dataframely-2.1.0 → dataframely-2.3.1}/README.md +0 -0
  42. {dataframely-2.1.0 → dataframely-2.3.1}/SECURITY.md +0 -0
  43. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_base_schema.py +0 -0
  44. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_compat.py +0 -0
  45. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_deprecation.py +0 -0
  46. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_filter.py +0 -0
  47. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_match_to_schema.py +0 -0
  48. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_native.pyi +0 -0
  49. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_plugin.py +0 -0
  50. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_polars.py +0 -0
  51. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_pydantic.py +0 -0
  52. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_rule.py +0 -0
  53. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_serialization.py +0 -0
  54. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_storage/__init__.py +0 -0
  55. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_storage/_base.py +0 -0
  56. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_storage/_exc.py +0 -0
  57. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_storage/constants.py +0 -0
  58. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_storage/delta.py +0 -0
  59. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_storage/parquet.py +0 -0
  60. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/_typing.py +0 -0
  61. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/collection/__init__.py +0 -0
  62. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/collection/filter_result.py +0 -0
  63. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/__init__.py +0 -0
  64. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/_base.py +0 -0
  65. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/_mixins.py +0 -0
  66. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/_registry.py +0 -0
  67. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/_utils.py +0 -0
  68. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/any.py +0 -0
  69. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/array.py +0 -0
  70. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/binary.py +0 -0
  71. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/bool.py +0 -0
  72. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/categorical.py +0 -0
  73. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/datetime.py +0 -0
  74. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/decimal.py +0 -0
  75. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/float.py +0 -0
  76. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/integer.py +0 -0
  77. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/list.py +0 -0
  78. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/object.py +0 -0
  79. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/string.py +0 -0
  80. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/columns/struct.py +0 -0
  81. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/config.py +0 -0
  82. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/functional.py +0 -0
  83. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/py.typed +0 -0
  84. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/random.py +0 -0
  85. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/testing/__init__.py +0 -0
  86. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/testing/const.py +0 -0
  87. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/testing/factory.py +0 -0
  88. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/testing/mask.py +0 -0
  89. {dataframely-2.1.0 → dataframely-2.3.1}/dataframely/testing/rules.py +0 -0
  90. {dataframely-2.1.0 → dataframely-2.3.1}/docker-compose.yml +0 -0
  91. {dataframely-2.1.0 → dataframely-2.3.1}/docs/_static/custom.css +0 -0
  92. {dataframely-2.1.0 → dataframely-2.3.1}/docs/_static/favicon.ico +0 -0
  93. {dataframely-2.1.0 → dataframely-2.3.1}/docs/_templates/autosummary/class.rst +0 -0
  94. {dataframely-2.1.0 → dataframely-2.3.1}/docs/_templates/autosummary/method.rst +0 -0
  95. {dataframely-2.1.0 → dataframely-2.3.1}/docs/_templates/classes/column.rst +0 -0
  96. {dataframely-2.1.0 → dataframely-2.3.1}/docs/_templates/classes/error.rst +0 -0
  97. {dataframely-2.1.0 → dataframely-2.3.1}/docs/_templates/classes/filter_result.rst +0 -0
  98. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/collection/generation.rst +0 -0
  99. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/collection/index.rst +0 -0
  100. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/collection/io.rst +0 -0
  101. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/collection/metadata.rst +0 -0
  102. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/collection/operations.rst +0 -0
  103. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/collection/validation.rst +0 -0
  104. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/columns/index.rst +0 -0
  105. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/errors/index.rst +0 -0
  106. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/filter_result/failure_info.rst +0 -0
  107. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/filter_result/index.rst +0 -0
  108. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/index.rst +0 -0
  109. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/misc/index.rst +0 -0
  110. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/schema/conversion.rst +0 -0
  111. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/schema/generation.rst +0 -0
  112. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/schema/index.rst +0 -0
  113. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/schema/io.rst +0 -0
  114. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/schema/metadata.rst +0 -0
  115. {dataframely-2.1.0 → dataframely-2.3.1}/docs/api/schema/validation.rst +0 -0
  116. {dataframely-2.1.0 → dataframely-2.3.1}/docs/conf.py +0 -0
  117. {dataframely-2.1.0 → dataframely-2.3.1}/docs/css/custom.css +0 -0
  118. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/development.md +0 -0
  119. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/examples/index.md +0 -0
  120. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/examples/real-world.ipynb +0 -0
  121. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/faq.md +0 -0
  122. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/features/column-metadata.md +0 -0
  123. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/features/data-generation.md +0 -0
  124. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/features/index.md +0 -0
  125. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/features/lazy-validation.md +0 -0
  126. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/features/primary-keys.md +0 -0
  127. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/features/serialization.md +0 -0
  128. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/features/sql-generation.md +0 -0
  129. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/index.md +0 -0
  130. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/migration/index.md +0 -0
  131. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/migration/v1-v2.md +0 -0
  132. {dataframely-2.1.0 → dataframely-2.3.1}/docs/guides/quickstart.md +0 -0
  133. {dataframely-2.1.0 → dataframely-2.3.1}/docs/index.md +0 -0
  134. {dataframely-2.1.0 → dataframely-2.3.1}/rust-toolchain.toml +0 -0
  135. {dataframely-2.1.0 → dataframely-2.3.1}/src/lib.rs +0 -0
  136. {dataframely-2.1.0 → dataframely-2.3.1}/src/polars_plugin/mod.rs +0 -0
  137. {dataframely-2.1.0 → dataframely-2.3.1}/src/polars_plugin/rule_failure.rs +0 -0
  138. {dataframely-2.1.0 → dataframely-2.3.1}/src/polars_plugin/utils.rs +0 -0
  139. {dataframely-2.1.0 → dataframely-2.3.1}/src/polars_plugin/validation_error.rs +0 -0
  140. {dataframely-2.1.0 → dataframely-2.3.1}/src/regex/errdefs.rs +0 -0
  141. {dataframely-2.1.0 → dataframely-2.3.1}/src/regex/mod.rs +0 -0
  142. {dataframely-2.1.0 → dataframely-2.3.1}/src/regex/repr.rs +0 -0
  143. {dataframely-2.1.0 → dataframely-2.3.1}/tests/benches/conftest.py +0 -0
  144. {dataframely-2.1.0 → dataframely-2.3.1}/tests/benches/test_collection.py +0 -0
  145. {dataframely-2.1.0 → dataframely-2.3.1}/tests/benches/test_failure.py +0 -0
  146. {dataframely-2.1.0 → dataframely-2.3.1}/tests/benches/test_schema.py +0 -0
  147. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_base.py +0 -0
  148. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_cast.py +0 -0
  149. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_collection_future_annotations.py +0 -0
  150. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_create_empty.py +0 -0
  151. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_filter_one_to_n.py +0 -0
  152. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_filter_validate.py +0 -0
  153. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_ignore_in_filter.py +0 -0
  154. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_implementation.py +0 -0
  155. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_join.py +0 -0
  156. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_matches.py +0 -0
  157. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_optional_members.py +0 -0
  158. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_repr.py +0 -0
  159. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_sample.py +0 -0
  160. {dataframely-2.1.0 → dataframely-2.3.1}/tests/collection/test_validate_input.py +0 -0
  161. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/__init__.py +0 -0
  162. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_any.py +0 -0
  163. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_array.py +0 -0
  164. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_binary.py +0 -0
  165. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_datetime.py +0 -0
  166. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_decimal.py +0 -0
  167. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_enum.py +0 -0
  168. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_float.py +0 -0
  169. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_integer.py +0 -0
  170. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_list.py +0 -0
  171. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_object.py +0 -0
  172. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_string.py +0 -0
  173. {dataframely-2.1.0 → dataframely-2.3.1}/tests/column_types/test_struct.py +0 -0
  174. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/__init__.py +0 -0
  175. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_alias.py +0 -0
  176. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_base.py +0 -0
  177. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_check.py +0 -0
  178. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_default_dtypes.py +0 -0
  179. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_matches.py +0 -0
  180. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_metadata.py +0 -0
  181. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_polars_schema.py +0 -0
  182. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_pyarrow.py +0 -0
  183. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_rules.py +0 -0
  184. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_sample.py +0 -0
  185. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_sqlalchemy_columns.py +0 -0
  186. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_str.py +0 -0
  187. {dataframely-2.1.0 → dataframely-2.3.1}/tests/columns/test_utils.py +0 -0
  188. {dataframely-2.1.0 → dataframely-2.3.1}/tests/conftest.py +0 -0
  189. {dataframely-2.1.0 → dataframely-2.3.1}/tests/core_validation/__init__.py +0 -0
  190. {dataframely-2.1.0 → dataframely-2.3.1}/tests/core_validation/test_match_to_schema.py +0 -0
  191. {dataframely-2.1.0 → dataframely-2.3.1}/tests/core_validation/test_rule_evaluation.py +0 -0
  192. {dataframely-2.1.0 → dataframely-2.3.1}/tests/failure_info/test_storage.py +0 -0
  193. {dataframely-2.1.0 → dataframely-2.3.1}/tests/functional/test_concat.py +0 -0
  194. {dataframely-2.1.0 → dataframely-2.3.1}/tests/functional/test_relationships.py +0 -0
  195. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_base.py +0 -0
  196. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_cast.py +0 -0
  197. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_create_empty.py +0 -0
  198. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_create_empty_if_none.py +0 -0
  199. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_filter.py +0 -0
  200. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_inheritance.py +0 -0
  201. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_matches.py +0 -0
  202. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_read_write_parquet.py +0 -0
  203. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_repr.py +0 -0
  204. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_rule_implementation.py +0 -0
  205. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_sample.py +0 -0
  206. {dataframely-2.1.0 → dataframely-2.3.1}/tests/schema/test_validate.py +0 -0
  207. {dataframely-2.1.0 → dataframely-2.3.1}/tests/storage/test_delta.py +0 -0
  208. {dataframely-2.1.0 → dataframely-2.3.1}/tests/test_compat.py +0 -0
  209. {dataframely-2.1.0 → dataframely-2.3.1}/tests/test_config.py +0 -0
  210. {dataframely-2.1.0 → dataframely-2.3.1}/tests/test_deprecation.py +0 -0
  211. {dataframely-2.1.0 → dataframely-2.3.1}/tests/test_factory.py +0 -0
  212. {dataframely-2.1.0 → dataframely-2.3.1}/tests/test_native_regex.py +0 -0
  213. {dataframely-2.1.0 → dataframely-2.3.1}/tests/test_pydantic.py +0 -0
  214. {dataframely-2.1.0 → dataframely-2.3.1}/tests/test_random.py +0 -0
  215. {dataframely-2.1.0 → dataframely-2.3.1}/tests/test_serialization.py +0 -0
  216. {dataframely-2.1.0 → dataframely-2.3.1}/tests/test_typing.py +0 -0
@@ -202,7 +202,7 @@ validated_df: dy.DataFrame[MySchema] = MySchema.validate(df, cast=True)
202
202
 
203
203
  1. **Python code**: Run `pixi run pre-commit run` before committing
204
204
  2. **Rust code**: Run `pixi run postinstall` to rebuild, then run tests
205
- 3. **Tests**: Ensure `pixi run test` passes
205
+ 3. **Tests**: Ensure `pixi run test` passes. If changes might affect storage backends, use `pixi run test -m s3`.
206
206
  4. **Documentation**: Update docstrings
207
207
  5. **API changes**: Ensure backward compatibility or document migration path
208
208
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dataframely
3
- Version: 2.1.0
3
+ Version: 2.3.1
4
4
  Classifier: Programming Language :: Python :: 3
5
5
  Classifier: Programming Language :: Python :: 3.10
6
6
  Classifier: Programming Language :: Python :: 3.11
@@ -9,7 +9,7 @@ Classifier: Programming Language :: Python :: 3.13
9
9
  Classifier: Programming Language :: Python :: 3.14
10
10
  Requires-Dist: fsspec>=2025.9
11
11
  Requires-Dist: numpy
12
- Requires-Dist: polars>=1.34
12
+ Requires-Dist: polars>=1.36
13
13
  Requires-Dist: typing-extensions ; python_full_version < '3.11'
14
14
  Requires-Dist: deltalake ; extra == 'deltalake'
15
15
  Requires-Dist: pyarrow ; extra == 'pyarrow'
@@ -51,6 +51,7 @@ from .columns import (
51
51
  UInt64,
52
52
  )
53
53
  from .config import Config
54
+ from .exc import DeserializationError
54
55
  from .filter_result import FailureInfo
55
56
  from .functional import (
56
57
  concat_collection_members,
@@ -106,4 +107,5 @@ __all__ = [
106
107
  "Array",
107
108
  "Object",
108
109
  "Validation",
110
+ "DeserializationError",
109
111
  ]
@@ -65,6 +65,11 @@ class CollectionMember:
65
65
  #: the collection's common primary key. Two members that share common column names
66
66
  #: may not both be inlined for sampling.
67
67
  inline_for_sampling: bool = False
68
+ #: Whether individual row failures in this member should be propagated to the
69
+ #: collection, i.e., cause the common primary key of the failures to be filtered
70
+ #: out from the entire collection. This setting is ignored if `ignored_in_filters`
71
+ #: is `True`.
72
+ propagate_row_failures: bool = False
68
73
 
69
74
 
70
75
  # --------------------------------------- UTILS -------------------------------------- #
@@ -250,6 +255,7 @@ class CollectionMeta(ABCMeta):
250
255
  is_optional=True,
251
256
  ignored_in_filters=collection_member.ignored_in_filters,
252
257
  inline_for_sampling=collection_member.inline_for_sampling,
258
+ propagate_row_failures=collection_member.propagate_row_failures,
253
259
  )
254
260
  elif issubclass(origin, TypedLazyFrame):
255
261
  # Happy path: required member
@@ -258,6 +264,7 @@ class CollectionMeta(ABCMeta):
258
264
  is_optional=False,
259
265
  ignored_in_filters=collection_member.ignored_in_filters,
260
266
  inline_for_sampling=collection_member.inline_for_sampling,
267
+ propagate_row_failures=collection_member.propagate_row_failures,
261
268
  )
262
269
  else:
263
270
  # Some other unknown annotation
@@ -333,6 +340,16 @@ class BaseCollection(metaclass=CollectionMeta):
333
340
  if not member.ignored_in_filters
334
341
  }
335
342
 
343
+ @classmethod
344
+ def _failure_propagating_members(cls) -> set[str]:
345
+ """The names of all members of the collection that propagate individual row
346
+ failures to the collection."""
347
+ return {
348
+ name
349
+ for name, member in cls.members().items()
350
+ if member.propagate_row_failures
351
+ }
352
+
336
353
  @classmethod
337
354
  def common_primary_key(cls) -> list[str]:
338
355
  """The primary keys shared by non ignored members of the collection."""
@@ -12,7 +12,7 @@ from collections.abc import Iterable, Mapping, Sequence
12
12
  from dataclasses import asdict
13
13
  from json import JSONDecodeError
14
14
  from pathlib import Path
15
- from typing import IO, Annotated, Any, Literal, cast
15
+ from typing import IO, Annotated, Any, Literal, cast, overload
16
16
 
17
17
  import polars as pl
18
18
  import polars.exceptions as plexc
@@ -33,7 +33,11 @@ from dataframely._storage.constants import COLLECTION_METADATA_KEY
33
33
  from dataframely._storage.delta import DeltaStorageBackend
34
34
  from dataframely._storage.parquet import ParquetStorageBackend
35
35
  from dataframely._typing import LazyFrame, Validation
36
- from dataframely.exc import ValidationError, ValidationRequiredError
36
+ from dataframely.exc import (
37
+ DeserializationError,
38
+ ValidationError,
39
+ ValidationRequiredError,
40
+ )
37
41
  from dataframely.filter_result import FailureInfo
38
42
  from dataframely.random import Generator
39
43
  from dataframely.schema import _schema_from_dict
@@ -569,7 +573,8 @@ class Collection(BaseCollection, ABC):
569
573
  # Once we've done that, we can apply the filters on this collection. To this end,
570
574
  # we iterate over all filters and store the filter results.
571
575
  filters = cls._filters()
572
- if len(filters) > 0:
576
+ failure_propagating_members = cls._failure_propagating_members()
577
+ if len(filters) > 0 or len(failure_propagating_members) > 0:
573
578
  result_cls = cls._init(results)
574
579
  primary_key = cls.common_primary_key()
575
580
 
@@ -582,6 +587,17 @@ class Collection(BaseCollection, ABC):
582
587
  .lazy()
583
588
  )
584
589
 
590
+ drop: dict[str, pl.LazyFrame] = {}
591
+ for failure_propagating_member in failure_propagating_members:
592
+ annotation_column = f"{failure_propagating_member}|failure_propagation"
593
+ drop[annotation_column] = (
594
+ failures[failure_propagating_member]
595
+ ._lf.select(primary_key)
596
+ .unique()
597
+ .pipe(collect_if, eager)
598
+ .lazy()
599
+ )
600
+
585
601
  # Now we can iterate over the results and left-join onto each individual
586
602
  # filter to obtain independent boolean indicators of whether to keep the row
587
603
  for member_name, filtered in results.items():
@@ -597,15 +613,23 @@ class Collection(BaseCollection, ABC):
597
613
  how="left",
598
614
  maintain_order="left",
599
615
  ).with_columns(pl.col(name).fill_null(False))
616
+ for name, filter_drop in drop.items():
617
+ lf_with_eval = lf_with_eval.join(
618
+ filter_drop.with_columns(pl.lit(False).alias(name)),
619
+ on=primary_key,
620
+ how="left",
621
+ maintain_order="left",
622
+ ).with_columns(pl.col(name).fill_null(True))
600
623
 
601
624
  lf_with_eval = lf_with_eval.pipe(collect_if, eager).lazy()
602
625
 
603
626
  # Filtering `lf_with_eval` by the rows for which all joins
604
627
  # "succeeded", we can identify the rows that pass all the filters. We
605
628
  # keep these rows for the result.
629
+ all_filter_columns = list(keep.keys()) + list(drop.keys())
606
630
  results[member_name] = lf_with_eval.filter(
607
- pl.all_horizontal(keep.keys())
608
- ).drop(keep.keys())
631
+ pl.all_horizontal(all_filter_columns)
632
+ ).drop(all_filter_columns)
609
633
 
610
634
  # Filtering `lf_with_eval` with the inverse condition, we find all
611
635
  # the problematic rows. We can build a single failure info object by
@@ -619,7 +643,7 @@ class Collection(BaseCollection, ABC):
619
643
  #
620
644
  failure = failures[member_name]
621
645
  filtered_failure = lf_with_eval.filter(
622
- ~pl.all_horizontal(keep.keys())
646
+ ~pl.all_horizontal(all_filter_columns)
623
647
  ).lazy()
624
648
 
625
649
  # If we cast previously, `failure` and `filtered_failure` have different
@@ -656,7 +680,7 @@ class Collection(BaseCollection, ABC):
656
680
 
657
681
  failures[member_name] = FailureInfo(
658
682
  lf=failure_lf,
659
- rule_columns=failure._rule_columns + list(keep.keys()),
683
+ rule_columns=failure._rule_columns + all_filter_columns,
660
684
  schema=failure.schema,
661
685
  )
662
686
 
@@ -891,13 +915,13 @@ class Collection(BaseCollection, ABC):
891
915
  - `"allow"`: The method tries to read the schema data from the parquet
892
916
  files. If the stored collection schema matches this collection
893
917
  schema, the collection is read without validation. If the stored
894
- schema mismatches this schema no metadata can be found in
918
+ schema mismatches this schema, no valid metadata can be found in
895
919
  the parquets, or the files have conflicting metadata,
896
920
  this method automatically runs :meth:`validate` with `cast=True`.
897
921
  - `"warn"`: The method behaves similarly to `"allow"`. However,
898
922
  it prints a warning if validation is necessary.
899
923
  - `"forbid"`: The method never runs validation automatically and only
900
- returns if the metadata stores a collection schema that matches
924
+ returns if the metadata stores a valid collection schema that matches
901
925
  this collection.
902
926
  - `"skip"`: The method never runs validation and simply reads the
903
927
  data, entrusting the user that the schema is valid. *Use this option
@@ -1184,7 +1208,12 @@ class Collection(BaseCollection, ABC):
1184
1208
  members=cls.member_schemas().keys(), **kwargs
1185
1209
  )
1186
1210
 
1187
- collection_types = _deserialize_types(serialized_collection_types)
1211
+ # Use strict=False when validation is "allow", "warn" or "skip" to tolerate
1212
+ # missing or broken collection metadata.
1213
+ strict = validation == "forbid"
1214
+ collection_types = _deserialize_types(
1215
+ serialized_collection_types, strict=strict
1216
+ )
1188
1217
  collection_type = _reconcile_collection_types(collection_types)
1189
1218
 
1190
1219
  if cls._requires_validation_for_reading_parquets(collection_type, validation):
@@ -1245,14 +1274,27 @@ def read_parquet_metadata_collection(
1245
1274
  """
1246
1275
  metadata = pl.read_parquet_metadata(source)
1247
1276
  if (schema_metadata := metadata.get(COLLECTION_METADATA_KEY)) is not None:
1248
- try:
1249
- return deserialize_collection(schema_metadata)
1250
- except (JSONDecodeError, plexc.ComputeError):
1251
- return None
1277
+ return deserialize_collection(schema_metadata, strict=False)
1252
1278
  return None
1253
1279
 
1254
1280
 
1255
- def deserialize_collection(data: str) -> type[Collection]:
1281
+ @overload
1282
+ def deserialize_collection(
1283
+ data: str, strict: Literal[True] = True
1284
+ ) -> type[Collection]: ...
1285
+
1286
+
1287
+ @overload
1288
+ def deserialize_collection(
1289
+ data: str, strict: Literal[False]
1290
+ ) -> type[Collection] | None: ...
1291
+
1292
+
1293
+ @overload
1294
+ def deserialize_collection(data: str, strict: bool) -> type[Collection] | None: ...
1295
+
1296
+
1297
+ def deserialize_collection(data: str, strict: bool = True) -> type[Collection] | None:
1256
1298
  """Deserialize a collection from a JSON string.
1257
1299
 
1258
1300
  This method allows to dynamically load a collection from its serialization, without
@@ -1260,12 +1302,14 @@ def deserialize_collection(data: str) -> type[Collection]:
1260
1302
 
1261
1303
  Args:
1262
1304
  data: The JSON string created via :meth:`Collection.serialize`.
1305
+ strict: Whether to raise an exception if the collection cannot be deserialized.
1263
1306
 
1264
1307
  Returns:
1265
1308
  The collection loaded from the JSON data.
1266
1309
 
1267
1310
  Raises:
1268
- ValueError: If the schema format version is not supported.
1311
+ DeserializationError: If the collection can not be deserialized
1312
+ and `strict=True`.
1269
1313
 
1270
1314
  Attention:
1271
1315
  The returned collection **cannot** be used to create instances of the
@@ -1280,34 +1324,41 @@ def deserialize_collection(data: str) -> type[Collection]:
1280
1324
  See also:
1281
1325
  :meth:`Collection.serialize` for additional information on serialization.
1282
1326
  """
1283
- decoded = json.loads(data, cls=SchemaJSONDecoder)
1284
- if (format := decoded["versions"]["format"]) != SERIALIZATION_FORMAT_VERSION:
1285
- raise ValueError(f"Unsupported schema format version: {format}")
1286
-
1287
- annotations: dict[str, Any] = {}
1288
- for name, info in decoded["members"].items():
1289
- lf_type = LazyFrame[_schema_from_dict(info["schema"])] # type: ignore
1290
- if info["is_optional"]:
1291
- lf_type = lf_type | None # type: ignore
1292
- annotations[name] = Annotated[
1293
- lf_type,
1294
- CollectionMember(
1295
- ignored_in_filters=info["ignored_in_filters"],
1296
- inline_for_sampling=info["inline_for_sampling"],
1297
- ),
1298
- ]
1299
-
1300
- return type(
1301
- f"{decoded['name']}_dynamic",
1302
- (Collection,),
1303
- {
1304
- "__annotations__": annotations,
1305
- **{
1306
- name: Filter(logic=lambda _, logic=logic: logic) # type: ignore
1307
- for name, logic in decoded["filters"].items()
1327
+ try:
1328
+ decoded = json.loads(data, cls=SchemaJSONDecoder)
1329
+ if (format := decoded["versions"]["format"]) != SERIALIZATION_FORMAT_VERSION:
1330
+ raise ValueError(f"Unsupported schema format version: {format}")
1331
+
1332
+ annotations: dict[str, Any] = {}
1333
+ for name, info in decoded["members"].items():
1334
+ lf_type = LazyFrame[_schema_from_dict(info["schema"])] # type: ignore
1335
+ if info["is_optional"]:
1336
+ lf_type = lf_type | None # type: ignore
1337
+ annotations[name] = Annotated[
1338
+ lf_type,
1339
+ CollectionMember(
1340
+ ignored_in_filters=info["ignored_in_filters"],
1341
+ inline_for_sampling=info["inline_for_sampling"],
1342
+ ),
1343
+ ]
1344
+
1345
+ return type(
1346
+ f"{decoded['name']}_dynamic",
1347
+ (Collection,),
1348
+ {
1349
+ "__annotations__": annotations,
1350
+ **{
1351
+ name: Filter(logic=lambda _, logic=logic: logic) # type: ignore
1352
+ for name, logic in decoded["filters"].items()
1353
+ },
1308
1354
  },
1309
- },
1310
- )
1355
+ )
1356
+ except (ValueError, TypeError, JSONDecodeError, plexc.ComputeError) as e:
1357
+ if strict:
1358
+ raise DeserializationError(
1359
+ "The Collection metadata could not be deserialized"
1360
+ ) from e
1361
+ return None
1311
1362
 
1312
1363
 
1313
1364
  # --------------------------------------- UTILS -------------------------------------- #
@@ -1333,14 +1384,15 @@ def _extract_keys_if_exist(
1333
1384
 
1334
1385
  def _deserialize_types(
1335
1386
  serialized_collection_types: Iterable[str | None],
1387
+ strict: bool = True,
1336
1388
  ) -> list[type[Collection]]:
1337
1389
  collection_types = []
1338
- collection_type: type[Collection] | None = None
1339
1390
  for t in serialized_collection_types:
1340
1391
  if t is None:
1341
1392
  continue
1342
- collection_type = deserialize_collection(t)
1343
- collection_types.append(collection_type)
1393
+ collection_type = deserialize_collection(t, strict=strict)
1394
+ if collection_type is not None:
1395
+ collection_types.append(collection_type)
1344
1396
 
1345
1397
  return collection_types
1346
1398
 
@@ -87,9 +87,9 @@ class Enum(Column):
87
87
 
88
88
  @property
89
89
  def pyarrow_dtype(self) -> pa.DataType:
90
- if len(self.categories) <= 2**8 - 2:
90
+ if len(self.categories) <= 2**8 - 1:
91
91
  dtype = pa.uint8()
92
- elif len(self.categories) <= 2**16 - 2:
92
+ elif len(self.categories) <= 2**16 - 1:
93
93
  dtype = pa.uint16()
94
94
  else:
95
95
  dtype = pa.uint32()
@@ -41,3 +41,10 @@ class AnnotationImplementationError(ImplementationError):
41
41
 
42
42
  class ValidationRequiredError(Exception):
43
43
  """Error raised when validation is required when reading a parquet file."""
44
+
45
+
46
+ # ---------------------------------- DESERIALIZATION --------------------------------- #
47
+
48
+
49
+ class DeserializationError(Exception):
50
+ """Error raised when deserialization of a schema or collection fails."""
@@ -10,7 +10,7 @@ from pathlib import Path
10
10
  from typing import IO, TYPE_CHECKING, Any, Generic, TypeVar
11
11
 
12
12
  import polars as pl
13
- from polars._typing import PartitioningScheme
13
+ from polars.io.partition import _SinkDirectory as SinkDirectory
14
14
 
15
15
  from dataframely._base_schema import BaseSchema
16
16
  from dataframely._compat import deltalake
@@ -164,7 +164,7 @@ class FailureInfo(Generic[S]):
164
164
  self._write(ParquetStorageBackend(), file=file, **kwargs)
165
165
 
166
166
  def sink_parquet(
167
- self, file: str | Path | IO[bytes] | PartitioningScheme, **kwargs: Any
167
+ self, file: str | Path | IO[bytes] | SinkDirectory, **kwargs: Any
168
168
  ) -> None:
169
169
  """Stream the failure info to a single parquet file.
170
170
 
@@ -14,7 +14,8 @@ from typing import IO, Any, Literal, overload
14
14
 
15
15
  import polars as pl
16
16
  import polars.exceptions as plexc
17
- from polars._typing import FileSource, PartitioningScheme
17
+ from polars._typing import FileSource
18
+ from polars.io.partition import _SinkDirectory as SinkDirectory
18
19
 
19
20
  from dataframely._compat import deltalake
20
21
 
@@ -40,7 +41,12 @@ from ._storage.parquet import (
40
41
  from ._typing import DataFrame, LazyFrame, Validation
41
42
  from .columns import Column, column_from_dict
42
43
  from .config import Config
43
- from .exc import SchemaError, ValidationError, ValidationRequiredError
44
+ from .exc import (
45
+ DeserializationError,
46
+ SchemaError,
47
+ ValidationError,
48
+ ValidationRequiredError,
49
+ )
44
50
  from .filter_result import FailureInfo, FilterResult, LazyFilterResult
45
51
  from .random import Generator
46
52
 
@@ -856,7 +862,7 @@ class Schema(BaseSchema, ABC):
856
862
  cls,
857
863
  lf: LazyFrame[Self],
858
864
  /,
859
- file: str | Path | IO[bytes] | PartitioningScheme,
865
+ file: str | Path | IO[bytes] | SinkDirectory,
860
866
  **kwargs: Any,
861
867
  ) -> None:
862
868
  """Stream a typed lazy frame with this schema to a parquet file.
@@ -1238,8 +1244,13 @@ class Schema(BaseSchema, ABC):
1238
1244
  validation: Validation,
1239
1245
  source: str,
1240
1246
  ) -> DataFrame[Self] | LazyFrame[Self]:
1247
+ # Use strict=False when validation is "allow", "warn" or "skip" to tolerate
1248
+ # deserialization failures from old serialized formats.
1249
+ strict = validation == "forbid"
1241
1250
  deserialized_schema = (
1242
- deserialize_schema(serialized_schema) if serialized_schema else None
1251
+ deserialize_schema(serialized_schema, strict=strict)
1252
+ if serialized_schema
1253
+ else None
1243
1254
  )
1244
1255
 
1245
1256
  # Smart validation
@@ -1347,6 +1358,10 @@ def deserialize_schema(data: str, strict: Literal[True] = True) -> type[Schema]:
1347
1358
  def deserialize_schema(data: str, strict: Literal[False]) -> type[Schema] | None: ...
1348
1359
 
1349
1360
 
1361
+ @overload
1362
+ def deserialize_schema(data: str, strict: bool) -> type[Schema] | None: ...
1363
+
1364
+
1350
1365
  def deserialize_schema(data: str, strict: bool = True) -> type[Schema] | None:
1351
1366
  """Deserialize a schema from a JSON string.
1352
1367
 
@@ -1375,9 +1390,11 @@ def deserialize_schema(data: str, strict: bool = True) -> type[Schema] | None:
1375
1390
  if (format := decoded["versions"]["format"]) != SERIALIZATION_FORMAT_VERSION:
1376
1391
  raise ValueError(f"Unsupported schema format version: {format}")
1377
1392
  return _schema_from_dict(decoded)
1378
- except (ValueError, JSONDecodeError, plexc.ComputeError) as e:
1393
+ except (ValueError, JSONDecodeError, plexc.ComputeError, TypeError) as e:
1379
1394
  if strict:
1380
- raise e from e
1395
+ raise DeserializationError(
1396
+ "The Schema metadata could not be deserialized"
1397
+ ) from e
1381
1398
  return None
1382
1399
 
1383
1400
 
@@ -29,11 +29,11 @@ class SchemaStorageTester(ABC):
29
29
  def write_typed(
30
30
  self, schema: type[S], df: dy.DataFrame[S], path: str, lazy: bool
31
31
  ) -> None:
32
- """Write a schema to the backend without recording schema information."""
32
+ """Write a schema to the backend and record schema information."""
33
33
 
34
34
  @abstractmethod
35
35
  def write_untyped(self, df: pl.DataFrame, path: str, lazy: bool) -> None:
36
- """Write a schema to the backend and record schema information."""
36
+ """Write a schema to the backend without recording schema information."""
37
37
 
38
38
  @overload
39
39
  def read(
@@ -45,12 +45,22 @@ class SchemaStorageTester(ABC):
45
45
  self, schema: type[S], path: str, lazy: Literal[False], validation: Validation
46
46
  ) -> dy.DataFrame[S]: ...
47
47
 
48
+ @overload
49
+ def read(
50
+ self, schema: type[S], path: str, lazy: bool, validation: Validation
51
+ ) -> dy.LazyFrame[S] | dy.DataFrame[S]: ...
52
+
48
53
  @abstractmethod
49
54
  def read(
50
55
  self, schema: type[S], path: str, lazy: bool, validation: Validation
51
56
  ) -> dy.LazyFrame[S] | dy.DataFrame[S]:
52
57
  """Read from the backend, using schema information if available."""
53
58
 
59
+ @abstractmethod
60
+ def set_metadata(self, path: str, metadata: dict[str, Any]) -> None:
61
+ """Overwrite the metadata stored at the given path with the provided
62
+ metadata."""
63
+
54
64
 
55
65
  class ParquetSchemaStorageTester(SchemaStorageTester):
56
66
  """Testing interface for the parquet storage functionality of Schema."""
@@ -83,6 +93,11 @@ class ParquetSchemaStorageTester(SchemaStorageTester):
83
93
  self, schema: type[S], path: str, lazy: Literal[False], validation: Validation
84
94
  ) -> dy.DataFrame[S]: ...
85
95
 
96
+ @overload
97
+ def read(
98
+ self, schema: type[S], path: str, lazy: bool, validation: Validation
99
+ ) -> dy.LazyFrame[S] | dy.DataFrame[S]: ...
100
+
86
101
  def read(
87
102
  self, schema: type[S], path: str, lazy: bool, validation: Validation
88
103
  ) -> dy.LazyFrame[S] | dy.DataFrame[S]:
@@ -93,6 +108,11 @@ class ParquetSchemaStorageTester(SchemaStorageTester):
93
108
  else:
94
109
  return schema.read_parquet(self._wrap_path(path), validation=validation)
95
110
 
111
+ def set_metadata(self, path: str, metadata: dict[str, Any]) -> None:
112
+ target = self._wrap_path(path)
113
+ data = pl.read_parquet(target)
114
+ data.write_parquet(target, metadata=metadata)
115
+
96
116
 
97
117
  class DeltaSchemaStorageTester(SchemaStorageTester):
98
118
  """Testing interface for the deltalake storage functionality of Schema."""
@@ -115,6 +135,11 @@ class DeltaSchemaStorageTester(SchemaStorageTester):
115
135
  self, schema: type[S], path: str, lazy: Literal[False], validation: Validation
116
136
  ) -> dy.DataFrame[S]: ...
117
137
 
138
+ @overload
139
+ def read(
140
+ self, schema: type[S], path: str, lazy: bool, validation: Validation
141
+ ) -> dy.LazyFrame[S] | dy.DataFrame[S]: ...
142
+
118
143
  def read(
119
144
  self, schema: type[S], path: str, lazy: bool, validation: Validation
120
145
  ) -> dy.DataFrame[S] | dy.LazyFrame[S]:
@@ -122,6 +147,18 @@ class DeltaSchemaStorageTester(SchemaStorageTester):
122
147
  return schema.scan_delta(path, validation=validation)
123
148
  return schema.read_delta(path, validation=validation)
124
149
 
150
+ def set_metadata(self, path: str, metadata: dict[str, Any]) -> None:
151
+ df = pl.read_delta(path)
152
+ df.head(0).write_delta(
153
+ path,
154
+ delta_write_options={
155
+ "commit_properties": deltalake.CommitProperties(
156
+ custom_metadata=metadata
157
+ ),
158
+ },
159
+ mode="overwrite",
160
+ )
161
+
125
162
 
126
163
  # ------------------------------- Collection -------------------------------------------
127
164
 
@@ -147,11 +184,36 @@ class CollectionStorageTester(ABC):
147
184
  def read(self, collection: type[C], path: str, lazy: bool, **kwargs: Any) -> C:
148
185
  """Read from the backend, using collection information if available."""
149
186
 
187
+ @abstractmethod
188
+ def set_metadata(self, path: str, metadata: dict[str, Any]) -> None:
189
+ """Overwrite the metadata stored at the given path with the provided
190
+ metadata."""
191
+
192
+ def _prefix_path(self, path: str, fs: AbstractFileSystem) -> str:
193
+ return f"{self._get_prefix(fs)}{path}"
194
+
195
+ @staticmethod
196
+ def _get_prefix(fs: AbstractFileSystem) -> str:
197
+ return (
198
+ ""
199
+ if fs.protocol == "file"
200
+ else (
201
+ f"{fs.protocol}://"
202
+ if isinstance(fs.protocol, str)
203
+ else f"{fs.protocol[0]}://"
204
+ )
205
+ )
206
+
150
207
 
151
208
  class ParquetCollectionStorageTester(CollectionStorageTester):
152
209
  def write_typed(
153
210
  self, collection: dy.Collection, path: str, lazy: bool, **kwargs: Any
154
211
  ) -> None:
212
+ if "metadata" in kwargs: # pragma: no cover
213
+ raise KeyError(
214
+ "`metadata` kwarg will be ignored in `write_typed`. Use `set_metadata`."
215
+ )
216
+
155
217
  # Polars does not support partitioning via kwarg on sink_parquet
156
218
  if lazy:
157
219
  kwargs.pop("partition_by", None)
@@ -164,6 +226,11 @@ class ParquetCollectionStorageTester(CollectionStorageTester):
164
226
  def write_untyped(
165
227
  self, collection: dy.Collection, path: str, lazy: bool, **kwargs: Any
166
228
  ) -> None:
229
+ if "metadata" in kwargs: # pragma: no cover
230
+ raise KeyError(
231
+ "Cannot set metadata through `write_untyped`. Use `set_metadata`."
232
+ )
233
+
167
234
  if lazy:
168
235
  collection.sink_parquet(path, **kwargs)
169
236
  else:
@@ -175,17 +242,8 @@ class ParquetCollectionStorageTester(CollectionStorageTester):
175
242
  df.write_parquet(file)
176
243
 
177
244
  fs: AbstractFileSystem = url_to_fs(path)[0]
178
- prefix = (
179
- ""
180
- if fs.protocol == "file"
181
- else (
182
- f"{fs.protocol}://"
183
- if isinstance(fs.protocol, str)
184
- else f"{fs.protocol[0]}://"
185
- )
186
- )
187
245
  for file in fs.glob(fs.sep.join([path, "**", "*.parquet"])):
188
- _delete_meta(f"{prefix}{file}")
246
+ _delete_meta(self._prefix_path(file, fs))
189
247
 
190
248
  def read(self, collection: type[C], path: str, lazy: bool, **kwargs: Any) -> C:
191
249
  if lazy:
@@ -193,11 +251,23 @@ class ParquetCollectionStorageTester(CollectionStorageTester):
193
251
  else:
194
252
  return collection.read_parquet(path, **kwargs)
195
253
 
254
+ def set_metadata(self, path: str, metadata: dict[str, Any]) -> None:
255
+ fs: AbstractFileSystem = url_to_fs(path)[0]
256
+ for file in fs.glob(fs.sep.join([path, "*.parquet"])):
257
+ file_path = self._prefix_path(file, fs)
258
+ df = pl.read_parquet(file_path)
259
+ df.write_parquet(file_path, metadata=metadata)
260
+
196
261
 
197
262
  class DeltaCollectionStorageTester(CollectionStorageTester):
198
263
  def write_typed(
199
264
  self, collection: dy.Collection, path: str, lazy: bool, **kwargs: Any
200
265
  ) -> None:
266
+ if "metadata" in kwargs: # pragma: no cover
267
+ raise KeyError(
268
+ "`metadata` kwarg will be ignored in `write_typed`. Use `set_metadata`."
269
+ )
270
+
201
271
  extra_kwargs = {}
202
272
  if partition_by := kwargs.pop("partition_by", None):
203
273
  extra_kwargs["delta_write_options"] = {"partition_by": partition_by}
@@ -207,6 +277,10 @@ class DeltaCollectionStorageTester(CollectionStorageTester):
207
277
  def write_untyped(
208
278
  self, collection: dy.Collection, path: str, lazy: bool, **kwargs: Any
209
279
  ) -> None:
280
+ if "metadata" in kwargs: # pragma: no cover
281
+ raise KeyError(
282
+ "Cannot set metadata through `write_untyped`. Use `set_metadata`."
283
+ )
210
284
  collection.write_delta(path, **kwargs)
211
285
 
212
286
  # For each member table, write an empty commit
@@ -222,6 +296,23 @@ class DeltaCollectionStorageTester(CollectionStorageTester):
222
296
  return collection.scan_delta(source=path, **kwargs)
223
297
  return collection.read_delta(source=path, **kwargs)
224
298
 
299
+ def set_metadata(self, path: str, metadata: dict[str, Any]) -> None:
300
+ fs: AbstractFileSystem = url_to_fs(path)[0]
301
+ # For delta, we need to update metadata on each member table
302
+ for entry in fs.ls(path):
303
+ member_path = self._prefix_path(entry, fs)
304
+ if fs.isdir(member_path):
305
+ df = pl.read_delta(member_path)
306
+ df.head(0).write_delta(
307
+ member_path,
308
+ delta_write_options={
309
+ "commit_properties": deltalake.CommitProperties(
310
+ custom_metadata=metadata
311
+ ),
312
+ },
313
+ mode="overwrite",
314
+ )
315
+
225
316
 
226
317
  # ------------------------------------ Failure info ------------------------------------
227
318
  class FailureInfoStorageTester(ABC):