dataframely 1.9.0__tar.gz → 1.10.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (200) hide show
  1. {dataframely-1.9.0 → dataframely-1.10.0}/.github/workflows/ci.yml +9 -1
  2. {dataframely-1.9.0 → dataframely-1.10.0}/Cargo.toml +1 -0
  3. {dataframely-1.9.0 → dataframely-1.10.0}/PKG-INFO +7 -1
  4. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_compat.py +29 -1
  5. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_serialization.py +0 -2
  6. dataframely-1.10.0/dataframely/_storage/__init__.py +8 -0
  7. dataframely-1.10.0/dataframely/_storage/_base.py +204 -0
  8. dataframely-1.10.0/dataframely/_storage/_exc.py +10 -0
  9. dataframely-1.10.0/dataframely/_storage/constants.py +6 -0
  10. dataframely-1.10.0/dataframely/_storage/delta.py +200 -0
  11. dataframely-1.10.0/dataframely/_storage/parquet.py +239 -0
  12. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/collection.py +240 -85
  13. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/decimal.py +6 -5
  14. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/failure.py +130 -31
  15. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/schema.py +263 -24
  16. dataframely-1.10.0/dataframely/testing/storage.py +293 -0
  17. {dataframely-1.9.0 → dataframely-1.10.0}/pixi.lock +6250 -6485
  18. {dataframely-1.9.0 → dataframely-1.10.0}/pixi.toml +9 -3
  19. {dataframely-1.9.0 → dataframely-1.10.0}/pyproject.toml +9 -1
  20. dataframely-1.10.0/tests/collection/test_storage.py +407 -0
  21. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_decimal.py +16 -3
  22. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_pyarrow.py +2 -0
  23. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_sql_schema.py +10 -10
  24. dataframely-1.10.0/tests/failure_info/test_storage.py +131 -0
  25. dataframely-1.10.0/tests/schema/test_read_write_parquet.py +27 -0
  26. dataframely-1.10.0/tests/schema/test_storage.py +250 -0
  27. dataframely-1.10.0/tests/storage/test_delta.py +54 -0
  28. dataframely-1.9.0/tests/collection/test_read_write_parquet.py +0 -398
  29. dataframely-1.9.0/tests/schema/test_read_write_parquet.py +0 -234
  30. dataframely-1.9.0/tests/test_failure_info.py +0 -109
  31. {dataframely-1.9.0 → dataframely-1.10.0}/.copier-answers.yml +0 -0
  32. {dataframely-1.9.0 → dataframely-1.10.0}/.envrc +0 -0
  33. {dataframely-1.9.0 → dataframely-1.10.0}/.gitattributes +0 -0
  34. {dataframely-1.9.0 → dataframely-1.10.0}/.github/CODEOWNERS +0 -0
  35. {dataframely-1.9.0 → dataframely-1.10.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  36. {dataframely-1.9.0 → dataframely-1.10.0}/.github/dependabot.yml +0 -0
  37. {dataframely-1.9.0 → dataframely-1.10.0}/.github/release-drafter.yml +0 -0
  38. {dataframely-1.9.0 → dataframely-1.10.0}/.github/workflows/build.yml +0 -0
  39. {dataframely-1.9.0 → dataframely-1.10.0}/.github/workflows/chore.yml +0 -0
  40. {dataframely-1.9.0 → dataframely-1.10.0}/.github/workflows/nightly.yml +0 -0
  41. {dataframely-1.9.0 → dataframely-1.10.0}/.github/workflows/scorecard.yml +0 -0
  42. {dataframely-1.9.0 → dataframely-1.10.0}/.gitignore +0 -0
  43. {dataframely-1.9.0 → dataframely-1.10.0}/.pre-commit-config.yaml +0 -0
  44. {dataframely-1.9.0 → dataframely-1.10.0}/.prettierignore +0 -0
  45. {dataframely-1.9.0 → dataframely-1.10.0}/.prettierrc +0 -0
  46. {dataframely-1.9.0 → dataframely-1.10.0}/.readthedocs.yml +0 -0
  47. {dataframely-1.9.0 → dataframely-1.10.0}/Cargo.lock +0 -0
  48. {dataframely-1.9.0 → dataframely-1.10.0}/LICENSE +0 -0
  49. {dataframely-1.9.0 → dataframely-1.10.0}/README.md +0 -0
  50. {dataframely-1.9.0 → dataframely-1.10.0}/SECURITY.md +0 -0
  51. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/__init__.py +0 -0
  52. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_base_collection.py +0 -0
  53. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_base_schema.py +0 -0
  54. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_deprecation.py +0 -0
  55. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_extre.pyi +0 -0
  56. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_filter.py +0 -0
  57. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_polars.py +0 -0
  58. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_rule.py +0 -0
  59. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_typing.py +0 -0
  60. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_validation.py +0 -0
  61. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/__init__.py +0 -0
  62. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/_base.py +0 -0
  63. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/_mixins.py +0 -0
  64. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/_registry.py +0 -0
  65. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/_utils.py +0 -0
  66. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/any.py +0 -0
  67. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/array.py +0 -0
  68. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/bool.py +0 -0
  69. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/categorical.py +0 -0
  70. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/datetime.py +0 -0
  71. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/enum.py +0 -0
  72. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/float.py +0 -0
  73. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/integer.py +0 -0
  74. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/list.py +0 -0
  75. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/object.py +0 -0
  76. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/string.py +0 -0
  77. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/struct.py +0 -0
  78. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/config.py +0 -0
  79. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/exc.py +0 -0
  80. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/functional.py +0 -0
  81. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/mypy.py +0 -0
  82. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/py.typed +0 -0
  83. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/random.py +0 -0
  84. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/__init__.py +0 -0
  85. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/const.py +0 -0
  86. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/factory.py +0 -0
  87. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/mask.py +0 -0
  88. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/rules.py +0 -0
  89. {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/typing.py +0 -0
  90. {dataframely-1.9.0 → dataframely-1.10.0}/docker-compose.yml +0 -0
  91. {dataframely-1.9.0 → dataframely-1.10.0}/docs/Makefile +0 -0
  92. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.collection.rst +0 -0
  93. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.any.rst +0 -0
  94. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.bool.rst +0 -0
  95. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.datetime.rst +0 -0
  96. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.decimal.rst +0 -0
  97. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.enum.rst +0 -0
  98. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.float.rst +0 -0
  99. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.integer.rst +0 -0
  100. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.list.rst +0 -0
  101. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.rst +0 -0
  102. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.string.rst +0 -0
  103. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.struct.rst +0 -0
  104. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.config.rst +0 -0
  105. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.exc.rst +0 -0
  106. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.failure.rst +0 -0
  107. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.functional.rst +0 -0
  108. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.mypy.rst +0 -0
  109. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.random.rst +0 -0
  110. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.rst +0 -0
  111. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.schema.rst +0 -0
  112. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.const.rst +0 -0
  113. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.factory.rst +0 -0
  114. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.mask.rst +0 -0
  115. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.rst +0 -0
  116. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.rules.rst +0 -0
  117. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.typing.rst +0 -0
  118. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/modules.rst +0 -0
  119. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_static/custom.css +0 -0
  120. {dataframely-1.9.0 → dataframely-1.10.0}/docs/_static/favicon.ico +0 -0
  121. {dataframely-1.9.0 → dataframely-1.10.0}/docs/conf.py +0 -0
  122. {dataframely-1.9.0 → dataframely-1.10.0}/docs/index.rst +0 -0
  123. {dataframely-1.9.0 → dataframely-1.10.0}/docs/make.bat +0 -0
  124. {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/development.rst +0 -0
  125. {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/examples/real-world.ipynb +0 -0
  126. {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/faq.rst +0 -0
  127. {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/installation.rst +0 -0
  128. {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/quickstart.rst +0 -0
  129. {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/versioning.rst +0 -0
  130. {dataframely-1.9.0 → dataframely-1.10.0}/src/errdefs.rs +0 -0
  131. {dataframely-1.9.0 → dataframely-1.10.0}/src/lib.rs +0 -0
  132. {dataframely-1.9.0 → dataframely-1.10.0}/src/regex_repr.rs +0 -0
  133. {dataframely-1.9.0 → dataframely-1.10.0}/tests/benches/conftest.py +0 -0
  134. {dataframely-1.9.0 → dataframely-1.10.0}/tests/benches/test_collection.py +0 -0
  135. {dataframely-1.9.0 → dataframely-1.10.0}/tests/benches/test_failure.py +0 -0
  136. {dataframely-1.9.0 → dataframely-1.10.0}/tests/benches/test_schema.py +0 -0
  137. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_base.py +0 -0
  138. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_cast.py +0 -0
  139. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_collection_future_annotations.py +0 -0
  140. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_create_empty.py +0 -0
  141. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_filter_one_to_n.py +0 -0
  142. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_filter_validate.py +0 -0
  143. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_ignore_in_filter.py +0 -0
  144. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_implementation.py +0 -0
  145. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_join.py +0 -0
  146. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_matches.py +0 -0
  147. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_optional_members.py +0 -0
  148. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_repr.py +0 -0
  149. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_sample.py +0 -0
  150. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_serialization.py +0 -0
  151. {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_validate_input.py +0 -0
  152. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/__init__.py +0 -0
  153. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_any.py +0 -0
  154. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_array.py +0 -0
  155. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_datetime.py +0 -0
  156. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_enum.py +0 -0
  157. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_float.py +0 -0
  158. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_integer.py +0 -0
  159. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_list.py +0 -0
  160. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_object.py +0 -0
  161. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_string.py +0 -0
  162. {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_struct.py +0 -0
  163. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/__init__.py +0 -0
  164. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_alias.py +0 -0
  165. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_check.py +0 -0
  166. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_default_dtypes.py +0 -0
  167. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_matches.py +0 -0
  168. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_metadata.py +0 -0
  169. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_polars_schema.py +0 -0
  170. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_rules.py +0 -0
  171. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_sample.py +0 -0
  172. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_str.py +0 -0
  173. {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_utils.py +0 -0
  174. {dataframely-1.9.0 → dataframely-1.10.0}/tests/core_validation/__init__.py +0 -0
  175. {dataframely-1.9.0 → dataframely-1.10.0}/tests/core_validation/test_column_validation.py +0 -0
  176. {dataframely-1.9.0 → dataframely-1.10.0}/tests/core_validation/test_dtype_validation.py +0 -0
  177. {dataframely-1.9.0 → dataframely-1.10.0}/tests/core_validation/test_rule_evaluation.py +0 -0
  178. {dataframely-1.9.0 → dataframely-1.10.0}/tests/functional/test_concat.py +0 -0
  179. {dataframely-1.9.0 → dataframely-1.10.0}/tests/functional/test_relationships.py +0 -0
  180. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_base.py +0 -0
  181. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_cast.py +0 -0
  182. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_create_empty.py +0 -0
  183. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_create_empty_if_none.py +0 -0
  184. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_filter.py +0 -0
  185. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_inheritance.py +0 -0
  186. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_matches.py +0 -0
  187. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_repr.py +0 -0
  188. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_rule_implementation.py +0 -0
  189. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_sample.py +0 -0
  190. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_serialization.py +0 -0
  191. {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_validate.py +0 -0
  192. {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_compat.py +0 -0
  193. {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_config.py +0 -0
  194. {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_deprecation.py +0 -0
  195. {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_exc.py +0 -0
  196. {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_extre.py +0 -0
  197. {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_factory.py +0 -0
  198. {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_random.py +0 -0
  199. {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_serialization.py +0 -0
  200. {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_typing.py +0 -0
@@ -41,6 +41,14 @@ jobs:
41
41
  matrix:
42
42
  os: [ubuntu-latest, windows-latest]
43
43
  environment: [py310, py311, py312, py313]
44
+ with_optionals: [false]
45
+ include:
46
+ - os: ubuntu-latest
47
+ environment: py313-optionals
48
+ with_optionals: true
49
+ - os: windows-latest
50
+ environment: py313-optionals
51
+ with_optionals: true
44
52
  steps:
45
53
  - name: Checkout branch
46
54
  uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2
@@ -51,7 +59,7 @@ jobs:
51
59
  - name: Install repository
52
60
  run: pixi run -e ${{ matrix.environment }} postinstall
53
61
  - name: Run pytest
54
- run: pixi run -e ${{ matrix.environment }} test-coverage --color=yes
62
+ run: pixi run -e ${{ matrix.environment }} test-coverage --color=yes ${{ matrix.with_optionals && '-m with_optionals' || '-m "not with_optionals"'}} --cov=dataframely --cov-report=xml
55
63
  - name: Upload codecov
56
64
  uses: codecov/codecov-action@18283e04ce6e62d37312384ff67231eb8fd56d24 # v5.4.3
57
65
  with:
@@ -2,6 +2,7 @@
2
2
  edition = "2021"
3
3
  name = "dataframely"
4
4
  version = "0.1.0"
5
+ readme = "README.md"
5
6
 
6
7
  [lib]
7
8
  crate-type = ["cdylib"]
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dataframely
3
- Version: 1.9.0
3
+ Version: 1.10.0
4
4
  Classifier: Programming Language :: Python :: 3
5
5
  Classifier: Programming Language :: Python :: 3.10
6
6
  Classifier: Programming Language :: Python :: 3.11
@@ -9,6 +9,12 @@ Classifier: Programming Language :: Python :: 3.13
9
9
  Requires-Dist: numpy
10
10
  Requires-Dist: polars>=1.32
11
11
  Requires-Dist: typing-extensions ; python_full_version < '3.11'
12
+ Requires-Dist: deltalake ; extra == 'deltalake'
13
+ Requires-Dist: sqlalchemy ; extra == 'sqlalchemy'
14
+ Requires-Dist: pyarrow ; extra == 'pyarrow'
15
+ Provides-Extra: deltalake
16
+ Provides-Extra: sqlalchemy
17
+ Provides-Extra: pyarrow
12
18
  License-File: LICENSE
13
19
  Summary: A declarative, polars-native data frame validation library
14
20
  Author-email: Andreas Albert <andreas.albert@quantco.com>, Daniel Elsner <daniel.elsner@quantco.com>, Oliver Borchert <oliver.borchert@quantco.com>
@@ -13,11 +13,24 @@ class _DummyModule: # pragma: no cover
13
13
  raise ValueError(f"Module '{self.module}' is not installed.")
14
14
 
15
15
 
16
+ # ------------------------------------ DELTALAKE ------------------------------------- #
17
+
18
+ try:
19
+ import deltalake
20
+ from deltalake import DeltaTable
21
+ except ImportError: # pragma: no cover
22
+ deltalake = _DummyModule("deltalake") # type: ignore
23
+
24
+ class DeltaTable: # type: ignore # noqa: N801
25
+ pass
16
26
  # ------------------------------------ SQLALCHEMY ------------------------------------ #
17
27
 
18
28
  try:
19
29
  import sqlalchemy as sa
20
30
  import sqlalchemy.dialects.mssql as sa_mssql
31
+ from sqlalchemy import Dialect
32
+ from sqlalchemy.dialects.mssql.pyodbc import MSDialect_pyodbc
33
+ from sqlalchemy.dialects.postgresql.psycopg2 import PGDialect_psycopg2
21
34
  from sqlalchemy.sql.type_api import TypeEngine as sa_TypeEngine
22
35
  except ImportError: # pragma: no cover
23
36
  sa = _DummyModule("sqlalchemy") # type: ignore
@@ -26,7 +39,14 @@ except ImportError: # pragma: no cover
26
39
  class sa_TypeEngine: # type: ignore # noqa: N801
27
40
  pass
28
41
 
42
+ class MSDialect_pyodbc: # type: ignore # noqa: N801
43
+ pass
44
+
45
+ class PGDialect_psycopg2: # type: ignore # noqa: N801
46
+ pass
29
47
 
48
+ class Dialect: # type: ignore # noqa: N801
49
+ pass
30
50
  # -------------------------------------- PYARROW ------------------------------------- #
31
51
 
32
52
  try:
@@ -36,4 +56,12 @@ except ImportError: # pragma: no cover
36
56
 
37
57
  # ------------------------------------------------------------------------------------ #
38
58
 
39
- __all__ = ["sa", "sa_mssql", "sa_TypeEngine", "pa"]
59
+ __all__ = [
60
+ "deltalake",
61
+ "sa",
62
+ "sa_mssql",
63
+ "sa_TypeEngine",
64
+ "pa",
65
+ "MSDialect_pyodbc",
66
+ "PGDialect_psycopg2",
67
+ ]
@@ -10,8 +10,6 @@ from typing import Any, cast
10
10
 
11
11
  import polars as pl
12
12
 
13
- SCHEMA_METADATA_KEY = "dataframely_schema"
14
- COLLECTION_METADATA_KEY = "dataframely_collection"
15
13
  SERIALIZATION_FORMAT_VERSION = "1"
16
14
 
17
15
 
@@ -0,0 +1,8 @@
1
+ # Copyright (c) QuantCo 2025-2025
2
+ # SPDX-License-Identifier: BSD-3-Clause
3
+
4
+ from ._base import StorageBackend
5
+
6
+ __all__ = [
7
+ "StorageBackend",
8
+ ]
@@ -0,0 +1,204 @@
1
+ # Copyright (c) QuantCo 2025-2025
2
+ # SPDX-License-Identifier: BSD-3-Clause
3
+
4
+ from abc import ABC, abstractmethod
5
+ from collections.abc import Iterable
6
+ from typing import Any
7
+
8
+ import polars as pl
9
+
10
+ SerializedSchema = str
11
+ SerializedCollection = str
12
+ SerializedRules = str
13
+
14
+
15
+ class StorageBackend(ABC):
16
+ """Base class for storage backends.
17
+
18
+ A storage backend encapsulates a way of serializing and deserializing dataframlely
19
+ data-/lazyframes and collections. This base class provides a unified interface for
20
+ all such use cases.
21
+
22
+ The interface is designed to operate on data provided as polars frames, and metadata
23
+ provided as serialized strings. This design is meant to limit the coupling between
24
+ the Schema/Collection classes and specifics of how data and metadata is stored.
25
+ """
26
+
27
+ # ----------------------------------- Schemas -------------------------------------
28
+ @abstractmethod
29
+ def sink_frame(
30
+ self, lf: pl.LazyFrame, serialized_schema: SerializedSchema, **kwargs: Any
31
+ ) -> None:
32
+ """Stream the contents of a dataframe, and its metadata to the storage backend.
33
+
34
+ Args:
35
+ lf: A frame containing the data to be stored.
36
+ serialized_schema: String-serialized schema information.
37
+ kwargs: Additional keyword arguments to pass to the underlying storage
38
+ implementation.
39
+ """
40
+
41
+ @abstractmethod
42
+ def write_frame(
43
+ self, df: pl.DataFrame, serialized_schema: SerializedSchema, **kwargs: Any
44
+ ) -> None:
45
+ """Write the contents of a dataframe, and its metadata to the storage backend.
46
+
47
+ Args:
48
+ df: A dataframe containing the data to be stored.
49
+ frame: String-serialized schema information.
50
+ kwargs: Additional keyword arguments to pass to the underlying storage
51
+ implementation.
52
+ """
53
+
54
+ @abstractmethod
55
+ def scan_frame(self, **kwargs: Any) -> tuple[pl.LazyFrame, SerializedSchema | None]:
56
+ """Lazily read frame data and metadata from the storage backend.
57
+
58
+ Args:
59
+ kwargs: Keyword arguments to pass to the underlying storage.
60
+ Refer to the individual implementation to see which keywords
61
+ are available.
62
+ Returns:
63
+ A tuple of the lazy frame data and metadata if available.
64
+ """
65
+
66
+ @abstractmethod
67
+ def read_frame(self, **kwargs: Any) -> tuple[pl.DataFrame, SerializedSchema | None]:
68
+ """Eagerly read frame data and metadata from the storage backend.
69
+
70
+ Args:
71
+ kwargs: Keyword arguments to pass to the underlying storage.
72
+ Refer to the individual implementation to see which keywords
73
+ are available.
74
+ Returns:
75
+ A tuple of the lazy frame data and metadata if available.
76
+ """
77
+
78
+ # ------------------------------ Collections ---------------------------------------
79
+ @abstractmethod
80
+ def sink_collection(
81
+ self,
82
+ dfs: dict[str, pl.LazyFrame],
83
+ serialized_collection: SerializedCollection,
84
+ serialized_schemas: dict[str, str],
85
+ **kwargs: Any,
86
+ ) -> None:
87
+ """Stream the members of this collection into the storage backend.
88
+
89
+ Args:
90
+ dfs: Dictionary containing the data to be stored.
91
+ serialized_collection: String-serialized information about the origin Collection.
92
+ serialized_schemas: String-serialized information about the individual Schemas
93
+ for each of the member frames. This information is also logically included
94
+ in the collection metadata, but it is passed separately here to ensure that
95
+ each member can also be read back as an individual frame.
96
+ """
97
+
98
+ @abstractmethod
99
+ def write_collection(
100
+ self,
101
+ dfs: dict[str, pl.LazyFrame],
102
+ serialized_collection: SerializedCollection,
103
+ serialized_schemas: dict[str, str],
104
+ **kwargs: Any,
105
+ ) -> None:
106
+ """Write the members of this collection into the storage backend.
107
+
108
+ Args:
109
+ dfs: Dictionary containing the data to be stored.
110
+ serialized_collection: String-serialized information about the origin Collection.
111
+ serialized_schemas: String-serialized information about the individual Schemas
112
+ for each of the member frames. This information is also logically included
113
+ in the collection metadata, but it is passed separately here to ensure that
114
+ each member can also be read back as an individual frame.
115
+ """
116
+
117
+ @abstractmethod
118
+ def scan_collection(
119
+ self, members: Iterable[str], **kwargs: Any
120
+ ) -> tuple[dict[str, pl.LazyFrame], list[SerializedCollection | None]]:
121
+ """Lazily read all collection members from the storage backend.
122
+
123
+ Args:
124
+ members: Collection member names to read.
125
+ kwargs: Additional keyword arguments to pass to the underlying storage.
126
+ Refer to the individual implementation to see which keywords are available.
127
+ Returns:
128
+ A tuple of the collection data and metadata if available.
129
+ Depending on the storage implementation, multiple copies of the metadata
130
+ may be available, which are returned as a list.
131
+ It is up to the caller to decide how to handle the presence/absence/consistency
132
+ of the returned values.
133
+ """
134
+
135
+ @abstractmethod
136
+ def read_collection(
137
+ self, members: Iterable[str], **kwargs: Any
138
+ ) -> tuple[dict[str, pl.LazyFrame], list[SerializedCollection | None]]:
139
+ """Lazily read all collection members from the storage backend.
140
+
141
+ Args:
142
+ members: Collection member names to read.
143
+ kwargs: Additional keyword arguments to pass to the underlying storage.
144
+ Refer to the individual implementation to see which keywords are available.
145
+ Returns:
146
+ A tuple of the collection data and metadata if available.
147
+ Depending on the storage implementation, multiple copies of the metadata
148
+ may be available, which are returned as a list.
149
+ It is up to the caller to decide how to handle the presence/absence/consistency
150
+ of the returned values.
151
+ """
152
+
153
+ # ------------------------------ Failure Info --------------------------------------
154
+ @abstractmethod
155
+ def sink_failure_info(
156
+ self,
157
+ lf: pl.LazyFrame,
158
+ serialized_rules: SerializedRules,
159
+ serialized_schema: SerializedSchema,
160
+ **kwargs: Any,
161
+ ) -> None:
162
+ """Stream the failure info to the storage backend.
163
+
164
+ Args:
165
+ lf: LazyFrame backing the failure info.
166
+ serialized_rules: JSON-serialized list of rule column names
167
+ used for validation.
168
+ serialized_schema: String-serialized schema information.
169
+ """
170
+
171
+ @abstractmethod
172
+ def write_failure_info(
173
+ self,
174
+ df: pl.DataFrame,
175
+ serialized_rules: SerializedRules,
176
+ serialized_schema: SerializedSchema,
177
+ **kwargs: Any,
178
+ ) -> None:
179
+ """Write the failure info to the storage backend.
180
+
181
+ Args:
182
+ df: DataFrame backing the failure info.
183
+ serialized_rules: JSON-serialized list of rule column names
184
+ used for validation.
185
+ serialized_schema: String-serialized schema information.
186
+ """
187
+
188
+ @abstractmethod
189
+ def scan_failure_info(
190
+ self, **kwargs: Any
191
+ ) -> tuple[pl.LazyFrame, SerializedRules, SerializedSchema]:
192
+ """Lazily read the failure info from the storage backend."""
193
+
194
+ def read_failure_info(
195
+ self, **kwargs: Any
196
+ ) -> tuple[pl.DataFrame, SerializedRules, SerializedSchema]:
197
+ """Read the failure info from the storage backend."""
198
+
199
+ lf, rule_metadata, schema_metadata = self.scan_failure_info(**kwargs)
200
+ return (
201
+ lf.collect(),
202
+ rule_metadata,
203
+ schema_metadata,
204
+ )
@@ -0,0 +1,10 @@
1
+ # Copyright (c) QuantCo 2025-2025
2
+ # SPDX-License-Identifier: BSD-3-Clause
3
+
4
+
5
+ def assert_failure_info_metadata(metadata: str | None) -> str:
6
+ if metadata:
7
+ return metadata
8
+ raise ValueError(
9
+ "The required FailureInfo metadata was not found in the storage backend."
10
+ )
@@ -0,0 +1,6 @@
1
+ # Copyright (c) QuantCo 2025-2025
2
+ # SPDX-License-Identifier: BSD-3-Clause
3
+
4
+ SCHEMA_METADATA_KEY = "dataframely_schema"
5
+ COLLECTION_METADATA_KEY = "dataframely_collection"
6
+ RULE_METADATA_KEY = "dataframely_rule_columns"
@@ -0,0 +1,200 @@
1
+ # Copyright (c) QuantCo 2025-2025
2
+ # SPDX-License-Identifier: BSD-3-Clause
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterable
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ import polars as pl
10
+
11
+ from dataframely._compat import deltalake
12
+
13
+ from ._base import (
14
+ SerializedCollection,
15
+ SerializedRules,
16
+ SerializedSchema,
17
+ StorageBackend,
18
+ )
19
+ from ._exc import assert_failure_info_metadata
20
+ from .constants import COLLECTION_METADATA_KEY, RULE_METADATA_KEY, SCHEMA_METADATA_KEY
21
+
22
+
23
+ class DeltaStorageBackend(StorageBackend):
24
+ def sink_frame(
25
+ self, lf: pl.LazyFrame, serialized_schema: SerializedSchema, **kwargs: Any
26
+ ) -> None:
27
+ _raise_on_lazy_write()
28
+
29
+ def write_frame(
30
+ self, df: pl.DataFrame, serialized_schema: SerializedSchema, **kwargs: Any
31
+ ) -> None:
32
+ target = kwargs.pop("target")
33
+ metadata = kwargs.pop("metadata", {})
34
+ delta_write_options = kwargs.pop("delta_write_options", {})
35
+
36
+ # Delta lake does not allow partitioning if there is only one column
37
+ # We dynamically remove this setting here to allow users to still specify it
38
+ # on the collection level without having to worry about each individual member
39
+ if len(df.columns) < 2:
40
+ delta_write_options.pop("partition_by", None)
41
+
42
+ df.write_delta(
43
+ target,
44
+ delta_write_options=(
45
+ delta_write_options
46
+ | {
47
+ "commit_properties": deltalake.CommitProperties(
48
+ custom_metadata=metadata
49
+ | {SCHEMA_METADATA_KEY: serialized_schema}
50
+ ),
51
+ }
52
+ ),
53
+ **kwargs,
54
+ )
55
+
56
+ def scan_frame(self, **kwargs: Any) -> tuple[pl.LazyFrame, SerializedSchema | None]:
57
+ table = _to_delta_table(kwargs.pop("source"))
58
+ serialized_schema = _read_serialized_schema(table)
59
+ df = pl.scan_delta(table, **kwargs)
60
+ return df, serialized_schema
61
+
62
+ def read_frame(self, **kwargs: Any) -> tuple[pl.DataFrame, SerializedSchema | None]:
63
+ table = _to_delta_table(kwargs.pop("source"))
64
+ serialized_schema = _read_serialized_schema(table)
65
+ df = pl.read_delta(table, **kwargs)
66
+ return df, serialized_schema
67
+
68
+ # ------------------------------ Collections ---------------------------------------
69
+ def sink_collection(
70
+ self,
71
+ dfs: dict[str, pl.LazyFrame],
72
+ serialized_collection: SerializedCollection,
73
+ serialized_schemas: dict[str, str],
74
+ **kwargs: Any,
75
+ ) -> None:
76
+ _raise_on_lazy_write()
77
+
78
+ def write_collection(
79
+ self,
80
+ dfs: dict[str, pl.LazyFrame],
81
+ serialized_collection: SerializedCollection,
82
+ serialized_schemas: dict[str, str],
83
+ **kwargs: Any,
84
+ ) -> None:
85
+ uri = Path(kwargs.pop("target"))
86
+
87
+ # The collection schema is serialized as part of the member parquet metadata
88
+ kwargs["metadata"] = kwargs.get("metadata", {}) | {
89
+ COLLECTION_METADATA_KEY: serialized_collection
90
+ }
91
+
92
+ for key, lf in dfs.items():
93
+ self.write_frame(
94
+ lf.collect(),
95
+ serialized_schema=serialized_schemas[key],
96
+ target=uri / key,
97
+ **kwargs,
98
+ )
99
+
100
+ def scan_collection(
101
+ self, members: Iterable[str], **kwargs: Any
102
+ ) -> tuple[dict[str, pl.LazyFrame], list[SerializedCollection | None]]:
103
+ uri = Path(kwargs.pop("source"))
104
+
105
+ data = {}
106
+ collection_types = []
107
+ for key in members:
108
+ member_uri = uri / key
109
+ if not deltalake.DeltaTable.is_deltatable(str(member_uri)):
110
+ continue
111
+ table = _to_delta_table(member_uri)
112
+ data[key] = pl.scan_delta(table, **kwargs)
113
+ collection_types.append(_read_serialized_collection(table))
114
+
115
+ return data, collection_types
116
+
117
+ def read_collection(
118
+ self, members: Iterable[str], **kwargs: Any
119
+ ) -> tuple[dict[str, pl.LazyFrame], list[SerializedCollection | None]]:
120
+ lazy, collection_types = self.scan_collection(members, **kwargs)
121
+ eager = {name: lf.collect().lazy() for name, lf in lazy.items()}
122
+ return eager, collection_types
123
+
124
+ # ------------------------------ Failure Info --------------------------------------
125
+ def sink_failure_info(
126
+ self,
127
+ lf: pl.LazyFrame,
128
+ serialized_rules: SerializedRules,
129
+ serialized_schema: SerializedSchema,
130
+ **kwargs: Any,
131
+ ) -> None:
132
+ _raise_on_lazy_write()
133
+
134
+ def write_failure_info(
135
+ self,
136
+ df: pl.DataFrame,
137
+ serialized_rules: SerializedRules,
138
+ serialized_schema: SerializedSchema,
139
+ **kwargs: Any,
140
+ ) -> None:
141
+ self.write_frame(
142
+ df,
143
+ serialized_schema,
144
+ metadata={
145
+ RULE_METADATA_KEY: serialized_rules,
146
+ },
147
+ **kwargs,
148
+ )
149
+
150
+ def scan_failure_info(
151
+ self, **kwargs: Any
152
+ ) -> tuple[pl.LazyFrame, SerializedRules, SerializedSchema]:
153
+ """Lazily read the failure info from the storage backend."""
154
+ table = _to_delta_table(kwargs.pop("source"))
155
+
156
+ # Metadata
157
+ serialized_rules = assert_failure_info_metadata(_read_serialized_rules(table))
158
+ serialized_schema = assert_failure_info_metadata(_read_serialized_schema(table))
159
+
160
+ # Data
161
+ lf = pl.scan_delta(table, **kwargs)
162
+
163
+ return lf, serialized_rules, serialized_schema
164
+
165
+
166
+ def _raise_on_lazy_write() -> None:
167
+ raise NotImplementedError("Lazy writes are not currently supported for deltalake.")
168
+
169
+
170
+ def _read_serialized_schema(table: deltalake.DeltaTable) -> SerializedSchema | None:
171
+ [last_commit] = table.history(limit=1)
172
+ return last_commit.get(SCHEMA_METADATA_KEY, None)
173
+
174
+
175
+ def _read_serialized_collection(
176
+ table: deltalake.DeltaTable,
177
+ ) -> SerializedCollection | None:
178
+ [last_commit] = table.history(limit=1)
179
+ return last_commit.get(COLLECTION_METADATA_KEY, None)
180
+
181
+
182
+ def _read_serialized_rules(
183
+ table: deltalake.DeltaTable,
184
+ ) -> SerializedRules | None:
185
+ [last_commit] = table.history(limit=1)
186
+ return last_commit.get(RULE_METADATA_KEY, None)
187
+
188
+
189
+ def _to_delta_table(
190
+ table: Path | str | deltalake.DeltaTable,
191
+ ) -> deltalake.DeltaTable:
192
+ from deltalake import DeltaTable
193
+
194
+ match table:
195
+ case DeltaTable():
196
+ return table
197
+ case str() | Path():
198
+ return DeltaTable(table)
199
+ case _:
200
+ raise TypeError(f"Unsupported type {table!r}")