dataframely 1.9.0__tar.gz → 1.10.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataframely-1.9.0 → dataframely-1.10.0}/.github/workflows/ci.yml +9 -1
- {dataframely-1.9.0 → dataframely-1.10.0}/Cargo.toml +1 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/PKG-INFO +7 -1
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_compat.py +29 -1
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_serialization.py +0 -2
- dataframely-1.10.0/dataframely/_storage/__init__.py +8 -0
- dataframely-1.10.0/dataframely/_storage/_base.py +204 -0
- dataframely-1.10.0/dataframely/_storage/_exc.py +10 -0
- dataframely-1.10.0/dataframely/_storage/constants.py +6 -0
- dataframely-1.10.0/dataframely/_storage/delta.py +200 -0
- dataframely-1.10.0/dataframely/_storage/parquet.py +239 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/collection.py +240 -85
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/decimal.py +6 -5
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/failure.py +130 -31
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/schema.py +263 -24
- dataframely-1.10.0/dataframely/testing/storage.py +293 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/pixi.lock +6250 -6485
- {dataframely-1.9.0 → dataframely-1.10.0}/pixi.toml +9 -3
- {dataframely-1.9.0 → dataframely-1.10.0}/pyproject.toml +9 -1
- dataframely-1.10.0/tests/collection/test_storage.py +407 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_decimal.py +16 -3
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_pyarrow.py +2 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_sql_schema.py +10 -10
- dataframely-1.10.0/tests/failure_info/test_storage.py +131 -0
- dataframely-1.10.0/tests/schema/test_read_write_parquet.py +27 -0
- dataframely-1.10.0/tests/schema/test_storage.py +250 -0
- dataframely-1.10.0/tests/storage/test_delta.py +54 -0
- dataframely-1.9.0/tests/collection/test_read_write_parquet.py +0 -398
- dataframely-1.9.0/tests/schema/test_read_write_parquet.py +0 -234
- dataframely-1.9.0/tests/test_failure_info.py +0 -109
- {dataframely-1.9.0 → dataframely-1.10.0}/.copier-answers.yml +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.envrc +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.gitattributes +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.github/CODEOWNERS +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.github/dependabot.yml +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.github/release-drafter.yml +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.github/workflows/build.yml +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.github/workflows/chore.yml +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.github/workflows/nightly.yml +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.github/workflows/scorecard.yml +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.gitignore +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.pre-commit-config.yaml +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.prettierignore +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.prettierrc +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/.readthedocs.yml +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/Cargo.lock +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/LICENSE +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/README.md +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/SECURITY.md +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/__init__.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_base_collection.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_base_schema.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_deprecation.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_extre.pyi +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_filter.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_polars.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_rule.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_typing.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/_validation.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/__init__.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/_base.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/_mixins.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/_registry.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/_utils.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/any.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/array.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/bool.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/categorical.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/datetime.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/enum.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/float.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/integer.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/list.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/object.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/string.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/columns/struct.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/config.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/exc.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/functional.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/mypy.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/py.typed +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/random.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/__init__.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/const.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/factory.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/mask.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/rules.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/dataframely/testing/typing.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docker-compose.yml +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/Makefile +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.collection.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.any.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.bool.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.datetime.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.decimal.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.enum.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.float.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.integer.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.list.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.string.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.columns.struct.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.config.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.exc.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.failure.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.functional.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.mypy.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.random.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.schema.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.const.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.factory.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.mask.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.rules.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/dataframely.testing.typing.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_api/modules.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_static/custom.css +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/_static/favicon.ico +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/conf.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/index.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/make.bat +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/development.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/examples/real-world.ipynb +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/faq.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/installation.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/quickstart.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/docs/sites/versioning.rst +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/src/errdefs.rs +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/src/lib.rs +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/src/regex_repr.rs +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/benches/conftest.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/benches/test_collection.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/benches/test_failure.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/benches/test_schema.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_base.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_cast.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_collection_future_annotations.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_create_empty.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_filter_one_to_n.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_filter_validate.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_ignore_in_filter.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_implementation.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_join.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_matches.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_optional_members.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_repr.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_sample.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_serialization.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/collection/test_validate_input.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/__init__.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_any.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_array.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_datetime.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_enum.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_float.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_integer.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_list.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_object.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_string.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/column_types/test_struct.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/__init__.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_alias.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_check.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_default_dtypes.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_matches.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_metadata.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_polars_schema.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_rules.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_sample.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_str.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/columns/test_utils.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/core_validation/__init__.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/core_validation/test_column_validation.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/core_validation/test_dtype_validation.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/core_validation/test_rule_evaluation.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/functional/test_concat.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/functional/test_relationships.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_base.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_cast.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_create_empty.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_create_empty_if_none.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_filter.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_inheritance.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_matches.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_repr.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_rule_implementation.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_sample.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_serialization.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/schema/test_validate.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_compat.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_config.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_deprecation.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_exc.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_extre.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_factory.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_random.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_serialization.py +0 -0
- {dataframely-1.9.0 → dataframely-1.10.0}/tests/test_typing.py +0 -0
|
@@ -41,6 +41,14 @@ jobs:
|
|
|
41
41
|
matrix:
|
|
42
42
|
os: [ubuntu-latest, windows-latest]
|
|
43
43
|
environment: [py310, py311, py312, py313]
|
|
44
|
+
with_optionals: [false]
|
|
45
|
+
include:
|
|
46
|
+
- os: ubuntu-latest
|
|
47
|
+
environment: py313-optionals
|
|
48
|
+
with_optionals: true
|
|
49
|
+
- os: windows-latest
|
|
50
|
+
environment: py313-optionals
|
|
51
|
+
with_optionals: true
|
|
44
52
|
steps:
|
|
45
53
|
- name: Checkout branch
|
|
46
54
|
uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2
|
|
@@ -51,7 +59,7 @@ jobs:
|
|
|
51
59
|
- name: Install repository
|
|
52
60
|
run: pixi run -e ${{ matrix.environment }} postinstall
|
|
53
61
|
- name: Run pytest
|
|
54
|
-
run: pixi run -e ${{ matrix.environment }} test-coverage --color=yes
|
|
62
|
+
run: pixi run -e ${{ matrix.environment }} test-coverage --color=yes ${{ matrix.with_optionals && '-m with_optionals' || '-m "not with_optionals"'}} --cov=dataframely --cov-report=xml
|
|
55
63
|
- name: Upload codecov
|
|
56
64
|
uses: codecov/codecov-action@18283e04ce6e62d37312384ff67231eb8fd56d24 # v5.4.3
|
|
57
65
|
with:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dataframely
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.10.0
|
|
4
4
|
Classifier: Programming Language :: Python :: 3
|
|
5
5
|
Classifier: Programming Language :: Python :: 3.10
|
|
6
6
|
Classifier: Programming Language :: Python :: 3.11
|
|
@@ -9,6 +9,12 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
9
9
|
Requires-Dist: numpy
|
|
10
10
|
Requires-Dist: polars>=1.32
|
|
11
11
|
Requires-Dist: typing-extensions ; python_full_version < '3.11'
|
|
12
|
+
Requires-Dist: deltalake ; extra == 'deltalake'
|
|
13
|
+
Requires-Dist: sqlalchemy ; extra == 'sqlalchemy'
|
|
14
|
+
Requires-Dist: pyarrow ; extra == 'pyarrow'
|
|
15
|
+
Provides-Extra: deltalake
|
|
16
|
+
Provides-Extra: sqlalchemy
|
|
17
|
+
Provides-Extra: pyarrow
|
|
12
18
|
License-File: LICENSE
|
|
13
19
|
Summary: A declarative, polars-native data frame validation library
|
|
14
20
|
Author-email: Andreas Albert <andreas.albert@quantco.com>, Daniel Elsner <daniel.elsner@quantco.com>, Oliver Borchert <oliver.borchert@quantco.com>
|
|
@@ -13,11 +13,24 @@ class _DummyModule: # pragma: no cover
|
|
|
13
13
|
raise ValueError(f"Module '{self.module}' is not installed.")
|
|
14
14
|
|
|
15
15
|
|
|
16
|
+
# ------------------------------------ DELTALAKE ------------------------------------- #
|
|
17
|
+
|
|
18
|
+
try:
|
|
19
|
+
import deltalake
|
|
20
|
+
from deltalake import DeltaTable
|
|
21
|
+
except ImportError: # pragma: no cover
|
|
22
|
+
deltalake = _DummyModule("deltalake") # type: ignore
|
|
23
|
+
|
|
24
|
+
class DeltaTable: # type: ignore # noqa: N801
|
|
25
|
+
pass
|
|
16
26
|
# ------------------------------------ SQLALCHEMY ------------------------------------ #
|
|
17
27
|
|
|
18
28
|
try:
|
|
19
29
|
import sqlalchemy as sa
|
|
20
30
|
import sqlalchemy.dialects.mssql as sa_mssql
|
|
31
|
+
from sqlalchemy import Dialect
|
|
32
|
+
from sqlalchemy.dialects.mssql.pyodbc import MSDialect_pyodbc
|
|
33
|
+
from sqlalchemy.dialects.postgresql.psycopg2 import PGDialect_psycopg2
|
|
21
34
|
from sqlalchemy.sql.type_api import TypeEngine as sa_TypeEngine
|
|
22
35
|
except ImportError: # pragma: no cover
|
|
23
36
|
sa = _DummyModule("sqlalchemy") # type: ignore
|
|
@@ -26,7 +39,14 @@ except ImportError: # pragma: no cover
|
|
|
26
39
|
class sa_TypeEngine: # type: ignore # noqa: N801
|
|
27
40
|
pass
|
|
28
41
|
|
|
42
|
+
class MSDialect_pyodbc: # type: ignore # noqa: N801
|
|
43
|
+
pass
|
|
44
|
+
|
|
45
|
+
class PGDialect_psycopg2: # type: ignore # noqa: N801
|
|
46
|
+
pass
|
|
29
47
|
|
|
48
|
+
class Dialect: # type: ignore # noqa: N801
|
|
49
|
+
pass
|
|
30
50
|
# -------------------------------------- PYARROW ------------------------------------- #
|
|
31
51
|
|
|
32
52
|
try:
|
|
@@ -36,4 +56,12 @@ except ImportError: # pragma: no cover
|
|
|
36
56
|
|
|
37
57
|
# ------------------------------------------------------------------------------------ #
|
|
38
58
|
|
|
39
|
-
__all__ = [
|
|
59
|
+
__all__ = [
|
|
60
|
+
"deltalake",
|
|
61
|
+
"sa",
|
|
62
|
+
"sa_mssql",
|
|
63
|
+
"sa_TypeEngine",
|
|
64
|
+
"pa",
|
|
65
|
+
"MSDialect_pyodbc",
|
|
66
|
+
"PGDialect_psycopg2",
|
|
67
|
+
]
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
# Copyright (c) QuantCo 2025-2025
|
|
2
|
+
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
|
+
|
|
4
|
+
from abc import ABC, abstractmethod
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import polars as pl
|
|
9
|
+
|
|
10
|
+
SerializedSchema = str
|
|
11
|
+
SerializedCollection = str
|
|
12
|
+
SerializedRules = str
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class StorageBackend(ABC):
|
|
16
|
+
"""Base class for storage backends.
|
|
17
|
+
|
|
18
|
+
A storage backend encapsulates a way of serializing and deserializing dataframlely
|
|
19
|
+
data-/lazyframes and collections. This base class provides a unified interface for
|
|
20
|
+
all such use cases.
|
|
21
|
+
|
|
22
|
+
The interface is designed to operate on data provided as polars frames, and metadata
|
|
23
|
+
provided as serialized strings. This design is meant to limit the coupling between
|
|
24
|
+
the Schema/Collection classes and specifics of how data and metadata is stored.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
# ----------------------------------- Schemas -------------------------------------
|
|
28
|
+
@abstractmethod
|
|
29
|
+
def sink_frame(
|
|
30
|
+
self, lf: pl.LazyFrame, serialized_schema: SerializedSchema, **kwargs: Any
|
|
31
|
+
) -> None:
|
|
32
|
+
"""Stream the contents of a dataframe, and its metadata to the storage backend.
|
|
33
|
+
|
|
34
|
+
Args:
|
|
35
|
+
lf: A frame containing the data to be stored.
|
|
36
|
+
serialized_schema: String-serialized schema information.
|
|
37
|
+
kwargs: Additional keyword arguments to pass to the underlying storage
|
|
38
|
+
implementation.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
@abstractmethod
|
|
42
|
+
def write_frame(
|
|
43
|
+
self, df: pl.DataFrame, serialized_schema: SerializedSchema, **kwargs: Any
|
|
44
|
+
) -> None:
|
|
45
|
+
"""Write the contents of a dataframe, and its metadata to the storage backend.
|
|
46
|
+
|
|
47
|
+
Args:
|
|
48
|
+
df: A dataframe containing the data to be stored.
|
|
49
|
+
frame: String-serialized schema information.
|
|
50
|
+
kwargs: Additional keyword arguments to pass to the underlying storage
|
|
51
|
+
implementation.
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
@abstractmethod
|
|
55
|
+
def scan_frame(self, **kwargs: Any) -> tuple[pl.LazyFrame, SerializedSchema | None]:
|
|
56
|
+
"""Lazily read frame data and metadata from the storage backend.
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
kwargs: Keyword arguments to pass to the underlying storage.
|
|
60
|
+
Refer to the individual implementation to see which keywords
|
|
61
|
+
are available.
|
|
62
|
+
Returns:
|
|
63
|
+
A tuple of the lazy frame data and metadata if available.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
@abstractmethod
|
|
67
|
+
def read_frame(self, **kwargs: Any) -> tuple[pl.DataFrame, SerializedSchema | None]:
|
|
68
|
+
"""Eagerly read frame data and metadata from the storage backend.
|
|
69
|
+
|
|
70
|
+
Args:
|
|
71
|
+
kwargs: Keyword arguments to pass to the underlying storage.
|
|
72
|
+
Refer to the individual implementation to see which keywords
|
|
73
|
+
are available.
|
|
74
|
+
Returns:
|
|
75
|
+
A tuple of the lazy frame data and metadata if available.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
# ------------------------------ Collections ---------------------------------------
|
|
79
|
+
@abstractmethod
|
|
80
|
+
def sink_collection(
|
|
81
|
+
self,
|
|
82
|
+
dfs: dict[str, pl.LazyFrame],
|
|
83
|
+
serialized_collection: SerializedCollection,
|
|
84
|
+
serialized_schemas: dict[str, str],
|
|
85
|
+
**kwargs: Any,
|
|
86
|
+
) -> None:
|
|
87
|
+
"""Stream the members of this collection into the storage backend.
|
|
88
|
+
|
|
89
|
+
Args:
|
|
90
|
+
dfs: Dictionary containing the data to be stored.
|
|
91
|
+
serialized_collection: String-serialized information about the origin Collection.
|
|
92
|
+
serialized_schemas: String-serialized information about the individual Schemas
|
|
93
|
+
for each of the member frames. This information is also logically included
|
|
94
|
+
in the collection metadata, but it is passed separately here to ensure that
|
|
95
|
+
each member can also be read back as an individual frame.
|
|
96
|
+
"""
|
|
97
|
+
|
|
98
|
+
@abstractmethod
|
|
99
|
+
def write_collection(
|
|
100
|
+
self,
|
|
101
|
+
dfs: dict[str, pl.LazyFrame],
|
|
102
|
+
serialized_collection: SerializedCollection,
|
|
103
|
+
serialized_schemas: dict[str, str],
|
|
104
|
+
**kwargs: Any,
|
|
105
|
+
) -> None:
|
|
106
|
+
"""Write the members of this collection into the storage backend.
|
|
107
|
+
|
|
108
|
+
Args:
|
|
109
|
+
dfs: Dictionary containing the data to be stored.
|
|
110
|
+
serialized_collection: String-serialized information about the origin Collection.
|
|
111
|
+
serialized_schemas: String-serialized information about the individual Schemas
|
|
112
|
+
for each of the member frames. This information is also logically included
|
|
113
|
+
in the collection metadata, but it is passed separately here to ensure that
|
|
114
|
+
each member can also be read back as an individual frame.
|
|
115
|
+
"""
|
|
116
|
+
|
|
117
|
+
@abstractmethod
|
|
118
|
+
def scan_collection(
|
|
119
|
+
self, members: Iterable[str], **kwargs: Any
|
|
120
|
+
) -> tuple[dict[str, pl.LazyFrame], list[SerializedCollection | None]]:
|
|
121
|
+
"""Lazily read all collection members from the storage backend.
|
|
122
|
+
|
|
123
|
+
Args:
|
|
124
|
+
members: Collection member names to read.
|
|
125
|
+
kwargs: Additional keyword arguments to pass to the underlying storage.
|
|
126
|
+
Refer to the individual implementation to see which keywords are available.
|
|
127
|
+
Returns:
|
|
128
|
+
A tuple of the collection data and metadata if available.
|
|
129
|
+
Depending on the storage implementation, multiple copies of the metadata
|
|
130
|
+
may be available, which are returned as a list.
|
|
131
|
+
It is up to the caller to decide how to handle the presence/absence/consistency
|
|
132
|
+
of the returned values.
|
|
133
|
+
"""
|
|
134
|
+
|
|
135
|
+
@abstractmethod
|
|
136
|
+
def read_collection(
|
|
137
|
+
self, members: Iterable[str], **kwargs: Any
|
|
138
|
+
) -> tuple[dict[str, pl.LazyFrame], list[SerializedCollection | None]]:
|
|
139
|
+
"""Lazily read all collection members from the storage backend.
|
|
140
|
+
|
|
141
|
+
Args:
|
|
142
|
+
members: Collection member names to read.
|
|
143
|
+
kwargs: Additional keyword arguments to pass to the underlying storage.
|
|
144
|
+
Refer to the individual implementation to see which keywords are available.
|
|
145
|
+
Returns:
|
|
146
|
+
A tuple of the collection data and metadata if available.
|
|
147
|
+
Depending on the storage implementation, multiple copies of the metadata
|
|
148
|
+
may be available, which are returned as a list.
|
|
149
|
+
It is up to the caller to decide how to handle the presence/absence/consistency
|
|
150
|
+
of the returned values.
|
|
151
|
+
"""
|
|
152
|
+
|
|
153
|
+
# ------------------------------ Failure Info --------------------------------------
|
|
154
|
+
@abstractmethod
|
|
155
|
+
def sink_failure_info(
|
|
156
|
+
self,
|
|
157
|
+
lf: pl.LazyFrame,
|
|
158
|
+
serialized_rules: SerializedRules,
|
|
159
|
+
serialized_schema: SerializedSchema,
|
|
160
|
+
**kwargs: Any,
|
|
161
|
+
) -> None:
|
|
162
|
+
"""Stream the failure info to the storage backend.
|
|
163
|
+
|
|
164
|
+
Args:
|
|
165
|
+
lf: LazyFrame backing the failure info.
|
|
166
|
+
serialized_rules: JSON-serialized list of rule column names
|
|
167
|
+
used for validation.
|
|
168
|
+
serialized_schema: String-serialized schema information.
|
|
169
|
+
"""
|
|
170
|
+
|
|
171
|
+
@abstractmethod
|
|
172
|
+
def write_failure_info(
|
|
173
|
+
self,
|
|
174
|
+
df: pl.DataFrame,
|
|
175
|
+
serialized_rules: SerializedRules,
|
|
176
|
+
serialized_schema: SerializedSchema,
|
|
177
|
+
**kwargs: Any,
|
|
178
|
+
) -> None:
|
|
179
|
+
"""Write the failure info to the storage backend.
|
|
180
|
+
|
|
181
|
+
Args:
|
|
182
|
+
df: DataFrame backing the failure info.
|
|
183
|
+
serialized_rules: JSON-serialized list of rule column names
|
|
184
|
+
used for validation.
|
|
185
|
+
serialized_schema: String-serialized schema information.
|
|
186
|
+
"""
|
|
187
|
+
|
|
188
|
+
@abstractmethod
|
|
189
|
+
def scan_failure_info(
|
|
190
|
+
self, **kwargs: Any
|
|
191
|
+
) -> tuple[pl.LazyFrame, SerializedRules, SerializedSchema]:
|
|
192
|
+
"""Lazily read the failure info from the storage backend."""
|
|
193
|
+
|
|
194
|
+
def read_failure_info(
|
|
195
|
+
self, **kwargs: Any
|
|
196
|
+
) -> tuple[pl.DataFrame, SerializedRules, SerializedSchema]:
|
|
197
|
+
"""Read the failure info from the storage backend."""
|
|
198
|
+
|
|
199
|
+
lf, rule_metadata, schema_metadata = self.scan_failure_info(**kwargs)
|
|
200
|
+
return (
|
|
201
|
+
lf.collect(),
|
|
202
|
+
rule_metadata,
|
|
203
|
+
schema_metadata,
|
|
204
|
+
)
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# Copyright (c) QuantCo 2025-2025
|
|
2
|
+
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def assert_failure_info_metadata(metadata: str | None) -> str:
|
|
6
|
+
if metadata:
|
|
7
|
+
return metadata
|
|
8
|
+
raise ValueError(
|
|
9
|
+
"The required FailureInfo metadata was not found in the storage backend."
|
|
10
|
+
)
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
# Copyright (c) QuantCo 2025-2025
|
|
2
|
+
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import polars as pl
|
|
10
|
+
|
|
11
|
+
from dataframely._compat import deltalake
|
|
12
|
+
|
|
13
|
+
from ._base import (
|
|
14
|
+
SerializedCollection,
|
|
15
|
+
SerializedRules,
|
|
16
|
+
SerializedSchema,
|
|
17
|
+
StorageBackend,
|
|
18
|
+
)
|
|
19
|
+
from ._exc import assert_failure_info_metadata
|
|
20
|
+
from .constants import COLLECTION_METADATA_KEY, RULE_METADATA_KEY, SCHEMA_METADATA_KEY
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class DeltaStorageBackend(StorageBackend):
|
|
24
|
+
def sink_frame(
|
|
25
|
+
self, lf: pl.LazyFrame, serialized_schema: SerializedSchema, **kwargs: Any
|
|
26
|
+
) -> None:
|
|
27
|
+
_raise_on_lazy_write()
|
|
28
|
+
|
|
29
|
+
def write_frame(
|
|
30
|
+
self, df: pl.DataFrame, serialized_schema: SerializedSchema, **kwargs: Any
|
|
31
|
+
) -> None:
|
|
32
|
+
target = kwargs.pop("target")
|
|
33
|
+
metadata = kwargs.pop("metadata", {})
|
|
34
|
+
delta_write_options = kwargs.pop("delta_write_options", {})
|
|
35
|
+
|
|
36
|
+
# Delta lake does not allow partitioning if there is only one column
|
|
37
|
+
# We dynamically remove this setting here to allow users to still specify it
|
|
38
|
+
# on the collection level without having to worry about each individual member
|
|
39
|
+
if len(df.columns) < 2:
|
|
40
|
+
delta_write_options.pop("partition_by", None)
|
|
41
|
+
|
|
42
|
+
df.write_delta(
|
|
43
|
+
target,
|
|
44
|
+
delta_write_options=(
|
|
45
|
+
delta_write_options
|
|
46
|
+
| {
|
|
47
|
+
"commit_properties": deltalake.CommitProperties(
|
|
48
|
+
custom_metadata=metadata
|
|
49
|
+
| {SCHEMA_METADATA_KEY: serialized_schema}
|
|
50
|
+
),
|
|
51
|
+
}
|
|
52
|
+
),
|
|
53
|
+
**kwargs,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
def scan_frame(self, **kwargs: Any) -> tuple[pl.LazyFrame, SerializedSchema | None]:
|
|
57
|
+
table = _to_delta_table(kwargs.pop("source"))
|
|
58
|
+
serialized_schema = _read_serialized_schema(table)
|
|
59
|
+
df = pl.scan_delta(table, **kwargs)
|
|
60
|
+
return df, serialized_schema
|
|
61
|
+
|
|
62
|
+
def read_frame(self, **kwargs: Any) -> tuple[pl.DataFrame, SerializedSchema | None]:
|
|
63
|
+
table = _to_delta_table(kwargs.pop("source"))
|
|
64
|
+
serialized_schema = _read_serialized_schema(table)
|
|
65
|
+
df = pl.read_delta(table, **kwargs)
|
|
66
|
+
return df, serialized_schema
|
|
67
|
+
|
|
68
|
+
# ------------------------------ Collections ---------------------------------------
|
|
69
|
+
def sink_collection(
|
|
70
|
+
self,
|
|
71
|
+
dfs: dict[str, pl.LazyFrame],
|
|
72
|
+
serialized_collection: SerializedCollection,
|
|
73
|
+
serialized_schemas: dict[str, str],
|
|
74
|
+
**kwargs: Any,
|
|
75
|
+
) -> None:
|
|
76
|
+
_raise_on_lazy_write()
|
|
77
|
+
|
|
78
|
+
def write_collection(
|
|
79
|
+
self,
|
|
80
|
+
dfs: dict[str, pl.LazyFrame],
|
|
81
|
+
serialized_collection: SerializedCollection,
|
|
82
|
+
serialized_schemas: dict[str, str],
|
|
83
|
+
**kwargs: Any,
|
|
84
|
+
) -> None:
|
|
85
|
+
uri = Path(kwargs.pop("target"))
|
|
86
|
+
|
|
87
|
+
# The collection schema is serialized as part of the member parquet metadata
|
|
88
|
+
kwargs["metadata"] = kwargs.get("metadata", {}) | {
|
|
89
|
+
COLLECTION_METADATA_KEY: serialized_collection
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
for key, lf in dfs.items():
|
|
93
|
+
self.write_frame(
|
|
94
|
+
lf.collect(),
|
|
95
|
+
serialized_schema=serialized_schemas[key],
|
|
96
|
+
target=uri / key,
|
|
97
|
+
**kwargs,
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
def scan_collection(
|
|
101
|
+
self, members: Iterable[str], **kwargs: Any
|
|
102
|
+
) -> tuple[dict[str, pl.LazyFrame], list[SerializedCollection | None]]:
|
|
103
|
+
uri = Path(kwargs.pop("source"))
|
|
104
|
+
|
|
105
|
+
data = {}
|
|
106
|
+
collection_types = []
|
|
107
|
+
for key in members:
|
|
108
|
+
member_uri = uri / key
|
|
109
|
+
if not deltalake.DeltaTable.is_deltatable(str(member_uri)):
|
|
110
|
+
continue
|
|
111
|
+
table = _to_delta_table(member_uri)
|
|
112
|
+
data[key] = pl.scan_delta(table, **kwargs)
|
|
113
|
+
collection_types.append(_read_serialized_collection(table))
|
|
114
|
+
|
|
115
|
+
return data, collection_types
|
|
116
|
+
|
|
117
|
+
def read_collection(
|
|
118
|
+
self, members: Iterable[str], **kwargs: Any
|
|
119
|
+
) -> tuple[dict[str, pl.LazyFrame], list[SerializedCollection | None]]:
|
|
120
|
+
lazy, collection_types = self.scan_collection(members, **kwargs)
|
|
121
|
+
eager = {name: lf.collect().lazy() for name, lf in lazy.items()}
|
|
122
|
+
return eager, collection_types
|
|
123
|
+
|
|
124
|
+
# ------------------------------ Failure Info --------------------------------------
|
|
125
|
+
def sink_failure_info(
|
|
126
|
+
self,
|
|
127
|
+
lf: pl.LazyFrame,
|
|
128
|
+
serialized_rules: SerializedRules,
|
|
129
|
+
serialized_schema: SerializedSchema,
|
|
130
|
+
**kwargs: Any,
|
|
131
|
+
) -> None:
|
|
132
|
+
_raise_on_lazy_write()
|
|
133
|
+
|
|
134
|
+
def write_failure_info(
|
|
135
|
+
self,
|
|
136
|
+
df: pl.DataFrame,
|
|
137
|
+
serialized_rules: SerializedRules,
|
|
138
|
+
serialized_schema: SerializedSchema,
|
|
139
|
+
**kwargs: Any,
|
|
140
|
+
) -> None:
|
|
141
|
+
self.write_frame(
|
|
142
|
+
df,
|
|
143
|
+
serialized_schema,
|
|
144
|
+
metadata={
|
|
145
|
+
RULE_METADATA_KEY: serialized_rules,
|
|
146
|
+
},
|
|
147
|
+
**kwargs,
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
def scan_failure_info(
|
|
151
|
+
self, **kwargs: Any
|
|
152
|
+
) -> tuple[pl.LazyFrame, SerializedRules, SerializedSchema]:
|
|
153
|
+
"""Lazily read the failure info from the storage backend."""
|
|
154
|
+
table = _to_delta_table(kwargs.pop("source"))
|
|
155
|
+
|
|
156
|
+
# Metadata
|
|
157
|
+
serialized_rules = assert_failure_info_metadata(_read_serialized_rules(table))
|
|
158
|
+
serialized_schema = assert_failure_info_metadata(_read_serialized_schema(table))
|
|
159
|
+
|
|
160
|
+
# Data
|
|
161
|
+
lf = pl.scan_delta(table, **kwargs)
|
|
162
|
+
|
|
163
|
+
return lf, serialized_rules, serialized_schema
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _raise_on_lazy_write() -> None:
|
|
167
|
+
raise NotImplementedError("Lazy writes are not currently supported for deltalake.")
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _read_serialized_schema(table: deltalake.DeltaTable) -> SerializedSchema | None:
|
|
171
|
+
[last_commit] = table.history(limit=1)
|
|
172
|
+
return last_commit.get(SCHEMA_METADATA_KEY, None)
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _read_serialized_collection(
|
|
176
|
+
table: deltalake.DeltaTable,
|
|
177
|
+
) -> SerializedCollection | None:
|
|
178
|
+
[last_commit] = table.history(limit=1)
|
|
179
|
+
return last_commit.get(COLLECTION_METADATA_KEY, None)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _read_serialized_rules(
|
|
183
|
+
table: deltalake.DeltaTable,
|
|
184
|
+
) -> SerializedRules | None:
|
|
185
|
+
[last_commit] = table.history(limit=1)
|
|
186
|
+
return last_commit.get(RULE_METADATA_KEY, None)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _to_delta_table(
|
|
190
|
+
table: Path | str | deltalake.DeltaTable,
|
|
191
|
+
) -> deltalake.DeltaTable:
|
|
192
|
+
from deltalake import DeltaTable
|
|
193
|
+
|
|
194
|
+
match table:
|
|
195
|
+
case DeltaTable():
|
|
196
|
+
return table
|
|
197
|
+
case str() | Path():
|
|
198
|
+
return DeltaTable(table)
|
|
199
|
+
case _:
|
|
200
|
+
raise TypeError(f"Unsupported type {table!r}")
|