dataframely 1.12.0__tar.gz → 1.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataframely-1.12.0 → dataframely-1.13.0}/PKG-INFO +3 -1
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_base_schema.py +41 -2
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_compat.py +15 -0
- dataframely-1.13.0/dataframely/_pydantic.py +115 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_typing.py +30 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/_base.py +0 -13
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/enum.py +4 -8
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/schema.py +0 -25
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/sites/quickstart.rst +1 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/pixi.lock +199 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/pixi.toml +2 -1
- {dataframely-1.12.0 → dataframely-1.13.0}/pyproject.toml +6 -2
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_repr.py +11 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_serialization.py +1 -0
- dataframely-1.13.0/tests/test_pydantic.py +164 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.copier-answers.yml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.envrc +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.gitattributes +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.github/CODEOWNERS +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.github/dependabot.yml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.github/release-drafter.yml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.github/workflows/build.yml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.github/workflows/chore.yml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.github/workflows/ci.yml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.github/workflows/nightly.yml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.github/workflows/scorecard.yml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.gitignore +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.pre-commit-config.yaml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.prettierignore +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.prettierrc +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/.readthedocs.yml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/Cargo.lock +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/Cargo.toml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/LICENSE +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/README.md +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/SECURITY.md +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/__init__.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_base_collection.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_deprecation.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_extre.pyi +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_filter.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_polars.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_rule.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_serialization.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_storage/__init__.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_storage/_base.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_storage/_exc.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_storage/constants.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_storage/delta.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_storage/parquet.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/_validation.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/collection.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/__init__.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/_mixins.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/_registry.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/_utils.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/any.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/array.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/binary.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/bool.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/categorical.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/datetime.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/decimal.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/float.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/integer.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/list.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/object.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/string.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/columns/struct.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/config.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/exc.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/failure.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/functional.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/mypy.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/py.typed +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/random.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/testing/__init__.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/testing/const.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/testing/factory.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/testing/mask.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/testing/rules.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/testing/storage.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/dataframely/testing/typing.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docker-compose.yml +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/Makefile +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.collection.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.any.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.bool.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.datetime.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.decimal.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.enum.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.float.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.integer.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.list.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.string.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.columns.struct.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.config.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.exc.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.failure.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.functional.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.mypy.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.random.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.schema.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.testing.const.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.testing.factory.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.testing.mask.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.testing.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.testing.rules.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/dataframely.testing.typing.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_api/modules.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_static/custom.css +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/_static/favicon.ico +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/conf.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/index.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/make.bat +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/sites/development.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/sites/examples/real-world.ipynb +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/sites/faq.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/sites/installation.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/docs/sites/versioning.rst +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/src/errdefs.rs +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/src/lib.rs +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/src/regex_repr.rs +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/benches/conftest.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/benches/test_collection.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/benches/test_failure.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/benches/test_schema.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_base.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_cast.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_collection_future_annotations.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_create_empty.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_filter_one_to_n.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_filter_validate.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_ignore_in_filter.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_implementation.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_join.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_matches.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_optional_members.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_repr.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_sample.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_serialization.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_storage.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/collection/test_validate_input.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/__init__.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_any.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_array.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_binary.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_datetime.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_decimal.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_enum.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_float.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_integer.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_list.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_object.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_string.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/column_types/test_struct.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/__init__.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_alias.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_check.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_default_dtypes.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_matches.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_metadata.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_polars_schema.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_pyarrow.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_rules.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_sample.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_sql_schema.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_str.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/columns/test_utils.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/core_validation/__init__.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/core_validation/test_column_validation.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/core_validation/test_dtype_validation.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/core_validation/test_rule_evaluation.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/failure_info/test_storage.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/functional/test_concat.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/functional/test_relationships.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_base.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_cast.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_create_empty.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_create_empty_if_none.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_filter.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_inheritance.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_matches.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_read_write_parquet.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_rule_implementation.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_sample.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_storage.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/schema/test_validate.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/storage/test_delta.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/test_compat.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/test_config.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/test_deprecation.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/test_exc.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/test_extre.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/test_factory.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/test_random.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/test_serialization.py +0 -0
- {dataframely-1.12.0 → dataframely-1.13.0}/tests/test_typing.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dataframely
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.13.0
|
|
4
4
|
Classifier: Programming Language :: Python :: 3
|
|
5
5
|
Classifier: Programming Language :: Python :: 3.10
|
|
6
6
|
Classifier: Programming Language :: Python :: 3.11
|
|
@@ -12,9 +12,11 @@ Requires-Dist: typing-extensions ; python_full_version < '3.11'
|
|
|
12
12
|
Requires-Dist: deltalake ; extra == 'deltalake'
|
|
13
13
|
Requires-Dist: sqlalchemy ; extra == 'sqlalchemy'
|
|
14
14
|
Requires-Dist: pyarrow ; extra == 'pyarrow'
|
|
15
|
+
Requires-Dist: pydantic>=2 ; extra == 'pydantic'
|
|
15
16
|
Provides-Extra: deltalake
|
|
16
17
|
Provides-Extra: sqlalchemy
|
|
17
18
|
Provides-Extra: pyarrow
|
|
19
|
+
Provides-Extra: pydantic
|
|
18
20
|
License-File: LICENSE
|
|
19
21
|
Summary: A declarative, polars-native data frame validation library
|
|
20
22
|
Author-email: Andreas Albert <andreas.albert@quantco.com>, Daniel Elsner <daniel.elsner@quantco.com>, Oliver Borchert <oliver.borchert@quantco.com>
|
|
@@ -5,10 +5,10 @@ from __future__ import annotations
|
|
|
5
5
|
|
|
6
6
|
import sys
|
|
7
7
|
import textwrap
|
|
8
|
-
from abc import ABCMeta
|
|
8
|
+
from abc import ABCMeta, abstractmethod
|
|
9
9
|
from copy import copy
|
|
10
10
|
from dataclasses import dataclass, field
|
|
11
|
-
from typing import Any
|
|
11
|
+
from typing import TYPE_CHECKING, Any
|
|
12
12
|
|
|
13
13
|
import polars as pl
|
|
14
14
|
|
|
@@ -21,6 +21,10 @@ if sys.version_info >= (3, 11):
|
|
|
21
21
|
else:
|
|
22
22
|
from typing_extensions import Self
|
|
23
23
|
|
|
24
|
+
|
|
25
|
+
if TYPE_CHECKING:
|
|
26
|
+
from ._typing import DataFrame
|
|
27
|
+
|
|
24
28
|
_COLUMN_ATTR = "__dataframely_columns__"
|
|
25
29
|
_RULE_ATTR = "__dataframely_rules__"
|
|
26
30
|
|
|
@@ -198,11 +202,46 @@ class BaseSchema(metaclass=SchemaMeta):
|
|
|
198
202
|
columns[name]._name = name
|
|
199
203
|
return columns
|
|
200
204
|
|
|
205
|
+
@classmethod
|
|
206
|
+
@abstractmethod
|
|
207
|
+
def polars_schema(cls) -> pl.Schema:
|
|
208
|
+
"""Obtain the polars schema for this schema.
|
|
209
|
+
|
|
210
|
+
Returns:
|
|
211
|
+
A :mod:`polars` schema that mirrors the schema defined by this class.
|
|
212
|
+
"""
|
|
213
|
+
|
|
201
214
|
@classmethod
|
|
202
215
|
def primary_keys(cls) -> list[str]:
|
|
203
216
|
"""The primary key columns in this schema (possibly empty)."""
|
|
204
217
|
return _primary_keys(cls.columns())
|
|
205
218
|
|
|
219
|
+
@classmethod
|
|
220
|
+
@abstractmethod
|
|
221
|
+
def validate(
|
|
222
|
+
cls, df: pl.DataFrame | pl.LazyFrame, /, *, cast: bool = False
|
|
223
|
+
) -> DataFrame[Self]:
|
|
224
|
+
"""Validate that a data frame satisfies the schema.
|
|
225
|
+
|
|
226
|
+
Args:
|
|
227
|
+
df: The data frame to validate.
|
|
228
|
+
cast: Whether columns with a wrong data type in the input data frame are
|
|
229
|
+
cast to the schema's defined data type if possible.
|
|
230
|
+
|
|
231
|
+
Returns:
|
|
232
|
+
The (collected) input data frame, wrapped in a generic version of the
|
|
233
|
+
input's data frame type to reflect schema adherence. The data frame is
|
|
234
|
+
guaranteed to maintain its order.
|
|
235
|
+
|
|
236
|
+
Raises:
|
|
237
|
+
ValidationError: If the input data frame does not satisfy the schema
|
|
238
|
+
definition.
|
|
239
|
+
|
|
240
|
+
Note:
|
|
241
|
+
This method _always_ collects the input data frame in order to raise
|
|
242
|
+
potential validation errors.
|
|
243
|
+
"""
|
|
244
|
+
|
|
206
245
|
@classmethod
|
|
207
246
|
def _validation_rules(cls, *, with_cast: bool) -> dict[str, Rule]:
|
|
208
247
|
return _build_rules(
|
|
@@ -54,6 +54,19 @@ try:
|
|
|
54
54
|
except ImportError: # pragma: no cover
|
|
55
55
|
pa = _DummyModule("pyarrow")
|
|
56
56
|
|
|
57
|
+
|
|
58
|
+
# -------------------------------------- PYDANTIC ------------------------------------ #
|
|
59
|
+
|
|
60
|
+
try:
|
|
61
|
+
import pydantic
|
|
62
|
+
except ImportError: # pragma: no cover
|
|
63
|
+
pydantic = _DummyModule("pydantic") # type: ignore
|
|
64
|
+
|
|
65
|
+
try:
|
|
66
|
+
from pydantic_core import core_schema as pydantic_core_schema # pragma: no cover
|
|
67
|
+
except ImportError:
|
|
68
|
+
pydantic_core_schema = _DummyModule("pydantic_core_schema") # type: ignore
|
|
69
|
+
|
|
57
70
|
# ------------------------------------------------------------------------------------ #
|
|
58
71
|
|
|
59
72
|
__all__ = [
|
|
@@ -64,4 +77,6 @@ __all__ = [
|
|
|
64
77
|
"pa",
|
|
65
78
|
"MSDialect_pyodbc",
|
|
66
79
|
"PGDialect_psycopg2",
|
|
80
|
+
"pydantic",
|
|
81
|
+
"pydantic_core_schema",
|
|
67
82
|
]
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
# Copyright (c) QuantCo 2025-2025
|
|
2
|
+
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from functools import partial
|
|
6
|
+
from typing import TYPE_CHECKING, Literal, TypeVar, get_args, get_origin, overload
|
|
7
|
+
|
|
8
|
+
import polars as pl
|
|
9
|
+
|
|
10
|
+
from ._base_schema import BaseSchema
|
|
11
|
+
from ._compat import pydantic, pydantic_core_schema
|
|
12
|
+
from .exc import ValidationError
|
|
13
|
+
|
|
14
|
+
if TYPE_CHECKING:
|
|
15
|
+
from ._typing import DataFrame, LazyFrame
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
_S = TypeVar("_S", bound=BaseSchema)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _dict_to_df(schema_type: type[BaseSchema], data: dict) -> pl.DataFrame:
|
|
22
|
+
return pl.from_dict(
|
|
23
|
+
data,
|
|
24
|
+
schema=schema_type.polars_schema(),
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _validate_df_schema(schema_type: type[_S], df: pl.DataFrame) -> DataFrame[_S]:
|
|
29
|
+
try:
|
|
30
|
+
return schema_type.validate(df, cast=False)
|
|
31
|
+
except ValidationError as e:
|
|
32
|
+
raise ValueError("DataFrame violates schema") from e
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _serialize_df(df: pl.DataFrame) -> dict:
|
|
36
|
+
return df.to_dict(as_series=False)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@overload
|
|
40
|
+
def get_pydantic_core_schema(
|
|
41
|
+
source_type: type[DataFrame],
|
|
42
|
+
_handler: pydantic.GetCoreSchemaHandler,
|
|
43
|
+
lazy: Literal[False],
|
|
44
|
+
) -> pydantic_core_schema.CoreSchema: ...
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@overload
|
|
48
|
+
def get_pydantic_core_schema(
|
|
49
|
+
source_type: type[LazyFrame],
|
|
50
|
+
_handler: pydantic.GetCoreSchemaHandler,
|
|
51
|
+
lazy: Literal[True],
|
|
52
|
+
) -> pydantic_core_schema.CoreSchema: ...
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def get_pydantic_core_schema(
|
|
56
|
+
source_type: type[DataFrame | LazyFrame],
|
|
57
|
+
_handler: pydantic.GetCoreSchemaHandler,
|
|
58
|
+
lazy: bool,
|
|
59
|
+
) -> pydantic_core_schema.CoreSchema:
|
|
60
|
+
# https://docs.pydantic.dev/2.11/concepts/types/#handling-custom-generic-classes
|
|
61
|
+
origin = get_origin(source_type)
|
|
62
|
+
if origin is None:
|
|
63
|
+
# used as `x: dy.DataFrame` without schema
|
|
64
|
+
raise TypeError("DataFrame must be parametrized with a schema")
|
|
65
|
+
|
|
66
|
+
schema_type: type[BaseSchema] = get_args(source_type)[0]
|
|
67
|
+
|
|
68
|
+
# accept a DataFrame, a LazyFrame, or a dict that is converted to a DataFrame
|
|
69
|
+
# (-> output: DataFrame or LazyFrame)
|
|
70
|
+
polars_schema = pydantic_core_schema.union_schema(
|
|
71
|
+
[
|
|
72
|
+
pydantic_core_schema.is_instance_schema(pl.DataFrame),
|
|
73
|
+
pydantic_core_schema.is_instance_schema(pl.LazyFrame),
|
|
74
|
+
pydantic_core_schema.chain_schema(
|
|
75
|
+
[
|
|
76
|
+
pydantic_core_schema.dict_schema(),
|
|
77
|
+
pydantic_core_schema.no_info_plain_validator_function(
|
|
78
|
+
partial(_dict_to_df, schema_type)
|
|
79
|
+
),
|
|
80
|
+
]
|
|
81
|
+
),
|
|
82
|
+
]
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
to_lazy_schema = []
|
|
86
|
+
if lazy:
|
|
87
|
+
# If the Pydantic field type is LazyFrame, add a step to convert
|
|
88
|
+
# the model back to a LazyFrame.
|
|
89
|
+
to_lazy_schema.append(
|
|
90
|
+
pydantic_core_schema.no_info_plain_validator_function(
|
|
91
|
+
lambda df: df.lazy(),
|
|
92
|
+
)
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
return pydantic_core_schema.chain_schema(
|
|
96
|
+
[
|
|
97
|
+
polars_schema,
|
|
98
|
+
pydantic_core_schema.no_info_plain_validator_function(
|
|
99
|
+
partial(_validate_df_schema, schema_type)
|
|
100
|
+
),
|
|
101
|
+
*to_lazy_schema,
|
|
102
|
+
],
|
|
103
|
+
serialization=pydantic_core_schema.plain_serializer_function_ser_schema(
|
|
104
|
+
_serialize_df
|
|
105
|
+
),
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def get_pydantic_json_schema(
|
|
110
|
+
handler: pydantic.GetJsonSchemaHandler,
|
|
111
|
+
) -> pydantic.json_schema.JsonSchemaValue:
|
|
112
|
+
from pydantic_core import core_schema
|
|
113
|
+
|
|
114
|
+
# This could be made more sophisticated by actually reflecting the schema.
|
|
115
|
+
return handler(core_schema.dict_schema())
|
|
@@ -9,6 +9,8 @@ from typing import TYPE_CHECKING, Any, Concatenate, Generic, Literal, ParamSpec,
|
|
|
9
9
|
import polars as pl
|
|
10
10
|
|
|
11
11
|
from ._base_schema import BaseSchema
|
|
12
|
+
from ._compat import pydantic, pydantic_core_schema
|
|
13
|
+
from ._pydantic import get_pydantic_core_schema, get_pydantic_json_schema
|
|
12
14
|
|
|
13
15
|
S = TypeVar("S", bound=BaseSchema, covariant=True)
|
|
14
16
|
|
|
@@ -70,6 +72,20 @@ class DataFrame(pl.DataFrame, Generic[S]):
|
|
|
70
72
|
def shrink_to_fit(self, *args: Any, **kwargs: Any) -> DataFrame[S]:
|
|
71
73
|
raise NotImplementedError # pragma: no cover
|
|
72
74
|
|
|
75
|
+
@classmethod
|
|
76
|
+
def __get_pydantic_core_schema__(
|
|
77
|
+
cls, source_type: Any, handler: pydantic.GetCoreSchemaHandler
|
|
78
|
+
) -> pydantic_core_schema.CoreSchema:
|
|
79
|
+
return get_pydantic_core_schema(source_type, handler, lazy=False)
|
|
80
|
+
|
|
81
|
+
@classmethod
|
|
82
|
+
def __get_pydantic_json_schema__(
|
|
83
|
+
cls,
|
|
84
|
+
_core_schema: pydantic_core_schema.CoreSchema,
|
|
85
|
+
handler: pydantic.GetJsonSchemaHandler,
|
|
86
|
+
) -> pydantic.json_schema.JsonSchemaValue:
|
|
87
|
+
return get_pydantic_json_schema(handler)
|
|
88
|
+
|
|
73
89
|
|
|
74
90
|
class LazyFrame(pl.LazyFrame, Generic[S]):
|
|
75
91
|
"""Generic wrapper around a :class:`polars.LazyFrame` to attach schema information.
|
|
@@ -113,3 +129,17 @@ class LazyFrame(pl.LazyFrame, Generic[S]):
|
|
|
113
129
|
@inherit_signature(pl.LazyFrame.set_sorted)
|
|
114
130
|
def set_sorted(self, *args: Any, **kwargs: Any) -> LazyFrame[S]:
|
|
115
131
|
raise NotImplementedError # pragma: no cover
|
|
132
|
+
|
|
133
|
+
@classmethod
|
|
134
|
+
def __get_pydantic_core_schema__(
|
|
135
|
+
cls, source_type: Any, handler: pydantic.GetCoreSchemaHandler
|
|
136
|
+
) -> pydantic_core_schema.CoreSchema:
|
|
137
|
+
return get_pydantic_core_schema(source_type, handler, lazy=True)
|
|
138
|
+
|
|
139
|
+
@classmethod
|
|
140
|
+
def __get_pydantic_json_schema__(
|
|
141
|
+
cls,
|
|
142
|
+
_core_schema: pydantic_core_schema.CoreSchema,
|
|
143
|
+
handler: pydantic.GetJsonSchemaHandler,
|
|
144
|
+
) -> pydantic.json_schema.JsonSchemaValue:
|
|
145
|
+
return get_pydantic_json_schema(handler)
|
|
@@ -381,15 +381,6 @@ class Column(ABC):
|
|
|
381
381
|
if name == "check":
|
|
382
382
|
return _compare_checks(lhs, rhs, column_expr)
|
|
383
383
|
|
|
384
|
-
lhs_is_series = isinstance(lhs, pl.Series)
|
|
385
|
-
rhs_is_series = isinstance(rhs, pl.Series)
|
|
386
|
-
|
|
387
|
-
if lhs_is_series != rhs_is_series:
|
|
388
|
-
return False
|
|
389
|
-
|
|
390
|
-
if lhs_is_series and rhs_is_series:
|
|
391
|
-
return _compare_series(lhs, rhs)
|
|
392
|
-
|
|
393
384
|
return lhs == rhs
|
|
394
385
|
|
|
395
386
|
# -------------------------------- DUNDER METHODS -------------------------------- #
|
|
@@ -413,10 +404,6 @@ class Column(ABC):
|
|
|
413
404
|
return self.__class__.__name__.lower()
|
|
414
405
|
|
|
415
406
|
|
|
416
|
-
def _compare_series(lhs: pl.Series, rhs: pl.Series) -> bool:
|
|
417
|
-
return (len(lhs) == len(rhs)) and lhs.equals(rhs)
|
|
418
|
-
|
|
419
|
-
|
|
420
407
|
def _compare_checks(lhs: Check | None, rhs: Check | None, expr: pl.Expr) -> bool:
|
|
421
408
|
match (lhs, rhs):
|
|
422
409
|
case (None, None):
|
|
@@ -67,12 +67,8 @@ class Enum(Column):
|
|
|
67
67
|
metadata=metadata,
|
|
68
68
|
)
|
|
69
69
|
if isclass(categories) and issubclass(categories, enum.Enum):
|
|
70
|
-
categories =
|
|
71
|
-
|
|
72
|
-
)
|
|
73
|
-
elif not isinstance(categories, pl.Series):
|
|
74
|
-
categories = pl.Series(values=categories)
|
|
75
|
-
self.categories = categories
|
|
70
|
+
categories = (item.value for item in categories)
|
|
71
|
+
self.categories = list(categories)
|
|
76
72
|
|
|
77
73
|
@property
|
|
78
74
|
def dtype(self) -> pl.DataType:
|
|
@@ -81,7 +77,7 @@ class Enum(Column):
|
|
|
81
77
|
def validate_dtype(self, dtype: PolarsDataType) -> bool:
|
|
82
78
|
if not isinstance(dtype, pl.Enum):
|
|
83
79
|
return False
|
|
84
|
-
return self.categories
|
|
80
|
+
return self.categories == dtype.categories.to_list()
|
|
85
81
|
|
|
86
82
|
def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
|
|
87
83
|
category_lengths = [len(c) for c in self.categories]
|
|
@@ -102,6 +98,6 @@ class Enum(Column):
|
|
|
102
98
|
def _sample_unchecked(self, generator: Generator, n: int) -> pl.Series:
|
|
103
99
|
return generator.sample_choice(
|
|
104
100
|
n,
|
|
105
|
-
choices=self.categories
|
|
101
|
+
choices=self.categories,
|
|
106
102
|
null_probability=self._null_probability,
|
|
107
103
|
).cast(self.dtype)
|
|
@@ -414,26 +414,6 @@ class Schema(BaseSchema, ABC):
|
|
|
414
414
|
def validate(
|
|
415
415
|
cls, df: pl.DataFrame | pl.LazyFrame, /, *, cast: bool = False
|
|
416
416
|
) -> DataFrame[Self]:
|
|
417
|
-
"""Validate that a data frame satisfies the schema.
|
|
418
|
-
|
|
419
|
-
Args:
|
|
420
|
-
df: The data frame to validate.
|
|
421
|
-
cast: Whether columns with a wrong data type in the input data frame are
|
|
422
|
-
cast to the schema's defined data type if possible.
|
|
423
|
-
|
|
424
|
-
Returns:
|
|
425
|
-
The (collected) input data frame, wrapped in a generic version of the
|
|
426
|
-
input's data frame type to reflect schema adherence. The data frame is
|
|
427
|
-
guaranteed to maintain its order.
|
|
428
|
-
|
|
429
|
-
Raises:
|
|
430
|
-
ValidationError: If the input data frame does not satisfy the schema
|
|
431
|
-
definition.
|
|
432
|
-
|
|
433
|
-
Note:
|
|
434
|
-
This method _always_ collects the input data frame in order to raise
|
|
435
|
-
potential validation errors.
|
|
436
|
-
"""
|
|
437
417
|
# We can dispatch to the `filter` method and raise an error if any row cannot
|
|
438
418
|
# be validated
|
|
439
419
|
df_valid, failures = cls.filter(df, cast=cast)
|
|
@@ -1118,11 +1098,6 @@ class Schema(BaseSchema, ABC):
|
|
|
1118
1098
|
|
|
1119
1099
|
@classmethod
|
|
1120
1100
|
def polars_schema(cls) -> pl.Schema:
|
|
1121
|
-
"""Obtain the polars schema for this schema.
|
|
1122
|
-
|
|
1123
|
-
Returns:
|
|
1124
|
-
A :mod:`polars` schema that mirrors the schema defined by this class.
|
|
1125
|
-
"""
|
|
1126
1101
|
return pl.Schema({name: col.dtype for name, col in cls.columns().items()})
|
|
1127
1102
|
|
|
1128
1103
|
@classmethod
|
|
@@ -253,6 +253,7 @@ Lastly, ``dataframely`` schemas can be used to integrate with external tools:
|
|
|
253
253
|
- ``HouseSchema.create_empty()`` creates an empty ``dy.DataFrame[HouseSchema]`` that can be used for testing
|
|
254
254
|
- ``HouseSchema.sql_schema()`` provides a list of `sqlalchemy <https://www.sqlalchemy.org>`_ columns that can be used to create SQL tables using types and constraints in line with the schema
|
|
255
255
|
- ``HouseSchema.pyarrow_schema()`` provides a `pyarrow <https://arrow.apache.org/docs/python/index.html>`_ schema with appropriate column dtypes and nullability information
|
|
256
|
+
- You can use ``dy.DataFrame[HouseSchema]`` (or the ``LazyFrame`` equivalent) as fields in `pydantic <https://pydantic.dev>`_ models, including support for validation and serialization. Integration with pydantic is unstable.
|
|
256
257
|
|
|
257
258
|
|
|
258
259
|
Outlook
|