dataframely 1.12.1__tar.gz → 1.14.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataframely-1.12.1 → dataframely-1.14.0}/.github/workflows/build.yml +7 -7
- {dataframely-1.12.1 → dataframely-1.14.0}/.github/workflows/chore.yml +1 -1
- {dataframely-1.12.1 → dataframely-1.14.0}/.github/workflows/ci.yml +5 -5
- {dataframely-1.12.1 → dataframely-1.14.0}/.github/workflows/nightly.yml +2 -2
- {dataframely-1.12.1 → dataframely-1.14.0}/.github/workflows/scorecard.yml +3 -3
- {dataframely-1.12.1 → dataframely-1.14.0}/PKG-INFO +3 -1
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_base_schema.py +64 -2
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_compat.py +15 -0
- dataframely-1.14.0/dataframely/_pydantic.py +115 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_typing.py +30 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/_mixins.py +1 -1
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/schema.py +0 -25
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.rst +16 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.testing.rst +8 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/index.rst +2 -1
- dataframely-1.14.0/docs/sites/faq.rst +30 -0
- dataframely-1.14.0/docs/sites/features/index.rst +7 -0
- dataframely-1.14.0/docs/sites/features/primary-keys.rst +47 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/sites/quickstart.rst +1 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/pixi.lock +6322 -7106
- {dataframely-1.12.1 → dataframely-1.14.0}/pixi.toml +2 -1
- {dataframely-1.12.1 → dataframely-1.14.0}/pyproject.toml +6 -2
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_integer.py +28 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_base.py +31 -0
- dataframely-1.14.0/tests/test_pydantic.py +164 -0
- dataframely-1.12.1/docs/sites/faq.rst +0 -5
- {dataframely-1.12.1 → dataframely-1.14.0}/.copier-answers.yml +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.envrc +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.gitattributes +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.github/CODEOWNERS +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.github/dependabot.yml +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.github/release-drafter.yml +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.gitignore +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.pre-commit-config.yaml +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.prettierignore +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.prettierrc +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/.readthedocs.yml +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/Cargo.lock +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/Cargo.toml +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/LICENSE +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/README.md +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/SECURITY.md +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/__init__.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_base_collection.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_deprecation.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_extre.pyi +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_filter.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_polars.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_rule.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_serialization.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_storage/__init__.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_storage/_base.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_storage/_exc.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_storage/constants.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_storage/delta.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_storage/parquet.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/_validation.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/collection.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/__init__.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/_base.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/_registry.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/_utils.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/any.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/array.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/binary.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/bool.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/categorical.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/datetime.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/decimal.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/enum.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/float.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/integer.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/list.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/object.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/string.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/columns/struct.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/config.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/exc.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/failure.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/functional.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/mypy.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/py.typed +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/random.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/testing/__init__.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/testing/const.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/testing/factory.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/testing/mask.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/testing/rules.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/testing/storage.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/dataframely/testing/typing.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docker-compose.yml +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/Makefile +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.collection.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.any.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.bool.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.datetime.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.decimal.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.enum.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.float.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.integer.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.list.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.string.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.columns.struct.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.config.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.exc.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.failure.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.functional.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.mypy.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.random.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.schema.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.testing.const.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.testing.factory.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.testing.mask.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.testing.rules.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/dataframely.testing.typing.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_api/modules.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_static/custom.css +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/_static/favicon.ico +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/conf.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/make.bat +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/sites/development.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/sites/examples/real-world.ipynb +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/sites/installation.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/docs/sites/versioning.rst +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/src/errdefs.rs +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/src/lib.rs +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/src/regex_repr.rs +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/benches/conftest.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/benches/test_collection.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/benches/test_failure.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/benches/test_schema.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_base.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_cast.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_collection_future_annotations.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_create_empty.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_filter_one_to_n.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_filter_validate.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_ignore_in_filter.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_implementation.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_join.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_matches.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_optional_members.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_repr.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_sample.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_serialization.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_storage.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/collection/test_validate_input.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/__init__.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_any.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_array.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_binary.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_datetime.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_decimal.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_enum.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_float.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_list.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_object.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_string.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/column_types/test_struct.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/__init__.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_alias.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_check.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_default_dtypes.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_matches.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_metadata.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_polars_schema.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_pyarrow.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_rules.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_sample.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_sql_schema.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_str.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/columns/test_utils.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/core_validation/__init__.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/core_validation/test_column_validation.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/core_validation/test_dtype_validation.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/core_validation/test_rule_evaluation.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/failure_info/test_storage.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/functional/test_concat.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/functional/test_relationships.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_cast.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_create_empty.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_create_empty_if_none.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_filter.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_inheritance.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_matches.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_read_write_parquet.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_repr.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_rule_implementation.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_sample.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_serialization.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_storage.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/schema/test_validate.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/storage/test_delta.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/test_compat.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/test_config.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/test_deprecation.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/test_exc.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/test_extre.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/test_factory.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/test_random.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/test_serialization.py +0 -0
- {dataframely-1.12.1 → dataframely-1.14.0}/tests/test_typing.py +0 -0
|
@@ -13,11 +13,11 @@ jobs:
|
|
|
13
13
|
permissions:
|
|
14
14
|
contents: read
|
|
15
15
|
steps:
|
|
16
|
-
- uses: actions/checkout@
|
|
16
|
+
- uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
|
|
17
17
|
with:
|
|
18
18
|
fetch-depth: 0
|
|
19
19
|
- name: Set up pixi
|
|
20
|
-
uses: prefix-dev/setup-pixi@
|
|
20
|
+
uses: prefix-dev/setup-pixi@194d461b21b6c5717c722ffc597fa91ed2ff29fa # v0.9.1
|
|
21
21
|
with:
|
|
22
22
|
environments: build
|
|
23
23
|
- name: Set version
|
|
@@ -48,17 +48,17 @@ jobs:
|
|
|
48
48
|
- target-platform: win-64
|
|
49
49
|
os: windows-latest
|
|
50
50
|
steps:
|
|
51
|
-
- uses: actions/checkout@
|
|
51
|
+
- uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
|
|
52
52
|
with:
|
|
53
53
|
fetch-depth: 0
|
|
54
54
|
- name: Set up pixi
|
|
55
|
-
uses: prefix-dev/setup-pixi@
|
|
55
|
+
uses: prefix-dev/setup-pixi@194d461b21b6c5717c722ffc597fa91ed2ff29fa # v0.9.1
|
|
56
56
|
with:
|
|
57
57
|
environments: build
|
|
58
58
|
- name: Set version
|
|
59
59
|
run: pixi run -e build set-version
|
|
60
60
|
- name: Build wheel
|
|
61
|
-
uses: PyO3/maturin-action@
|
|
61
|
+
uses: PyO3/maturin-action@86b9d133d34bc1b40018696f782949dac11bd380 # v1.49.4
|
|
62
62
|
with:
|
|
63
63
|
command: build
|
|
64
64
|
args: --out dist -i python3.10
|
|
@@ -80,9 +80,9 @@ jobs:
|
|
|
80
80
|
id-token: write
|
|
81
81
|
environment: pypi
|
|
82
82
|
steps:
|
|
83
|
-
- uses: actions/download-artifact@
|
|
83
|
+
- uses: actions/download-artifact@634f93cb2916e3fdff6788551b99b062d0335ce0 # v5.0.0
|
|
84
84
|
with:
|
|
85
85
|
path: dist
|
|
86
86
|
merge-multiple: true
|
|
87
87
|
- name: Publish package on PyPi
|
|
88
|
-
uses: pypa/gh-action-pypi-publish@
|
|
88
|
+
uses: pypa/gh-action-pypi-publish@ed0c53931b1dc9bd32cbe73a98c7f6766f8a527e # v1.13.0
|
|
@@ -21,7 +21,7 @@ jobs:
|
|
|
21
21
|
steps:
|
|
22
22
|
- name: Check valid conventional commit message
|
|
23
23
|
id: lint
|
|
24
|
-
uses: amannn/action-semantic-pull-request@
|
|
24
|
+
uses: amannn/action-semantic-pull-request@48f256284bd46cdaab1048c3721360e808335d50 # v6.1.1
|
|
25
25
|
with:
|
|
26
26
|
subjectPattern: ^[A-Z].+[^. ]$ # subject must start with uppercase letter and may not end with a dot/space
|
|
27
27
|
env:
|
|
@@ -19,12 +19,12 @@ jobs:
|
|
|
19
19
|
runs-on: ubuntu-latest
|
|
20
20
|
steps:
|
|
21
21
|
- name: Checkout branch
|
|
22
|
-
uses: actions/checkout@
|
|
22
|
+
uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
|
|
23
23
|
with:
|
|
24
24
|
# needed for 'pre-commit-mirrors-insert-license'
|
|
25
25
|
fetch-depth: 0
|
|
26
26
|
- name: Set up pixi
|
|
27
|
-
uses: prefix-dev/setup-pixi@
|
|
27
|
+
uses: prefix-dev/setup-pixi@194d461b21b6c5717c722ffc597fa91ed2ff29fa # v0.9.1
|
|
28
28
|
with:
|
|
29
29
|
environments: default lint
|
|
30
30
|
- name: Install repository
|
|
@@ -51,9 +51,9 @@ jobs:
|
|
|
51
51
|
with_optionals: true
|
|
52
52
|
steps:
|
|
53
53
|
- name: Checkout branch
|
|
54
|
-
uses: actions/checkout@
|
|
54
|
+
uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
|
|
55
55
|
- name: Set up pixi
|
|
56
|
-
uses: prefix-dev/setup-pixi@
|
|
56
|
+
uses: prefix-dev/setup-pixi@194d461b21b6c5717c722ffc597fa91ed2ff29fa # v0.9.1
|
|
57
57
|
with:
|
|
58
58
|
environments: ${{ matrix.environment }}
|
|
59
59
|
- name: Install repository
|
|
@@ -61,7 +61,7 @@ jobs:
|
|
|
61
61
|
- name: Run pytest
|
|
62
62
|
run: pixi run -e ${{ matrix.environment }} test-coverage --color=yes ${{ matrix.with_optionals && '-m with_optionals' || '-m "not with_optionals"'}} --cov=dataframely --cov-report=xml
|
|
63
63
|
- name: Upload codecov
|
|
64
|
-
uses: codecov/codecov-action@
|
|
64
|
+
uses: codecov/codecov-action@5a1091511ad55cbe89839c7260b706298ca349f7 # v5.5.1
|
|
65
65
|
with:
|
|
66
66
|
files: ./coverage.xml
|
|
67
67
|
token: ${{ secrets.CODECOV_TOKEN }}
|
|
@@ -23,9 +23,9 @@ jobs:
|
|
|
23
23
|
os: [ubuntu-latest, windows-latest]
|
|
24
24
|
steps:
|
|
25
25
|
- name: Checkout branch
|
|
26
|
-
uses: actions/checkout@
|
|
26
|
+
uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
|
|
27
27
|
- name: Set up pixi
|
|
28
|
-
uses: prefix-dev/setup-pixi@
|
|
28
|
+
uses: prefix-dev/setup-pixi@194d461b21b6c5717c722ffc597fa91ed2ff29fa # v0.9.1
|
|
29
29
|
with:
|
|
30
30
|
environments: nightly
|
|
31
31
|
- name: Install polars nightly
|
|
@@ -35,12 +35,12 @@ jobs:
|
|
|
35
35
|
|
|
36
36
|
steps:
|
|
37
37
|
- name: "Checkout code"
|
|
38
|
-
uses: actions/checkout@
|
|
38
|
+
uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.0
|
|
39
39
|
with:
|
|
40
40
|
persist-credentials: false
|
|
41
41
|
|
|
42
42
|
- name: "Run analysis"
|
|
43
|
-
uses: ossf/scorecard-action@
|
|
43
|
+
uses: ossf/scorecard-action@4eaacf0543bb3f2c246792bd56e8cdeffafb205a # v2.4.3
|
|
44
44
|
with:
|
|
45
45
|
results_file: results.sarif
|
|
46
46
|
results_format: sarif
|
|
@@ -74,6 +74,6 @@ jobs:
|
|
|
74
74
|
# Upload the results to GitHub's code scanning dashboard (optional).
|
|
75
75
|
# Commenting out will disable upload of results to your repo's Code Scanning dashboard
|
|
76
76
|
- name: "Upload to code-scanning"
|
|
77
|
-
uses: github/codeql-action/upload-sarif@
|
|
77
|
+
uses: github/codeql-action/upload-sarif@3599b3baa15b485a2e49ef411a7a4bb2452e7f93 # v3.29.5
|
|
78
78
|
with:
|
|
79
79
|
sarif_file: results.sarif
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dataframely
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.14.0
|
|
4
4
|
Classifier: Programming Language :: Python :: 3
|
|
5
5
|
Classifier: Programming Language :: Python :: 3.10
|
|
6
6
|
Classifier: Programming Language :: Python :: 3.11
|
|
@@ -12,9 +12,11 @@ Requires-Dist: typing-extensions ; python_full_version < '3.11'
|
|
|
12
12
|
Requires-Dist: deltalake ; extra == 'deltalake'
|
|
13
13
|
Requires-Dist: sqlalchemy ; extra == 'sqlalchemy'
|
|
14
14
|
Requires-Dist: pyarrow ; extra == 'pyarrow'
|
|
15
|
+
Requires-Dist: pydantic>=2 ; extra == 'pydantic'
|
|
15
16
|
Provides-Extra: deltalake
|
|
16
17
|
Provides-Extra: sqlalchemy
|
|
17
18
|
Provides-Extra: pyarrow
|
|
19
|
+
Provides-Extra: pydantic
|
|
18
20
|
License-File: LICENSE
|
|
19
21
|
Summary: A declarative, polars-native data frame validation library
|
|
20
22
|
Author-email: Andreas Albert <andreas.albert@quantco.com>, Daniel Elsner <daniel.elsner@quantco.com>, Oliver Borchert <oliver.borchert@quantco.com>
|
|
@@ -5,10 +5,10 @@ from __future__ import annotations
|
|
|
5
5
|
|
|
6
6
|
import sys
|
|
7
7
|
import textwrap
|
|
8
|
-
from abc import ABCMeta
|
|
8
|
+
from abc import ABCMeta, abstractmethod
|
|
9
9
|
from copy import copy
|
|
10
10
|
from dataclasses import dataclass, field
|
|
11
|
-
from typing import Any
|
|
11
|
+
from typing import TYPE_CHECKING, Any
|
|
12
12
|
|
|
13
13
|
import polars as pl
|
|
14
14
|
|
|
@@ -21,6 +21,10 @@ if sys.version_info >= (3, 11):
|
|
|
21
21
|
else:
|
|
22
22
|
from typing_extensions import Self
|
|
23
23
|
|
|
24
|
+
|
|
25
|
+
if TYPE_CHECKING:
|
|
26
|
+
from ._typing import DataFrame
|
|
27
|
+
|
|
24
28
|
_COLUMN_ATTR = "__dataframely_columns__"
|
|
25
29
|
_RULE_ATTR = "__dataframely_rules__"
|
|
26
30
|
|
|
@@ -133,6 +137,29 @@ class SchemaMeta(ABCMeta):
|
|
|
133
137
|
f"which are not in the schema: {missing_list}."
|
|
134
138
|
)
|
|
135
139
|
|
|
140
|
+
# 3) Check that all members are non-pathological (i.e., user errors).
|
|
141
|
+
for attr, value in namespace.items():
|
|
142
|
+
if attr.startswith("__"):
|
|
143
|
+
continue
|
|
144
|
+
# Check for tuple of column (commonly caused by trailing comma)
|
|
145
|
+
if (
|
|
146
|
+
isinstance(value, tuple)
|
|
147
|
+
and len(value) > 0
|
|
148
|
+
and isinstance(value[0], Column)
|
|
149
|
+
):
|
|
150
|
+
raise TypeError(
|
|
151
|
+
f"Column '{attr}' is defined as a tuple of dy.Column. "
|
|
152
|
+
f"Did you accidentally add a trailing comma?"
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
# Check for column type instead of instance (e.g., dy.Float64 instead of dy.Float64())
|
|
156
|
+
if isinstance(value, type) and issubclass(value, Column):
|
|
157
|
+
raise TypeError(
|
|
158
|
+
f"Column '{attr}' is a type, not an instance. "
|
|
159
|
+
f"Schema members must be of type Column not type[Column]. "
|
|
160
|
+
f"Did you forget to add parentheses?"
|
|
161
|
+
)
|
|
162
|
+
|
|
136
163
|
return super().__new__(mcs, name, bases, namespace, *args, **kwargs)
|
|
137
164
|
|
|
138
165
|
def __getattribute__(cls, name: str) -> Any:
|
|
@@ -198,11 +225,46 @@ class BaseSchema(metaclass=SchemaMeta):
|
|
|
198
225
|
columns[name]._name = name
|
|
199
226
|
return columns
|
|
200
227
|
|
|
228
|
+
@classmethod
|
|
229
|
+
@abstractmethod
|
|
230
|
+
def polars_schema(cls) -> pl.Schema:
|
|
231
|
+
"""Obtain the polars schema for this schema.
|
|
232
|
+
|
|
233
|
+
Returns:
|
|
234
|
+
A :mod:`polars` schema that mirrors the schema defined by this class.
|
|
235
|
+
"""
|
|
236
|
+
|
|
201
237
|
@classmethod
|
|
202
238
|
def primary_keys(cls) -> list[str]:
|
|
203
239
|
"""The primary key columns in this schema (possibly empty)."""
|
|
204
240
|
return _primary_keys(cls.columns())
|
|
205
241
|
|
|
242
|
+
@classmethod
|
|
243
|
+
@abstractmethod
|
|
244
|
+
def validate(
|
|
245
|
+
cls, df: pl.DataFrame | pl.LazyFrame, /, *, cast: bool = False
|
|
246
|
+
) -> DataFrame[Self]:
|
|
247
|
+
"""Validate that a data frame satisfies the schema.
|
|
248
|
+
|
|
249
|
+
Args:
|
|
250
|
+
df: The data frame to validate.
|
|
251
|
+
cast: Whether columns with a wrong data type in the input data frame are
|
|
252
|
+
cast to the schema's defined data type if possible.
|
|
253
|
+
|
|
254
|
+
Returns:
|
|
255
|
+
The (collected) input data frame, wrapped in a generic version of the
|
|
256
|
+
input's data frame type to reflect schema adherence. The data frame is
|
|
257
|
+
guaranteed to maintain its order.
|
|
258
|
+
|
|
259
|
+
Raises:
|
|
260
|
+
ValidationError: If the input data frame does not satisfy the schema
|
|
261
|
+
definition.
|
|
262
|
+
|
|
263
|
+
Note:
|
|
264
|
+
This method _always_ collects the input data frame in order to raise
|
|
265
|
+
potential validation errors.
|
|
266
|
+
"""
|
|
267
|
+
|
|
206
268
|
@classmethod
|
|
207
269
|
def _validation_rules(cls, *, with_cast: bool) -> dict[str, Rule]:
|
|
208
270
|
return _build_rules(
|
|
@@ -54,6 +54,19 @@ try:
|
|
|
54
54
|
except ImportError: # pragma: no cover
|
|
55
55
|
pa = _DummyModule("pyarrow")
|
|
56
56
|
|
|
57
|
+
|
|
58
|
+
# -------------------------------------- PYDANTIC ------------------------------------ #
|
|
59
|
+
|
|
60
|
+
try:
|
|
61
|
+
import pydantic
|
|
62
|
+
except ImportError: # pragma: no cover
|
|
63
|
+
pydantic = _DummyModule("pydantic") # type: ignore
|
|
64
|
+
|
|
65
|
+
try:
|
|
66
|
+
from pydantic_core import core_schema as pydantic_core_schema # pragma: no cover
|
|
67
|
+
except ImportError:
|
|
68
|
+
pydantic_core_schema = _DummyModule("pydantic_core_schema") # type: ignore
|
|
69
|
+
|
|
57
70
|
# ------------------------------------------------------------------------------------ #
|
|
58
71
|
|
|
59
72
|
__all__ = [
|
|
@@ -64,4 +77,6 @@ __all__ = [
|
|
|
64
77
|
"pa",
|
|
65
78
|
"MSDialect_pyodbc",
|
|
66
79
|
"PGDialect_psycopg2",
|
|
80
|
+
"pydantic",
|
|
81
|
+
"pydantic_core_schema",
|
|
67
82
|
]
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
# Copyright (c) QuantCo 2025-2025
|
|
2
|
+
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from functools import partial
|
|
6
|
+
from typing import TYPE_CHECKING, Literal, TypeVar, get_args, get_origin, overload
|
|
7
|
+
|
|
8
|
+
import polars as pl
|
|
9
|
+
|
|
10
|
+
from ._base_schema import BaseSchema
|
|
11
|
+
from ._compat import pydantic, pydantic_core_schema
|
|
12
|
+
from .exc import ValidationError
|
|
13
|
+
|
|
14
|
+
if TYPE_CHECKING:
|
|
15
|
+
from ._typing import DataFrame, LazyFrame
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
_S = TypeVar("_S", bound=BaseSchema)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _dict_to_df(schema_type: type[BaseSchema], data: dict) -> pl.DataFrame:
|
|
22
|
+
return pl.from_dict(
|
|
23
|
+
data,
|
|
24
|
+
schema=schema_type.polars_schema(),
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _validate_df_schema(schema_type: type[_S], df: pl.DataFrame) -> DataFrame[_S]:
|
|
29
|
+
try:
|
|
30
|
+
return schema_type.validate(df, cast=False)
|
|
31
|
+
except ValidationError as e:
|
|
32
|
+
raise ValueError("DataFrame violates schema") from e
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _serialize_df(df: pl.DataFrame) -> dict:
|
|
36
|
+
return df.to_dict(as_series=False)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@overload
|
|
40
|
+
def get_pydantic_core_schema(
|
|
41
|
+
source_type: type[DataFrame],
|
|
42
|
+
_handler: pydantic.GetCoreSchemaHandler,
|
|
43
|
+
lazy: Literal[False],
|
|
44
|
+
) -> pydantic_core_schema.CoreSchema: ...
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@overload
|
|
48
|
+
def get_pydantic_core_schema(
|
|
49
|
+
source_type: type[LazyFrame],
|
|
50
|
+
_handler: pydantic.GetCoreSchemaHandler,
|
|
51
|
+
lazy: Literal[True],
|
|
52
|
+
) -> pydantic_core_schema.CoreSchema: ...
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def get_pydantic_core_schema(
|
|
56
|
+
source_type: type[DataFrame | LazyFrame],
|
|
57
|
+
_handler: pydantic.GetCoreSchemaHandler,
|
|
58
|
+
lazy: bool,
|
|
59
|
+
) -> pydantic_core_schema.CoreSchema:
|
|
60
|
+
# https://docs.pydantic.dev/2.11/concepts/types/#handling-custom-generic-classes
|
|
61
|
+
origin = get_origin(source_type)
|
|
62
|
+
if origin is None:
|
|
63
|
+
# used as `x: dy.DataFrame` without schema
|
|
64
|
+
raise TypeError("DataFrame must be parametrized with a schema")
|
|
65
|
+
|
|
66
|
+
schema_type: type[BaseSchema] = get_args(source_type)[0]
|
|
67
|
+
|
|
68
|
+
# accept a DataFrame, a LazyFrame, or a dict that is converted to a DataFrame
|
|
69
|
+
# (-> output: DataFrame or LazyFrame)
|
|
70
|
+
polars_schema = pydantic_core_schema.union_schema(
|
|
71
|
+
[
|
|
72
|
+
pydantic_core_schema.is_instance_schema(pl.DataFrame),
|
|
73
|
+
pydantic_core_schema.is_instance_schema(pl.LazyFrame),
|
|
74
|
+
pydantic_core_schema.chain_schema(
|
|
75
|
+
[
|
|
76
|
+
pydantic_core_schema.dict_schema(),
|
|
77
|
+
pydantic_core_schema.no_info_plain_validator_function(
|
|
78
|
+
partial(_dict_to_df, schema_type)
|
|
79
|
+
),
|
|
80
|
+
]
|
|
81
|
+
),
|
|
82
|
+
]
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
to_lazy_schema = []
|
|
86
|
+
if lazy:
|
|
87
|
+
# If the Pydantic field type is LazyFrame, add a step to convert
|
|
88
|
+
# the model back to a LazyFrame.
|
|
89
|
+
to_lazy_schema.append(
|
|
90
|
+
pydantic_core_schema.no_info_plain_validator_function(
|
|
91
|
+
lambda df: df.lazy(),
|
|
92
|
+
)
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
return pydantic_core_schema.chain_schema(
|
|
96
|
+
[
|
|
97
|
+
polars_schema,
|
|
98
|
+
pydantic_core_schema.no_info_plain_validator_function(
|
|
99
|
+
partial(_validate_df_schema, schema_type)
|
|
100
|
+
),
|
|
101
|
+
*to_lazy_schema,
|
|
102
|
+
],
|
|
103
|
+
serialization=pydantic_core_schema.plain_serializer_function_ser_schema(
|
|
104
|
+
_serialize_df
|
|
105
|
+
),
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def get_pydantic_json_schema(
|
|
110
|
+
handler: pydantic.GetJsonSchemaHandler,
|
|
111
|
+
) -> pydantic.json_schema.JsonSchemaValue:
|
|
112
|
+
from pydantic_core import core_schema
|
|
113
|
+
|
|
114
|
+
# This could be made more sophisticated by actually reflecting the schema.
|
|
115
|
+
return handler(core_schema.dict_schema())
|
|
@@ -9,6 +9,8 @@ from typing import TYPE_CHECKING, Any, Concatenate, Generic, Literal, ParamSpec,
|
|
|
9
9
|
import polars as pl
|
|
10
10
|
|
|
11
11
|
from ._base_schema import BaseSchema
|
|
12
|
+
from ._compat import pydantic, pydantic_core_schema
|
|
13
|
+
from ._pydantic import get_pydantic_core_schema, get_pydantic_json_schema
|
|
12
14
|
|
|
13
15
|
S = TypeVar("S", bound=BaseSchema, covariant=True)
|
|
14
16
|
|
|
@@ -70,6 +72,20 @@ class DataFrame(pl.DataFrame, Generic[S]):
|
|
|
70
72
|
def shrink_to_fit(self, *args: Any, **kwargs: Any) -> DataFrame[S]:
|
|
71
73
|
raise NotImplementedError # pragma: no cover
|
|
72
74
|
|
|
75
|
+
@classmethod
|
|
76
|
+
def __get_pydantic_core_schema__(
|
|
77
|
+
cls, source_type: Any, handler: pydantic.GetCoreSchemaHandler
|
|
78
|
+
) -> pydantic_core_schema.CoreSchema:
|
|
79
|
+
return get_pydantic_core_schema(source_type, handler, lazy=False)
|
|
80
|
+
|
|
81
|
+
@classmethod
|
|
82
|
+
def __get_pydantic_json_schema__(
|
|
83
|
+
cls,
|
|
84
|
+
_core_schema: pydantic_core_schema.CoreSchema,
|
|
85
|
+
handler: pydantic.GetJsonSchemaHandler,
|
|
86
|
+
) -> pydantic.json_schema.JsonSchemaValue:
|
|
87
|
+
return get_pydantic_json_schema(handler)
|
|
88
|
+
|
|
73
89
|
|
|
74
90
|
class LazyFrame(pl.LazyFrame, Generic[S]):
|
|
75
91
|
"""Generic wrapper around a :class:`polars.LazyFrame` to attach schema information.
|
|
@@ -113,3 +129,17 @@ class LazyFrame(pl.LazyFrame, Generic[S]):
|
|
|
113
129
|
@inherit_signature(pl.LazyFrame.set_sorted)
|
|
114
130
|
def set_sorted(self, *args: Any, **kwargs: Any) -> LazyFrame[S]:
|
|
115
131
|
raise NotImplementedError # pragma: no cover
|
|
132
|
+
|
|
133
|
+
@classmethod
|
|
134
|
+
def __get_pydantic_core_schema__(
|
|
135
|
+
cls, source_type: Any, handler: pydantic.GetCoreSchemaHandler
|
|
136
|
+
) -> pydantic_core_schema.CoreSchema:
|
|
137
|
+
return get_pydantic_core_schema(source_type, handler, lazy=True)
|
|
138
|
+
|
|
139
|
+
@classmethod
|
|
140
|
+
def __get_pydantic_json_schema__(
|
|
141
|
+
cls,
|
|
142
|
+
_core_schema: pydantic_core_schema.CoreSchema,
|
|
143
|
+
handler: pydantic.GetJsonSchemaHandler,
|
|
144
|
+
) -> pydantic.json_schema.JsonSchemaValue:
|
|
145
|
+
return get_pydantic_json_schema(handler)
|
|
@@ -76,7 +76,7 @@ class OrdinalMixin(Generic[T], Base):
|
|
|
76
76
|
result["min_exclusive"] = expr > self.min_exclusive # type: ignore
|
|
77
77
|
if self.max is not None:
|
|
78
78
|
result["max"] = expr <= self.max # type: ignore
|
|
79
|
-
if self.max_exclusive:
|
|
79
|
+
if self.max_exclusive is not None:
|
|
80
80
|
result["max_exclusive"] = expr < self.max_exclusive # type: ignore
|
|
81
81
|
return result
|
|
82
82
|
|
|
@@ -414,26 +414,6 @@ class Schema(BaseSchema, ABC):
|
|
|
414
414
|
def validate(
|
|
415
415
|
cls, df: pl.DataFrame | pl.LazyFrame, /, *, cast: bool = False
|
|
416
416
|
) -> DataFrame[Self]:
|
|
417
|
-
"""Validate that a data frame satisfies the schema.
|
|
418
|
-
|
|
419
|
-
Args:
|
|
420
|
-
df: The data frame to validate.
|
|
421
|
-
cast: Whether columns with a wrong data type in the input data frame are
|
|
422
|
-
cast to the schema's defined data type if possible.
|
|
423
|
-
|
|
424
|
-
Returns:
|
|
425
|
-
The (collected) input data frame, wrapped in a generic version of the
|
|
426
|
-
input's data frame type to reflect schema adherence. The data frame is
|
|
427
|
-
guaranteed to maintain its order.
|
|
428
|
-
|
|
429
|
-
Raises:
|
|
430
|
-
ValidationError: If the input data frame does not satisfy the schema
|
|
431
|
-
definition.
|
|
432
|
-
|
|
433
|
-
Note:
|
|
434
|
-
This method _always_ collects the input data frame in order to raise
|
|
435
|
-
potential validation errors.
|
|
436
|
-
"""
|
|
437
417
|
# We can dispatch to the `filter` method and raise an error if any row cannot
|
|
438
418
|
# be validated
|
|
439
419
|
df_valid, failures = cls.filter(df, cast=cast)
|
|
@@ -1118,11 +1098,6 @@ class Schema(BaseSchema, ABC):
|
|
|
1118
1098
|
|
|
1119
1099
|
@classmethod
|
|
1120
1100
|
def polars_schema(cls) -> pl.Schema:
|
|
1121
|
-
"""Obtain the polars schema for this schema.
|
|
1122
|
-
|
|
1123
|
-
Returns:
|
|
1124
|
-
A :mod:`polars` schema that mirrors the schema defined by this class.
|
|
1125
|
-
"""
|
|
1126
1101
|
return pl.Schema({name: col.dtype for name, col in cls.columns().items()})
|
|
1127
1102
|
|
|
1128
1103
|
@classmethod
|
|
@@ -25,6 +25,14 @@ dataframely.columns.array module
|
|
|
25
25
|
:show-inheritance:
|
|
26
26
|
:undoc-members:
|
|
27
27
|
|
|
28
|
+
dataframely.columns.binary module
|
|
29
|
+
---------------------------------
|
|
30
|
+
|
|
31
|
+
.. automodule:: dataframely.columns.binary
|
|
32
|
+
:members:
|
|
33
|
+
:show-inheritance:
|
|
34
|
+
:undoc-members:
|
|
35
|
+
|
|
28
36
|
dataframely.columns.bool module
|
|
29
37
|
-------------------------------
|
|
30
38
|
|
|
@@ -33,6 +41,14 @@ dataframely.columns.bool module
|
|
|
33
41
|
:show-inheritance:
|
|
34
42
|
:undoc-members:
|
|
35
43
|
|
|
44
|
+
dataframely.columns.categorical module
|
|
45
|
+
--------------------------------------
|
|
46
|
+
|
|
47
|
+
.. automodule:: dataframely.columns.categorical
|
|
48
|
+
:members:
|
|
49
|
+
:show-inheritance:
|
|
50
|
+
:undoc-members:
|
|
51
|
+
|
|
36
52
|
dataframely.columns.datetime module
|
|
37
53
|
-----------------------------------
|
|
38
54
|
|
|
@@ -41,6 +41,14 @@ dataframely.testing.rules module
|
|
|
41
41
|
:show-inheritance:
|
|
42
42
|
:undoc-members:
|
|
43
43
|
|
|
44
|
+
dataframely.testing.storage module
|
|
45
|
+
----------------------------------
|
|
46
|
+
|
|
47
|
+
.. automodule:: dataframely.testing.storage
|
|
48
|
+
:members:
|
|
49
|
+
:show-inheritance:
|
|
50
|
+
:undoc-members:
|
|
51
|
+
|
|
44
52
|
dataframely.testing.typing module
|
|
45
53
|
---------------------------------
|
|
46
54
|
|
|
@@ -22,11 +22,12 @@ Contents
|
|
|
22
22
|
|
|
23
23
|
.. toctree::
|
|
24
24
|
:caption: Contents
|
|
25
|
-
:maxdepth:
|
|
25
|
+
:maxdepth: 2
|
|
26
26
|
|
|
27
27
|
Installation <sites/installation.rst>
|
|
28
28
|
Quickstart <sites/quickstart.rst>
|
|
29
29
|
Real-world Example <sites/examples/real-world.ipynb>
|
|
30
|
+
Features <sites/features/index.rst>
|
|
30
31
|
FAQ <sites/faq.rst>
|
|
31
32
|
Development Guide <sites/development.rst>
|
|
32
33
|
Versioning <sites/versioning.rst>
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
FAQ
|
|
2
|
+
===
|
|
3
|
+
|
|
4
|
+
Whenever you find out something that you were surprised by or needed some non-trivial
|
|
5
|
+
thinking, please add it here.
|
|
6
|
+
|
|
7
|
+
How do I define additional unique keys in a ``dy.Schema``?
|
|
8
|
+
----------------------------------------------------------
|
|
9
|
+
|
|
10
|
+
By default, ``dataframely`` only supports defining a single non-nullable (composite) primary key in ``dy.Schema``.
|
|
11
|
+
However, in some scenarios it may be useful to define additional unique keys (which support nullable fields and/or which are additionally unique).
|
|
12
|
+
|
|
13
|
+
Consider the following example, which demonstrates two rules: one for validating that a field is entirely unique, and another for validating that a field, when provided, is unique.
|
|
14
|
+
|
|
15
|
+
::
|
|
16
|
+
|
|
17
|
+
class UserSchema(dy.Schema):
|
|
18
|
+
user_id = dy.UInt64(primary_key=True, nullable=False)
|
|
19
|
+
username = dy.String(nullable=False)
|
|
20
|
+
email = dy.String(nullable=True) # Must be unique, or null.
|
|
21
|
+
|
|
22
|
+
@dy.rule(group_by=["username"])
|
|
23
|
+
def unique_username() -> pl.Expr:
|
|
24
|
+
"""Username, a non-nullable field, must be total unique."""
|
|
25
|
+
return pl.len() == 1
|
|
26
|
+
|
|
27
|
+
@dy.rule()
|
|
28
|
+
def unique_email_or_null() -> pl.Expr:
|
|
29
|
+
"""Email must be unique, if provided."""
|
|
30
|
+
return pl.col("email").is_null() | pl.col("email").is_unique()
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
Primary keys
|
|
2
|
+
============
|
|
3
|
+
|
|
4
|
+
Defining primary keys in ``dy.Schema``
|
|
5
|
+
--------------------------------------
|
|
6
|
+
|
|
7
|
+
When working with tabular data, it is often useful to define a `primary key <https://en.wikipedia.org/wiki/Primary_key>`_. A primary key is a set of one or multiple column, the combined values of which form a unique identifier for every record in a table.
|
|
8
|
+
|
|
9
|
+
Dataframely supports marking columns as part of the primary key when defining a ``dy.Schema`` by setting ``primary_key=True`` on the respective column(s).
|
|
10
|
+
|
|
11
|
+
.. note::
|
|
12
|
+
|
|
13
|
+
Primary key columns must not be nullable.
|
|
14
|
+
|
|
15
|
+
Single primary keys
|
|
16
|
+
^^^^^^^^^^^^^^^^^^^
|
|
17
|
+
|
|
18
|
+
For example, when managing data about users, we might use an ``id`` column to uniquely identify users:
|
|
19
|
+
|
|
20
|
+
::
|
|
21
|
+
|
|
22
|
+
class UserSchema(dy.Schema):
|
|
23
|
+
name = dy.String(primary_key=True)
|
|
24
|
+
name = dy.String()
|
|
25
|
+
|
|
26
|
+
When we later validate data with this schema, ``dataframely`` checks that the values of the primary key are unique, i.e. there are no two users with the same value of ``id``. Having multiple users with the same ``name`` but different ``id`` but be allowed in this case.
|
|
27
|
+
|
|
28
|
+
Composite primary keys
|
|
29
|
+
^^^^^^^^^^^^^^^^^^^^^^
|
|
30
|
+
|
|
31
|
+
In another scenario, we might be tracking line items on invoices. We have many invoices, and each invoice may contain any number of line items. To uniquely identify a line item, we need to specify the invoice, as well as the line items position within the invoice. To encode this, we set ``primary_key=True`` on both the ``invoice_id`` and ``item_id`` columns:
|
|
32
|
+
|
|
33
|
+
::
|
|
34
|
+
|
|
35
|
+
class LineItemSchema(dy.Schema):
|
|
36
|
+
invoice_id = dy.Int64(primary_key=True)
|
|
37
|
+
item_id = dy.Int64(primary_key=True)
|
|
38
|
+
price = dy.Decimal()
|
|
39
|
+
|
|
40
|
+
Validation will now ensure that all pairs of (``invoice_id``, ``item_id``) are unique.
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
Primary keys in ``dy.Collection``
|
|
44
|
+
---------------------------------
|
|
45
|
+
|
|
46
|
+
The central idea behind ``dy.Collection`` is to unify multiple tables relating to the same set of underlying entities.
|
|
47
|
+
This is useful because it allows us to write ``dy.filter``s that use information from multiple tables to identify whether the underlying entity is valid or not. If any ``dy.filter``s are defined, ``dataframely`` requires the tables in a ``dy.Collection`` to have an overlapping primary key (i.e., there must be at least one column that is a primary key in all tables).
|
|
@@ -253,6 +253,7 @@ Lastly, ``dataframely`` schemas can be used to integrate with external tools:
|
|
|
253
253
|
- ``HouseSchema.create_empty()`` creates an empty ``dy.DataFrame[HouseSchema]`` that can be used for testing
|
|
254
254
|
- ``HouseSchema.sql_schema()`` provides a list of `sqlalchemy <https://www.sqlalchemy.org>`_ columns that can be used to create SQL tables using types and constraints in line with the schema
|
|
255
255
|
- ``HouseSchema.pyarrow_schema()`` provides a `pyarrow <https://arrow.apache.org/docs/python/index.html>`_ schema with appropriate column dtypes and nullability information
|
|
256
|
+
- You can use ``dy.DataFrame[HouseSchema]`` (or the ``LazyFrame`` equivalent) as fields in `pydantic <https://pydantic.dev>`_ models, including support for validation and serialization. Integration with pydantic is unstable.
|
|
256
257
|
|
|
257
258
|
|
|
258
259
|
Outlook
|