dataframely 2.3.0__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/copilot-instructions.md +23 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/PKG-INFO +2 -2
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_base_schema.py +13 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/decimal.py +5 -1
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/enum.py +2 -2
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/list.py +5 -2
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/string.py +4 -2
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/filter_result.py +2 -2
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/schema.py +61 -21
- {dataframely-2.3.0 → dataframely-2.4.0}/pixi.lock +137 -146
- {dataframely-2.3.0 → dataframely-2.4.0}/pixi.toml +1 -1
- {dataframely-2.3.0 → dataframely-2.4.0}/pyproject.toml +2 -2
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_sample.py +13 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_decimal.py +56 -1
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_list.py +13 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_str.py +17 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_base.py +16 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_sample.py +26 -17
- {dataframely-2.3.0 → dataframely-2.4.0}/.copier-answers.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.envrc +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.gitattributes +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/CODEOWNERS +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/dependabot.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/instructions/tests.instructions.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/release-drafter.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/workflows/build.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/workflows/chore.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/workflows/ci.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/workflows/copilot-setup-steps.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/workflows/nightly.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.github/workflows/scorecard.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.gitignore +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.pre-commit-config.yaml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.prettierignore +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.prettierrc +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/.readthedocs.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/Cargo.lock +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/Cargo.toml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/LICENSE +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/README.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/SECURITY.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/__init__.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_compat.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_deprecation.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_filter.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_match_to_schema.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_native.pyi +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_plugin.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_polars.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_pydantic.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_rule.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_serialization.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_storage/__init__.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_storage/_base.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_storage/_exc.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_storage/constants.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_storage/delta.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_storage/parquet.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/_typing.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/collection/__init__.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/collection/_base.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/collection/collection.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/collection/filter_result.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/__init__.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/_base.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/_mixins.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/_registry.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/_utils.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/any.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/array.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/binary.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/bool.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/categorical.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/datetime.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/float.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/integer.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/object.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/columns/struct.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/config.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/exc.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/functional.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/py.typed +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/random.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/testing/__init__.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/testing/const.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/testing/factory.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/testing/mask.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/testing/rules.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/dataframely/testing/storage.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docker-compose.yml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/_static/custom.css +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/_static/favicon.ico +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/_templates/autosummary/class.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/_templates/autosummary/method.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/_templates/classes/column.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/_templates/classes/error.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/_templates/classes/filter_result.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/collection/generation.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/collection/index.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/collection/io.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/collection/metadata.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/collection/operations.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/collection/validation.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/columns/index.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/errors/index.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/filter_result/failure_info.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/filter_result/index.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/index.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/misc/index.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/schema/conversion.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/schema/generation.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/schema/index.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/schema/io.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/schema/metadata.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/api/schema/validation.rst +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/conf.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/css/custom.css +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/development.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/examples/index.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/examples/real-world.ipynb +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/faq.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/features/column-metadata.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/features/data-generation.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/features/index.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/features/lazy-validation.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/features/primary-keys.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/features/serialization.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/features/sql-generation.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/index.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/migration/index.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/migration/v1-v2.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/guides/quickstart.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/docs/index.md +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/rust-toolchain.toml +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/src/lib.rs +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/src/polars_plugin/mod.rs +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/src/polars_plugin/rule_failure.rs +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/src/polars_plugin/utils.rs +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/src/polars_plugin/validation_error.rs +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/src/regex/errdefs.rs +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/src/regex/mod.rs +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/src/regex/repr.rs +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/benches/conftest.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/benches/test_collection.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/benches/test_failure.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/benches/test_schema.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_base.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_cast.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_collection_future_annotations.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_create_empty.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_filter_one_to_n.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_filter_validate.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_ignore_in_filter.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_implementation.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_join.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_matches.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_optional_members.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_propagate_row_failures.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_repr.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_serialization.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_storage.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/collection/test_validate_input.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/__init__.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_any.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_array.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_binary.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_datetime.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_enum.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_float.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_integer.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_object.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_string.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/column_types/test_struct.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/__init__.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_alias.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_base.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_check.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_default_dtypes.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_matches.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_metadata.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_polars_schema.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_pyarrow.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_rules.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_sample.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_sqlalchemy_columns.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/columns/test_utils.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/conftest.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/core_validation/__init__.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/core_validation/test_match_to_schema.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/core_validation/test_rule_evaluation.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/failure_info/test_storage.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/functional/test_concat.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/functional/test_relationships.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_cast.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_create_empty.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_create_empty_if_none.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_filter.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_inheritance.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_matches.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_read_write_parquet.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_repr.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_rule_implementation.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_serialization.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_storage.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/schema/test_validate.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/storage/test_delta.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/test_compat.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/test_config.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/test_deprecation.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/test_factory.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/test_native_regex.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/test_pydantic.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/test_random.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/test_serialization.py +0 -0
- {dataframely-2.3.0 → dataframely-2.4.0}/tests/test_typing.py +0 -0
|
@@ -206,6 +206,29 @@ validated_df: dy.DataFrame[MySchema] = MySchema.validate(df, cast=True)
|
|
|
206
206
|
4. **Documentation**: Update docstrings
|
|
207
207
|
5. **API changes**: Ensure backward compatibility or document migration path
|
|
208
208
|
|
|
209
|
+
### Pull request titles (required)
|
|
210
|
+
|
|
211
|
+
Pull request titles must follow the Conventional Commits format: `<type>[!]: <Subject>`
|
|
212
|
+
|
|
213
|
+
Allowed `type` values:
|
|
214
|
+
|
|
215
|
+
- `feat`: A new feature
|
|
216
|
+
- `fix`: A bug fix
|
|
217
|
+
- `docs`: Documentation only changes
|
|
218
|
+
- `style`: Changes that do not affect the meaning of the code (white-space, formatting, missing semi-colons, etc)
|
|
219
|
+
- `refactor`: A code change that neither fixes a bug nor adds a feature
|
|
220
|
+
- `perf`: A code change that improves performance
|
|
221
|
+
- `test`: Adding missing tests or correcting existing tests
|
|
222
|
+
- `build`: Changes that affect the build system or external dependencies
|
|
223
|
+
- `ci`: Changes to our CI configuration files and scripts
|
|
224
|
+
- `chore`: Other changes that don't modify src or test files
|
|
225
|
+
- `revert`: Reverts a previous commit
|
|
226
|
+
|
|
227
|
+
Additional rules:
|
|
228
|
+
|
|
229
|
+
- Use `!` only for **breaking changes**
|
|
230
|
+
- `Subject` must start with an **uppercase** letter and must **not** end with `.` or a trailing space
|
|
231
|
+
|
|
209
232
|
## Performance Considerations
|
|
210
233
|
|
|
211
234
|
- Validation uses native polars expressions for performance
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dataframely
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.4.0
|
|
4
4
|
Classifier: Programming Language :: Python :: 3
|
|
5
5
|
Classifier: Programming Language :: Python :: 3.10
|
|
6
6
|
Classifier: Programming Language :: Python :: 3.11
|
|
@@ -9,7 +9,7 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
9
9
|
Classifier: Programming Language :: Python :: 3.14
|
|
10
10
|
Requires-Dist: fsspec>=2025.9
|
|
11
11
|
Requires-Dist: numpy
|
|
12
|
-
Requires-Dist: polars>=1.
|
|
12
|
+
Requires-Dist: polars>=1.36
|
|
13
13
|
Requires-Dist: typing-extensions ; python_full_version < '3.11'
|
|
14
14
|
Requires-Dist: deltalake ; extra == 'deltalake'
|
|
15
15
|
Requires-Dist: pyarrow ; extra == 'pyarrow'
|
|
@@ -162,6 +162,19 @@ class SchemaMeta(ABCMeta):
|
|
|
162
162
|
f"Did you forget to add parentheses?"
|
|
163
163
|
)
|
|
164
164
|
|
|
165
|
+
# Check for pl.DataType instance or type (e.g., pl.String() or pl.String instead of dy.String())
|
|
166
|
+
if isinstance(value, pl.DataType) or (
|
|
167
|
+
isinstance(value, type) and issubclass(value, pl.DataType)
|
|
168
|
+
):
|
|
169
|
+
value_type = "instance" if isinstance(value, pl.DataType) else "type"
|
|
170
|
+
example = (
|
|
171
|
+
"pl.String()" if isinstance(value, pl.DataType) else "pl.String"
|
|
172
|
+
)
|
|
173
|
+
raise TypeError(
|
|
174
|
+
f"Schema member '{attr}' is a polars DataType {value_type}. "
|
|
175
|
+
f"Use dataframely column types (e.g., dy.String()) instead of polars types (e.g., {example})."
|
|
176
|
+
)
|
|
177
|
+
|
|
165
178
|
return cls
|
|
166
179
|
|
|
167
180
|
if not TYPE_CHECKING:
|
|
@@ -98,7 +98,11 @@ class Decimal(OrdinalMixin[decimal.Decimal], Column):
|
|
|
98
98
|
return pl.Decimal(self.precision, self.scale)
|
|
99
99
|
|
|
100
100
|
def validate_dtype(self, dtype: PolarsDataType) -> bool:
|
|
101
|
-
return
|
|
101
|
+
return (
|
|
102
|
+
isinstance(dtype, pl.Decimal)
|
|
103
|
+
and dtype.scale == self.scale
|
|
104
|
+
and (self.precision is None or dtype.precision == self.precision)
|
|
105
|
+
)
|
|
102
106
|
|
|
103
107
|
def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
|
|
104
108
|
if self.scale and not self.precision:
|
|
@@ -87,9 +87,9 @@ class Enum(Column):
|
|
|
87
87
|
|
|
88
88
|
@property
|
|
89
89
|
def pyarrow_dtype(self) -> pa.DataType:
|
|
90
|
-
if len(self.categories) <= 2**8 -
|
|
90
|
+
if len(self.categories) <= 2**8 - 1:
|
|
91
91
|
dtype = pa.uint8()
|
|
92
|
-
elif len(self.categories) <= 2**16 -
|
|
92
|
+
elif len(self.categories) <= 2**16 - 1:
|
|
93
93
|
dtype = pa.uint16()
|
|
94
94
|
else:
|
|
95
95
|
dtype = pa.uint32()
|
|
@@ -131,9 +131,12 @@ class List(Column):
|
|
|
131
131
|
def _sample_unchecked(self, generator: Generator, n: int) -> pl.Series:
|
|
132
132
|
# First, sample the number of items per list element
|
|
133
133
|
# NOTE: We default to 32 for the upper bound as we need some kind of reasonable
|
|
134
|
-
# upper bound if none is set.
|
|
134
|
+
# upper bound if none is set. If min_length is greater than 32, we use
|
|
135
|
+
# min_length as the default upper bound instead.
|
|
136
|
+
min_len = self.min_length or 0
|
|
137
|
+
default_max = max(32, min_len)
|
|
135
138
|
element_lengths = generator.sample_int(
|
|
136
|
-
n, min=
|
|
139
|
+
n, min=min_len, max=(self.max_length or default_max) + 1
|
|
137
140
|
)
|
|
138
141
|
|
|
139
142
|
# Then, we can sample the inner elements in a flat series
|
|
@@ -14,6 +14,8 @@ from dataframely.random import Generator
|
|
|
14
14
|
from ._base import Check, Column
|
|
15
15
|
from ._registry import register
|
|
16
16
|
|
|
17
|
+
DEFAULT_SAMPLING_REGEX = r"[0-9a-zA-Z]"
|
|
18
|
+
|
|
17
19
|
|
|
18
20
|
@register
|
|
19
21
|
class String(Column):
|
|
@@ -126,9 +128,9 @@ class String(Column):
|
|
|
126
128
|
str_max = f"{self.max_length}" if self.max_length is not None else ""
|
|
127
129
|
# NOTE: We generate single-byte unicode characters here as validation uses
|
|
128
130
|
# `len_bytes()`. Potentially we need to be more accurate at some point...
|
|
129
|
-
regex = f"
|
|
131
|
+
regex = f"{DEFAULT_SAMPLING_REGEX}{{{str_min},{str_max}}}"
|
|
130
132
|
else:
|
|
131
|
-
regex =
|
|
133
|
+
regex = rf"{DEFAULT_SAMPLING_REGEX}*"
|
|
132
134
|
|
|
133
135
|
return generator.sample_string(
|
|
134
136
|
n,
|
|
@@ -10,7 +10,7 @@ from pathlib import Path
|
|
|
10
10
|
from typing import IO, TYPE_CHECKING, Any, Generic, TypeVar
|
|
11
11
|
|
|
12
12
|
import polars as pl
|
|
13
|
-
from polars.
|
|
13
|
+
from polars.io.partition import _SinkDirectory as SinkDirectory
|
|
14
14
|
|
|
15
15
|
from dataframely._base_schema import BaseSchema
|
|
16
16
|
from dataframely._compat import deltalake
|
|
@@ -164,7 +164,7 @@ class FailureInfo(Generic[S]):
|
|
|
164
164
|
self._write(ParquetStorageBackend(), file=file, **kwargs)
|
|
165
165
|
|
|
166
166
|
def sink_parquet(
|
|
167
|
-
self, file: str | Path | IO[bytes] |
|
|
167
|
+
self, file: str | Path | IO[bytes] | SinkDirectory, **kwargs: Any
|
|
168
168
|
) -> None:
|
|
169
169
|
"""Stream the failure info to a single parquet file.
|
|
170
170
|
|
|
@@ -7,14 +7,15 @@ import json
|
|
|
7
7
|
import sys
|
|
8
8
|
import warnings
|
|
9
9
|
from abc import ABC
|
|
10
|
-
from collections.abc import
|
|
10
|
+
from collections.abc import Mapping, Sequence
|
|
11
11
|
from json import JSONDecodeError
|
|
12
12
|
from pathlib import Path
|
|
13
13
|
from typing import IO, Any, Literal, overload
|
|
14
14
|
|
|
15
15
|
import polars as pl
|
|
16
16
|
import polars.exceptions as plexc
|
|
17
|
-
from polars._typing import FileSource
|
|
17
|
+
from polars._typing import FileSource
|
|
18
|
+
from polars.io.partition import _SinkDirectory as SinkDirectory
|
|
18
19
|
|
|
19
20
|
from dataframely._compat import deltalake
|
|
20
21
|
|
|
@@ -176,7 +177,7 @@ class Schema(BaseSchema, ABC):
|
|
|
176
177
|
num_rows: int | None = None,
|
|
177
178
|
*,
|
|
178
179
|
overrides: (
|
|
179
|
-
Mapping[str,
|
|
180
|
+
Mapping[str, Sequence[Any] | Any] | Sequence[Mapping[str, Any]] | None
|
|
180
181
|
) = None,
|
|
181
182
|
generator: Generator | None = None,
|
|
182
183
|
) -> DataFrame[Self]:
|
|
@@ -233,26 +234,22 @@ class Schema(BaseSchema, ABC):
|
|
|
233
234
|
g = generator or Generator()
|
|
234
235
|
|
|
235
236
|
# Precondition: valid overrides. We put them into a data frame to remember which
|
|
236
|
-
# values have been used in the algorithm below.
|
|
237
|
-
|
|
237
|
+
# values have been used in the algorithm below. When the user passes a sequence
|
|
238
|
+
# of mappings, they do not require to have the same keys. Hence, we have to
|
|
239
|
+
# remember that the data frame has "holes".
|
|
240
|
+
missing_override_indices: dict[str, pl.Series] = {}
|
|
241
|
+
if overrides is not None:
|
|
238
242
|
override_keys = (
|
|
239
|
-
set(overrides)
|
|
243
|
+
set(overrides)
|
|
244
|
+
if isinstance(overrides, Mapping)
|
|
245
|
+
else (
|
|
246
|
+
set.union(*[set(o.keys()) for o in overrides])
|
|
247
|
+
if len(overrides) > 0
|
|
248
|
+
else set()
|
|
249
|
+
)
|
|
240
250
|
)
|
|
241
|
-
if isinstance(overrides, Sequence):
|
|
242
|
-
# Check that overrides entries are consistent. Not necessary for mapping
|
|
243
|
-
# overrides as polars checks the series lists upon data frame construction.
|
|
244
|
-
inconsistent_override_keys = [
|
|
245
|
-
index
|
|
246
|
-
for index, current in enumerate(overrides)
|
|
247
|
-
if set(current) != override_keys
|
|
248
|
-
]
|
|
249
|
-
if len(inconsistent_override_keys) > 0:
|
|
250
|
-
raise ValueError(
|
|
251
|
-
"The `overrides` entries at the following indices "
|
|
252
|
-
"do not provide the same keys as the first entry: "
|
|
253
|
-
f"{inconsistent_override_keys}."
|
|
254
|
-
)
|
|
255
251
|
|
|
252
|
+
# Check that all override keys refer to valid columns
|
|
256
253
|
column_names = set(cls.column_names())
|
|
257
254
|
if not override_keys.issubset(column_names):
|
|
258
255
|
raise ValueError(
|
|
@@ -260,6 +257,19 @@ class Schema(BaseSchema, ABC):
|
|
|
260
257
|
"which are not in the schema."
|
|
261
258
|
)
|
|
262
259
|
|
|
260
|
+
# Remember the "holes" of the inputs if overrides are provided as a sequence
|
|
261
|
+
if isinstance(overrides, Sequence):
|
|
262
|
+
for key in override_keys:
|
|
263
|
+
indices = [
|
|
264
|
+
i for i, override in enumerate(overrides) if key not in override
|
|
265
|
+
]
|
|
266
|
+
if len(indices) > 0:
|
|
267
|
+
missing_override_indices[key] = pl.Series(indices)
|
|
268
|
+
|
|
269
|
+
# NOTE: Even if the user-provided overrides have "holes", we can still just
|
|
270
|
+
# create the data frame. Polars will fill the missing values with nulls, we
|
|
271
|
+
# will replace them later during sampling. If we were to already replace
|
|
272
|
+
# them here, we would not be able to resample these values.
|
|
263
273
|
values = pl.DataFrame(
|
|
264
274
|
overrides,
|
|
265
275
|
schema={
|
|
@@ -322,6 +332,7 @@ class Schema(BaseSchema, ABC):
|
|
|
322
332
|
used_values=values.slice(0, 0),
|
|
323
333
|
remaining_values=values,
|
|
324
334
|
override_expressions=override_expressions,
|
|
335
|
+
missing_value_indices=missing_override_indices,
|
|
325
336
|
)
|
|
326
337
|
|
|
327
338
|
sampling_rounds = 1
|
|
@@ -359,6 +370,7 @@ class Schema(BaseSchema, ABC):
|
|
|
359
370
|
used_values=used_values,
|
|
360
371
|
remaining_values=remaining_values,
|
|
361
372
|
override_expressions=override_expressions,
|
|
373
|
+
missing_value_indices=missing_override_indices,
|
|
362
374
|
)
|
|
363
375
|
sampling_rounds += 1
|
|
364
376
|
|
|
@@ -387,6 +399,7 @@ class Schema(BaseSchema, ABC):
|
|
|
387
399
|
used_values: pl.DataFrame,
|
|
388
400
|
remaining_values: pl.DataFrame,
|
|
389
401
|
override_expressions: list[pl.Expr],
|
|
402
|
+
missing_value_indices: dict[str, pl.Series],
|
|
390
403
|
) -> tuple[pl.DataFrame, pl.DataFrame, pl.DataFrame]:
|
|
391
404
|
"""Private method to sample a data frame with the schema including subsequent
|
|
392
405
|
filtering.
|
|
@@ -405,6 +418,33 @@ class Schema(BaseSchema, ABC):
|
|
|
405
418
|
}
|
|
406
419
|
)
|
|
407
420
|
|
|
421
|
+
# If we have missing value indices, we need to sample new values for the
|
|
422
|
+
# indices that overlap with indices in the remaining values and replace them
|
|
423
|
+
# in the sampled data frame.
|
|
424
|
+
for name, indices in missing_value_indices.items():
|
|
425
|
+
remapped_indices = (
|
|
426
|
+
indices.to_frame("idx")
|
|
427
|
+
.join(
|
|
428
|
+
remaining_values.select("__row_index__").with_row_index(
|
|
429
|
+
"__row_index_loop__"
|
|
430
|
+
),
|
|
431
|
+
left_on="idx",
|
|
432
|
+
right_on="__row_index__",
|
|
433
|
+
)
|
|
434
|
+
.select("__row_index_loop__")
|
|
435
|
+
.to_series()
|
|
436
|
+
)
|
|
437
|
+
if (num := len(remapped_indices)) > 0:
|
|
438
|
+
sampled_values = cls.columns()[name].sample(generator, num)
|
|
439
|
+
sampled = sampled.with_columns(
|
|
440
|
+
sampled[name]
|
|
441
|
+
# NOTE: We need to sort here as `scatter` requires sorted indices.
|
|
442
|
+
# Due to concatenations in `remaining_values`, the indices can go
|
|
443
|
+
# out of order.
|
|
444
|
+
.scatter(remapped_indices.sort(), sampled_values)
|
|
445
|
+
.alias(name)
|
|
446
|
+
)
|
|
447
|
+
|
|
408
448
|
combined_dataframe = pl.concat([previous_result, sampled])
|
|
409
449
|
# Pre-process columns before filtering.
|
|
410
450
|
combined_dataframe = combined_dataframe.with_columns(override_expressions)
|
|
@@ -861,7 +901,7 @@ class Schema(BaseSchema, ABC):
|
|
|
861
901
|
cls,
|
|
862
902
|
lf: LazyFrame[Self],
|
|
863
903
|
/,
|
|
864
|
-
file: str | Path | IO[bytes] |
|
|
904
|
+
file: str | Path | IO[bytes] | SinkDirectory,
|
|
865
905
|
**kwargs: Any,
|
|
866
906
|
) -> None:
|
|
867
907
|
"""Stream a typed lazy frame with this schema to a parquet file.
|