dataframely 2.3.1__tar.gz → 2.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/copilot-instructions.md +23 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/build.yml +8 -8
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/ci.yml +7 -7
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/copilot-setup-steps.yml +3 -3
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/nightly.yml +2 -2
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/scorecard.yml +3 -3
- {dataframely-2.3.1 → dataframely-2.5.0}/PKG-INFO +1 -1
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_base_schema.py +13 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_compat.py +4 -1
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/array.py +12 -3
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/binary.py +6 -4
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/decimal.py +5 -1
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/list.py +13 -5
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/string.py +4 -2
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/struct.py +7 -4
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/schema.py +58 -19
- {dataframely-2.3.1 → dataframely-2.5.0}/pixi.lock +7506 -7616
- {dataframely-2.3.1 → dataframely-2.5.0}/pyproject.toml +1 -1
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_sample.py +13 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_decimal.py +56 -1
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_list.py +13 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_sqlalchemy_columns.py +9 -4
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_str.py +17 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_base.py +16 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_sample.py +26 -17
- {dataframely-2.3.1 → dataframely-2.5.0}/.copier-answers.yml +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.envrc +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.gitattributes +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/CODEOWNERS +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/dependabot.yml +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/instructions/tests.instructions.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/release-drafter.yml +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.github/workflows/chore.yml +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.gitignore +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.pre-commit-config.yaml +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.prettierignore +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.prettierrc +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/.readthedocs.yml +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/Cargo.lock +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/Cargo.toml +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/LICENSE +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/README.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/SECURITY.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/__init__.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_deprecation.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_filter.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_match_to_schema.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_native.pyi +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_plugin.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_polars.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_pydantic.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_rule.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_serialization.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/__init__.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/_base.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/_exc.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/constants.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/delta.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_storage/parquet.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/_typing.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/collection/__init__.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/collection/_base.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/collection/collection.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/collection/filter_result.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/__init__.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/_base.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/_mixins.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/_registry.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/_utils.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/any.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/bool.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/categorical.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/datetime.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/enum.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/float.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/integer.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/columns/object.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/config.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/exc.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/filter_result.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/functional.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/py.typed +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/random.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/__init__.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/const.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/factory.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/mask.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/rules.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/dataframely/testing/storage.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docker-compose.yml +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/_static/custom.css +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/_static/favicon.ico +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/_templates/autosummary/class.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/_templates/autosummary/method.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/_templates/classes/column.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/_templates/classes/error.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/_templates/classes/filter_result.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/generation.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/index.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/io.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/metadata.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/operations.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/collection/validation.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/columns/index.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/errors/index.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/filter_result/failure_info.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/filter_result/index.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/index.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/misc/index.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/conversion.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/generation.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/index.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/io.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/metadata.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/api/schema/validation.rst +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/conf.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/css/custom.css +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/development.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/examples/index.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/examples/real-world.ipynb +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/faq.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/column-metadata.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/data-generation.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/index.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/lazy-validation.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/primary-keys.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/serialization.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/features/sql-generation.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/index.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/migration/index.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/migration/v1-v2.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/guides/quickstart.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/docs/index.md +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/pixi.toml +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/rust-toolchain.toml +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/src/lib.rs +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/src/polars_plugin/mod.rs +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/src/polars_plugin/rule_failure.rs +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/src/polars_plugin/utils.rs +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/src/polars_plugin/validation_error.rs +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/src/regex/errdefs.rs +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/src/regex/mod.rs +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/src/regex/repr.rs +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/benches/conftest.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/benches/test_collection.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/benches/test_failure.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/benches/test_schema.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_base.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_cast.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_collection_future_annotations.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_create_empty.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_filter_one_to_n.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_filter_validate.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_ignore_in_filter.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_implementation.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_join.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_matches.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_optional_members.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_propagate_row_failures.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_repr.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_serialization.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_storage.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/collection/test_validate_input.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/__init__.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_any.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_array.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_binary.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_datetime.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_enum.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_float.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_integer.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_object.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_string.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/column_types/test_struct.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/__init__.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_alias.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_base.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_check.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_default_dtypes.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_matches.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_metadata.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_polars_schema.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_pyarrow.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_rules.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_sample.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/columns/test_utils.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/conftest.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/core_validation/__init__.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/core_validation/test_match_to_schema.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/core_validation/test_rule_evaluation.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/failure_info/test_storage.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/functional/test_concat.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/functional/test_relationships.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_cast.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_create_empty.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_create_empty_if_none.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_filter.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_inheritance.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_matches.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_read_write_parquet.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_repr.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_rule_implementation.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_serialization.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_storage.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/schema/test_validate.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/storage/test_delta.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_compat.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_config.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_deprecation.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_factory.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_native_regex.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_pydantic.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_random.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_serialization.py +0 -0
- {dataframely-2.3.1 → dataframely-2.5.0}/tests/test_typing.py +0 -0
|
@@ -206,6 +206,29 @@ validated_df: dy.DataFrame[MySchema] = MySchema.validate(df, cast=True)
|
|
|
206
206
|
4. **Documentation**: Update docstrings
|
|
207
207
|
5. **API changes**: Ensure backward compatibility or document migration path
|
|
208
208
|
|
|
209
|
+
### Pull request titles (required)
|
|
210
|
+
|
|
211
|
+
Pull request titles must follow the Conventional Commits format: `<type>[!]: <Subject>`
|
|
212
|
+
|
|
213
|
+
Allowed `type` values:
|
|
214
|
+
|
|
215
|
+
- `feat`: A new feature
|
|
216
|
+
- `fix`: A bug fix
|
|
217
|
+
- `docs`: Documentation only changes
|
|
218
|
+
- `style`: Changes that do not affect the meaning of the code (white-space, formatting, missing semi-colons, etc)
|
|
219
|
+
- `refactor`: A code change that neither fixes a bug nor adds a feature
|
|
220
|
+
- `perf`: A code change that improves performance
|
|
221
|
+
- `test`: Adding missing tests or correcting existing tests
|
|
222
|
+
- `build`: Changes that affect the build system or external dependencies
|
|
223
|
+
- `ci`: Changes to our CI configuration files and scripts
|
|
224
|
+
- `chore`: Other changes that don't modify src or test files
|
|
225
|
+
- `revert`: Reverts a previous commit
|
|
226
|
+
|
|
227
|
+
Additional rules:
|
|
228
|
+
|
|
229
|
+
- Use `!` only for **breaking changes**
|
|
230
|
+
- `Subject` must start with an **uppercase** letter and must **not** end with `.` or a trailing space
|
|
231
|
+
|
|
209
232
|
## Performance Considerations
|
|
210
233
|
|
|
211
234
|
- Validation uses native polars expressions for performance
|
|
@@ -13,11 +13,11 @@ jobs:
|
|
|
13
13
|
permissions:
|
|
14
14
|
contents: read
|
|
15
15
|
steps:
|
|
16
|
-
- uses: actions/checkout@
|
|
16
|
+
- uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
|
|
17
17
|
with:
|
|
18
18
|
fetch-depth: 0
|
|
19
19
|
- name: Set up pixi
|
|
20
|
-
uses: prefix-dev/setup-pixi@
|
|
20
|
+
uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
|
|
21
21
|
with:
|
|
22
22
|
environments: build
|
|
23
23
|
- name: Set version
|
|
@@ -25,7 +25,7 @@ jobs:
|
|
|
25
25
|
- name: Build project
|
|
26
26
|
run: pixi run -e build build-sdist
|
|
27
27
|
- name: Upload package
|
|
28
|
-
uses: actions/upload-artifact@
|
|
28
|
+
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0
|
|
29
29
|
with:
|
|
30
30
|
name: sdist
|
|
31
31
|
path: dist/*
|
|
@@ -48,16 +48,16 @@ jobs:
|
|
|
48
48
|
- target-platform: win-64
|
|
49
49
|
os: windows-latest
|
|
50
50
|
steps:
|
|
51
|
-
- uses: actions/checkout@
|
|
51
|
+
- uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
|
|
52
52
|
with:
|
|
53
53
|
fetch-depth: 0
|
|
54
54
|
- name: Set up pixi
|
|
55
|
-
uses: prefix-dev/setup-pixi@
|
|
55
|
+
uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
|
|
56
56
|
with:
|
|
57
57
|
environments: build
|
|
58
58
|
- name: Set version
|
|
59
59
|
run: pixi run -e build set-version
|
|
60
|
-
- uses: actions/setup-python@
|
|
60
|
+
- uses: actions/setup-python@83679a892e2d95755f2dac6acb0bfd1e9ac5d548 # v6.1.0
|
|
61
61
|
with:
|
|
62
62
|
python-version: "3.10"
|
|
63
63
|
- name: Build wheel
|
|
@@ -70,7 +70,7 @@ jobs:
|
|
|
70
70
|
- name: Check package
|
|
71
71
|
run: pixi run -e build check-wheel
|
|
72
72
|
- name: Upload package
|
|
73
|
-
uses: actions/upload-artifact@
|
|
73
|
+
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0
|
|
74
74
|
with:
|
|
75
75
|
name: wheel-${{ matrix.target-platform }}
|
|
76
76
|
path: dist/*
|
|
@@ -84,7 +84,7 @@ jobs:
|
|
|
84
84
|
id-token: write
|
|
85
85
|
environment: pypi
|
|
86
86
|
steps:
|
|
87
|
-
- uses: actions/download-artifact@
|
|
87
|
+
- uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7.0.0
|
|
88
88
|
with:
|
|
89
89
|
path: dist
|
|
90
90
|
merge-multiple: true
|
|
@@ -19,18 +19,18 @@ jobs:
|
|
|
19
19
|
runs-on: ubuntu-latest
|
|
20
20
|
steps:
|
|
21
21
|
- name: Checkout branch
|
|
22
|
-
uses: actions/checkout@
|
|
22
|
+
uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
|
|
23
23
|
with:
|
|
24
24
|
# needed for 'pre-commit-mirrors-insert-license'
|
|
25
25
|
fetch-depth: 0
|
|
26
26
|
- name: Set up pixi
|
|
27
|
-
uses: prefix-dev/setup-pixi@
|
|
27
|
+
uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
|
|
28
28
|
with:
|
|
29
29
|
environments: default lint
|
|
30
30
|
- name: Install Rust
|
|
31
31
|
run: rustup show
|
|
32
32
|
- name: Cache Rust dependencies
|
|
33
|
-
uses: Swatinem/rust-cache@
|
|
33
|
+
uses: Swatinem/rust-cache@779680da715d629ac1d338a641029a2f4372abb5 # v2.8.2
|
|
34
34
|
- name: pre-commit
|
|
35
35
|
run: pixi run pre-commit-run --color=always --show-diff-on-failure
|
|
36
36
|
|
|
@@ -56,9 +56,9 @@ jobs:
|
|
|
56
56
|
with_optionals: true
|
|
57
57
|
steps:
|
|
58
58
|
- name: Checkout branch
|
|
59
|
-
uses: actions/checkout@
|
|
59
|
+
uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
|
|
60
60
|
- name: Set up pixi
|
|
61
|
-
uses: prefix-dev/setup-pixi@
|
|
61
|
+
uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
|
|
62
62
|
with:
|
|
63
63
|
environments: ${{ matrix.environment }}
|
|
64
64
|
# FIXME: Remove when `s3_server` fixture does not start a process anymore
|
|
@@ -66,13 +66,13 @@ jobs:
|
|
|
66
66
|
- name: Install Rust
|
|
67
67
|
run: rustup show
|
|
68
68
|
- name: Cache Rust dependencies
|
|
69
|
-
uses: Swatinem/rust-cache@
|
|
69
|
+
uses: Swatinem/rust-cache@779680da715d629ac1d338a641029a2f4372abb5 # v2.8.2
|
|
70
70
|
- name: Install repository
|
|
71
71
|
run: pixi run -e ${{ matrix.environment }} postinstall
|
|
72
72
|
- name: Run pytest
|
|
73
73
|
run: pixi run -e ${{ matrix.environment }} test-coverage --color=yes ${{ matrix.with_optionals && '-m with_optionals' || '-m "not with_optionals"'}} --cov=dataframely --cov-report=xml
|
|
74
74
|
- name: Upload codecov
|
|
75
|
-
uses: codecov/codecov-action@
|
|
75
|
+
uses: codecov/codecov-action@671740ac38dd9b0130fbe1cec585b89eea48d3de # v5.5.2
|
|
76
76
|
with:
|
|
77
77
|
files: ./coverage.xml
|
|
78
78
|
token: ${{ secrets.CODECOV_TOKEN }}
|
|
@@ -13,14 +13,14 @@ jobs:
|
|
|
13
13
|
id-token: write
|
|
14
14
|
steps:
|
|
15
15
|
- name: Checkout branch
|
|
16
|
-
uses: actions/checkout@
|
|
16
|
+
uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
|
|
17
17
|
- name: Set up pixi
|
|
18
|
-
uses: prefix-dev/setup-pixi@
|
|
18
|
+
uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
|
|
19
19
|
with:
|
|
20
20
|
environments: default
|
|
21
21
|
- name: Install Rust
|
|
22
22
|
run: rustup show
|
|
23
23
|
- name: Cache Rust dependencies
|
|
24
|
-
uses: Swatinem/rust-cache@
|
|
24
|
+
uses: Swatinem/rust-cache@779680da715d629ac1d338a641029a2f4372abb5 # v2.8.2
|
|
25
25
|
- name: Install repository
|
|
26
26
|
run: pixi run postinstall
|
|
@@ -23,9 +23,9 @@ jobs:
|
|
|
23
23
|
os: [ubuntu-latest, windows-latest]
|
|
24
24
|
steps:
|
|
25
25
|
- name: Checkout branch
|
|
26
|
-
uses: actions/checkout@
|
|
26
|
+
uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
|
|
27
27
|
- name: Set up pixi
|
|
28
|
-
uses: prefix-dev/setup-pixi@
|
|
28
|
+
uses: prefix-dev/setup-pixi@82d477f15f3a381dbcc8adc1206ce643fe110fb7 # v0.9.3
|
|
29
29
|
with:
|
|
30
30
|
environments: nightly
|
|
31
31
|
- name: Install polars nightly
|
|
@@ -35,7 +35,7 @@ jobs:
|
|
|
35
35
|
|
|
36
36
|
steps:
|
|
37
37
|
- name: "Checkout code"
|
|
38
|
-
uses: actions/checkout@
|
|
38
|
+
uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
|
|
39
39
|
with:
|
|
40
40
|
persist-credentials: false
|
|
41
41
|
|
|
@@ -65,7 +65,7 @@ jobs:
|
|
|
65
65
|
# Upload the results as artifacts (optional). Commenting out will disable uploads of run results in SARIF
|
|
66
66
|
# format to the repository Actions tab.
|
|
67
67
|
- name: "Upload artifact"
|
|
68
|
-
uses: actions/upload-artifact@
|
|
68
|
+
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0
|
|
69
69
|
with:
|
|
70
70
|
name: SARIF file
|
|
71
71
|
path: results.sarif
|
|
@@ -74,6 +74,6 @@ jobs:
|
|
|
74
74
|
# Upload the results to GitHub's code scanning dashboard (optional).
|
|
75
75
|
# Commenting out will disable upload of results to your repo's Code Scanning dashboard
|
|
76
76
|
- name: "Upload to code-scanning"
|
|
77
|
-
uses: github/codeql-action/upload-sarif@
|
|
77
|
+
uses: github/codeql-action/upload-sarif@5d4e8d1aca955e8d8589aabd499c5cae939e33c7 # v3.29.5
|
|
78
78
|
with:
|
|
79
79
|
sarif_file: results.sarif
|
|
@@ -162,6 +162,19 @@ class SchemaMeta(ABCMeta):
|
|
|
162
162
|
f"Did you forget to add parentheses?"
|
|
163
163
|
)
|
|
164
164
|
|
|
165
|
+
# Check for pl.DataType instance or type (e.g., pl.String() or pl.String instead of dy.String())
|
|
166
|
+
if isinstance(value, pl.DataType) or (
|
|
167
|
+
isinstance(value, type) and issubclass(value, pl.DataType)
|
|
168
|
+
):
|
|
169
|
+
value_type = "instance" if isinstance(value, pl.DataType) else "type"
|
|
170
|
+
example = (
|
|
171
|
+
"pl.String()" if isinstance(value, pl.DataType) else "pl.String"
|
|
172
|
+
)
|
|
173
|
+
raise TypeError(
|
|
174
|
+
f"Schema member '{attr}' is a polars DataType {value_type}. "
|
|
175
|
+
f"Use dataframely column types (e.g., dy.String()) instead of polars types (e.g., {example})."
|
|
176
|
+
)
|
|
177
|
+
|
|
165
178
|
return cls
|
|
166
179
|
|
|
167
180
|
if not TYPE_CHECKING:
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
# Copyright (c) QuantCo 2025-
|
|
1
|
+
# Copyright (c) QuantCo 2025-2026
|
|
2
2
|
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
3
|
|
|
4
4
|
|
|
@@ -29,6 +29,7 @@ except ImportError:
|
|
|
29
29
|
try:
|
|
30
30
|
import sqlalchemy as sa
|
|
31
31
|
import sqlalchemy.dialects.mssql as sa_mssql
|
|
32
|
+
import sqlalchemy.dialects.postgresql as sa_postgresql
|
|
32
33
|
from sqlalchemy import Dialect
|
|
33
34
|
from sqlalchemy.dialects.mssql.pyodbc import MSDialect_pyodbc
|
|
34
35
|
from sqlalchemy.dialects.postgresql.psycopg2 import PGDialect_psycopg2
|
|
@@ -36,6 +37,7 @@ try:
|
|
|
36
37
|
except ImportError:
|
|
37
38
|
sa = _DummyModule("sqlalchemy") # type: ignore
|
|
38
39
|
sa_mssql = _DummyModule("sqlalchemy") # type: ignore
|
|
40
|
+
sa_postgresql = _DummyModule("sqlalchemy") # type: ignore
|
|
39
41
|
|
|
40
42
|
class sa_TypeEngine: # type: ignore # noqa: N801
|
|
41
43
|
pass
|
|
@@ -81,6 +83,7 @@ __all__ = [
|
|
|
81
83
|
"pydantic_core_schema",
|
|
82
84
|
"pydantic",
|
|
83
85
|
"sa_mssql",
|
|
86
|
+
"sa_postgresql",
|
|
84
87
|
"sa_TypeEngine",
|
|
85
88
|
"sa",
|
|
86
89
|
]
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
# Copyright (c) QuantCo 2025-
|
|
1
|
+
# Copyright (c) QuantCo 2025-2026
|
|
2
2
|
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
@@ -97,8 +97,17 @@ class Array(Column):
|
|
|
97
97
|
}
|
|
98
98
|
|
|
99
99
|
def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
|
|
100
|
-
|
|
101
|
-
|
|
100
|
+
match dialect.name:
|
|
101
|
+
case "postgresql":
|
|
102
|
+
# Note that the length of the array in each dimension is not supported in SQLAlchemy
|
|
103
|
+
# That is because PostgreSQL does not enforce the length anyway
|
|
104
|
+
return sa.ARRAY(
|
|
105
|
+
self.inner.sqlalchemy_dtype(dialect), dimensions=len(self.shape)
|
|
106
|
+
)
|
|
107
|
+
case _:
|
|
108
|
+
raise NotImplementedError(
|
|
109
|
+
f"SQL column cannot have 'Array' type for dialect '{dialect}'."
|
|
110
|
+
)
|
|
102
111
|
|
|
103
112
|
def _pyarrow_field_of_shape(self, shape: Sequence[int]) -> pa.Field:
|
|
104
113
|
if shape:
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
# Copyright (c) QuantCo 2025-
|
|
1
|
+
# Copyright (c) QuantCo 2025-2026
|
|
2
2
|
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
@@ -21,9 +21,11 @@ class Binary(Column):
|
|
|
21
21
|
return pl.Binary()
|
|
22
22
|
|
|
23
23
|
def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
24
|
+
match dialect.name:
|
|
25
|
+
case "mssql":
|
|
26
|
+
return sa.VARBINARY()
|
|
27
|
+
case _:
|
|
28
|
+
return sa.LargeBinary()
|
|
27
29
|
|
|
28
30
|
@property
|
|
29
31
|
def pyarrow_dtype(self) -> pa.DataType:
|
|
@@ -98,7 +98,11 @@ class Decimal(OrdinalMixin[decimal.Decimal], Column):
|
|
|
98
98
|
return pl.Decimal(self.precision, self.scale)
|
|
99
99
|
|
|
100
100
|
def validate_dtype(self, dtype: PolarsDataType) -> bool:
|
|
101
|
-
return
|
|
101
|
+
return (
|
|
102
|
+
isinstance(dtype, pl.Decimal)
|
|
103
|
+
and dtype.scale == self.scale
|
|
104
|
+
and (self.precision is None or dtype.precision == self.precision)
|
|
105
|
+
)
|
|
102
106
|
|
|
103
107
|
def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
|
|
104
108
|
if self.scale and not self.precision:
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
# Copyright (c) QuantCo 2025-
|
|
1
|
+
# Copyright (c) QuantCo 2025-2026
|
|
2
2
|
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
@@ -120,8 +120,13 @@ class List(Column):
|
|
|
120
120
|
}
|
|
121
121
|
|
|
122
122
|
def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
|
|
123
|
-
|
|
124
|
-
|
|
123
|
+
match dialect.name:
|
|
124
|
+
case "postgresql":
|
|
125
|
+
return sa.ARRAY(self.inner.sqlalchemy_dtype(dialect))
|
|
126
|
+
case _:
|
|
127
|
+
raise NotImplementedError(
|
|
128
|
+
f"SQL column cannot have 'List' type for dialect '{dialect}'."
|
|
129
|
+
)
|
|
125
130
|
|
|
126
131
|
@property
|
|
127
132
|
def pyarrow_dtype(self) -> pa.DataType:
|
|
@@ -131,9 +136,12 @@ class List(Column):
|
|
|
131
136
|
def _sample_unchecked(self, generator: Generator, n: int) -> pl.Series:
|
|
132
137
|
# First, sample the number of items per list element
|
|
133
138
|
# NOTE: We default to 32 for the upper bound as we need some kind of reasonable
|
|
134
|
-
# upper bound if none is set.
|
|
139
|
+
# upper bound if none is set. If min_length is greater than 32, we use
|
|
140
|
+
# min_length as the default upper bound instead.
|
|
141
|
+
min_len = self.min_length or 0
|
|
142
|
+
default_max = max(32, min_len)
|
|
135
143
|
element_lengths = generator.sample_int(
|
|
136
|
-
n, min=
|
|
144
|
+
n, min=min_len, max=(self.max_length or default_max) + 1
|
|
137
145
|
)
|
|
138
146
|
|
|
139
147
|
# Then, we can sample the inner elements in a flat series
|
|
@@ -14,6 +14,8 @@ from dataframely.random import Generator
|
|
|
14
14
|
from ._base import Check, Column
|
|
15
15
|
from ._registry import register
|
|
16
16
|
|
|
17
|
+
DEFAULT_SAMPLING_REGEX = r"[0-9a-zA-Z]"
|
|
18
|
+
|
|
17
19
|
|
|
18
20
|
@register
|
|
19
21
|
class String(Column):
|
|
@@ -126,9 +128,9 @@ class String(Column):
|
|
|
126
128
|
str_max = f"{self.max_length}" if self.max_length is not None else ""
|
|
127
129
|
# NOTE: We generate single-byte unicode characters here as validation uses
|
|
128
130
|
# `len_bytes()`. Potentially we need to be more accurate at some point...
|
|
129
|
-
regex = f"
|
|
131
|
+
regex = f"{DEFAULT_SAMPLING_REGEX}{{{str_min},{str_max}}}"
|
|
130
132
|
else:
|
|
131
|
-
regex =
|
|
133
|
+
regex = rf"{DEFAULT_SAMPLING_REGEX}*"
|
|
132
134
|
|
|
133
135
|
return generator.sample_string(
|
|
134
136
|
n,
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
# Copyright (c) QuantCo 2025-
|
|
1
|
+
# Copyright (c) QuantCo 2025-2026
|
|
2
2
|
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
@@ -8,7 +8,7 @@ from typing import Any, cast
|
|
|
8
8
|
|
|
9
9
|
import polars as pl
|
|
10
10
|
|
|
11
|
-
from dataframely._compat import pa, sa, sa_TypeEngine
|
|
11
|
+
from dataframely._compat import pa, sa, sa_postgresql, sa_TypeEngine
|
|
12
12
|
from dataframely._polars import PolarsDataType
|
|
13
13
|
from dataframely.random import Generator
|
|
14
14
|
|
|
@@ -107,8 +107,11 @@ class Struct(Column):
|
|
|
107
107
|
}
|
|
108
108
|
|
|
109
109
|
def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
|
|
110
|
-
|
|
111
|
-
|
|
110
|
+
match dialect.name:
|
|
111
|
+
case "postgresql":
|
|
112
|
+
return sa_postgresql.JSONB()
|
|
113
|
+
case _:
|
|
114
|
+
raise NotImplementedError("SQL column cannot have 'Struct' type.")
|
|
112
115
|
|
|
113
116
|
@property
|
|
114
117
|
def pyarrow_dtype(self) -> pa.DataType:
|
|
@@ -7,7 +7,7 @@ import json
|
|
|
7
7
|
import sys
|
|
8
8
|
import warnings
|
|
9
9
|
from abc import ABC
|
|
10
|
-
from collections.abc import
|
|
10
|
+
from collections.abc import Mapping, Sequence
|
|
11
11
|
from json import JSONDecodeError
|
|
12
12
|
from pathlib import Path
|
|
13
13
|
from typing import IO, Any, Literal, overload
|
|
@@ -177,7 +177,7 @@ class Schema(BaseSchema, ABC):
|
|
|
177
177
|
num_rows: int | None = None,
|
|
178
178
|
*,
|
|
179
179
|
overrides: (
|
|
180
|
-
Mapping[str,
|
|
180
|
+
Mapping[str, Sequence[Any] | Any] | Sequence[Mapping[str, Any]] | None
|
|
181
181
|
) = None,
|
|
182
182
|
generator: Generator | None = None,
|
|
183
183
|
) -> DataFrame[Self]:
|
|
@@ -234,26 +234,22 @@ class Schema(BaseSchema, ABC):
|
|
|
234
234
|
g = generator or Generator()
|
|
235
235
|
|
|
236
236
|
# Precondition: valid overrides. We put them into a data frame to remember which
|
|
237
|
-
# values have been used in the algorithm below.
|
|
238
|
-
|
|
237
|
+
# values have been used in the algorithm below. When the user passes a sequence
|
|
238
|
+
# of mappings, they do not require to have the same keys. Hence, we have to
|
|
239
|
+
# remember that the data frame has "holes".
|
|
240
|
+
missing_override_indices: dict[str, pl.Series] = {}
|
|
241
|
+
if overrides is not None:
|
|
239
242
|
override_keys = (
|
|
240
|
-
set(overrides)
|
|
243
|
+
set(overrides)
|
|
244
|
+
if isinstance(overrides, Mapping)
|
|
245
|
+
else (
|
|
246
|
+
set.union(*[set(o.keys()) for o in overrides])
|
|
247
|
+
if len(overrides) > 0
|
|
248
|
+
else set()
|
|
249
|
+
)
|
|
241
250
|
)
|
|
242
|
-
if isinstance(overrides, Sequence):
|
|
243
|
-
# Check that overrides entries are consistent. Not necessary for mapping
|
|
244
|
-
# overrides as polars checks the series lists upon data frame construction.
|
|
245
|
-
inconsistent_override_keys = [
|
|
246
|
-
index
|
|
247
|
-
for index, current in enumerate(overrides)
|
|
248
|
-
if set(current) != override_keys
|
|
249
|
-
]
|
|
250
|
-
if len(inconsistent_override_keys) > 0:
|
|
251
|
-
raise ValueError(
|
|
252
|
-
"The `overrides` entries at the following indices "
|
|
253
|
-
"do not provide the same keys as the first entry: "
|
|
254
|
-
f"{inconsistent_override_keys}."
|
|
255
|
-
)
|
|
256
251
|
|
|
252
|
+
# Check that all override keys refer to valid columns
|
|
257
253
|
column_names = set(cls.column_names())
|
|
258
254
|
if not override_keys.issubset(column_names):
|
|
259
255
|
raise ValueError(
|
|
@@ -261,6 +257,19 @@ class Schema(BaseSchema, ABC):
|
|
|
261
257
|
"which are not in the schema."
|
|
262
258
|
)
|
|
263
259
|
|
|
260
|
+
# Remember the "holes" of the inputs if overrides are provided as a sequence
|
|
261
|
+
if isinstance(overrides, Sequence):
|
|
262
|
+
for key in override_keys:
|
|
263
|
+
indices = [
|
|
264
|
+
i for i, override in enumerate(overrides) if key not in override
|
|
265
|
+
]
|
|
266
|
+
if len(indices) > 0:
|
|
267
|
+
missing_override_indices[key] = pl.Series(indices)
|
|
268
|
+
|
|
269
|
+
# NOTE: Even if the user-provided overrides have "holes", we can still just
|
|
270
|
+
# create the data frame. Polars will fill the missing values with nulls, we
|
|
271
|
+
# will replace them later during sampling. If we were to already replace
|
|
272
|
+
# them here, we would not be able to resample these values.
|
|
264
273
|
values = pl.DataFrame(
|
|
265
274
|
overrides,
|
|
266
275
|
schema={
|
|
@@ -323,6 +332,7 @@ class Schema(BaseSchema, ABC):
|
|
|
323
332
|
used_values=values.slice(0, 0),
|
|
324
333
|
remaining_values=values,
|
|
325
334
|
override_expressions=override_expressions,
|
|
335
|
+
missing_value_indices=missing_override_indices,
|
|
326
336
|
)
|
|
327
337
|
|
|
328
338
|
sampling_rounds = 1
|
|
@@ -360,6 +370,7 @@ class Schema(BaseSchema, ABC):
|
|
|
360
370
|
used_values=used_values,
|
|
361
371
|
remaining_values=remaining_values,
|
|
362
372
|
override_expressions=override_expressions,
|
|
373
|
+
missing_value_indices=missing_override_indices,
|
|
363
374
|
)
|
|
364
375
|
sampling_rounds += 1
|
|
365
376
|
|
|
@@ -388,6 +399,7 @@ class Schema(BaseSchema, ABC):
|
|
|
388
399
|
used_values: pl.DataFrame,
|
|
389
400
|
remaining_values: pl.DataFrame,
|
|
390
401
|
override_expressions: list[pl.Expr],
|
|
402
|
+
missing_value_indices: dict[str, pl.Series],
|
|
391
403
|
) -> tuple[pl.DataFrame, pl.DataFrame, pl.DataFrame]:
|
|
392
404
|
"""Private method to sample a data frame with the schema including subsequent
|
|
393
405
|
filtering.
|
|
@@ -406,6 +418,33 @@ class Schema(BaseSchema, ABC):
|
|
|
406
418
|
}
|
|
407
419
|
)
|
|
408
420
|
|
|
421
|
+
# If we have missing value indices, we need to sample new values for the
|
|
422
|
+
# indices that overlap with indices in the remaining values and replace them
|
|
423
|
+
# in the sampled data frame.
|
|
424
|
+
for name, indices in missing_value_indices.items():
|
|
425
|
+
remapped_indices = (
|
|
426
|
+
indices.to_frame("idx")
|
|
427
|
+
.join(
|
|
428
|
+
remaining_values.select("__row_index__").with_row_index(
|
|
429
|
+
"__row_index_loop__"
|
|
430
|
+
),
|
|
431
|
+
left_on="idx",
|
|
432
|
+
right_on="__row_index__",
|
|
433
|
+
)
|
|
434
|
+
.select("__row_index_loop__")
|
|
435
|
+
.to_series()
|
|
436
|
+
)
|
|
437
|
+
if (num := len(remapped_indices)) > 0:
|
|
438
|
+
sampled_values = cls.columns()[name].sample(generator, num)
|
|
439
|
+
sampled = sampled.with_columns(
|
|
440
|
+
sampled[name]
|
|
441
|
+
# NOTE: We need to sort here as `scatter` requires sorted indices.
|
|
442
|
+
# Due to concatenations in `remaining_values`, the indices can go
|
|
443
|
+
# out of order.
|
|
444
|
+
.scatter(remapped_indices.sort(), sampled_values)
|
|
445
|
+
.alias(name)
|
|
446
|
+
)
|
|
447
|
+
|
|
409
448
|
combined_dataframe = pl.concat([previous_result, sampled])
|
|
410
449
|
# Pre-process columns before filtering.
|
|
411
450
|
combined_dataframe = combined_dataframe.with_columns(override_expressions)
|