dataframely 1.10.0__tar.gz → 1.12.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataframely-1.10.0 → dataframely-1.12.0}/PKG-INFO +1 -1
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/__init__.py +2 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/__init__.py +2 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/_base.py +23 -7
- dataframely-1.12.0/dataframely/columns/binary.py +38 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/enum.py +17 -6
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/random.py +25 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/testing/const.py +1 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/pyproject.toml +1 -1
- dataframely-1.12.0/tests/column_types/test_binary.py +24 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_enum.py +48 -1
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_check.py +16 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_matches.py +12 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_sql_schema.py +2 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/test_random.py +14 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.copier-answers.yml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.envrc +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.gitattributes +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.github/CODEOWNERS +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.github/dependabot.yml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.github/release-drafter.yml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.github/workflows/build.yml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.github/workflows/chore.yml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.github/workflows/ci.yml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.github/workflows/nightly.yml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.github/workflows/scorecard.yml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.gitignore +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.pre-commit-config.yaml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.prettierignore +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.prettierrc +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/.readthedocs.yml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/Cargo.lock +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/Cargo.toml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/LICENSE +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/README.md +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/SECURITY.md +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_base_collection.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_base_schema.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_compat.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_deprecation.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_extre.pyi +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_filter.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_polars.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_rule.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_serialization.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_storage/__init__.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_storage/_base.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_storage/_exc.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_storage/constants.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_storage/delta.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_storage/parquet.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_typing.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/_validation.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/collection.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/_mixins.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/_registry.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/_utils.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/any.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/array.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/bool.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/categorical.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/datetime.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/decimal.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/float.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/integer.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/list.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/object.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/string.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/columns/struct.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/config.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/exc.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/failure.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/functional.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/mypy.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/py.typed +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/schema.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/testing/__init__.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/testing/factory.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/testing/mask.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/testing/rules.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/testing/storage.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/dataframely/testing/typing.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docker-compose.yml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/Makefile +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.collection.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.any.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.bool.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.datetime.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.decimal.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.enum.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.float.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.integer.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.list.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.string.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.columns.struct.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.config.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.exc.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.failure.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.functional.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.mypy.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.random.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.schema.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.testing.const.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.testing.factory.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.testing.mask.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.testing.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.testing.rules.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/dataframely.testing.typing.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_api/modules.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_static/custom.css +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/_static/favicon.ico +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/conf.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/index.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/make.bat +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/sites/development.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/sites/examples/real-world.ipynb +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/sites/faq.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/sites/installation.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/sites/quickstart.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/docs/sites/versioning.rst +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/pixi.lock +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/pixi.toml +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/src/errdefs.rs +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/src/lib.rs +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/src/regex_repr.rs +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/benches/conftest.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/benches/test_collection.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/benches/test_failure.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/benches/test_schema.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_base.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_cast.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_collection_future_annotations.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_create_empty.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_filter_one_to_n.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_filter_validate.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_ignore_in_filter.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_implementation.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_join.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_matches.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_optional_members.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_repr.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_sample.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_serialization.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_storage.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_validate_input.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/__init__.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_any.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_array.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_datetime.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_decimal.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_float.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_integer.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_list.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_object.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_string.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/column_types/test_struct.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/__init__.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_alias.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_default_dtypes.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_metadata.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_polars_schema.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_pyarrow.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_rules.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_sample.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_str.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/columns/test_utils.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/core_validation/__init__.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/core_validation/test_column_validation.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/core_validation/test_dtype_validation.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/core_validation/test_rule_evaluation.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/failure_info/test_storage.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/functional/test_concat.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/functional/test_relationships.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_base.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_cast.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_create_empty.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_create_empty_if_none.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_filter.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_inheritance.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_matches.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_read_write_parquet.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_repr.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_rule_implementation.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_sample.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_serialization.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_storage.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/schema/test_validate.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/storage/test_delta.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/test_compat.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/test_config.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/test_deprecation.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/test_exc.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/test_extre.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/test_factory.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/test_serialization.py +0 -0
- {dataframely-1.10.0 → dataframely-1.12.0}/tests/test_typing.py +0 -0
|
@@ -23,6 +23,7 @@ from .collection import (
|
|
|
23
23
|
from .columns import (
|
|
24
24
|
Any,
|
|
25
25
|
Array,
|
|
26
|
+
Binary,
|
|
26
27
|
Bool,
|
|
27
28
|
Categorical,
|
|
28
29
|
Column,
|
|
@@ -77,6 +78,7 @@ __all__ = [
|
|
|
77
78
|
"read_parquet_metadata_schema",
|
|
78
79
|
"read_parquet_metadata_collection",
|
|
79
80
|
"Any",
|
|
81
|
+
"Binary",
|
|
80
82
|
"Bool",
|
|
81
83
|
"Categorical",
|
|
82
84
|
"Column",
|
|
@@ -5,6 +5,7 @@ from ._base import Column
|
|
|
5
5
|
from ._registry import column_from_dict
|
|
6
6
|
from .any import Any
|
|
7
7
|
from .array import Array
|
|
8
|
+
from .binary import Binary
|
|
8
9
|
from .bool import Bool
|
|
9
10
|
from .categorical import Categorical
|
|
10
11
|
from .datetime import Date, Datetime, Duration, Time
|
|
@@ -22,6 +23,7 @@ __all__ = [
|
|
|
22
23
|
"column_from_dict",
|
|
23
24
|
"Any",
|
|
24
25
|
"Array",
|
|
26
|
+
"Binary",
|
|
25
27
|
"Bool",
|
|
26
28
|
"Categorical",
|
|
27
29
|
"Date",
|
|
@@ -7,7 +7,7 @@ import inspect
|
|
|
7
7
|
import sys
|
|
8
8
|
from abc import ABC, abstractmethod
|
|
9
9
|
from collections import Counter
|
|
10
|
-
from collections.abc import Callable
|
|
10
|
+
from collections.abc import Callable, Mapping, Sequence
|
|
11
11
|
from typing import Any, TypeAlias, cast
|
|
12
12
|
|
|
13
13
|
import polars as pl
|
|
@@ -27,8 +27,8 @@ else:
|
|
|
27
27
|
|
|
28
28
|
Check: TypeAlias = (
|
|
29
29
|
Callable[[pl.Expr], pl.Expr]
|
|
30
|
-
|
|
|
31
|
-
|
|
|
30
|
+
| Sequence[Callable[[pl.Expr], pl.Expr]]
|
|
31
|
+
| Mapping[str, Callable[[pl.Expr], pl.Expr]]
|
|
32
32
|
)
|
|
33
33
|
|
|
34
34
|
# ------------------------------------------------------------------------------------ #
|
|
@@ -141,12 +141,14 @@ class Column(ABC):
|
|
|
141
141
|
result["nullability"] = expr.is_not_null()
|
|
142
142
|
|
|
143
143
|
if self.check is not None:
|
|
144
|
-
if isinstance(self.check,
|
|
144
|
+
if isinstance(self.check, Mapping):
|
|
145
145
|
for rule_name, rule_callable in self.check.items():
|
|
146
146
|
result[f"check__{rule_name}"] = rule_callable(expr)
|
|
147
147
|
else:
|
|
148
148
|
list_of_rules = (
|
|
149
|
-
|
|
149
|
+
list(self.check)
|
|
150
|
+
if isinstance(self.check, Sequence)
|
|
151
|
+
else [self.check]
|
|
150
152
|
)
|
|
151
153
|
# Get unique names for rules from callables
|
|
152
154
|
rule_names = self._derive_check_rule_names(list_of_rules)
|
|
@@ -378,6 +380,16 @@ class Column(ABC):
|
|
|
378
380
|
) -> bool:
|
|
379
381
|
if name == "check":
|
|
380
382
|
return _compare_checks(lhs, rhs, column_expr)
|
|
383
|
+
|
|
384
|
+
lhs_is_series = isinstance(lhs, pl.Series)
|
|
385
|
+
rhs_is_series = isinstance(rhs, pl.Series)
|
|
386
|
+
|
|
387
|
+
if lhs_is_series != rhs_is_series:
|
|
388
|
+
return False
|
|
389
|
+
|
|
390
|
+
if lhs_is_series and rhs_is_series:
|
|
391
|
+
return _compare_series(lhs, rhs)
|
|
392
|
+
|
|
381
393
|
return lhs == rhs
|
|
382
394
|
|
|
383
395
|
# -------------------------------- DUNDER METHODS -------------------------------- #
|
|
@@ -401,6 +413,10 @@ class Column(ABC):
|
|
|
401
413
|
return self.__class__.__name__.lower()
|
|
402
414
|
|
|
403
415
|
|
|
416
|
+
def _compare_series(lhs: pl.Series, rhs: pl.Series) -> bool:
|
|
417
|
+
return (len(lhs) == len(rhs)) and lhs.equals(rhs)
|
|
418
|
+
|
|
419
|
+
|
|
404
420
|
def _compare_checks(lhs: Check | None, rhs: Check | None, expr: pl.Expr) -> bool:
|
|
405
421
|
match (lhs, rhs):
|
|
406
422
|
case (None, None):
|
|
@@ -423,9 +439,9 @@ def _check_to_expr(check: Check | None, expr: pl.Expr) -> Any | None:
|
|
|
423
439
|
match check:
|
|
424
440
|
case None:
|
|
425
441
|
return None
|
|
426
|
-
case
|
|
442
|
+
case Sequence():
|
|
427
443
|
return [c(expr) for c in check]
|
|
428
|
-
case
|
|
444
|
+
case Mapping():
|
|
429
445
|
return {key: c(expr) for key, c in check.items()}
|
|
430
446
|
case _ if callable(check):
|
|
431
447
|
return check(expr)
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# Copyright (c) QuantCo 2025-2025
|
|
2
|
+
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import polars as pl
|
|
7
|
+
|
|
8
|
+
from dataframely._compat import pa, sa, sa_TypeEngine
|
|
9
|
+
from dataframely.random import Generator
|
|
10
|
+
|
|
11
|
+
from ._base import Column
|
|
12
|
+
from ._registry import register
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@register
|
|
16
|
+
class Binary(Column):
|
|
17
|
+
"""A column of binary values."""
|
|
18
|
+
|
|
19
|
+
@property
|
|
20
|
+
def dtype(self) -> pl.DataType:
|
|
21
|
+
return pl.Binary()
|
|
22
|
+
|
|
23
|
+
def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
|
|
24
|
+
if dialect.name == "mssql":
|
|
25
|
+
return sa.VARBINARY()
|
|
26
|
+
return sa.LargeBinary()
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def pyarrow_dtype(self) -> pa.DataType:
|
|
30
|
+
return pa.large_binary()
|
|
31
|
+
|
|
32
|
+
def _sample_unchecked(self, generator: Generator, n: int) -> pl.Series:
|
|
33
|
+
return generator.sample_binary(
|
|
34
|
+
n,
|
|
35
|
+
min_bytes=0,
|
|
36
|
+
max_bytes=32,
|
|
37
|
+
null_probability=self._null_probability,
|
|
38
|
+
)
|
|
@@ -3,7 +3,9 @@
|
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
|
-
|
|
6
|
+
import enum
|
|
7
|
+
from collections.abc import Iterable
|
|
8
|
+
from inspect import isclass
|
|
7
9
|
from typing import Any
|
|
8
10
|
|
|
9
11
|
import polars as pl
|
|
@@ -22,7 +24,7 @@ class Enum(Column):
|
|
|
22
24
|
|
|
23
25
|
def __init__(
|
|
24
26
|
self,
|
|
25
|
-
categories:
|
|
27
|
+
categories: pl.Series | Iterable[str] | type[enum.Enum],
|
|
26
28
|
*,
|
|
27
29
|
nullable: bool | None = None,
|
|
28
30
|
primary_key: bool = False,
|
|
@@ -32,7 +34,8 @@ class Enum(Column):
|
|
|
32
34
|
):
|
|
33
35
|
"""
|
|
34
36
|
Args:
|
|
35
|
-
categories: The
|
|
37
|
+
categories: The set of valid categories for the enum, or an existing Python
|
|
38
|
+
string-valued enum.
|
|
36
39
|
nullable: Whether this column may contain null values.
|
|
37
40
|
Explicitly set `nullable=True` if you want your column to be nullable.
|
|
38
41
|
In a future release, `nullable=False` will be the default if `nullable`
|
|
@@ -63,7 +66,13 @@ class Enum(Column):
|
|
|
63
66
|
alias=alias,
|
|
64
67
|
metadata=metadata,
|
|
65
68
|
)
|
|
66
|
-
|
|
69
|
+
if isclass(categories) and issubclass(categories, enum.Enum):
|
|
70
|
+
categories = pl.Series(
|
|
71
|
+
values=[getattr(v, "value", v) for v in categories.__members__.values()]
|
|
72
|
+
)
|
|
73
|
+
elif not isinstance(categories, pl.Series):
|
|
74
|
+
categories = pl.Series(values=categories)
|
|
75
|
+
self.categories = categories
|
|
67
76
|
|
|
68
77
|
@property
|
|
69
78
|
def dtype(self) -> pl.DataType:
|
|
@@ -72,7 +81,7 @@ class Enum(Column):
|
|
|
72
81
|
def validate_dtype(self, dtype: PolarsDataType) -> bool:
|
|
73
82
|
if not isinstance(dtype, pl.Enum):
|
|
74
83
|
return False
|
|
75
|
-
return self.categories
|
|
84
|
+
return self.categories.equals(dtype.categories)
|
|
76
85
|
|
|
77
86
|
def sqlalchemy_dtype(self, dialect: sa.Dialect) -> sa_TypeEngine:
|
|
78
87
|
category_lengths = [len(c) for c in self.categories]
|
|
@@ -92,5 +101,7 @@ class Enum(Column):
|
|
|
92
101
|
|
|
93
102
|
def _sample_unchecked(self, generator: Generator, n: int) -> pl.Series:
|
|
94
103
|
return generator.sample_choice(
|
|
95
|
-
n,
|
|
104
|
+
n,
|
|
105
|
+
choices=self.categories.to_list(),
|
|
106
|
+
null_probability=self._null_probability,
|
|
96
107
|
).cast(self.dtype)
|
|
@@ -148,6 +148,31 @@ class Generator:
|
|
|
148
148
|
pl.Series(samples, dtype=pl.String), null_probability
|
|
149
149
|
)
|
|
150
150
|
|
|
151
|
+
def sample_binary(
|
|
152
|
+
self,
|
|
153
|
+
n: int = 1,
|
|
154
|
+
*,
|
|
155
|
+
min_bytes: int,
|
|
156
|
+
max_bytes: int,
|
|
157
|
+
null_probability: float = 0.0,
|
|
158
|
+
) -> pl.Series:
|
|
159
|
+
"""Sample a list of binary values in the specified length range.
|
|
160
|
+
|
|
161
|
+
Args:
|
|
162
|
+
n: The number of binary values to sample.
|
|
163
|
+
min_bytes: The minimum number of bytes for each value.
|
|
164
|
+
max_bytes: The maximum number of bytes for each value.
|
|
165
|
+
null_probability: The probability of an element being ``null``.
|
|
166
|
+
|
|
167
|
+
Returns:
|
|
168
|
+
A series with ``n`` elements of dtype ``Binary``.
|
|
169
|
+
"""
|
|
170
|
+
lengths = self.numpy_generator.integers(min_bytes, max_bytes + 1, size=n)
|
|
171
|
+
samples = [self.numpy_generator.bytes(length) for length in lengths]
|
|
172
|
+
return self._apply_null_mask(
|
|
173
|
+
pl.Series(samples, dtype=pl.Binary), null_probability
|
|
174
|
+
)
|
|
175
|
+
|
|
151
176
|
def sample_choice(
|
|
152
177
|
self,
|
|
153
178
|
n: int = 1,
|
|
@@ -25,7 +25,7 @@ description = "A declarative, polars-native data frame validation library"
|
|
|
25
25
|
name = "dataframely"
|
|
26
26
|
readme = "README.md"
|
|
27
27
|
requires-python = ">=3.10"
|
|
28
|
-
version = "1.
|
|
28
|
+
version = "1.12.0"
|
|
29
29
|
|
|
30
30
|
[project.optional-dependencies]
|
|
31
31
|
deltalake = ["deltalake"]
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Copyright (c) QuantCo 2025-2025
|
|
2
|
+
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
|
+
|
|
4
|
+
import polars as pl
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
import dataframely as dy
|
|
8
|
+
from dataframely.columns import Column
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class BinarySchema(dy.Schema):
|
|
12
|
+
a = dy.Binary()
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@pytest.mark.parametrize(
|
|
16
|
+
("column", "dtype", "is_valid"),
|
|
17
|
+
[
|
|
18
|
+
(dy.Binary(), pl.Binary(), True),
|
|
19
|
+
(dy.Binary(), pl.String(), False),
|
|
20
|
+
(dy.Binary(), pl.Null(), False),
|
|
21
|
+
],
|
|
22
|
+
)
|
|
23
|
+
def test_validate_dtype(column: Column, dtype: pl.DataType, is_valid: bool) -> None:
|
|
24
|
+
assert column.validate_dtype(dtype) == is_valid
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
# Copyright (c) QuantCo 2025-2025
|
|
2
2
|
# SPDX-License-Identifier: BSD-3-Clause
|
|
3
|
-
|
|
3
|
+
import enum
|
|
4
|
+
from collections.abc import Iterable
|
|
5
|
+
from enum import Enum
|
|
4
6
|
from typing import Any
|
|
5
7
|
|
|
6
8
|
import polars as pl
|
|
@@ -61,3 +63,48 @@ def test_different_sequences(type1: type, type2: type) -> None:
|
|
|
61
63
|
S = create_schema("test", {"x": dy.Enum(type1(allowed))})
|
|
62
64
|
df = pl.DataFrame({"x": pl.Series(["a", "b"], dtype=pl.Enum(type2(allowed)))})
|
|
63
65
|
S.validate(df)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_enum_of_enum_136() -> None:
|
|
69
|
+
class Categories(str, Enum):
|
|
70
|
+
a = "a"
|
|
71
|
+
b = "b"
|
|
72
|
+
|
|
73
|
+
assert pl.Enum(Categories) == dy.Enum(Categories).dtype
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def test_enum_of_series() -> None:
|
|
77
|
+
categories = pl.Series(["a", "b"])
|
|
78
|
+
assert pl.Enum(categories) == dy.Enum(categories).dtype
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def test_enum_of_iterable() -> None:
|
|
82
|
+
categories = (x for x in ["a", "b"])
|
|
83
|
+
assert pl.Enum(["a", "b"]) == dy.Enum(categories).dtype
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@pytest.mark.parametrize(
|
|
87
|
+
"categories1",
|
|
88
|
+
[
|
|
89
|
+
["a", "b"],
|
|
90
|
+
("a", "b"),
|
|
91
|
+
pl.Series(["a", "b"]),
|
|
92
|
+
Enum("Categories", {"a": "a", "b": "b"}),
|
|
93
|
+
],
|
|
94
|
+
)
|
|
95
|
+
@pytest.mark.parametrize(
|
|
96
|
+
"categories2",
|
|
97
|
+
[
|
|
98
|
+
["a", "b"],
|
|
99
|
+
("a", "b"),
|
|
100
|
+
pl.Series(["a", "b"]),
|
|
101
|
+
Enum("Categories", {"a": "a", "b": "b"}),
|
|
102
|
+
],
|
|
103
|
+
)
|
|
104
|
+
def test_sequences_and_enums(
|
|
105
|
+
categories1: pl.Series | Iterable[str] | type[enum.Enum],
|
|
106
|
+
categories2: pl.Series | Iterable[str] | type[enum.Enum],
|
|
107
|
+
) -> None:
|
|
108
|
+
S = create_schema("test", {"x": dy.Enum(categories1)})
|
|
109
|
+
df = pl.DataFrame({"x": pl.Series(["a", "b"], dtype=pl.Enum(categories2))})
|
|
110
|
+
S.validate(df)
|
|
@@ -12,6 +12,22 @@ class CheckSchema(dy.Schema):
|
|
|
12
12
|
b = dy.String(min_length=3, check=lambda col: col.str.contains("x"))
|
|
13
13
|
|
|
14
14
|
|
|
15
|
+
def test_check_argument_covariant() -> None:
|
|
16
|
+
# The interesting part of this test is mypy accepting it statically,
|
|
17
|
+
# not the runtime behavior.
|
|
18
|
+
def check_all_or_none(expr: pl.Expr) -> pl.Expr:
|
|
19
|
+
return (expr == "all").all() | (expr != "all").all()
|
|
20
|
+
|
|
21
|
+
check_dict = {"all_or_none": check_all_or_none}
|
|
22
|
+
|
|
23
|
+
class CheckDictSchema(dy.Schema):
|
|
24
|
+
column = dy.String(check=check_dict)
|
|
25
|
+
|
|
26
|
+
df = pl.DataFrame({"column": ["all", "all", "all"]})
|
|
27
|
+
_, failures = CheckDictSchema.filter(df)
|
|
28
|
+
assert failures.counts() == {}
|
|
29
|
+
|
|
30
|
+
|
|
15
31
|
def test_check() -> None:
|
|
16
32
|
df = pl.DataFrame({"a": [7, 3, 15], "b": ["abc", "xyz", "x"]})
|
|
17
33
|
_, failures = CheckSchema.filter(df)
|
|
@@ -58,7 +58,19 @@ import dataframely as dy
|
|
|
58
58
|
dy.Datetime(time_zone=dt.timezone(dt.timedelta(hours=0))),
|
|
59
59
|
True,
|
|
60
60
|
),
|
|
61
|
+
(dy.Enum(["a", "b"]), dy.Enum(["a", "b"]), True),
|
|
62
|
+
(dy.Enum(["a", "b"]), dy.Enum(["a", "b", "c"]), False),
|
|
61
63
|
],
|
|
62
64
|
)
|
|
63
65
|
def test_matches(lhs: dy.Column, rhs: dy.Column, expected: bool) -> None:
|
|
64
66
|
assert lhs.matches(rhs, expr=pl.element()) == expected
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_matches_enum_attribute_type_mismatch() -> None:
|
|
70
|
+
# Comparison should fail if the `other` column has
|
|
71
|
+
# a `category` member, but its dtype is not `pl.Series`
|
|
72
|
+
col1 = dy.Enum(["a", "b"])
|
|
73
|
+
col2 = dy.Enum(["a", "b"])
|
|
74
|
+
col2.categories = "this_is_not_a_series" # type: ignore
|
|
75
|
+
|
|
76
|
+
assert not col1.matches(col2, pl.element())
|
|
@@ -15,6 +15,7 @@ pytestmark = pytest.mark.with_optionals
|
|
|
15
15
|
("column", "datatype"),
|
|
16
16
|
[
|
|
17
17
|
(dy.Any(), "SQL_VARIANT"),
|
|
18
|
+
(dy.Binary(), "VARBINARY(max)"),
|
|
18
19
|
(dy.Bool(), "BIT"),
|
|
19
20
|
(dy.Date(), "DATE"),
|
|
20
21
|
(dy.Datetime(), "DATETIME2(6)"),
|
|
@@ -61,6 +62,7 @@ def test_mssql_datatype(column: Column, datatype: str) -> None:
|
|
|
61
62
|
@pytest.mark.parametrize(
|
|
62
63
|
("column", "datatype"),
|
|
63
64
|
[
|
|
65
|
+
(dy.Binary(), "BYTEA"),
|
|
64
66
|
(dy.Bool(), "BOOLEAN"),
|
|
65
67
|
(dy.Date(), "DATE"),
|
|
66
68
|
(dy.Datetime(), "TIMESTAMP WITHOUT TIME ZONE"),
|
|
@@ -40,6 +40,7 @@ def test_seeding_nonconstant() -> None:
|
|
|
40
40
|
lambda generator, n: generator.sample_float(n, min=0, max=5),
|
|
41
41
|
lambda generator, n: generator.sample_string(n, regex="[abc]"),
|
|
42
42
|
lambda generator, n: generator.sample_choice(n, choices=[1, 2, 3]),
|
|
43
|
+
lambda generator, n: generator.sample_binary(n, min_bytes=1, max_bytes=10),
|
|
43
44
|
lambda generator, n: generator.sample_time(n, min=dt.time(0, 0), max=None),
|
|
44
45
|
lambda generator, n: generator.sample_date(
|
|
45
46
|
n, min=dt.date(1970, 1, 1), max=None
|
|
@@ -75,6 +76,9 @@ def test_sample_correct_n(
|
|
|
75
76
|
lambda generator, n, prob: generator.sample_choice(
|
|
76
77
|
n, choices=[1, 2, 3], null_probability=prob
|
|
77
78
|
),
|
|
79
|
+
lambda generator, n, prob: generator.sample_binary(
|
|
80
|
+
n, min_bytes=1, max_bytes=10, null_probability=prob
|
|
81
|
+
),
|
|
78
82
|
lambda generator, n, prob: generator.sample_time(
|
|
79
83
|
n, min=dt.time(0, 0), max=None, null_probability=prob
|
|
80
84
|
),
|
|
@@ -131,6 +135,16 @@ def test_sample_string(generator: Generator) -> None:
|
|
|
131
135
|
assert (samples.str.len_bytes() == 2).all()
|
|
132
136
|
|
|
133
137
|
|
|
138
|
+
def test_sample_binary(generator: Generator) -> None:
|
|
139
|
+
samples = generator.sample_binary(100, min_bytes=1, max_bytes=10)
|
|
140
|
+
assert (
|
|
141
|
+
samples.to_frame("s").select(pl.col("s").bin.size("b") >= 1).to_series().all()
|
|
142
|
+
)
|
|
143
|
+
assert (
|
|
144
|
+
samples.to_frame("s").select(pl.col("s").bin.size("b") <= 10).to_series().all()
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
134
148
|
def test_sample_choice(generator: Generator) -> None:
|
|
135
149
|
samples = generator.sample_choice(100_000, choices=[1, 2, 3])
|
|
136
150
|
assert np.allclose(
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{dataframely-1.10.0 → dataframely-1.12.0}/tests/collection/test_collection_future_annotations.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|