dppd 0.27__tar.gz → 0.30__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dppd-0.27 → dppd-0.30}/LICENSE.txt +1 -1
- {dppd-0.27/src/dppd.egg-info → dppd-0.30}/PKG-INFO +20 -12
- dppd-0.30/pyproject.toml +57 -0
- dppd-0.30/setup.cfg +4 -0
- {dppd-0.27 → dppd-0.30}/src/dppd/__init__.py +1 -1
- {dppd-0.27 → dppd-0.30}/src/dppd/single_verbs.py +34 -14
- {dppd-0.27 → dppd-0.30/src/dppd.egg-info}/PKG-INFO +20 -12
- {dppd-0.27 → dppd-0.30}/src/dppd.egg-info/SOURCES.txt +7 -4
- {dppd-0.27 → dppd-0.30}/src/dppd.egg-info/requires.txt +7 -11
- dppd-0.30/tests/test_base.py +352 -0
- dppd-0.30/tests/test_pandas_forwards.py +169 -0
- dppd-0.30/tests/test_reshaping.py +173 -0
- dppd-0.30/tests/test_select.py +405 -0
- dppd-0.30/tests/test_single_verbs.py +1031 -0
- dppd-0.27/setup.cfg +0 -94
- dppd-0.27/setup.py +0 -21
- dppd-0.27/src/dppd.egg-info/not-zip-safe +0 -1
- {dppd-0.27 → dppd-0.30}/AUTHORS.rst +0 -0
- {dppd-0.27 → dppd-0.30}/README.md +0 -0
- {dppd-0.27 → dppd-0.30}/src/dppd/base.py +0 -0
- {dppd-0.27 → dppd-0.30}/src/dppd/column_spec.py +0 -0
- {dppd-0.27 → dppd-0.30}/src/dppd/non_df_verbs.py +0 -0
- {dppd-0.27 → dppd-0.30}/src/dppd.egg-info/dependency_links.txt +0 -0
- {dppd-0.27 → dppd-0.30}/src/dppd.egg-info/top_level.txt +0 -0
|
@@ -1,20 +1,28 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: dppd
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.30
|
|
4
4
|
Summary: A pythonic dplyr clone
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
Classifier: Development Status :: 4 - Beta
|
|
11
|
-
Classifier: Programming Language :: Python
|
|
12
|
-
Requires-Python: >=3.6
|
|
5
|
+
Author-email: Florian Finkernagel <finkernagel@imt.uni-marburg.de>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Documentation, https://dppd.readthedocs.io/en/latest/
|
|
8
|
+
Project-URL: Repository, https://github.com/TyberiusPrime/dppd
|
|
9
|
+
Requires-Python: >=3.9
|
|
13
10
|
Description-Content-Type: text/markdown
|
|
14
|
-
Provides-Extra: testing
|
|
15
|
-
Provides-Extra: doc
|
|
16
11
|
License-File: LICENSE.txt
|
|
17
12
|
License-File: AUTHORS.rst
|
|
13
|
+
Requires-Dist: natsort
|
|
14
|
+
Requires-Dist: numpy
|
|
15
|
+
Requires-Dist: pandas>=2
|
|
16
|
+
Requires-Dist: wrapt
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: build; extra == "dev"
|
|
19
|
+
Requires-Dist: numpydoc; extra == "dev"
|
|
20
|
+
Requires-Dist: plotnine; extra == "dev"
|
|
21
|
+
Requires-Dist: pytest; extra == "dev"
|
|
22
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
23
|
+
Requires-Dist: sphinx; extra == "dev"
|
|
24
|
+
Requires-Dist: sphinx-bootstrap-theme; extra == "dev"
|
|
25
|
+
Dynamic: license-file
|
|
18
26
|
|
|
19
27
|
# dppd
|
|
20
28
|
|
dppd-0.30/pyproject.toml
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "dppd"
|
|
3
|
+
version = "0.30"
|
|
4
|
+
description = "A pythonic dplyr clone"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.9"
|
|
7
|
+
authors = [
|
|
8
|
+
{name = "Florian Finkernagel", email = "finkernagel@imt.uni-marburg.de"}
|
|
9
|
+
]
|
|
10
|
+
license="MIT"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"natsort",
|
|
13
|
+
"numpy",
|
|
14
|
+
"pandas>=2",
|
|
15
|
+
"wrapt",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[project.urls]
|
|
19
|
+
Documentation = "https://dppd.readthedocs.io/en/latest/"
|
|
20
|
+
Repository = "https://github.com/TyberiusPrime/dppd"
|
|
21
|
+
|
|
22
|
+
[build-system]
|
|
23
|
+
requires = ["setuptools >= 61.0"]
|
|
24
|
+
build-backend = "setuptools.build_meta"
|
|
25
|
+
|
|
26
|
+
[project.optional-dependencies]
|
|
27
|
+
dev = [
|
|
28
|
+
"build",
|
|
29
|
+
"numpydoc",
|
|
30
|
+
"plotnine",
|
|
31
|
+
"pytest",
|
|
32
|
+
"pytest-cov",
|
|
33
|
+
"sphinx",
|
|
34
|
+
"sphinx-bootstrap-theme",
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
[tool.pytest.ini_options]
|
|
38
|
+
# Options for py.test:
|
|
39
|
+
# Specify command line options as you would do when invoking py.test directly.
|
|
40
|
+
# e.g. --cov-report html (or xml) for html/xml output or --junitxml junit.xml
|
|
41
|
+
# in order to write a coverage file that can be read by Jenkins.
|
|
42
|
+
addopts = """
|
|
43
|
+
--cov dppd --cov-report term-missing
|
|
44
|
+
--verbose
|
|
45
|
+
"""
|
|
46
|
+
norecursedirs = [
|
|
47
|
+
"dist",
|
|
48
|
+
"build",
|
|
49
|
+
".tox",
|
|
50
|
+
]
|
|
51
|
+
testpaths = "tests"
|
|
52
|
+
filterwarnings = [
|
|
53
|
+
"ignore:::statsmodels.base.wrapper:100",
|
|
54
|
+
"ignore:::patsy.constraint:13",
|
|
55
|
+
"ignore:::matplotlib.backends.backend_wx:",
|
|
56
|
+
]
|
|
57
|
+
|
dppd-0.30/setup.cfg
ADDED
|
@@ -4,6 +4,6 @@ from .base import dppd, register_verb, register_type_methods_as_verbs
|
|
|
4
4
|
from . import single_verbs # noqa:F401
|
|
5
5
|
from . import non_df_verbs # noqa:F401
|
|
6
6
|
|
|
7
|
-
__version__ = "0.
|
|
7
|
+
__version__ = "0.30"
|
|
8
8
|
|
|
9
9
|
__all_ = [dppd, register_verb, register_type_methods_as_verbs, __version__]
|
|
@@ -23,14 +23,20 @@ register_type_methods_as_verbs(SeriesGroupBy, [])
|
|
|
23
23
|
|
|
24
24
|
|
|
25
25
|
def group_variables(grp):
|
|
26
|
-
|
|
26
|
+
if hasattr(grp, "_grouper"):
|
|
27
|
+
return grp._grouper.names
|
|
28
|
+
else:
|
|
29
|
+
return grp.grouper.names
|
|
27
30
|
|
|
28
31
|
|
|
29
32
|
def group_extract_params(grp):
|
|
30
|
-
if grp.axis != 0:
|
|
33
|
+
if hasattr(grp, "axis") and grp.axis != 0:
|
|
31
34
|
raise ValueError(f"Verbs assume that groupby is on axis=0, was {grp.axis}")
|
|
32
35
|
res = {"by": group_variables(grp)}
|
|
33
|
-
|
|
36
|
+
attrs = ["squeeze", "level", "as_index", "sort", "group_keys", "observed"]
|
|
37
|
+
if pd.__version__ < "2.1.":
|
|
38
|
+
attrs.append("axis")
|
|
39
|
+
for k in attrs:
|
|
34
40
|
if hasattr(grp, k):
|
|
35
41
|
res[k] = getattr(grp, k)
|
|
36
42
|
else: # pragma: no cover
|
|
@@ -56,7 +62,10 @@ def _print(obj):
|
|
|
56
62
|
@register_verb(name="debug", types=None)
|
|
57
63
|
def _debug(obj, k=5):
|
|
58
64
|
d = obj.iloc[np.r_[0:k, -k:0]]
|
|
59
|
-
|
|
65
|
+
try:
|
|
66
|
+
display(d) # noqa: F821 - Jupyter only, needs to import.
|
|
67
|
+
except NameError:
|
|
68
|
+
print(d)
|
|
60
69
|
return obj
|
|
61
70
|
|
|
62
71
|
|
|
@@ -354,9 +363,12 @@ def mutate_DataFrameGroupBy(grp, **kwargs):
|
|
|
354
363
|
try:
|
|
355
364
|
r = v[group_key]
|
|
356
365
|
except KeyError:
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
366
|
+
try:
|
|
367
|
+
r = v[group_key,]
|
|
368
|
+
except KeyError:
|
|
369
|
+
raise KeyError(
|
|
370
|
+
f"Grouped mutate results did not contain data for {group_key}. Keys where {v.keys()}"
|
|
371
|
+
)
|
|
360
372
|
r = pd.Series(r, index=sub_index)
|
|
361
373
|
parts.append(r)
|
|
362
374
|
parts = pd.concat(parts)
|
|
@@ -451,7 +463,10 @@ def filter_by(obj, filter_arg):
|
|
|
451
463
|
for idx, sub_df in df.groupby(groups):
|
|
452
464
|
# if not idx in filter_arg and not isinstance(tuple(idx)):
|
|
453
465
|
# idx = (idx,)
|
|
454
|
-
|
|
466
|
+
try:
|
|
467
|
+
keep = filter_arg[idx]
|
|
468
|
+
except KeyError:
|
|
469
|
+
keep = filter_arg[idx[0]]
|
|
455
470
|
parts.append(sub_df[keep])
|
|
456
471
|
result = pd.concat(parts, axis=0)
|
|
457
472
|
elif isinstance(filter_arg, str):
|
|
@@ -543,7 +558,7 @@ def summarize(obj, *args):
|
|
|
543
558
|
result = result[groups + [x for x in result.columns if x not in groups]]
|
|
544
559
|
# restore category to categories
|
|
545
560
|
for g in groups:
|
|
546
|
-
if
|
|
561
|
+
if isinstance(df.dtypes[g], pd.CategoricalDtype):
|
|
547
562
|
result = result.assign(
|
|
548
563
|
**{
|
|
549
564
|
g: pd.Categorical(
|
|
@@ -603,7 +618,7 @@ def do(obj, func, *args, **kwargs):
|
|
|
603
618
|
result = result[groups + [x for x in result.columns if x not in groups]]
|
|
604
619
|
# restore category to categories
|
|
605
620
|
for g in groups:
|
|
606
|
-
if
|
|
621
|
+
if isinstance(df.dtypes[g], pd.CategoricalDtype):
|
|
607
622
|
result = result.assign(
|
|
608
623
|
**{
|
|
609
624
|
g: pd.Categorical(
|
|
@@ -767,7 +782,10 @@ def seperate(df, column, new_names, sep=".", remove=False):
|
|
|
767
782
|
|
|
768
783
|
@register_verb("print", types=DataFrameGroupBy)
|
|
769
784
|
def print_DataFrameGroupBy(grps):
|
|
770
|
-
|
|
785
|
+
if hasattr(grps, "_grouper"):
|
|
786
|
+
print("groups: %s" % (grps._grouper.names))
|
|
787
|
+
else:
|
|
788
|
+
print("groups: %s" % (grps.grouper.names))
|
|
771
789
|
print(grps._selected_obj)
|
|
772
790
|
return grps
|
|
773
791
|
|
|
@@ -809,7 +827,7 @@ def arrange_DataFrameGroupBy(grp, column_spec, kind="quicksort", na_position="la
|
|
|
809
827
|
columns = grp_params["by"].copy()
|
|
810
828
|
ascending = [True] * len(columns)
|
|
811
829
|
columns += [x[0] for x in cols_plus_inversed]
|
|
812
|
-
ascending += [
|
|
830
|
+
ascending += [not x[1] for x in cols_plus_inversed]
|
|
813
831
|
df_out = df.sort_values(
|
|
814
832
|
columns, ascending=ascending, kind=kind, na_position=na_position
|
|
815
833
|
)
|
|
@@ -951,7 +969,9 @@ def norm_0_to_1(df, axis=1, keep_nan=False):
|
|
|
951
969
|
a1 = 0
|
|
952
970
|
a2 = 1
|
|
953
971
|
df_normed = df.sub(df.min(axis=a1), axis=a2)
|
|
954
|
-
df_normed
|
|
972
|
+
assert df_normed.min().min() == 0.0
|
|
973
|
+
df_normed = df_normed.div(df_normed.max(axis=a1), axis=a2)
|
|
974
|
+
assert df_normed.max().max() == 1.0
|
|
955
975
|
if not keep_nan:
|
|
956
976
|
df_normed = df_normed[~pd.isnull(df_normed).any(axis=1)]
|
|
957
977
|
return df_normed
|
|
@@ -1004,7 +1024,7 @@ def pca_dataframe(df, whiten=False, random_state=None, n_components=2):
|
|
|
1004
1024
|
df_fit = pd.DataFrame(p.fit_transform(df))
|
|
1005
1025
|
cols = ["1st", "2nd"]
|
|
1006
1026
|
if n_components > 2:
|
|
1007
|
-
cols.append(
|
|
1027
|
+
cols.append("3rd")
|
|
1008
1028
|
for ii in range(3, n_components):
|
|
1009
1029
|
cols.append(f"{ii+1}th")
|
|
1010
1030
|
df_fit.columns = cols
|
|
@@ -1,20 +1,28 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: dppd
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.30
|
|
4
4
|
Summary: A pythonic dplyr clone
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
Classifier: Development Status :: 4 - Beta
|
|
11
|
-
Classifier: Programming Language :: Python
|
|
12
|
-
Requires-Python: >=3.6
|
|
5
|
+
Author-email: Florian Finkernagel <finkernagel@imt.uni-marburg.de>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Documentation, https://dppd.readthedocs.io/en/latest/
|
|
8
|
+
Project-URL: Repository, https://github.com/TyberiusPrime/dppd
|
|
9
|
+
Requires-Python: >=3.9
|
|
13
10
|
Description-Content-Type: text/markdown
|
|
14
|
-
Provides-Extra: testing
|
|
15
|
-
Provides-Extra: doc
|
|
16
11
|
License-File: LICENSE.txt
|
|
17
12
|
License-File: AUTHORS.rst
|
|
13
|
+
Requires-Dist: natsort
|
|
14
|
+
Requires-Dist: numpy
|
|
15
|
+
Requires-Dist: pandas>=2
|
|
16
|
+
Requires-Dist: wrapt
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: build; extra == "dev"
|
|
19
|
+
Requires-Dist: numpydoc; extra == "dev"
|
|
20
|
+
Requires-Dist: plotnine; extra == "dev"
|
|
21
|
+
Requires-Dist: pytest; extra == "dev"
|
|
22
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
23
|
+
Requires-Dist: sphinx; extra == "dev"
|
|
24
|
+
Requires-Dist: sphinx-bootstrap-theme; extra == "dev"
|
|
25
|
+
Dynamic: license-file
|
|
18
26
|
|
|
19
27
|
# dppd
|
|
20
28
|
|
|
@@ -1,8 +1,7 @@
|
|
|
1
1
|
AUTHORS.rst
|
|
2
2
|
LICENSE.txt
|
|
3
3
|
README.md
|
|
4
|
-
|
|
5
|
-
setup.py
|
|
4
|
+
pyproject.toml
|
|
6
5
|
src/dppd/__init__.py
|
|
7
6
|
src/dppd/base.py
|
|
8
7
|
src/dppd/column_spec.py
|
|
@@ -11,6 +10,10 @@ src/dppd/single_verbs.py
|
|
|
11
10
|
src/dppd.egg-info/PKG-INFO
|
|
12
11
|
src/dppd.egg-info/SOURCES.txt
|
|
13
12
|
src/dppd.egg-info/dependency_links.txt
|
|
14
|
-
src/dppd.egg-info/not-zip-safe
|
|
15
13
|
src/dppd.egg-info/requires.txt
|
|
16
|
-
src/dppd.egg-info/top_level.txt
|
|
14
|
+
src/dppd.egg-info/top_level.txt
|
|
15
|
+
tests/test_base.py
|
|
16
|
+
tests/test_pandas_forwards.py
|
|
17
|
+
tests/test_reshaping.py
|
|
18
|
+
tests/test_select.py
|
|
19
|
+
tests/test_single_verbs.py
|
|
@@ -1,17 +1,13 @@
|
|
|
1
|
-
pandas>=0.22
|
|
2
|
-
numpy
|
|
3
1
|
natsort
|
|
2
|
+
numpy
|
|
3
|
+
pandas>=2
|
|
4
4
|
wrapt
|
|
5
5
|
|
|
6
|
-
[
|
|
7
|
-
|
|
8
|
-
sphinx-bootstrap-theme
|
|
6
|
+
[dev]
|
|
7
|
+
build
|
|
9
8
|
numpydoc
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
[testing]
|
|
9
|
+
plotnine
|
|
13
10
|
pytest
|
|
14
11
|
pytest-cov
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
flake8
|
|
12
|
+
sphinx
|
|
13
|
+
sphinx-bootstrap-theme
|
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
from dppd import dppd, register_verb
|
|
6
|
+
from dppd.base import register_property
|
|
7
|
+
import pandas as pd
|
|
8
|
+
import numpy as np
|
|
9
|
+
import pandas.testing
|
|
10
|
+
import wrapt
|
|
11
|
+
from plotnine.data import mtcars, diamonds
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
__author__ = "Florian Finkernagel"
|
|
15
|
+
__copyright__ = "Florian Finkernagel"
|
|
16
|
+
__license__ = "mit"
|
|
17
|
+
|
|
18
|
+
assert_series_equal = pandas.testing.assert_series_equal
|
|
19
|
+
assert_frame_equal = pandas.testing.assert_frame_equal
|
|
20
|
+
|
|
21
|
+
dp, X = dppd()
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def test_noop():
|
|
25
|
+
df = pd.DataFrame({"a": list(range(10))})
|
|
26
|
+
actual = dp(df)
|
|
27
|
+
actual = actual.pd
|
|
28
|
+
assert isinstance(actual, pd.DataFrame)
|
|
29
|
+
should = df
|
|
30
|
+
assert_frame_equal(should, actual)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_non_df_result():
|
|
34
|
+
import dppd.base
|
|
35
|
+
|
|
36
|
+
df = pd.DataFrame({"a": list(range(10))})
|
|
37
|
+
shape = dp(df).head(5).shape
|
|
38
|
+
assert not isinstance(shape, dppd.base.DPPDAwareProxy)
|
|
39
|
+
assert shape == (5, 1)
|
|
40
|
+
real_pd = shape.pd
|
|
41
|
+
assert isinstance(real_pd, tuple)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_nested_dp_pd_calls():
|
|
45
|
+
df = pd.DataFrame({"a": list(range(10))})
|
|
46
|
+
actual = dp(df).head(5).concat(dp(df).tail(4).pd).pd
|
|
47
|
+
should = pd.concat([df.head(5), df.tail(4)], axis=0)
|
|
48
|
+
assert_frame_equal(should, actual)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def test_dp_pd_calls_nested_in_function_calls():
|
|
52
|
+
df = pd.DataFrame(
|
|
53
|
+
{"a": list(range(10)), "bb": list(range(10)), "ccc": list(range(10))}
|
|
54
|
+
).set_index("a")
|
|
55
|
+
|
|
56
|
+
def shu():
|
|
57
|
+
return dp(df).head(1).pd
|
|
58
|
+
|
|
59
|
+
def sha():
|
|
60
|
+
return dp(df).tail(1).pd
|
|
61
|
+
|
|
62
|
+
should = pd.concat([df.head(1), df.tail(1), df.head(1)])
|
|
63
|
+
actual = dp(shu()).concat(sha()).concat(shu()).pd
|
|
64
|
+
assert_frame_equal(should, actual)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def test_redefining_verb_vars():
|
|
68
|
+
def noop(df):
|
|
69
|
+
return df
|
|
70
|
+
|
|
71
|
+
register_verb("test_redefining_verb_vars_noop")(noop)
|
|
72
|
+
with pytest.warns(UserWarning):
|
|
73
|
+
register_verb("test_redefining_verb_vars_noop")(noop)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def test_property_shadowed():
|
|
77
|
+
with pytest.warns(UserWarning):
|
|
78
|
+
register_property("select", pd.DataFrame)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def test_property_list_of_types():
|
|
82
|
+
import dppd.base
|
|
83
|
+
|
|
84
|
+
class A:
|
|
85
|
+
pass
|
|
86
|
+
|
|
87
|
+
class B:
|
|
88
|
+
pass
|
|
89
|
+
|
|
90
|
+
register_property("test_property_list_of_types", [A, B])
|
|
91
|
+
assert "test_property_list_of_types" in dppd.base.property_registry[A]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def test_verb_shadows_property():
|
|
95
|
+
def noop(df):
|
|
96
|
+
return df
|
|
97
|
+
|
|
98
|
+
register_property("test_verb_shadows_property_noop")
|
|
99
|
+
|
|
100
|
+
with pytest.warns(UserWarning):
|
|
101
|
+
register_verb("test_verb_shadows_property_noop")(noop)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def test_register_verb_raises_on_non_identifier():
|
|
105
|
+
with pytest.raises(TypeError):
|
|
106
|
+
register_verb(name="Hello world")(lambda x: x)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def test_register_verb_aliases():
|
|
110
|
+
import dppd.base
|
|
111
|
+
|
|
112
|
+
def shu():
|
|
113
|
+
pass
|
|
114
|
+
|
|
115
|
+
register_verb(["test_register_verb_aliases", "test_register_verb_aliases2"])(shu)
|
|
116
|
+
assert ("test_register_verb_aliases", None) in dppd.base.verb_registry
|
|
117
|
+
assert ("test_register_verb_aliases2", None) in dppd.base.verb_registry
|
|
118
|
+
assert (
|
|
119
|
+
dppd.base.verb_registry[("test_register_verb_aliases", None)]
|
|
120
|
+
is dppd.base.verb_registry[("test_register_verb_aliases2", None)]
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def test_verb_returning_non_df():
|
|
125
|
+
df = pd.DataFrame({"a": [str(x) for x in (range(10))]})
|
|
126
|
+
register_verb("da_length")(lambda df: len(df))
|
|
127
|
+
assert 10 == dp(df).da_length()
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def test_no_context_manager_non_df_returning_verbs():
|
|
131
|
+
df = pd.DataFrame({"a": [str(x) for x in (range(10))]})
|
|
132
|
+
dp(df)
|
|
133
|
+
dp().head(1)
|
|
134
|
+
assert len(X) == 1
|
|
135
|
+
assert X.shape == (1, 1)
|
|
136
|
+
actual = dp().pd
|
|
137
|
+
assert actual.shape == (1, 1)
|
|
138
|
+
assert actual.__len__() == 1
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def test_no_attribute_no_verb_raises_attribute_error():
|
|
142
|
+
df = pd.DataFrame({"a": [str(x) for x in (range(10))]})
|
|
143
|
+
with pytest.raises(AttributeError):
|
|
144
|
+
dp(df).shu()
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def test_no_attribute_no_verb_raises_attribute_error_context_manager():
|
|
148
|
+
df = pd.DataFrame({"a": [str(x) for x in (range(10))]})
|
|
149
|
+
with pytest.raises(AttributeError):
|
|
150
|
+
with dppd(df) as (dp, X):
|
|
151
|
+
dp.shu()
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def test_dp_on_empty_stack_raises():
|
|
155
|
+
dp, X = dppd()
|
|
156
|
+
with pytest.raises(ValueError):
|
|
157
|
+
dp()
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def test_dp_continuation():
|
|
161
|
+
df = pd.DataFrame(
|
|
162
|
+
{"a": [str(x) for x in (range(10))], "bb": 10, "ccc": list(range(20, 30))}
|
|
163
|
+
).set_index("a")
|
|
164
|
+
dp(df).head(5)
|
|
165
|
+
dp().tail(1)
|
|
166
|
+
actual = dp().pd
|
|
167
|
+
should = df.iloc[4:5]
|
|
168
|
+
assert_frame_equal(should, actual)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def test_context_manager():
|
|
172
|
+
df = pd.DataFrame(
|
|
173
|
+
{"a": [str(x) for x in (range(10))], "bb": 10, "ccc": list(range(20, 30))}
|
|
174
|
+
).set_index("a")
|
|
175
|
+
with dppd(df) as (d, X):
|
|
176
|
+
d.head(5)
|
|
177
|
+
d.tail(1)
|
|
178
|
+
should = df.iloc[4:5]
|
|
179
|
+
assert_frame_equal(X, should)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def test_context_manager_with_non_df_args():
|
|
183
|
+
df = pd.DataFrame(
|
|
184
|
+
{"a": [str(x) for x in (range(10))], "bb": 10, "ccc": list(range(20, 30))}
|
|
185
|
+
).set_index("a")
|
|
186
|
+
with dppd(df) as (d, X):
|
|
187
|
+
d.head(5)
|
|
188
|
+
assert d.shape == (5, 2)
|
|
189
|
+
d.tail(1)
|
|
190
|
+
should = df.iloc[4:5]
|
|
191
|
+
assert_frame_equal(X, should)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def test_context_manager_totally_to_pandas():
|
|
195
|
+
df = pd.DataFrame(
|
|
196
|
+
{"a": [str(x) for x in (range(10))], "bb": 10, "ccc": list(range(20, 30))}
|
|
197
|
+
).set_index("a")
|
|
198
|
+
with dppd(df) as (d, X):
|
|
199
|
+
d.head(5)
|
|
200
|
+
assert d.shape == (5, 2)
|
|
201
|
+
d.tail(1)
|
|
202
|
+
should = df.iloc[4:5]
|
|
203
|
+
assert_frame_equal(X, should)
|
|
204
|
+
assert isinstance(X, wrapt.ObjectProxy)
|
|
205
|
+
X = X.pd
|
|
206
|
+
assert not isinstance(X, wrapt.ObjectProxy)
|
|
207
|
+
assert_frame_equal(X, should)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def test_interleaved_context_managers():
|
|
211
|
+
with dppd(mtcars) as (dpX, X):
|
|
212
|
+
with dppd(diamonds) as (dpY, Y):
|
|
213
|
+
dpX.groupby("cyl")
|
|
214
|
+
dpY.filter_by(Y.cut == "Ideal")
|
|
215
|
+
dpX.summarize(("hp", np.mean, "mean_hp"))
|
|
216
|
+
dpY.summarize(("price", np.max, "max_price"))
|
|
217
|
+
should_X = (
|
|
218
|
+
mtcars.groupby("cyl")[["hp"]].agg("mean").rename(columns={"hp": "mean_hp"})
|
|
219
|
+
).reset_index()
|
|
220
|
+
should_Y = (
|
|
221
|
+
pd.DataFrame(diamonds[diamonds.cut == "Ideal"].max()[["price"]])
|
|
222
|
+
.transpose()
|
|
223
|
+
.rename(columns={"price": "max_price"})
|
|
224
|
+
)
|
|
225
|
+
should_Y["max_price"] = should_Y["max_price"].astype(int)
|
|
226
|
+
assert_frame_equal(X, should_X)
|
|
227
|
+
assert_frame_equal(Y, should_Y)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def test_context_manager_chain():
|
|
231
|
+
with dppd(mtcars) as (dp, X):
|
|
232
|
+
dp.mutate(kw=X.hp * 0.7457)
|
|
233
|
+
with dppd(X) as (dp, X):
|
|
234
|
+
dp.mutate(watt=X.kw * 1000)
|
|
235
|
+
assert "watt" in X.columns
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def test_mixing_context_manager_and_dp():
|
|
239
|
+
with dppd(mtcars) as (dpY, Y):
|
|
240
|
+
dpY.sort_values("hp")
|
|
241
|
+
dp(diamonds).filter_by(X.cut == "ideal")
|
|
242
|
+
dpY.filter_by(Y.cyl.isin([4, 6]))
|
|
243
|
+
actual_diamonds = dp().sort_values("price").head().pd
|
|
244
|
+
actual_mtcars_full = dpY.pd
|
|
245
|
+
dpY.head()
|
|
246
|
+
actual_mtcars = dpY.pd
|
|
247
|
+
should_diamonds = diamonds[diamonds.cut == "ideal"].sort_values("price").head()
|
|
248
|
+
should_mtcars = mtcars.sort_values("hp")
|
|
249
|
+
should_mtcars_full = should_mtcars[should_mtcars["cyl"].isin([4, 6])]
|
|
250
|
+
should_mtcars = should_mtcars_full.head()
|
|
251
|
+
assert_frame_equal(should_diamonds, actual_diamonds)
|
|
252
|
+
assert_frame_equal(should_mtcars, actual_mtcars)
|
|
253
|
+
assert_frame_equal(should_mtcars_full, actual_mtcars_full)
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def test_dppd_raises_on_non_dataframe():
|
|
257
|
+
with pytest.raises(ValueError):
|
|
258
|
+
dp(5)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def test_straight_dp_raises():
|
|
262
|
+
dp, X = dppd()
|
|
263
|
+
with pytest.raises(ValueError):
|
|
264
|
+
dp.select(["hp", "cyl"])
|
|
265
|
+
|
|
266
|
+
with pytest.raises(ValueError):
|
|
267
|
+
dp.loc[5]
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def test_stacking():
|
|
271
|
+
dp, X = dppd()
|
|
272
|
+
dp(mtcars).select(["name", "hp", "cyl"])
|
|
273
|
+
b = dp(mtcars).select("hp").pd
|
|
274
|
+
assert_frame_equal(b, mtcars[["hp"]])
|
|
275
|
+
assert_frame_equal(X, mtcars[["name", "hp", "cyl"]])
|
|
276
|
+
c = dp.pd
|
|
277
|
+
assert_frame_equal(c, mtcars[["name", "hp", "cyl"]])
|
|
278
|
+
assert X == None # noqa:E711 since it's the proxy, is will fail
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def test_forking():
|
|
282
|
+
dp, X = dppd()
|
|
283
|
+
a = dp(mtcars).select(["name", "hp", "cyl"])
|
|
284
|
+
b = dp.unselect("hp").select(X.name).head().pd
|
|
285
|
+
with pytest.raises(AttributeError):
|
|
286
|
+
c = a.select(X.hp).head().pd
|
|
287
|
+
c = dp(a).select(X.hp).head().pd
|
|
288
|
+
assert_series_equal(c["hp"], mtcars["hp"].head())
|
|
289
|
+
assert_series_equal(b["name"], mtcars["name"].head())
|
|
290
|
+
assert X == None # noqa:E711 since it's the proxy, is will fail
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def test_forking_context_manager():
|
|
294
|
+
with dppd(mtcars) as (dp, X):
|
|
295
|
+
a = dp.select(["name", "hp", "cyl"])
|
|
296
|
+
b = dp.select("name").head().pd
|
|
297
|
+
c = a.select("hp").head().pd
|
|
298
|
+
dp.head()
|
|
299
|
+
assert_series_equal(c["hp"], mtcars["hp"].head())
|
|
300
|
+
assert_series_equal(b["name"], mtcars["name"].head())
|
|
301
|
+
assert_frame_equal(X, mtcars[["hp"]].head())
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def test_dp_on_dp():
|
|
305
|
+
import wrapt
|
|
306
|
+
|
|
307
|
+
a = dp(mtcars)
|
|
308
|
+
b = dp(a)
|
|
309
|
+
assert not isinstance(a.df, wrapt.ObjectProxy)
|
|
310
|
+
assert not isinstance(b.df, wrapt.ObjectProxy)
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def test_descend_on_None_raises():
|
|
314
|
+
with pytest.raises(ValueError):
|
|
315
|
+
dp(mtcars)._descend(None)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def test_series_methods():
|
|
319
|
+
actual = dp(mtcars).sum().to_frame().pd
|
|
320
|
+
should = mtcars.sum().to_frame()
|
|
321
|
+
assert_frame_equal(should, actual)
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def test_group_on_series_raises():
|
|
325
|
+
with pytest.raises(KeyError):
|
|
326
|
+
dp(mtcars).sum().groupby("no_such_columns")
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def test_dir():
|
|
330
|
+
from dppd import base
|
|
331
|
+
|
|
332
|
+
dp, X = dppd()
|
|
333
|
+
actual = set(dir(dp(mtcars)))
|
|
334
|
+
should_min = set(base.property_registry[pd.DataFrame])
|
|
335
|
+
delta = should_min.difference(actual)
|
|
336
|
+
print(sorted(actual))
|
|
337
|
+
print(sorted(delta))
|
|
338
|
+
assert not len(delta)
|
|
339
|
+
assert len(actual) > len(should_min)
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def test_version_is_correct():
|
|
343
|
+
from pathlib import Path
|
|
344
|
+
try:
|
|
345
|
+
import tomllib
|
|
346
|
+
import dppd as org_dppd
|
|
347
|
+
|
|
348
|
+
c = tomllib.load(open(Path(__file__).parent.parent / "pyproject.toml", "rb"))
|
|
349
|
+
version = c["project"]["version"]
|
|
350
|
+
assert version == org_dppd.__version__
|
|
351
|
+
except ImportError:
|
|
352
|
+
pass # python < 3.11
|