dppd 0.27__tar.gz → 0.30__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  The MIT License (MIT)
2
2
 
3
- Copyright (c) 2018 Florian Finkernagel
3
+ Copyright (c) 2025 Florian Finkernagel
4
4
 
5
5
  Permission is hereby granted, free of charge, to any person obtaining a copy
6
6
  of this software and associated documentation files (the "Software"), to deal
@@ -1,20 +1,28 @@
1
- Metadata-Version: 2.1
1
+ Metadata-Version: 2.4
2
2
  Name: dppd
3
- Version: 0.27
3
+ Version: 0.30
4
4
  Summary: A pythonic dplyr clone
5
- Home-page: https://github.com/TyberiusPrime/dppd
6
- Author: Florian Finkernagel
7
- Author-email: finkernagel@imt.uni-marburg.de
8
- License: mit
9
- Platform: any
10
- Classifier: Development Status :: 4 - Beta
11
- Classifier: Programming Language :: Python
12
- Requires-Python: >=3.6
5
+ Author-email: Florian Finkernagel <finkernagel@imt.uni-marburg.de>
6
+ License-Expression: MIT
7
+ Project-URL: Documentation, https://dppd.readthedocs.io/en/latest/
8
+ Project-URL: Repository, https://github.com/TyberiusPrime/dppd
9
+ Requires-Python: >=3.9
13
10
  Description-Content-Type: text/markdown
14
- Provides-Extra: testing
15
- Provides-Extra: doc
16
11
  License-File: LICENSE.txt
17
12
  License-File: AUTHORS.rst
13
+ Requires-Dist: natsort
14
+ Requires-Dist: numpy
15
+ Requires-Dist: pandas>=2
16
+ Requires-Dist: wrapt
17
+ Provides-Extra: dev
18
+ Requires-Dist: build; extra == "dev"
19
+ Requires-Dist: numpydoc; extra == "dev"
20
+ Requires-Dist: plotnine; extra == "dev"
21
+ Requires-Dist: pytest; extra == "dev"
22
+ Requires-Dist: pytest-cov; extra == "dev"
23
+ Requires-Dist: sphinx; extra == "dev"
24
+ Requires-Dist: sphinx-bootstrap-theme; extra == "dev"
25
+ Dynamic: license-file
18
26
 
19
27
  # dppd
20
28
 
@@ -0,0 +1,57 @@
1
+ [project]
2
+ name = "dppd"
3
+ version = "0.30"
4
+ description = "A pythonic dplyr clone"
5
+ readme = "README.md"
6
+ requires-python = ">=3.9"
7
+ authors = [
8
+ {name = "Florian Finkernagel", email = "finkernagel@imt.uni-marburg.de"}
9
+ ]
10
+ license="MIT"
11
+ dependencies = [
12
+ "natsort",
13
+ "numpy",
14
+ "pandas>=2",
15
+ "wrapt",
16
+ ]
17
+
18
+ [project.urls]
19
+ Documentation = "https://dppd.readthedocs.io/en/latest/"
20
+ Repository = "https://github.com/TyberiusPrime/dppd"
21
+
22
+ [build-system]
23
+ requires = ["setuptools >= 61.0"]
24
+ build-backend = "setuptools.build_meta"
25
+
26
+ [project.optional-dependencies]
27
+ dev = [
28
+ "build",
29
+ "numpydoc",
30
+ "plotnine",
31
+ "pytest",
32
+ "pytest-cov",
33
+ "sphinx",
34
+ "sphinx-bootstrap-theme",
35
+ ]
36
+
37
+ [tool.pytest.ini_options]
38
+ # Options for py.test:
39
+ # Specify command line options as you would do when invoking py.test directly.
40
+ # e.g. --cov-report html (or xml) for html/xml output or --junitxml junit.xml
41
+ # in order to write a coverage file that can be read by Jenkins.
42
+ addopts = """
43
+ --cov dppd --cov-report term-missing
44
+ --verbose
45
+ """
46
+ norecursedirs = [
47
+ "dist",
48
+ "build",
49
+ ".tox",
50
+ ]
51
+ testpaths = "tests"
52
+ filterwarnings = [
53
+ "ignore:::statsmodels.base.wrapper:100",
54
+ "ignore:::patsy.constraint:13",
55
+ "ignore:::matplotlib.backends.backend_wx:",
56
+ ]
57
+
dppd-0.30/setup.cfg ADDED
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -4,6 +4,6 @@ from .base import dppd, register_verb, register_type_methods_as_verbs
4
4
  from . import single_verbs # noqa:F401
5
5
  from . import non_df_verbs # noqa:F401
6
6
 
7
- __version__ = "0.27"
7
+ __version__ = "0.30"
8
8
 
9
9
  __all_ = [dppd, register_verb, register_type_methods_as_verbs, __version__]
@@ -23,14 +23,20 @@ register_type_methods_as_verbs(SeriesGroupBy, [])
23
23
 
24
24
 
25
25
  def group_variables(grp):
26
- return grp.grouper.names
26
+ if hasattr(grp, "_grouper"):
27
+ return grp._grouper.names
28
+ else:
29
+ return grp.grouper.names
27
30
 
28
31
 
29
32
  def group_extract_params(grp):
30
- if grp.axis != 0:
33
+ if hasattr(grp, "axis") and grp.axis != 0:
31
34
  raise ValueError(f"Verbs assume that groupby is on axis=0, was {grp.axis}")
32
35
  res = {"by": group_variables(grp)}
33
- for k in ["squeeze", "axis", "level", "as_index", "sort", "group_keys", "observed"]:
36
+ attrs = ["squeeze", "level", "as_index", "sort", "group_keys", "observed"]
37
+ if pd.__version__ < "2.1.":
38
+ attrs.append("axis")
39
+ for k in attrs:
34
40
  if hasattr(grp, k):
35
41
  res[k] = getattr(grp, k)
36
42
  else: # pragma: no cover
@@ -56,7 +62,10 @@ def _print(obj):
56
62
  @register_verb(name="debug", types=None)
57
63
  def _debug(obj, k=5):
58
64
  d = obj.iloc[np.r_[0:k, -k:0]]
59
- print(d)
65
+ try:
66
+ display(d) # noqa: F821 - Jupyter only, needs to import.
67
+ except NameError:
68
+ print(d)
60
69
  return obj
61
70
 
62
71
 
@@ -354,9 +363,12 @@ def mutate_DataFrameGroupBy(grp, **kwargs):
354
363
  try:
355
364
  r = v[group_key]
356
365
  except KeyError:
357
- raise KeyError(
358
- f"Grouped mutate results did not contain data for {group_key}"
359
- )
366
+ try:
367
+ r = v[group_key,]
368
+ except KeyError:
369
+ raise KeyError(
370
+ f"Grouped mutate results did not contain data for {group_key}. Keys where {v.keys()}"
371
+ )
360
372
  r = pd.Series(r, index=sub_index)
361
373
  parts.append(r)
362
374
  parts = pd.concat(parts)
@@ -451,7 +463,10 @@ def filter_by(obj, filter_arg):
451
463
  for idx, sub_df in df.groupby(groups):
452
464
  # if not idx in filter_arg and not isinstance(tuple(idx)):
453
465
  # idx = (idx,)
454
- keep = filter_arg[idx]
466
+ try:
467
+ keep = filter_arg[idx]
468
+ except KeyError:
469
+ keep = filter_arg[idx[0]]
455
470
  parts.append(sub_df[keep])
456
471
  result = pd.concat(parts, axis=0)
457
472
  elif isinstance(filter_arg, str):
@@ -543,7 +558,7 @@ def summarize(obj, *args):
543
558
  result = result[groups + [x for x in result.columns if x not in groups]]
544
559
  # restore category to categories
545
560
  for g in groups:
546
- if pd.api.types.is_categorical_dtype(df[g]):
561
+ if isinstance(df.dtypes[g], pd.CategoricalDtype):
547
562
  result = result.assign(
548
563
  **{
549
564
  g: pd.Categorical(
@@ -603,7 +618,7 @@ def do(obj, func, *args, **kwargs):
603
618
  result = result[groups + [x for x in result.columns if x not in groups]]
604
619
  # restore category to categories
605
620
  for g in groups:
606
- if pd.api.types.is_categorical_dtype(df[g]):
621
+ if isinstance(df.dtypes[g], pd.CategoricalDtype):
607
622
  result = result.assign(
608
623
  **{
609
624
  g: pd.Categorical(
@@ -767,7 +782,10 @@ def seperate(df, column, new_names, sep=".", remove=False):
767
782
 
768
783
  @register_verb("print", types=DataFrameGroupBy)
769
784
  def print_DataFrameGroupBy(grps):
770
- print("groups: %s" % (grps.grouper.names))
785
+ if hasattr(grps, "_grouper"):
786
+ print("groups: %s" % (grps._grouper.names))
787
+ else:
788
+ print("groups: %s" % (grps.grouper.names))
771
789
  print(grps._selected_obj)
772
790
  return grps
773
791
 
@@ -809,7 +827,7 @@ def arrange_DataFrameGroupBy(grp, column_spec, kind="quicksort", na_position="la
809
827
  columns = grp_params["by"].copy()
810
828
  ascending = [True] * len(columns)
811
829
  columns += [x[0] for x in cols_plus_inversed]
812
- ascending += [~x[1] for x in cols_plus_inversed]
830
+ ascending += [not x[1] for x in cols_plus_inversed]
813
831
  df_out = df.sort_values(
814
832
  columns, ascending=ascending, kind=kind, na_position=na_position
815
833
  )
@@ -951,7 +969,9 @@ def norm_0_to_1(df, axis=1, keep_nan=False):
951
969
  a1 = 0
952
970
  a2 = 1
953
971
  df_normed = df.sub(df.min(axis=a1), axis=a2)
954
- df_normed = df.div(df_normed.max(axis=a1), axis=a2)
972
+ assert df_normed.min().min() == 0.0
973
+ df_normed = df_normed.div(df_normed.max(axis=a1), axis=a2)
974
+ assert df_normed.max().max() == 1.0
955
975
  if not keep_nan:
956
976
  df_normed = df_normed[~pd.isnull(df_normed).any(axis=1)]
957
977
  return df_normed
@@ -1004,7 +1024,7 @@ def pca_dataframe(df, whiten=False, random_state=None, n_components=2):
1004
1024
  df_fit = pd.DataFrame(p.fit_transform(df))
1005
1025
  cols = ["1st", "2nd"]
1006
1026
  if n_components > 2:
1007
- cols.append('3rd')
1027
+ cols.append("3rd")
1008
1028
  for ii in range(3, n_components):
1009
1029
  cols.append(f"{ii+1}th")
1010
1030
  df_fit.columns = cols
@@ -1,20 +1,28 @@
1
- Metadata-Version: 2.1
1
+ Metadata-Version: 2.4
2
2
  Name: dppd
3
- Version: 0.27
3
+ Version: 0.30
4
4
  Summary: A pythonic dplyr clone
5
- Home-page: https://github.com/TyberiusPrime/dppd
6
- Author: Florian Finkernagel
7
- Author-email: finkernagel@imt.uni-marburg.de
8
- License: mit
9
- Platform: any
10
- Classifier: Development Status :: 4 - Beta
11
- Classifier: Programming Language :: Python
12
- Requires-Python: >=3.6
5
+ Author-email: Florian Finkernagel <finkernagel@imt.uni-marburg.de>
6
+ License-Expression: MIT
7
+ Project-URL: Documentation, https://dppd.readthedocs.io/en/latest/
8
+ Project-URL: Repository, https://github.com/TyberiusPrime/dppd
9
+ Requires-Python: >=3.9
13
10
  Description-Content-Type: text/markdown
14
- Provides-Extra: testing
15
- Provides-Extra: doc
16
11
  License-File: LICENSE.txt
17
12
  License-File: AUTHORS.rst
13
+ Requires-Dist: natsort
14
+ Requires-Dist: numpy
15
+ Requires-Dist: pandas>=2
16
+ Requires-Dist: wrapt
17
+ Provides-Extra: dev
18
+ Requires-Dist: build; extra == "dev"
19
+ Requires-Dist: numpydoc; extra == "dev"
20
+ Requires-Dist: plotnine; extra == "dev"
21
+ Requires-Dist: pytest; extra == "dev"
22
+ Requires-Dist: pytest-cov; extra == "dev"
23
+ Requires-Dist: sphinx; extra == "dev"
24
+ Requires-Dist: sphinx-bootstrap-theme; extra == "dev"
25
+ Dynamic: license-file
18
26
 
19
27
  # dppd
20
28
 
@@ -1,8 +1,7 @@
1
1
  AUTHORS.rst
2
2
  LICENSE.txt
3
3
  README.md
4
- setup.cfg
5
- setup.py
4
+ pyproject.toml
6
5
  src/dppd/__init__.py
7
6
  src/dppd/base.py
8
7
  src/dppd/column_spec.py
@@ -11,6 +10,10 @@ src/dppd/single_verbs.py
11
10
  src/dppd.egg-info/PKG-INFO
12
11
  src/dppd.egg-info/SOURCES.txt
13
12
  src/dppd.egg-info/dependency_links.txt
14
- src/dppd.egg-info/not-zip-safe
15
13
  src/dppd.egg-info/requires.txt
16
- src/dppd.egg-info/top_level.txt
14
+ src/dppd.egg-info/top_level.txt
15
+ tests/test_base.py
16
+ tests/test_pandas_forwards.py
17
+ tests/test_reshaping.py
18
+ tests/test_select.py
19
+ tests/test_single_verbs.py
@@ -1,17 +1,13 @@
1
- pandas>=0.22
2
- numpy
3
1
  natsort
2
+ numpy
3
+ pandas>=2
4
4
  wrapt
5
5
 
6
- [doc]
7
- sphinx
8
- sphinx-bootstrap-theme
6
+ [dev]
7
+ build
9
8
  numpydoc
10
- pandas
11
-
12
- [testing]
9
+ plotnine
13
10
  pytest
14
11
  pytest-cov
15
- plotnine
16
- pandas<2.0
17
- flake8
12
+ sphinx
13
+ sphinx-bootstrap-theme
@@ -0,0 +1,352 @@
1
+ #!/usr/bin/env python
2
+ # -*- coding: utf-8 -*-
3
+
4
+ import pytest
5
+ from dppd import dppd, register_verb
6
+ from dppd.base import register_property
7
+ import pandas as pd
8
+ import numpy as np
9
+ import pandas.testing
10
+ import wrapt
11
+ from plotnine.data import mtcars, diamonds
12
+
13
+
14
+ __author__ = "Florian Finkernagel"
15
+ __copyright__ = "Florian Finkernagel"
16
+ __license__ = "mit"
17
+
18
+ assert_series_equal = pandas.testing.assert_series_equal
19
+ assert_frame_equal = pandas.testing.assert_frame_equal
20
+
21
+ dp, X = dppd()
22
+
23
+
24
+ def test_noop():
25
+ df = pd.DataFrame({"a": list(range(10))})
26
+ actual = dp(df)
27
+ actual = actual.pd
28
+ assert isinstance(actual, pd.DataFrame)
29
+ should = df
30
+ assert_frame_equal(should, actual)
31
+
32
+
33
+ def test_non_df_result():
34
+ import dppd.base
35
+
36
+ df = pd.DataFrame({"a": list(range(10))})
37
+ shape = dp(df).head(5).shape
38
+ assert not isinstance(shape, dppd.base.DPPDAwareProxy)
39
+ assert shape == (5, 1)
40
+ real_pd = shape.pd
41
+ assert isinstance(real_pd, tuple)
42
+
43
+
44
+ def test_nested_dp_pd_calls():
45
+ df = pd.DataFrame({"a": list(range(10))})
46
+ actual = dp(df).head(5).concat(dp(df).tail(4).pd).pd
47
+ should = pd.concat([df.head(5), df.tail(4)], axis=0)
48
+ assert_frame_equal(should, actual)
49
+
50
+
51
+ def test_dp_pd_calls_nested_in_function_calls():
52
+ df = pd.DataFrame(
53
+ {"a": list(range(10)), "bb": list(range(10)), "ccc": list(range(10))}
54
+ ).set_index("a")
55
+
56
+ def shu():
57
+ return dp(df).head(1).pd
58
+
59
+ def sha():
60
+ return dp(df).tail(1).pd
61
+
62
+ should = pd.concat([df.head(1), df.tail(1), df.head(1)])
63
+ actual = dp(shu()).concat(sha()).concat(shu()).pd
64
+ assert_frame_equal(should, actual)
65
+
66
+
67
+ def test_redefining_verb_vars():
68
+ def noop(df):
69
+ return df
70
+
71
+ register_verb("test_redefining_verb_vars_noop")(noop)
72
+ with pytest.warns(UserWarning):
73
+ register_verb("test_redefining_verb_vars_noop")(noop)
74
+
75
+
76
+ def test_property_shadowed():
77
+ with pytest.warns(UserWarning):
78
+ register_property("select", pd.DataFrame)
79
+
80
+
81
+ def test_property_list_of_types():
82
+ import dppd.base
83
+
84
+ class A:
85
+ pass
86
+
87
+ class B:
88
+ pass
89
+
90
+ register_property("test_property_list_of_types", [A, B])
91
+ assert "test_property_list_of_types" in dppd.base.property_registry[A]
92
+
93
+
94
+ def test_verb_shadows_property():
95
+ def noop(df):
96
+ return df
97
+
98
+ register_property("test_verb_shadows_property_noop")
99
+
100
+ with pytest.warns(UserWarning):
101
+ register_verb("test_verb_shadows_property_noop")(noop)
102
+
103
+
104
+ def test_register_verb_raises_on_non_identifier():
105
+ with pytest.raises(TypeError):
106
+ register_verb(name="Hello world")(lambda x: x)
107
+
108
+
109
+ def test_register_verb_aliases():
110
+ import dppd.base
111
+
112
+ def shu():
113
+ pass
114
+
115
+ register_verb(["test_register_verb_aliases", "test_register_verb_aliases2"])(shu)
116
+ assert ("test_register_verb_aliases", None) in dppd.base.verb_registry
117
+ assert ("test_register_verb_aliases2", None) in dppd.base.verb_registry
118
+ assert (
119
+ dppd.base.verb_registry[("test_register_verb_aliases", None)]
120
+ is dppd.base.verb_registry[("test_register_verb_aliases2", None)]
121
+ )
122
+
123
+
124
+ def test_verb_returning_non_df():
125
+ df = pd.DataFrame({"a": [str(x) for x in (range(10))]})
126
+ register_verb("da_length")(lambda df: len(df))
127
+ assert 10 == dp(df).da_length()
128
+
129
+
130
+ def test_no_context_manager_non_df_returning_verbs():
131
+ df = pd.DataFrame({"a": [str(x) for x in (range(10))]})
132
+ dp(df)
133
+ dp().head(1)
134
+ assert len(X) == 1
135
+ assert X.shape == (1, 1)
136
+ actual = dp().pd
137
+ assert actual.shape == (1, 1)
138
+ assert actual.__len__() == 1
139
+
140
+
141
+ def test_no_attribute_no_verb_raises_attribute_error():
142
+ df = pd.DataFrame({"a": [str(x) for x in (range(10))]})
143
+ with pytest.raises(AttributeError):
144
+ dp(df).shu()
145
+
146
+
147
+ def test_no_attribute_no_verb_raises_attribute_error_context_manager():
148
+ df = pd.DataFrame({"a": [str(x) for x in (range(10))]})
149
+ with pytest.raises(AttributeError):
150
+ with dppd(df) as (dp, X):
151
+ dp.shu()
152
+
153
+
154
+ def test_dp_on_empty_stack_raises():
155
+ dp, X = dppd()
156
+ with pytest.raises(ValueError):
157
+ dp()
158
+
159
+
160
+ def test_dp_continuation():
161
+ df = pd.DataFrame(
162
+ {"a": [str(x) for x in (range(10))], "bb": 10, "ccc": list(range(20, 30))}
163
+ ).set_index("a")
164
+ dp(df).head(5)
165
+ dp().tail(1)
166
+ actual = dp().pd
167
+ should = df.iloc[4:5]
168
+ assert_frame_equal(should, actual)
169
+
170
+
171
+ def test_context_manager():
172
+ df = pd.DataFrame(
173
+ {"a": [str(x) for x in (range(10))], "bb": 10, "ccc": list(range(20, 30))}
174
+ ).set_index("a")
175
+ with dppd(df) as (d, X):
176
+ d.head(5)
177
+ d.tail(1)
178
+ should = df.iloc[4:5]
179
+ assert_frame_equal(X, should)
180
+
181
+
182
+ def test_context_manager_with_non_df_args():
183
+ df = pd.DataFrame(
184
+ {"a": [str(x) for x in (range(10))], "bb": 10, "ccc": list(range(20, 30))}
185
+ ).set_index("a")
186
+ with dppd(df) as (d, X):
187
+ d.head(5)
188
+ assert d.shape == (5, 2)
189
+ d.tail(1)
190
+ should = df.iloc[4:5]
191
+ assert_frame_equal(X, should)
192
+
193
+
194
+ def test_context_manager_totally_to_pandas():
195
+ df = pd.DataFrame(
196
+ {"a": [str(x) for x in (range(10))], "bb": 10, "ccc": list(range(20, 30))}
197
+ ).set_index("a")
198
+ with dppd(df) as (d, X):
199
+ d.head(5)
200
+ assert d.shape == (5, 2)
201
+ d.tail(1)
202
+ should = df.iloc[4:5]
203
+ assert_frame_equal(X, should)
204
+ assert isinstance(X, wrapt.ObjectProxy)
205
+ X = X.pd
206
+ assert not isinstance(X, wrapt.ObjectProxy)
207
+ assert_frame_equal(X, should)
208
+
209
+
210
+ def test_interleaved_context_managers():
211
+ with dppd(mtcars) as (dpX, X):
212
+ with dppd(diamonds) as (dpY, Y):
213
+ dpX.groupby("cyl")
214
+ dpY.filter_by(Y.cut == "Ideal")
215
+ dpX.summarize(("hp", np.mean, "mean_hp"))
216
+ dpY.summarize(("price", np.max, "max_price"))
217
+ should_X = (
218
+ mtcars.groupby("cyl")[["hp"]].agg("mean").rename(columns={"hp": "mean_hp"})
219
+ ).reset_index()
220
+ should_Y = (
221
+ pd.DataFrame(diamonds[diamonds.cut == "Ideal"].max()[["price"]])
222
+ .transpose()
223
+ .rename(columns={"price": "max_price"})
224
+ )
225
+ should_Y["max_price"] = should_Y["max_price"].astype(int)
226
+ assert_frame_equal(X, should_X)
227
+ assert_frame_equal(Y, should_Y)
228
+
229
+
230
+ def test_context_manager_chain():
231
+ with dppd(mtcars) as (dp, X):
232
+ dp.mutate(kw=X.hp * 0.7457)
233
+ with dppd(X) as (dp, X):
234
+ dp.mutate(watt=X.kw * 1000)
235
+ assert "watt" in X.columns
236
+
237
+
238
+ def test_mixing_context_manager_and_dp():
239
+ with dppd(mtcars) as (dpY, Y):
240
+ dpY.sort_values("hp")
241
+ dp(diamonds).filter_by(X.cut == "ideal")
242
+ dpY.filter_by(Y.cyl.isin([4, 6]))
243
+ actual_diamonds = dp().sort_values("price").head().pd
244
+ actual_mtcars_full = dpY.pd
245
+ dpY.head()
246
+ actual_mtcars = dpY.pd
247
+ should_diamonds = diamonds[diamonds.cut == "ideal"].sort_values("price").head()
248
+ should_mtcars = mtcars.sort_values("hp")
249
+ should_mtcars_full = should_mtcars[should_mtcars["cyl"].isin([4, 6])]
250
+ should_mtcars = should_mtcars_full.head()
251
+ assert_frame_equal(should_diamonds, actual_diamonds)
252
+ assert_frame_equal(should_mtcars, actual_mtcars)
253
+ assert_frame_equal(should_mtcars_full, actual_mtcars_full)
254
+
255
+
256
+ def test_dppd_raises_on_non_dataframe():
257
+ with pytest.raises(ValueError):
258
+ dp(5)
259
+
260
+
261
+ def test_straight_dp_raises():
262
+ dp, X = dppd()
263
+ with pytest.raises(ValueError):
264
+ dp.select(["hp", "cyl"])
265
+
266
+ with pytest.raises(ValueError):
267
+ dp.loc[5]
268
+
269
+
270
+ def test_stacking():
271
+ dp, X = dppd()
272
+ dp(mtcars).select(["name", "hp", "cyl"])
273
+ b = dp(mtcars).select("hp").pd
274
+ assert_frame_equal(b, mtcars[["hp"]])
275
+ assert_frame_equal(X, mtcars[["name", "hp", "cyl"]])
276
+ c = dp.pd
277
+ assert_frame_equal(c, mtcars[["name", "hp", "cyl"]])
278
+ assert X == None # noqa:E711 since it's the proxy, is will fail
279
+
280
+
281
+ def test_forking():
282
+ dp, X = dppd()
283
+ a = dp(mtcars).select(["name", "hp", "cyl"])
284
+ b = dp.unselect("hp").select(X.name).head().pd
285
+ with pytest.raises(AttributeError):
286
+ c = a.select(X.hp).head().pd
287
+ c = dp(a).select(X.hp).head().pd
288
+ assert_series_equal(c["hp"], mtcars["hp"].head())
289
+ assert_series_equal(b["name"], mtcars["name"].head())
290
+ assert X == None # noqa:E711 since it's the proxy, is will fail
291
+
292
+
293
+ def test_forking_context_manager():
294
+ with dppd(mtcars) as (dp, X):
295
+ a = dp.select(["name", "hp", "cyl"])
296
+ b = dp.select("name").head().pd
297
+ c = a.select("hp").head().pd
298
+ dp.head()
299
+ assert_series_equal(c["hp"], mtcars["hp"].head())
300
+ assert_series_equal(b["name"], mtcars["name"].head())
301
+ assert_frame_equal(X, mtcars[["hp"]].head())
302
+
303
+
304
+ def test_dp_on_dp():
305
+ import wrapt
306
+
307
+ a = dp(mtcars)
308
+ b = dp(a)
309
+ assert not isinstance(a.df, wrapt.ObjectProxy)
310
+ assert not isinstance(b.df, wrapt.ObjectProxy)
311
+
312
+
313
+ def test_descend_on_None_raises():
314
+ with pytest.raises(ValueError):
315
+ dp(mtcars)._descend(None)
316
+
317
+
318
+ def test_series_methods():
319
+ actual = dp(mtcars).sum().to_frame().pd
320
+ should = mtcars.sum().to_frame()
321
+ assert_frame_equal(should, actual)
322
+
323
+
324
+ def test_group_on_series_raises():
325
+ with pytest.raises(KeyError):
326
+ dp(mtcars).sum().groupby("no_such_columns")
327
+
328
+
329
+ def test_dir():
330
+ from dppd import base
331
+
332
+ dp, X = dppd()
333
+ actual = set(dir(dp(mtcars)))
334
+ should_min = set(base.property_registry[pd.DataFrame])
335
+ delta = should_min.difference(actual)
336
+ print(sorted(actual))
337
+ print(sorted(delta))
338
+ assert not len(delta)
339
+ assert len(actual) > len(should_min)
340
+
341
+
342
+ def test_version_is_correct():
343
+ from pathlib import Path
344
+ try:
345
+ import tomllib
346
+ import dppd as org_dppd
347
+
348
+ c = tomllib.load(open(Path(__file__).parent.parent / "pyproject.toml", "rb"))
349
+ version = c["project"]["version"]
350
+ assert version == org_dppd.__version__
351
+ except ImportError:
352
+ pass # python < 3.11