dppd 0.26__tar.gz → 0.30__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dppd-0.26 → dppd-0.30}/LICENSE.txt +1 -1
- {dppd-0.26/src/dppd.egg-info → dppd-0.30}/PKG-INFO +20 -12
- dppd-0.30/pyproject.toml +57 -0
- dppd-0.30/setup.cfg +4 -0
- {dppd-0.26 → dppd-0.30}/src/dppd/__init__.py +1 -1
- {dppd-0.26 → dppd-0.30}/src/dppd/single_verbs.py +43 -17
- {dppd-0.26 → dppd-0.30/src/dppd.egg-info}/PKG-INFO +20 -12
- {dppd-0.26 → dppd-0.30}/src/dppd.egg-info/SOURCES.txt +7 -4
- {dppd-0.26 → dppd-0.30}/src/dppd.egg-info/requires.txt +7 -11
- dppd-0.30/tests/test_base.py +352 -0
- dppd-0.30/tests/test_pandas_forwards.py +169 -0
- dppd-0.30/tests/test_reshaping.py +173 -0
- dppd-0.30/tests/test_select.py +405 -0
- dppd-0.30/tests/test_single_verbs.py +1031 -0
- dppd-0.26/setup.cfg +0 -94
- dppd-0.26/setup.py +0 -21
- dppd-0.26/src/dppd.egg-info/not-zip-safe +0 -1
- {dppd-0.26 → dppd-0.30}/AUTHORS.rst +0 -0
- {dppd-0.26 → dppd-0.30}/README.md +0 -0
- {dppd-0.26 → dppd-0.30}/src/dppd/base.py +0 -0
- {dppd-0.26 → dppd-0.30}/src/dppd/column_spec.py +0 -0
- {dppd-0.26 → dppd-0.30}/src/dppd/non_df_verbs.py +0 -0
- {dppd-0.26 → dppd-0.30}/src/dppd.egg-info/dependency_links.txt +0 -0
- {dppd-0.26 → dppd-0.30}/src/dppd.egg-info/top_level.txt +0 -0
|
@@ -1,20 +1,28 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: dppd
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.30
|
|
4
4
|
Summary: A pythonic dplyr clone
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
Classifier: Development Status :: 4 - Beta
|
|
11
|
-
Classifier: Programming Language :: Python
|
|
12
|
-
Requires-Python: >=3.6
|
|
5
|
+
Author-email: Florian Finkernagel <finkernagel@imt.uni-marburg.de>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Documentation, https://dppd.readthedocs.io/en/latest/
|
|
8
|
+
Project-URL: Repository, https://github.com/TyberiusPrime/dppd
|
|
9
|
+
Requires-Python: >=3.9
|
|
13
10
|
Description-Content-Type: text/markdown
|
|
14
|
-
Provides-Extra: testing
|
|
15
|
-
Provides-Extra: doc
|
|
16
11
|
License-File: LICENSE.txt
|
|
17
12
|
License-File: AUTHORS.rst
|
|
13
|
+
Requires-Dist: natsort
|
|
14
|
+
Requires-Dist: numpy
|
|
15
|
+
Requires-Dist: pandas>=2
|
|
16
|
+
Requires-Dist: wrapt
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: build; extra == "dev"
|
|
19
|
+
Requires-Dist: numpydoc; extra == "dev"
|
|
20
|
+
Requires-Dist: plotnine; extra == "dev"
|
|
21
|
+
Requires-Dist: pytest; extra == "dev"
|
|
22
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
23
|
+
Requires-Dist: sphinx; extra == "dev"
|
|
24
|
+
Requires-Dist: sphinx-bootstrap-theme; extra == "dev"
|
|
25
|
+
Dynamic: license-file
|
|
18
26
|
|
|
19
27
|
# dppd
|
|
20
28
|
|
dppd-0.30/pyproject.toml
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "dppd"
|
|
3
|
+
version = "0.30"
|
|
4
|
+
description = "A pythonic dplyr clone"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.9"
|
|
7
|
+
authors = [
|
|
8
|
+
{name = "Florian Finkernagel", email = "finkernagel@imt.uni-marburg.de"}
|
|
9
|
+
]
|
|
10
|
+
license="MIT"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"natsort",
|
|
13
|
+
"numpy",
|
|
14
|
+
"pandas>=2",
|
|
15
|
+
"wrapt",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[project.urls]
|
|
19
|
+
Documentation = "https://dppd.readthedocs.io/en/latest/"
|
|
20
|
+
Repository = "https://github.com/TyberiusPrime/dppd"
|
|
21
|
+
|
|
22
|
+
[build-system]
|
|
23
|
+
requires = ["setuptools >= 61.0"]
|
|
24
|
+
build-backend = "setuptools.build_meta"
|
|
25
|
+
|
|
26
|
+
[project.optional-dependencies]
|
|
27
|
+
dev = [
|
|
28
|
+
"build",
|
|
29
|
+
"numpydoc",
|
|
30
|
+
"plotnine",
|
|
31
|
+
"pytest",
|
|
32
|
+
"pytest-cov",
|
|
33
|
+
"sphinx",
|
|
34
|
+
"sphinx-bootstrap-theme",
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
[tool.pytest.ini_options]
|
|
38
|
+
# Options for py.test:
|
|
39
|
+
# Specify command line options as you would do when invoking py.test directly.
|
|
40
|
+
# e.g. --cov-report html (or xml) for html/xml output or --junitxml junit.xml
|
|
41
|
+
# in order to write a coverage file that can be read by Jenkins.
|
|
42
|
+
addopts = """
|
|
43
|
+
--cov dppd --cov-report term-missing
|
|
44
|
+
--verbose
|
|
45
|
+
"""
|
|
46
|
+
norecursedirs = [
|
|
47
|
+
"dist",
|
|
48
|
+
"build",
|
|
49
|
+
".tox",
|
|
50
|
+
]
|
|
51
|
+
testpaths = "tests"
|
|
52
|
+
filterwarnings = [
|
|
53
|
+
"ignore:::statsmodels.base.wrapper:100",
|
|
54
|
+
"ignore:::patsy.constraint:13",
|
|
55
|
+
"ignore:::matplotlib.backends.backend_wx:",
|
|
56
|
+
]
|
|
57
|
+
|
dppd-0.30/setup.cfg
ADDED
|
@@ -4,6 +4,6 @@ from .base import dppd, register_verb, register_type_methods_as_verbs
|
|
|
4
4
|
from . import single_verbs # noqa:F401
|
|
5
5
|
from . import non_df_verbs # noqa:F401
|
|
6
6
|
|
|
7
|
-
__version__ = "0.
|
|
7
|
+
__version__ = "0.30"
|
|
8
8
|
|
|
9
9
|
__all_ = [dppd, register_verb, register_type_methods_as_verbs, __version__]
|
|
@@ -23,14 +23,20 @@ register_type_methods_as_verbs(SeriesGroupBy, [])
|
|
|
23
23
|
|
|
24
24
|
|
|
25
25
|
def group_variables(grp):
|
|
26
|
-
|
|
26
|
+
if hasattr(grp, "_grouper"):
|
|
27
|
+
return grp._grouper.names
|
|
28
|
+
else:
|
|
29
|
+
return grp.grouper.names
|
|
27
30
|
|
|
28
31
|
|
|
29
32
|
def group_extract_params(grp):
|
|
30
|
-
if grp.axis != 0:
|
|
33
|
+
if hasattr(grp, "axis") and grp.axis != 0:
|
|
31
34
|
raise ValueError(f"Verbs assume that groupby is on axis=0, was {grp.axis}")
|
|
32
35
|
res = {"by": group_variables(grp)}
|
|
33
|
-
|
|
36
|
+
attrs = ["squeeze", "level", "as_index", "sort", "group_keys", "observed"]
|
|
37
|
+
if pd.__version__ < "2.1.":
|
|
38
|
+
attrs.append("axis")
|
|
39
|
+
for k in attrs:
|
|
34
40
|
if hasattr(grp, k):
|
|
35
41
|
res[k] = getattr(grp, k)
|
|
36
42
|
else: # pragma: no cover
|
|
@@ -56,7 +62,10 @@ def _print(obj):
|
|
|
56
62
|
@register_verb(name="debug", types=None)
|
|
57
63
|
def _debug(obj, k=5):
|
|
58
64
|
d = obj.iloc[np.r_[0:k, -k:0]]
|
|
59
|
-
|
|
65
|
+
try:
|
|
66
|
+
display(d) # noqa: F821 - Jupyter only, needs to import.
|
|
67
|
+
except NameError:
|
|
68
|
+
print(d)
|
|
60
69
|
return obj
|
|
61
70
|
|
|
62
71
|
|
|
@@ -354,9 +363,12 @@ def mutate_DataFrameGroupBy(grp, **kwargs):
|
|
|
354
363
|
try:
|
|
355
364
|
r = v[group_key]
|
|
356
365
|
except KeyError:
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
366
|
+
try:
|
|
367
|
+
r = v[group_key,]
|
|
368
|
+
except KeyError:
|
|
369
|
+
raise KeyError(
|
|
370
|
+
f"Grouped mutate results did not contain data for {group_key}. Keys where {v.keys()}"
|
|
371
|
+
)
|
|
360
372
|
r = pd.Series(r, index=sub_index)
|
|
361
373
|
parts.append(r)
|
|
362
374
|
parts = pd.concat(parts)
|
|
@@ -451,7 +463,10 @@ def filter_by(obj, filter_arg):
|
|
|
451
463
|
for idx, sub_df in df.groupby(groups):
|
|
452
464
|
# if not idx in filter_arg and not isinstance(tuple(idx)):
|
|
453
465
|
# idx = (idx,)
|
|
454
|
-
|
|
466
|
+
try:
|
|
467
|
+
keep = filter_arg[idx]
|
|
468
|
+
except KeyError:
|
|
469
|
+
keep = filter_arg[idx[0]]
|
|
455
470
|
parts.append(sub_df[keep])
|
|
456
471
|
result = pd.concat(parts, axis=0)
|
|
457
472
|
elif isinstance(filter_arg, str):
|
|
@@ -543,7 +558,7 @@ def summarize(obj, *args):
|
|
|
543
558
|
result = result[groups + [x for x in result.columns if x not in groups]]
|
|
544
559
|
# restore category to categories
|
|
545
560
|
for g in groups:
|
|
546
|
-
if
|
|
561
|
+
if isinstance(df.dtypes[g], pd.CategoricalDtype):
|
|
547
562
|
result = result.assign(
|
|
548
563
|
**{
|
|
549
564
|
g: pd.Categorical(
|
|
@@ -603,7 +618,7 @@ def do(obj, func, *args, **kwargs):
|
|
|
603
618
|
result = result[groups + [x for x in result.columns if x not in groups]]
|
|
604
619
|
# restore category to categories
|
|
605
620
|
for g in groups:
|
|
606
|
-
if
|
|
621
|
+
if isinstance(df.dtypes[g], pd.CategoricalDtype):
|
|
607
622
|
result = result.assign(
|
|
608
623
|
**{
|
|
609
624
|
g: pd.Categorical(
|
|
@@ -767,7 +782,10 @@ def seperate(df, column, new_names, sep=".", remove=False):
|
|
|
767
782
|
|
|
768
783
|
@register_verb("print", types=DataFrameGroupBy)
|
|
769
784
|
def print_DataFrameGroupBy(grps):
|
|
770
|
-
|
|
785
|
+
if hasattr(grps, "_grouper"):
|
|
786
|
+
print("groups: %s" % (grps._grouper.names))
|
|
787
|
+
else:
|
|
788
|
+
print("groups: %s" % (grps.grouper.names))
|
|
771
789
|
print(grps._selected_obj)
|
|
772
790
|
return grps
|
|
773
791
|
|
|
@@ -809,7 +827,7 @@ def arrange_DataFrameGroupBy(grp, column_spec, kind="quicksort", na_position="la
|
|
|
809
827
|
columns = grp_params["by"].copy()
|
|
810
828
|
ascending = [True] * len(columns)
|
|
811
829
|
columns += [x[0] for x in cols_plus_inversed]
|
|
812
|
-
ascending += [
|
|
830
|
+
ascending += [not x[1] for x in cols_plus_inversed]
|
|
813
831
|
df_out = df.sort_values(
|
|
814
832
|
columns, ascending=ascending, kind=kind, na_position=na_position
|
|
815
833
|
)
|
|
@@ -939,7 +957,7 @@ def to_frame_dict(d, **kwargs):
|
|
|
939
957
|
|
|
940
958
|
|
|
941
959
|
@register_verb("norm_0_to_1", types=pd.DataFrame)
|
|
942
|
-
def norm_0_to_1(df, axis=1):
|
|
960
|
+
def norm_0_to_1(df, axis=1, keep_nan=False):
|
|
943
961
|
"""Normalize a (numeric) data frame so that
|
|
944
962
|
it goes from 0 to 1 in each row (axis=1) or column (axis=0)
|
|
945
963
|
Usefully for PCA, correlation, etc. because then
|
|
@@ -951,7 +969,11 @@ def norm_0_to_1(df, axis=1):
|
|
|
951
969
|
a1 = 0
|
|
952
970
|
a2 = 1
|
|
953
971
|
df_normed = df.sub(df.min(axis=a1), axis=a2)
|
|
954
|
-
df_normed
|
|
972
|
+
assert df_normed.min().min() == 0.0
|
|
973
|
+
df_normed = df_normed.div(df_normed.max(axis=a1), axis=a2)
|
|
974
|
+
assert df_normed.max().max() == 1.0
|
|
975
|
+
if not keep_nan:
|
|
976
|
+
df_normed = df_normed[~pd.isnull(df_normed).any(axis=1)]
|
|
955
977
|
return df_normed
|
|
956
978
|
|
|
957
979
|
|
|
@@ -1000,7 +1022,12 @@ def pca_dataframe(df, whiten=False, random_state=None, n_components=2):
|
|
|
1000
1022
|
|
|
1001
1023
|
p = PCA(n_components=n_components, whiten=whiten, random_state=random_state)
|
|
1002
1024
|
df_fit = pd.DataFrame(p.fit_transform(df))
|
|
1003
|
-
|
|
1025
|
+
cols = ["1st", "2nd"]
|
|
1026
|
+
if n_components > 2:
|
|
1027
|
+
cols.append("3rd")
|
|
1028
|
+
for ii in range(3, n_components):
|
|
1029
|
+
cols.append(f"{ii+1}th")
|
|
1030
|
+
df_fit.columns = cols
|
|
1004
1031
|
df_fit.index = df.index
|
|
1005
1032
|
df_fit.index.name = "sample"
|
|
1006
1033
|
df_fit = df_fit.reset_index()
|
|
@@ -1012,7 +1039,6 @@ def pca_dataframe(df, whiten=False, random_state=None, n_components=2):
|
|
|
1012
1039
|
|
|
1013
1040
|
@register_verb("insert", types=pd.DataFrame, ignore_redefine=True)
|
|
1014
1041
|
def insert_return_self(df, loc, column, value, **kwargs):
|
|
1015
|
-
"""DataFrame.insert, but return self.
|
|
1016
|
-
"""
|
|
1042
|
+
"""DataFrame.insert, but return self."""
|
|
1017
1043
|
df.insert(loc, column, value, **kwargs)
|
|
1018
1044
|
return df
|
|
@@ -1,20 +1,28 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: dppd
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.30
|
|
4
4
|
Summary: A pythonic dplyr clone
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
Classifier: Development Status :: 4 - Beta
|
|
11
|
-
Classifier: Programming Language :: Python
|
|
12
|
-
Requires-Python: >=3.6
|
|
5
|
+
Author-email: Florian Finkernagel <finkernagel@imt.uni-marburg.de>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Documentation, https://dppd.readthedocs.io/en/latest/
|
|
8
|
+
Project-URL: Repository, https://github.com/TyberiusPrime/dppd
|
|
9
|
+
Requires-Python: >=3.9
|
|
13
10
|
Description-Content-Type: text/markdown
|
|
14
|
-
Provides-Extra: testing
|
|
15
|
-
Provides-Extra: doc
|
|
16
11
|
License-File: LICENSE.txt
|
|
17
12
|
License-File: AUTHORS.rst
|
|
13
|
+
Requires-Dist: natsort
|
|
14
|
+
Requires-Dist: numpy
|
|
15
|
+
Requires-Dist: pandas>=2
|
|
16
|
+
Requires-Dist: wrapt
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: build; extra == "dev"
|
|
19
|
+
Requires-Dist: numpydoc; extra == "dev"
|
|
20
|
+
Requires-Dist: plotnine; extra == "dev"
|
|
21
|
+
Requires-Dist: pytest; extra == "dev"
|
|
22
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
23
|
+
Requires-Dist: sphinx; extra == "dev"
|
|
24
|
+
Requires-Dist: sphinx-bootstrap-theme; extra == "dev"
|
|
25
|
+
Dynamic: license-file
|
|
18
26
|
|
|
19
27
|
# dppd
|
|
20
28
|
|
|
@@ -1,8 +1,7 @@
|
|
|
1
1
|
AUTHORS.rst
|
|
2
2
|
LICENSE.txt
|
|
3
3
|
README.md
|
|
4
|
-
|
|
5
|
-
setup.py
|
|
4
|
+
pyproject.toml
|
|
6
5
|
src/dppd/__init__.py
|
|
7
6
|
src/dppd/base.py
|
|
8
7
|
src/dppd/column_spec.py
|
|
@@ -11,6 +10,10 @@ src/dppd/single_verbs.py
|
|
|
11
10
|
src/dppd.egg-info/PKG-INFO
|
|
12
11
|
src/dppd.egg-info/SOURCES.txt
|
|
13
12
|
src/dppd.egg-info/dependency_links.txt
|
|
14
|
-
src/dppd.egg-info/not-zip-safe
|
|
15
13
|
src/dppd.egg-info/requires.txt
|
|
16
|
-
src/dppd.egg-info/top_level.txt
|
|
14
|
+
src/dppd.egg-info/top_level.txt
|
|
15
|
+
tests/test_base.py
|
|
16
|
+
tests/test_pandas_forwards.py
|
|
17
|
+
tests/test_reshaping.py
|
|
18
|
+
tests/test_select.py
|
|
19
|
+
tests/test_single_verbs.py
|
|
@@ -1,17 +1,13 @@
|
|
|
1
|
-
pandas>=0.22
|
|
2
|
-
numpy
|
|
3
1
|
natsort
|
|
2
|
+
numpy
|
|
3
|
+
pandas>=2
|
|
4
4
|
wrapt
|
|
5
5
|
|
|
6
|
-
[
|
|
7
|
-
|
|
8
|
-
sphinx-bootstrap-theme
|
|
6
|
+
[dev]
|
|
7
|
+
build
|
|
9
8
|
numpydoc
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
[testing]
|
|
9
|
+
plotnine
|
|
13
10
|
pytest
|
|
14
11
|
pytest-cov
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
flake8
|
|
12
|
+
sphinx
|
|
13
|
+
sphinx-bootstrap-theme
|