dppd 0.26__tar.gz → 0.30__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  The MIT License (MIT)
2
2
 
3
- Copyright (c) 2018 Florian Finkernagel
3
+ Copyright (c) 2025 Florian Finkernagel
4
4
 
5
5
  Permission is hereby granted, free of charge, to any person obtaining a copy
6
6
  of this software and associated documentation files (the "Software"), to deal
@@ -1,20 +1,28 @@
1
- Metadata-Version: 2.1
1
+ Metadata-Version: 2.4
2
2
  Name: dppd
3
- Version: 0.26
3
+ Version: 0.30
4
4
  Summary: A pythonic dplyr clone
5
- Home-page: https://github.com/TyberiusPrime/dppd
6
- Author: Florian Finkernagel
7
- Author-email: finkernagel@imt.uni-marburg.de
8
- License: mit
9
- Platform: any
10
- Classifier: Development Status :: 4 - Beta
11
- Classifier: Programming Language :: Python
12
- Requires-Python: >=3.6
5
+ Author-email: Florian Finkernagel <finkernagel@imt.uni-marburg.de>
6
+ License-Expression: MIT
7
+ Project-URL: Documentation, https://dppd.readthedocs.io/en/latest/
8
+ Project-URL: Repository, https://github.com/TyberiusPrime/dppd
9
+ Requires-Python: >=3.9
13
10
  Description-Content-Type: text/markdown
14
- Provides-Extra: testing
15
- Provides-Extra: doc
16
11
  License-File: LICENSE.txt
17
12
  License-File: AUTHORS.rst
13
+ Requires-Dist: natsort
14
+ Requires-Dist: numpy
15
+ Requires-Dist: pandas>=2
16
+ Requires-Dist: wrapt
17
+ Provides-Extra: dev
18
+ Requires-Dist: build; extra == "dev"
19
+ Requires-Dist: numpydoc; extra == "dev"
20
+ Requires-Dist: plotnine; extra == "dev"
21
+ Requires-Dist: pytest; extra == "dev"
22
+ Requires-Dist: pytest-cov; extra == "dev"
23
+ Requires-Dist: sphinx; extra == "dev"
24
+ Requires-Dist: sphinx-bootstrap-theme; extra == "dev"
25
+ Dynamic: license-file
18
26
 
19
27
  # dppd
20
28
 
@@ -0,0 +1,57 @@
1
+ [project]
2
+ name = "dppd"
3
+ version = "0.30"
4
+ description = "A pythonic dplyr clone"
5
+ readme = "README.md"
6
+ requires-python = ">=3.9"
7
+ authors = [
8
+ {name = "Florian Finkernagel", email = "finkernagel@imt.uni-marburg.de"}
9
+ ]
10
+ license="MIT"
11
+ dependencies = [
12
+ "natsort",
13
+ "numpy",
14
+ "pandas>=2",
15
+ "wrapt",
16
+ ]
17
+
18
+ [project.urls]
19
+ Documentation = "https://dppd.readthedocs.io/en/latest/"
20
+ Repository = "https://github.com/TyberiusPrime/dppd"
21
+
22
+ [build-system]
23
+ requires = ["setuptools >= 61.0"]
24
+ build-backend = "setuptools.build_meta"
25
+
26
+ [project.optional-dependencies]
27
+ dev = [
28
+ "build",
29
+ "numpydoc",
30
+ "plotnine",
31
+ "pytest",
32
+ "pytest-cov",
33
+ "sphinx",
34
+ "sphinx-bootstrap-theme",
35
+ ]
36
+
37
+ [tool.pytest.ini_options]
38
+ # Options for py.test:
39
+ # Specify command line options as you would do when invoking py.test directly.
40
+ # e.g. --cov-report html (or xml) for html/xml output or --junitxml junit.xml
41
+ # in order to write a coverage file that can be read by Jenkins.
42
+ addopts = """
43
+ --cov dppd --cov-report term-missing
44
+ --verbose
45
+ """
46
+ norecursedirs = [
47
+ "dist",
48
+ "build",
49
+ ".tox",
50
+ ]
51
+ testpaths = "tests"
52
+ filterwarnings = [
53
+ "ignore:::statsmodels.base.wrapper:100",
54
+ "ignore:::patsy.constraint:13",
55
+ "ignore:::matplotlib.backends.backend_wx:",
56
+ ]
57
+
dppd-0.30/setup.cfg ADDED
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -4,6 +4,6 @@ from .base import dppd, register_verb, register_type_methods_as_verbs
4
4
  from . import single_verbs # noqa:F401
5
5
  from . import non_df_verbs # noqa:F401
6
6
 
7
- __version__ = "0.26"
7
+ __version__ = "0.30"
8
8
 
9
9
  __all_ = [dppd, register_verb, register_type_methods_as_verbs, __version__]
@@ -23,14 +23,20 @@ register_type_methods_as_verbs(SeriesGroupBy, [])
23
23
 
24
24
 
25
25
  def group_variables(grp):
26
- return grp.grouper.names
26
+ if hasattr(grp, "_grouper"):
27
+ return grp._grouper.names
28
+ else:
29
+ return grp.grouper.names
27
30
 
28
31
 
29
32
  def group_extract_params(grp):
30
- if grp.axis != 0:
33
+ if hasattr(grp, "axis") and grp.axis != 0:
31
34
  raise ValueError(f"Verbs assume that groupby is on axis=0, was {grp.axis}")
32
35
  res = {"by": group_variables(grp)}
33
- for k in ["squeeze", "axis", "level", "as_index", "sort", "group_keys", "observed"]:
36
+ attrs = ["squeeze", "level", "as_index", "sort", "group_keys", "observed"]
37
+ if pd.__version__ < "2.1.":
38
+ attrs.append("axis")
39
+ for k in attrs:
34
40
  if hasattr(grp, k):
35
41
  res[k] = getattr(grp, k)
36
42
  else: # pragma: no cover
@@ -56,7 +62,10 @@ def _print(obj):
56
62
  @register_verb(name="debug", types=None)
57
63
  def _debug(obj, k=5):
58
64
  d = obj.iloc[np.r_[0:k, -k:0]]
59
- print(d)
65
+ try:
66
+ display(d) # noqa: F821 - Jupyter only, needs to import.
67
+ except NameError:
68
+ print(d)
60
69
  return obj
61
70
 
62
71
 
@@ -354,9 +363,12 @@ def mutate_DataFrameGroupBy(grp, **kwargs):
354
363
  try:
355
364
  r = v[group_key]
356
365
  except KeyError:
357
- raise KeyError(
358
- f"Grouped mutate results did not contain data for {group_key}"
359
- )
366
+ try:
367
+ r = v[group_key,]
368
+ except KeyError:
369
+ raise KeyError(
370
+ f"Grouped mutate results did not contain data for {group_key}. Keys where {v.keys()}"
371
+ )
360
372
  r = pd.Series(r, index=sub_index)
361
373
  parts.append(r)
362
374
  parts = pd.concat(parts)
@@ -451,7 +463,10 @@ def filter_by(obj, filter_arg):
451
463
  for idx, sub_df in df.groupby(groups):
452
464
  # if not idx in filter_arg and not isinstance(tuple(idx)):
453
465
  # idx = (idx,)
454
- keep = filter_arg[idx]
466
+ try:
467
+ keep = filter_arg[idx]
468
+ except KeyError:
469
+ keep = filter_arg[idx[0]]
455
470
  parts.append(sub_df[keep])
456
471
  result = pd.concat(parts, axis=0)
457
472
  elif isinstance(filter_arg, str):
@@ -543,7 +558,7 @@ def summarize(obj, *args):
543
558
  result = result[groups + [x for x in result.columns if x not in groups]]
544
559
  # restore category to categories
545
560
  for g in groups:
546
- if pd.api.types.is_categorical_dtype(df[g]):
561
+ if isinstance(df.dtypes[g], pd.CategoricalDtype):
547
562
  result = result.assign(
548
563
  **{
549
564
  g: pd.Categorical(
@@ -603,7 +618,7 @@ def do(obj, func, *args, **kwargs):
603
618
  result = result[groups + [x for x in result.columns if x not in groups]]
604
619
  # restore category to categories
605
620
  for g in groups:
606
- if pd.api.types.is_categorical_dtype(df[g]):
621
+ if isinstance(df.dtypes[g], pd.CategoricalDtype):
607
622
  result = result.assign(
608
623
  **{
609
624
  g: pd.Categorical(
@@ -767,7 +782,10 @@ def seperate(df, column, new_names, sep=".", remove=False):
767
782
 
768
783
  @register_verb("print", types=DataFrameGroupBy)
769
784
  def print_DataFrameGroupBy(grps):
770
- print("groups: %s" % (grps.grouper.names))
785
+ if hasattr(grps, "_grouper"):
786
+ print("groups: %s" % (grps._grouper.names))
787
+ else:
788
+ print("groups: %s" % (grps.grouper.names))
771
789
  print(grps._selected_obj)
772
790
  return grps
773
791
 
@@ -809,7 +827,7 @@ def arrange_DataFrameGroupBy(grp, column_spec, kind="quicksort", na_position="la
809
827
  columns = grp_params["by"].copy()
810
828
  ascending = [True] * len(columns)
811
829
  columns += [x[0] for x in cols_plus_inversed]
812
- ascending += [~x[1] for x in cols_plus_inversed]
830
+ ascending += [not x[1] for x in cols_plus_inversed]
813
831
  df_out = df.sort_values(
814
832
  columns, ascending=ascending, kind=kind, na_position=na_position
815
833
  )
@@ -939,7 +957,7 @@ def to_frame_dict(d, **kwargs):
939
957
 
940
958
 
941
959
  @register_verb("norm_0_to_1", types=pd.DataFrame)
942
- def norm_0_to_1(df, axis=1):
960
+ def norm_0_to_1(df, axis=1, keep_nan=False):
943
961
  """Normalize a (numeric) data frame so that
944
962
  it goes from 0 to 1 in each row (axis=1) or column (axis=0)
945
963
  Usefully for PCA, correlation, etc. because then
@@ -951,7 +969,11 @@ def norm_0_to_1(df, axis=1):
951
969
  a1 = 0
952
970
  a2 = 1
953
971
  df_normed = df.sub(df.min(axis=a1), axis=a2)
954
- df_normed = df.div(df.max(axis=a1), axis=a2)
972
+ assert df_normed.min().min() == 0.0
973
+ df_normed = df_normed.div(df_normed.max(axis=a1), axis=a2)
974
+ assert df_normed.max().max() == 1.0
975
+ if not keep_nan:
976
+ df_normed = df_normed[~pd.isnull(df_normed).any(axis=1)]
955
977
  return df_normed
956
978
 
957
979
 
@@ -1000,7 +1022,12 @@ def pca_dataframe(df, whiten=False, random_state=None, n_components=2):
1000
1022
 
1001
1023
  p = PCA(n_components=n_components, whiten=whiten, random_state=random_state)
1002
1024
  df_fit = pd.DataFrame(p.fit_transform(df))
1003
- df_fit.columns = ["1st", "2nd"]
1025
+ cols = ["1st", "2nd"]
1026
+ if n_components > 2:
1027
+ cols.append("3rd")
1028
+ for ii in range(3, n_components):
1029
+ cols.append(f"{ii+1}th")
1030
+ df_fit.columns = cols
1004
1031
  df_fit.index = df.index
1005
1032
  df_fit.index.name = "sample"
1006
1033
  df_fit = df_fit.reset_index()
@@ -1012,7 +1039,6 @@ def pca_dataframe(df, whiten=False, random_state=None, n_components=2):
1012
1039
 
1013
1040
  @register_verb("insert", types=pd.DataFrame, ignore_redefine=True)
1014
1041
  def insert_return_self(df, loc, column, value, **kwargs):
1015
- """DataFrame.insert, but return self.
1016
- """
1042
+ """DataFrame.insert, but return self."""
1017
1043
  df.insert(loc, column, value, **kwargs)
1018
1044
  return df
@@ -1,20 +1,28 @@
1
- Metadata-Version: 2.1
1
+ Metadata-Version: 2.4
2
2
  Name: dppd
3
- Version: 0.26
3
+ Version: 0.30
4
4
  Summary: A pythonic dplyr clone
5
- Home-page: https://github.com/TyberiusPrime/dppd
6
- Author: Florian Finkernagel
7
- Author-email: finkernagel@imt.uni-marburg.de
8
- License: mit
9
- Platform: any
10
- Classifier: Development Status :: 4 - Beta
11
- Classifier: Programming Language :: Python
12
- Requires-Python: >=3.6
5
+ Author-email: Florian Finkernagel <finkernagel@imt.uni-marburg.de>
6
+ License-Expression: MIT
7
+ Project-URL: Documentation, https://dppd.readthedocs.io/en/latest/
8
+ Project-URL: Repository, https://github.com/TyberiusPrime/dppd
9
+ Requires-Python: >=3.9
13
10
  Description-Content-Type: text/markdown
14
- Provides-Extra: testing
15
- Provides-Extra: doc
16
11
  License-File: LICENSE.txt
17
12
  License-File: AUTHORS.rst
13
+ Requires-Dist: natsort
14
+ Requires-Dist: numpy
15
+ Requires-Dist: pandas>=2
16
+ Requires-Dist: wrapt
17
+ Provides-Extra: dev
18
+ Requires-Dist: build; extra == "dev"
19
+ Requires-Dist: numpydoc; extra == "dev"
20
+ Requires-Dist: plotnine; extra == "dev"
21
+ Requires-Dist: pytest; extra == "dev"
22
+ Requires-Dist: pytest-cov; extra == "dev"
23
+ Requires-Dist: sphinx; extra == "dev"
24
+ Requires-Dist: sphinx-bootstrap-theme; extra == "dev"
25
+ Dynamic: license-file
18
26
 
19
27
  # dppd
20
28
 
@@ -1,8 +1,7 @@
1
1
  AUTHORS.rst
2
2
  LICENSE.txt
3
3
  README.md
4
- setup.cfg
5
- setup.py
4
+ pyproject.toml
6
5
  src/dppd/__init__.py
7
6
  src/dppd/base.py
8
7
  src/dppd/column_spec.py
@@ -11,6 +10,10 @@ src/dppd/single_verbs.py
11
10
  src/dppd.egg-info/PKG-INFO
12
11
  src/dppd.egg-info/SOURCES.txt
13
12
  src/dppd.egg-info/dependency_links.txt
14
- src/dppd.egg-info/not-zip-safe
15
13
  src/dppd.egg-info/requires.txt
16
- src/dppd.egg-info/top_level.txt
14
+ src/dppd.egg-info/top_level.txt
15
+ tests/test_base.py
16
+ tests/test_pandas_forwards.py
17
+ tests/test_reshaping.py
18
+ tests/test_select.py
19
+ tests/test_single_verbs.py
@@ -1,17 +1,13 @@
1
- pandas>=0.22
2
- numpy
3
1
  natsort
2
+ numpy
3
+ pandas>=2
4
4
  wrapt
5
5
 
6
- [doc]
7
- sphinx
8
- sphinx-bootstrap-theme
6
+ [dev]
7
+ build
9
8
  numpydoc
10
- pandas
11
-
12
- [testing]
9
+ plotnine
13
10
  pytest
14
11
  pytest-cov
15
- plotnine
16
- pandas<2.0
17
- flake8
12
+ sphinx
13
+ sphinx-bootstrap-theme