pytae 2.3.3__tar.gz → 2.3.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. {pytae-2.3.3 → pytae-2.3.4}/.ipynb_checkpoints/pyproject-checkpoint.toml +1 -1
  2. {pytae-2.3.3 → pytae-2.3.4}/PKG-INFO +1 -1
  3. {pytae-2.3.3 → pytae-2.3.4}/pyproject.toml +1 -1
  4. pytae-2.3.4/src/.ipynb_checkpoints/other_utilities-checkpoint.ipynb +152 -0
  5. {pytae-2.3.3 → pytae-2.3.4}/src/.ipynb_checkpoints/plot_x-checkpoint.ipynb +2 -2
  6. {pytae-2.3.3 → pytae-2.3.4}/src/other_utilities.ipynb +55 -1
  7. {pytae-2.3.3 → pytae-2.3.4}/src/plot_x.ipynb +2 -2
  8. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/.ipynb_checkpoints/other_utilities-checkpoint.py +27 -29
  9. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/other_utilities.py +27 -29
  10. pytae-2.3.3/src/.ipynb_checkpoints/other_utilities-checkpoint.ipynb +0 -814
  11. {pytae-2.3.3 → pytae-2.3.4}/.github/workflows/.ipynb_checkpoints/publish-to-test-pypi-checkpoint.yml +0 -0
  12. {pytae-2.3.3 → pytae-2.3.4}/.github/workflows/publish-to-test-pypi.yml +0 -0
  13. {pytae-2.3.3 → pytae-2.3.4}/.gitignore +0 -0
  14. {pytae-2.3.3 → pytae-2.3.4}/.ipynb_checkpoints/LICENSE-checkpoint +0 -0
  15. {pytae-2.3.3 → pytae-2.3.4}/.ipynb_checkpoints/README-checkpoint.md +0 -0
  16. {pytae-2.3.3 → pytae-2.3.4}/.ipynb_checkpoints/requirements-checkpoint.txt +0 -0
  17. {pytae-2.3.3 → pytae-2.3.4}/LICENSE +0 -0
  18. {pytae-2.3.3 → pytae-2.3.4}/README.md +0 -0
  19. {pytae-2.3.3 → pytae-2.3.4}/requirements.txt +0 -0
  20. {pytae-2.3.3 → pytae-2.3.4}/src/.ipynb_checkpoints/agg_df-checkpoint.ipynb +0 -0
  21. {pytae-2.3.3 → pytae-2.3.4}/src/.ipynb_checkpoints/get_data-checkpoint.ipynb +0 -0
  22. {pytae-2.3.3 → pytae-2.3.4}/src/.ipynb_checkpoints/shape-checkpoint.ipynb +0 -0
  23. {pytae-2.3.3 → pytae-2.3.4}/src/agg_df.ipynb +0 -0
  24. {pytae-2.3.3 → pytae-2.3.4}/src/get_data.ipynb +0 -0
  25. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/.ipynb_checkpoints/__init__-checkpoint.py +0 -0
  26. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/.ipynb_checkpoints/agg_df-checkpoint.py +0 -0
  27. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/.ipynb_checkpoints/get_data-checkpoint.py +0 -0
  28. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/.ipynb_checkpoints/plot_x-checkpoint.py +0 -0
  29. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/.ipynb_checkpoints/shape-checkpoint.py +0 -0
  30. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/__init__.py +0 -0
  31. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/agg_df.py +0 -0
  32. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/get_data.py +0 -0
  33. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/plot_x.py +0 -0
  34. {pytae-2.3.3 → pytae-2.3.4}/src/pytae/shape.py +0 -0
  35. {pytae-2.3.3 → pytae-2.3.4}/src/shape.ipynb +0 -0
  36. {pytae-2.3.3 → pytae-2.3.4}/tests/.ipynb_checkpoints/test_agg_df-checkpoint.py +0 -0
  37. {pytae-2.3.3 → pytae-2.3.4}/tests/.ipynb_checkpoints/test_shape-checkpoint.py +0 -0
  38. {pytae-2.3.3 → pytae-2.3.4}/tests/test_agg_df.py +0 -0
  39. {pytae-2.3.3 → pytae-2.3.4}/tests/test_shape.py +0 -0
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pytae"
7
- version = "2.3.3"
7
+ version = "2.3.4"
8
8
  description = "A package to enhance common day to day tasks in the life of a data scientist"
9
9
  authors = [{ name = "Madhukar Kumar Jha", email = "madhukar.k.jha@gmail.com" }]
10
10
  readme = "README.md"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: pytae
3
- Version: 2.3.3
3
+ Version: 2.3.4
4
4
  Dynamic: Requires-Dist
5
5
  Summary: A package to enhance common day to day tasks in the life of a data scientist
6
6
  Project-URL: Homepage, https://github.com/maddytae/pytae
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pytae"
7
- version = "2.3.3"
7
+ version = "2.3.4"
8
8
  description = "A package to enhance common day to day tasks in the life of a data scientist"
9
9
  authors = [{ name = "Madhukar Kumar Jha", email = "madhukar.k.jha@gmail.com" }]
10
10
  readme = "README.md"
@@ -0,0 +1,152 @@
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "code",
5
+ "execution_count": null,
6
+ "id": "af0e75c1-7748-44e7-974e-34112d92d813",
7
+ "metadata": {},
8
+ "outputs": [],
9
+ "source": [
10
+ "import numpy as np\n",
11
+ "import pandas as pd\n",
12
+ "import pytae as pt\n",
13
+ "import seaborn as sns\n",
14
+ "penguins = sns.load_dataset('penguins')"
15
+ ]
16
+ },
17
+ {
18
+ "cell_type": "code",
19
+ "execution_count": null,
20
+ "id": "e42e01b5-4b00-4f95-acec-274deda67a3c",
21
+ "metadata": {},
22
+ "outputs": [],
23
+ "source": [
24
+ "# get a list of columns\n",
25
+ "penguins.cols()"
26
+ ]
27
+ },
28
+ {
29
+ "cell_type": "code",
30
+ "execution_count": null,
31
+ "id": "e7136bed-d2e1-48c5-a337-8e31fc020a98",
32
+ "metadata": {},
33
+ "outputs": [],
34
+ "source": [
35
+ "#add a group count\n",
36
+ "df = (\n",
37
+ " penguins\n",
38
+ " .group_x()\n",
39
+ ")\n",
40
+ "df.head()"
41
+ ]
42
+ },
43
+ {
44
+ "cell_type": "code",
45
+ "execution_count": null,
46
+ "id": "87e3dd20-d5e1-4219-8aee-e842a3d52194",
47
+ "metadata": {},
48
+ "outputs": [],
49
+ "source": [
50
+ "# handle missing i.e. str missing or NaN is changed to . and numeric NaN to 0\n",
51
+ "\n",
52
+ "df = (\n",
53
+ " penguins\n",
54
+ " .handle_missing(fillna='$')\n",
55
+ ")\n",
56
+ "df.head()\n"
57
+ ]
58
+ },
59
+ {
60
+ "cell_type": "code",
61
+ "execution_count": null,
62
+ "id": "b17a9574-35ff-4fe2-b40f-054a0b5e5003",
63
+ "metadata": {},
64
+ "outputs": [],
65
+ "source": [
66
+ "df.clip()"
67
+ ]
68
+ },
69
+ {
70
+ "cell_type": "code",
71
+ "execution_count": null,
72
+ "id": "b39ba218-6361-47eb-91b2-c6f35b7a130e",
73
+ "metadata": {},
74
+ "outputs": [],
75
+ "source": [
76
+ "penguins\n"
77
+ ]
78
+ },
79
+ {
80
+ "cell_type": "code",
81
+ "execution_count": null,
82
+ "id": "cf3bf0ae-b4ea-45ad-a358-c3eb5042e039",
83
+ "metadata": {},
84
+ "outputs": [],
85
+ "source": [
86
+ "penguins.select('spe|sex|^n|^bi')"
87
+ ]
88
+ },
89
+ {
90
+ "cell_type": "code",
91
+ "execution_count": null,
92
+ "id": "3a74dbdf-4afa-4b9c-af1e-22cd016bc39a",
93
+ "metadata": {},
94
+ "outputs": [],
95
+ "source": [
96
+ "penguins.select(['species','n'])"
97
+ ]
98
+ },
99
+ {
100
+ "cell_type": "code",
101
+ "execution_count": null,
102
+ "id": "f29560f9-bb6e-4b0f-a143-8c75be726609",
103
+ "metadata": {},
104
+ "outputs": [],
105
+ "source": [
106
+ "df = pd.DataFrame(np.arange(12).reshape(3, 4),\n",
107
+ " columns=['A', 'B_X', 'B', 'C'])\n",
108
+ "\n",
109
+ "print('original df')\n",
110
+ "print(df)\n",
111
+ "\n",
112
+ "print('\\nselect using a list')\n",
113
+ "print(df.select(['A','B_X']))\n",
114
+ "\n",
115
+ "print('\\nselect using a regex')\n",
116
+ "print(df.select('^B|C'))\n",
117
+ "\n",
118
+ "print('\\nselect using a tuple of list and regex')\n",
119
+ "print(df.select((['A','B_X'],'^B|C')))"
120
+ ]
121
+ },
122
+ {
123
+ "cell_type": "code",
124
+ "execution_count": null,
125
+ "id": "efde3bab-b430-4785-bfff-6ff1f515e6a6",
126
+ "metadata": {},
127
+ "outputs": [],
128
+ "source": []
129
+ }
130
+ ],
131
+ "metadata": {
132
+ "kernelspec": {
133
+ "display_name": "py_env",
134
+ "language": "python",
135
+ "name": "py_env"
136
+ },
137
+ "language_info": {
138
+ "codemirror_mode": {
139
+ "name": "ipython",
140
+ "version": 3
141
+ },
142
+ "file_extension": ".py",
143
+ "mimetype": "text/x-python",
144
+ "name": "python",
145
+ "nbconvert_exporter": "python",
146
+ "pygments_lexer": "ipython3",
147
+ "version": "3.9.6"
148
+ }
149
+ },
150
+ "nbformat": 4,
151
+ "nbformat_minor": 5
152
+ }
@@ -141,7 +141,7 @@
141
141
  },
142
142
  {
143
143
  "cell_type": "code",
144
- "execution_count": 5,
144
+ "execution_count": 6,
145
145
  "id": "4c45df2e-49f3-4540-a1bf-caf19a3a6d45",
146
146
  "metadata": {},
147
147
  "outputs": [
@@ -152,7 +152,7 @@
152
152
  "<Figure size 1200x600 with 1 Axes>"
153
153
  ]
154
154
  },
155
- "execution_count": 5,
155
+ "execution_count": 6,
156
156
  "metadata": {},
157
157
  "output_type": "execute_result"
158
158
  }
@@ -7,6 +7,7 @@
7
7
  "metadata": {},
8
8
  "outputs": [],
9
9
  "source": [
10
+ "import numpy as np\n",
10
11
  "import pandas as pd\n",
11
12
  "import pytae as pt\n",
12
13
  "import seaborn as sns\n",
@@ -783,9 +784,62 @@
783
784
  },
784
785
  {
785
786
  "cell_type": "code",
786
- "execution_count": null,
787
+ "execution_count": 9,
787
788
  "id": "f29560f9-bb6e-4b0f-a143-8c75be726609",
788
789
  "metadata": {},
790
+ "outputs": [
791
+ {
792
+ "name": "stdout",
793
+ "output_type": "stream",
794
+ "text": [
795
+ "original df\n",
796
+ " A B_X B C\n",
797
+ "0 0 1 2 3\n",
798
+ "1 4 5 6 7\n",
799
+ "2 8 9 10 11\n",
800
+ "\n",
801
+ "select using a list\n",
802
+ " A B_X\n",
803
+ "0 0 1\n",
804
+ "1 4 5\n",
805
+ "2 8 9\n",
806
+ "\n",
807
+ "select using a regex\n",
808
+ " B_X B C\n",
809
+ "0 1 2 3\n",
810
+ "1 5 6 7\n",
811
+ "2 9 10 11\n",
812
+ "\n",
813
+ "select using a tuple of list and regex\n",
814
+ " B C A B_X\n",
815
+ "0 2 3 0 1\n",
816
+ "1 6 7 4 5\n",
817
+ "2 10 11 8 9\n"
818
+ ]
819
+ }
820
+ ],
821
+ "source": [
822
+ "df = pd.DataFrame(np.arange(12).reshape(3, 4),\n",
823
+ " columns=['A', 'B_X', 'B', 'C'])\n",
824
+ "\n",
825
+ "print('original df')\n",
826
+ "print(df)\n",
827
+ "\n",
828
+ "print('\\nselect using a list')\n",
829
+ "print(df.select(['A','B_X']))\n",
830
+ "\n",
831
+ "print('\\nselect using a regex')\n",
832
+ "print(df.select('^B|C'))\n",
833
+ "\n",
834
+ "print('\\nselect using a tuple of list and regex')\n",
835
+ "print(df.select((['A','B_X'],'^B|C')))"
836
+ ]
837
+ },
838
+ {
839
+ "cell_type": "code",
840
+ "execution_count": null,
841
+ "id": "efde3bab-b430-4785-bfff-6ff1f515e6a6",
842
+ "metadata": {},
789
843
  "outputs": [],
790
844
  "source": []
791
845
  }
@@ -141,7 +141,7 @@
141
141
  },
142
142
  {
143
143
  "cell_type": "code",
144
- "execution_count": 5,
144
+ "execution_count": 6,
145
145
  "id": "4c45df2e-49f3-4540-a1bf-caf19a3a6d45",
146
146
  "metadata": {},
147
147
  "outputs": [
@@ -152,7 +152,7 @@
152
152
  "<Figure size 1200x600 with 1 Axes>"
153
153
  ]
154
154
  },
155
- "execution_count": 5,
155
+ "execution_count": 6,
156
156
  "metadata": {},
157
157
  "output_type": "execute_result"
158
158
  }
@@ -23,50 +23,48 @@ def handle_missing(self,fillna='.'):
23
23
  def cols(self):#this is for more general situations
24
24
  return sorted(self.columns.to_list())
25
25
 
26
- #select is more intuitive than filter. It needs a regex expression which is quite powerful
27
- #ex
28
- def select(self, cols_or_regex):
26
+ #select is more intuitive than filter. It can handle a list or a regex or a tuple containing list and regex
27
+ def select(self, cols_or_regex=None):
29
28
  '''
30
- Select columns based on a list of column names or a regex pattern.
29
+ Select columns based on a list of column names, a regex pattern, or a tuple of both.
31
30
 
32
31
  Parameters:
33
32
  self (pd.DataFrame): The DataFrame from which to select columns.
34
- cols_or_regex (list or str): List of column names or a regex pattern.
33
+ cols_or_regex (list, str, or tuple):
34
+ - List of column names
35
+ - Regex pattern
36
+ - Tuple containing a list of column names and a regex pattern
35
37
 
36
38
  Returns:
37
39
  pd.DataFrame: A DataFrame with the selected columns.
38
40
  '''
39
- if isinstance(cols_or_regex, list):
40
- # Ensure all columns in the list exist in the DataFrame
41
+
42
+ if isinstance(cols_or_regex, tuple):
43
+ if len(cols_or_regex) != 2:
44
+ raise ValueError("Tuple must contain exactly two elements: a list of columns and a regex pattern")
45
+ cols, regex = cols_or_regex
46
+ if not isinstance(cols, list) or not isinstance(regex, str):
47
+ raise TypeError("First element of tuple must be a list, and the second element must be a string (regex pattern)")
48
+ # Select columns based on the list and regex independently
49
+ selected_cols = list(set(cols + self.filter(regex=regex).columns.tolist()))
50
+ return self[selected_cols]
51
+
52
+ elif isinstance(cols_or_regex, list):
41
53
  missing_cols = [col for col in cols_or_regex if col not in self.columns]
42
54
  if missing_cols:
43
55
  raise KeyError(f"Columns not found in the DataFrame: {missing_cols}")
44
56
  return self[cols_or_regex]
57
+
45
58
  elif isinstance(cols_or_regex, str):
46
59
  return self.filter(regex=cols_or_regex)
60
+
47
61
  else:
48
- raise TypeError("cols_or_regex must be a list or a string")
49
-
50
- # # Example usage
51
- # data = {
52
- # 'apple': [1, 2, 3],
53
- # 'banana': [4, 5, 6],
54
- # 'apricot': [7, 8, 9],
55
- # 'cherry': [10, 11, 12]
56
- # }
57
-
58
- # # Create a DataFrame
59
- # df = pd.DataFrame(data)
60
-
61
- # # Select columns using a list of column names
62
- # selected_cols_list = select(df, ['apple', 'banana'])
63
- # print("Selected columns using list:")
64
- # print(selected_cols_list)
65
-
66
- # # Select columns using a regex pattern
67
- # selected_cols_regex = select(df, '^(apple|banana|cherry)$')
68
- # print("\nSelected columns using regex:")
69
- # print(selected_cols_regex)
62
+ raise TypeError("cols_or_regex must be a list, a regex string, or a tuple of (list, regex)")
63
+
64
+ # Add the method to the DataFrame class
65
+ pd.DataFrame.select = select
66
+
67
+
70
68
 
71
69
 
72
70
 
@@ -23,50 +23,48 @@ def handle_missing(self,fillna='.'):
23
23
  def cols(self):#this is for more general situations
24
24
  return sorted(self.columns.to_list())
25
25
 
26
- #select is more intuitive than filter. It needs a regex expression which is quite powerful
27
- #ex
28
- def select(self, cols_or_regex):
26
+ #select is more intuitive than filter. It can handle a list or a regex or a tuple containing list and regex
27
+ def select(self, cols_or_regex=None):
29
28
  '''
30
- Select columns based on a list of column names or a regex pattern.
29
+ Select columns based on a list of column names, a regex pattern, or a tuple of both.
31
30
 
32
31
  Parameters:
33
32
  self (pd.DataFrame): The DataFrame from which to select columns.
34
- cols_or_regex (list or str): List of column names or a regex pattern.
33
+ cols_or_regex (list, str, or tuple):
34
+ - List of column names
35
+ - Regex pattern
36
+ - Tuple containing a list of column names and a regex pattern
35
37
 
36
38
  Returns:
37
39
  pd.DataFrame: A DataFrame with the selected columns.
38
40
  '''
39
- if isinstance(cols_or_regex, list):
40
- # Ensure all columns in the list exist in the DataFrame
41
+
42
+ if isinstance(cols_or_regex, tuple):
43
+ if len(cols_or_regex) != 2:
44
+ raise ValueError("Tuple must contain exactly two elements: a list of columns and a regex pattern")
45
+ cols, regex = cols_or_regex
46
+ if not isinstance(cols, list) or not isinstance(regex, str):
47
+ raise TypeError("First element of tuple must be a list, and the second element must be a string (regex pattern)")
48
+ # Select columns based on the list and regex independently
49
+ selected_cols = list(set(cols + self.filter(regex=regex).columns.tolist()))
50
+ return self[selected_cols]
51
+
52
+ elif isinstance(cols_or_regex, list):
41
53
  missing_cols = [col for col in cols_or_regex if col not in self.columns]
42
54
  if missing_cols:
43
55
  raise KeyError(f"Columns not found in the DataFrame: {missing_cols}")
44
56
  return self[cols_or_regex]
57
+
45
58
  elif isinstance(cols_or_regex, str):
46
59
  return self.filter(regex=cols_or_regex)
60
+
47
61
  else:
48
- raise TypeError("cols_or_regex must be a list or a string")
49
-
50
- # # Example usage
51
- # data = {
52
- # 'apple': [1, 2, 3],
53
- # 'banana': [4, 5, 6],
54
- # 'apricot': [7, 8, 9],
55
- # 'cherry': [10, 11, 12]
56
- # }
57
-
58
- # # Create a DataFrame
59
- # df = pd.DataFrame(data)
60
-
61
- # # Select columns using a list of column names
62
- # selected_cols_list = select(df, ['apple', 'banana'])
63
- # print("Selected columns using list:")
64
- # print(selected_cols_list)
65
-
66
- # # Select columns using a regex pattern
67
- # selected_cols_regex = select(df, '^(apple|banana|cherry)$')
68
- # print("\nSelected columns using regex:")
69
- # print(selected_cols_regex)
62
+ raise TypeError("cols_or_regex must be a list, a regex string, or a tuple of (list, regex)")
63
+
64
+ # Add the method to the DataFrame class
65
+ pd.DataFrame.select = select
66
+
67
+
70
68
 
71
69
 
72
70