google-data-utils 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Zomi Learner
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,124 @@
1
+ Metadata-Version: 2.4
2
+ Name: google-data-utils
3
+ Version: 0.1.0
4
+ Summary: Load public Google Sheets, CSV, and Excel files into pandas
5
+ Author: Zomi Learner
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/ZomiLearner/google-data-utils
8
+ Requires-Python: >=3.8
9
+ Description-Content-Type: text/markdown
10
+ License-File: LICENSE
11
+ Requires-Dist: pandas>=1.5
12
+ Requires-Dist: openpyxl>=3.0
13
+ Provides-Extra: dev
14
+ Requires-Dist: pytest>=8; extra == "dev"
15
+ Dynamic: license-file
16
+
17
+ # google-data-utils
18
+
19
+ Simple utilities for loading public Google Drive CSV files, Excel files, and Google Sheets into pandas DataFrames.
20
+
21
+ > **Note**
22
+ >
23
+ > This package only supports publicly accessible Google Drive files and Google Sheets. Authentication, OAuth, and Google API credentials are not required.
24
+
25
+ ## Features
26
+
27
+ - Read public Google Drive CSV files into a DataFrame
28
+ - Read public Google Drive Excel files into a DataFrame
29
+ - Read public Google Sheets into a DataFrame
30
+ - Validate Google Drive and Google Sheets URLs
31
+ - Extract Google Drive file IDs and Google Sheets IDs
32
+ - No API keys required
33
+ - No OAuth required
34
+ - No Google Cloud setup required
35
+
36
+ ## Installation
37
+
38
+ ```bash
39
+ pip install google-data-utils
40
+ ```
41
+
42
+ ## Usage
43
+
44
+ ### Google Drive CSV
45
+
46
+ ```bash
47
+ from google_data_utils import google_csv_to_df
48
+
49
+ df = google_csv_to_df(
50
+ "https://drive.google.com/file/d/FILE_ID/view?usp=sharing"
51
+ )
52
+ ```
53
+
54
+ ### Google Drive Excel
55
+
56
+ ```bash
57
+ from google_data_utils import google_excel_to_df
58
+
59
+ df = google_excel_to_df(
60
+ "https://drive.google.com/file/d/FILE_ID/view?usp=sharing"
61
+ )
62
+ ```
63
+
64
+ ### Google Sheets
65
+
66
+ ```bash
67
+ from google_data_utils import google_sheet_to_df
68
+
69
+ df = google_sheet_to_df(
70
+ "https://docs.google.com/spreadsheets/d/SHEET_ID/edit#gid=0"
71
+ )
72
+ ```
73
+
74
+ ### URL Validation
75
+
76
+ ```bash
77
+ from google_data_utils import (
78
+ is_google_drive_url,
79
+ is_google_sheet_url,
80
+ )
81
+
82
+ is_google_drive_url(
83
+ "https://drive.google.com/file/d/FILE_ID/view"
84
+ )
85
+ # True
86
+
87
+ is_google_sheet_url(
88
+ "https://docs.google.com/spreadsheets/d/SHEET_ID/edit"
89
+ )
90
+ # True
91
+ ```
92
+
93
+ ### Extract IDs
94
+
95
+ ```bash
96
+ from google_data_utils import (
97
+ extract_drive_file_id,
98
+ extract_sheet_id,
99
+ )
100
+
101
+ extract_drive_file_id(
102
+ "https://drive.google.com/file/d/FILE_ID/view"
103
+ )
104
+ # FILE_ID
105
+
106
+ extract_sheet_id(
107
+ "https://docs.google.com/spreadsheets/d/SHEET_ID/edit"
108
+ )
109
+ # SHEET_ID
110
+ ```
111
+
112
+ ### Requirements
113
+
114
+ ```text
115
+ Python 3.8+
116
+ pandas
117
+ openpyxl
118
+ ```
119
+
120
+ ### License
121
+
122
+ ```text
123
+ MIT License
124
+ ```
@@ -0,0 +1,108 @@
1
+ # google-data-utils
2
+
3
+ Simple utilities for loading public Google Drive CSV files, Excel files, and Google Sheets into pandas DataFrames.
4
+
5
+ > **Note**
6
+ >
7
+ > This package only supports publicly accessible Google Drive files and Google Sheets. Authentication, OAuth, and Google API credentials are not required.
8
+
9
+ ## Features
10
+
11
+ - Read public Google Drive CSV files into a DataFrame
12
+ - Read public Google Drive Excel files into a DataFrame
13
+ - Read public Google Sheets into a DataFrame
14
+ - Validate Google Drive and Google Sheets URLs
15
+ - Extract Google Drive file IDs and Google Sheets IDs
16
+ - No API keys required
17
+ - No OAuth required
18
+ - No Google Cloud setup required
19
+
20
+ ## Installation
21
+
22
+ ```bash
23
+ pip install google-data-utils
24
+ ```
25
+
26
+ ## Usage
27
+
28
+ ### Google Drive CSV
29
+
30
+ ```bash
31
+ from google_data_utils import google_csv_to_df
32
+
33
+ df = google_csv_to_df(
34
+ "https://drive.google.com/file/d/FILE_ID/view?usp=sharing"
35
+ )
36
+ ```
37
+
38
+ ### Google Drive Excel
39
+
40
+ ```bash
41
+ from google_data_utils import google_excel_to_df
42
+
43
+ df = google_excel_to_df(
44
+ "https://drive.google.com/file/d/FILE_ID/view?usp=sharing"
45
+ )
46
+ ```
47
+
48
+ ### Google Sheets
49
+
50
+ ```bash
51
+ from google_data_utils import google_sheet_to_df
52
+
53
+ df = google_sheet_to_df(
54
+ "https://docs.google.com/spreadsheets/d/SHEET_ID/edit#gid=0"
55
+ )
56
+ ```
57
+
58
+ ### URL Validation
59
+
60
+ ```bash
61
+ from google_data_utils import (
62
+ is_google_drive_url,
63
+ is_google_sheet_url,
64
+ )
65
+
66
+ is_google_drive_url(
67
+ "https://drive.google.com/file/d/FILE_ID/view"
68
+ )
69
+ # True
70
+
71
+ is_google_sheet_url(
72
+ "https://docs.google.com/spreadsheets/d/SHEET_ID/edit"
73
+ )
74
+ # True
75
+ ```
76
+
77
+ ### Extract IDs
78
+
79
+ ```bash
80
+ from google_data_utils import (
81
+ extract_drive_file_id,
82
+ extract_sheet_id,
83
+ )
84
+
85
+ extract_drive_file_id(
86
+ "https://drive.google.com/file/d/FILE_ID/view"
87
+ )
88
+ # FILE_ID
89
+
90
+ extract_sheet_id(
91
+ "https://docs.google.com/spreadsheets/d/SHEET_ID/edit"
92
+ )
93
+ # SHEET_ID
94
+ ```
95
+
96
+ ### Requirements
97
+
98
+ ```text
99
+ Python 3.8+
100
+ pandas
101
+ openpyxl
102
+ ```
103
+
104
+ ### License
105
+
106
+ ```text
107
+ MIT License
108
+ ```
@@ -0,0 +1,31 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "google-data-utils"
7
+ version = "0.1.0"
8
+ description = "Load public Google Sheets, CSV, and Excel files into pandas"
9
+ readme = "README.md"
10
+ requires-python = ">=3.8"
11
+ license = {text = "MIT"}
12
+ authors = [
13
+ {name = "Zomi Learner"}
14
+ ]
15
+ dependencies = [
16
+ "pandas>=1.5",
17
+ "openpyxl>=3.0"
18
+ ]
19
+
20
+ [project.optional-dependencies]
21
+ dev = [
22
+ "pytest>=8"
23
+ ]
24
+
25
+
26
+
27
+ [tool.setuptools.packages.find]
28
+ where = ["src"]
29
+
30
+ [project.urls]
31
+ Homepage = "https://github.com/ZomiLearner/google-data-utils"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,22 @@
1
+ from .public import (
2
+ google_csv_to_df,
3
+ google_excel_to_df,
4
+ google_sheet_to_df,
5
+ )
6
+
7
+ from .validators import (
8
+ extract_drive_file_id,
9
+ extract_sheet_id,
10
+ is_google_drive_url,
11
+ is_google_sheet_url,
12
+ )
13
+
14
+ __all__ = [
15
+ "google_csv_to_df",
16
+ "google_excel_to_df",
17
+ "google_sheet_to_df",
18
+ "extract_drive_file_id",
19
+ "extract_sheet_id",
20
+ "is_google_drive_url",
21
+ "is_google_sheet_url",
22
+ ]
@@ -0,0 +1,38 @@
1
+ # public.py
2
+
3
+ import pandas as pd
4
+
5
+ from .validators import (
6
+ extract_drive_file_id,
7
+ extract_sheet_id,
8
+ )
9
+
10
+
11
+
12
+ def google_csv_to_df(url, **kwargs):
13
+ file_id = extract_drive_file_id(url)
14
+
15
+ return pd.read_csv(
16
+ f"https://drive.google.com/uc?id={file_id}",
17
+ **kwargs,
18
+ )
19
+
20
+
21
+ def google_excel_to_df(url, **kwargs):
22
+ file_id = extract_drive_file_id(url)
23
+
24
+ return pd.read_excel(
25
+ f"https://drive.google.com/uc?id={file_id}",
26
+ **kwargs,
27
+ )
28
+
29
+
30
+ def google_sheet_to_df(url, **kwargs):
31
+ sheet_id = extract_sheet_id(url)
32
+
33
+ csv_url = (
34
+ f"https://docs.google.com/spreadsheets/d/"
35
+ f"{sheet_id}/export?format=csv"
36
+ )
37
+
38
+ return pd.read_csv(csv_url, **kwargs)
@@ -0,0 +1,87 @@
1
+ # validators.py
2
+
3
+ from urllib.parse import urlparse
4
+ import re
5
+
6
+ class InvalidGoogleUrl(ValueError):
7
+ pass
8
+
9
+
10
+ _DRIVE_FILE_RE = re.compile(
11
+ r"^/file/d/([a-zA-Z0-9_-]+)"
12
+ )
13
+
14
+ _SHEET_RE = re.compile(
15
+ r"^/spreadsheets/d/([a-zA-Z0-9_-]+)"
16
+ )
17
+
18
+ def is_google_drive_url(url: str) -> bool:
19
+ try:
20
+ extract_drive_file_id(url)
21
+ return True
22
+ except InvalidGoogleUrl:
23
+ return False
24
+
25
+
26
+ def is_google_sheet_url(url: str) -> bool:
27
+ try:
28
+ extract_sheet_id(url)
29
+ return True
30
+ except InvalidGoogleUrl:
31
+ return False
32
+
33
+ def extract_drive_file_id(url: str) -> str:
34
+ """
35
+ Accepts:
36
+ https://drive.google.com/file/d/FILE_ID/view
37
+ """
38
+
39
+ parsed = urlparse(url)
40
+
41
+ if parsed.netloc != "drive.google.com":
42
+ raise InvalidGoogleUrl(
43
+ "Expected a drive.google.com URL"
44
+ )
45
+
46
+ match = _DRIVE_FILE_RE.match(parsed.path)
47
+
48
+ if not match:
49
+ raise InvalidGoogleUrl(
50
+ "Could not extract Drive file ID"
51
+ )
52
+
53
+ return match.group(1)
54
+
55
+
56
+ def extract_sheet_id(url: str) -> str:
57
+ """
58
+ Accepts:
59
+ https://docs.google.com/spreadsheets/d/SHEET_ID/edit
60
+ """
61
+
62
+ parsed = urlparse(url)
63
+
64
+ if parsed.netloc != "docs.google.com":
65
+ raise InvalidGoogleUrl(
66
+ "Expected a docs.google.com URL"
67
+ )
68
+
69
+ match = _SHEET_RE.match(parsed.path)
70
+
71
+ if not match:
72
+ raise InvalidGoogleUrl(
73
+ "Could not extract Sheet ID"
74
+ )
75
+
76
+ return match.group(1)
77
+
78
+ def drive_download_url(url: str) -> str:
79
+ file_id = extract_drive_file_id(url)
80
+ return f"https://drive.google.com/uc?id={file_id}"
81
+
82
+ def sheet_csv_url(url: str) -> str:
83
+ sheet_id = extract_sheet_id(url)
84
+ return (
85
+ f"https://docs.google.com/spreadsheets/d/"
86
+ f"{sheet_id}/export?format=csv"
87
+ )
@@ -0,0 +1,124 @@
1
+ Metadata-Version: 2.4
2
+ Name: google-data-utils
3
+ Version: 0.1.0
4
+ Summary: Load public Google Sheets, CSV, and Excel files into pandas
5
+ Author: Zomi Learner
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/ZomiLearner/google-data-utils
8
+ Requires-Python: >=3.8
9
+ Description-Content-Type: text/markdown
10
+ License-File: LICENSE
11
+ Requires-Dist: pandas>=1.5
12
+ Requires-Dist: openpyxl>=3.0
13
+ Provides-Extra: dev
14
+ Requires-Dist: pytest>=8; extra == "dev"
15
+ Dynamic: license-file
16
+
17
+ # google-data-utils
18
+
19
+ Simple utilities for loading public Google Drive CSV files, Excel files, and Google Sheets into pandas DataFrames.
20
+
21
+ > **Note**
22
+ >
23
+ > This package only supports publicly accessible Google Drive files and Google Sheets. Authentication, OAuth, and Google API credentials are not required.
24
+
25
+ ## Features
26
+
27
+ - Read public Google Drive CSV files into a DataFrame
28
+ - Read public Google Drive Excel files into a DataFrame
29
+ - Read public Google Sheets into a DataFrame
30
+ - Validate Google Drive and Google Sheets URLs
31
+ - Extract Google Drive file IDs and Google Sheets IDs
32
+ - No API keys required
33
+ - No OAuth required
34
+ - No Google Cloud setup required
35
+
36
+ ## Installation
37
+
38
+ ```bash
39
+ pip install google-data-utils
40
+ ```
41
+
42
+ ## Usage
43
+
44
+ ### Google Drive CSV
45
+
46
+ ```bash
47
+ from google_data_utils import google_csv_to_df
48
+
49
+ df = google_csv_to_df(
50
+ "https://drive.google.com/file/d/FILE_ID/view?usp=sharing"
51
+ )
52
+ ```
53
+
54
+ ### Google Drive Excel
55
+
56
+ ```bash
57
+ from google_data_utils import google_excel_to_df
58
+
59
+ df = google_excel_to_df(
60
+ "https://drive.google.com/file/d/FILE_ID/view?usp=sharing"
61
+ )
62
+ ```
63
+
64
+ ### Google Sheets
65
+
66
+ ```bash
67
+ from google_data_utils import google_sheet_to_df
68
+
69
+ df = google_sheet_to_df(
70
+ "https://docs.google.com/spreadsheets/d/SHEET_ID/edit#gid=0"
71
+ )
72
+ ```
73
+
74
+ ### URL Validation
75
+
76
+ ```bash
77
+ from google_data_utils import (
78
+ is_google_drive_url,
79
+ is_google_sheet_url,
80
+ )
81
+
82
+ is_google_drive_url(
83
+ "https://drive.google.com/file/d/FILE_ID/view"
84
+ )
85
+ # True
86
+
87
+ is_google_sheet_url(
88
+ "https://docs.google.com/spreadsheets/d/SHEET_ID/edit"
89
+ )
90
+ # True
91
+ ```
92
+
93
+ ### Extract IDs
94
+
95
+ ```bash
96
+ from google_data_utils import (
97
+ extract_drive_file_id,
98
+ extract_sheet_id,
99
+ )
100
+
101
+ extract_drive_file_id(
102
+ "https://drive.google.com/file/d/FILE_ID/view"
103
+ )
104
+ # FILE_ID
105
+
106
+ extract_sheet_id(
107
+ "https://docs.google.com/spreadsheets/d/SHEET_ID/edit"
108
+ )
109
+ # SHEET_ID
110
+ ```
111
+
112
+ ### Requirements
113
+
114
+ ```text
115
+ Python 3.8+
116
+ pandas
117
+ openpyxl
118
+ ```
119
+
120
+ ### License
121
+
122
+ ```text
123
+ MIT License
124
+ ```
@@ -0,0 +1,14 @@
1
+ LICENSE
2
+ README.md
3
+ pyproject.toml
4
+ src/google_data_utils/__init__.py
5
+ src/google_data_utils/public.py
6
+ src/google_data_utils/validators.py
7
+ src/google_data_utils.egg-info/PKG-INFO
8
+ src/google_data_utils.egg-info/SOURCES.txt
9
+ src/google_data_utils.egg-info/dependency_links.txt
10
+ src/google_data_utils.egg-info/requires.txt
11
+ src/google_data_utils.egg-info/top_level.txt
12
+ tests/test_drive.py
13
+ tests/test_sheet.py
14
+ tests/test_validators.py
@@ -0,0 +1,5 @@
1
+ pandas>=1.5
2
+ openpyxl>=3.0
3
+
4
+ [dev]
5
+ pytest>=8
@@ -0,0 +1 @@
1
+ google_data_utils
@@ -0,0 +1,69 @@
1
+ import pytest
2
+
3
+ from google_data_utils.validators import (
4
+ InvalidGoogleUrl,
5
+ extract_drive_file_id,
6
+ is_google_drive_url,
7
+ drive_download_url,
8
+ )
9
+
10
+
11
+ def test_extract_drive_file_id():
12
+ url = (
13
+ "https://drive.google.com/file/d/"
14
+ "ABC123_XYZ/view?usp=sharing"
15
+ )
16
+
17
+ assert extract_drive_file_id(url) == "ABC123_XYZ"
18
+
19
+
20
+ def test_extract_drive_file_id_invalid_domain():
21
+ url = (
22
+ "https://example.com/file/d/"
23
+ "ABC123/view"
24
+ )
25
+
26
+ with pytest.raises(InvalidGoogleUrl):
27
+ extract_drive_file_id(url)
28
+
29
+
30
+ def test_extract_drive_file_id_invalid_path():
31
+ url = "https://drive.google.com/"
32
+
33
+ with pytest.raises(InvalidGoogleUrl):
34
+ extract_drive_file_id(url)
35
+
36
+
37
+ def test_is_google_drive_url_valid():
38
+ url = (
39
+ "https://drive.google.com/file/d/"
40
+ "ABC123/view"
41
+ )
42
+
43
+ assert is_google_drive_url(url) is True
44
+
45
+
46
+ def test_is_google_drive_url_invalid_domain():
47
+ url = (
48
+ "https://example.com/file/d/"
49
+ "ABC123/view"
50
+ )
51
+
52
+ assert is_google_drive_url(url) is False
53
+
54
+
55
+ def test_is_google_drive_url_invalid_path():
56
+ url = "https://drive.google.com"
57
+
58
+ assert is_google_drive_url(url) is False
59
+
60
+
61
+ def test_drive_download_url():
62
+ url = (
63
+ "https://drive.google.com/file/d/"
64
+ "ABC123/view"
65
+ )
66
+ assert (
67
+ drive_download_url(url)
68
+ == "https://drive.google.com/uc?id=ABC123"
69
+ )
@@ -0,0 +1,77 @@
1
+ # tests/test_sheet.py
2
+
3
+ import pytest
4
+
5
+ from google_data_utils.validators import (
6
+ InvalidGoogleUrl,
7
+ extract_sheet_id,
8
+ is_google_sheet_url,
9
+ )
10
+
11
+
12
+ def test_extract_sheet_id():
13
+ url = (
14
+ "https://docs.google.com/spreadsheets/d/"
15
+ "ABC123_XYZ/edit#gid=0"
16
+ )
17
+
18
+ assert extract_sheet_id(url) == "ABC123_XYZ"
19
+
20
+
21
+ def test_extract_sheet_id_export_url():
22
+ url = (
23
+ "https://docs.google.com/spreadsheets/d/"
24
+ "ABC123_XYZ/export?format=csv"
25
+ )
26
+
27
+ assert extract_sheet_id(url) == "ABC123_XYZ"
28
+
29
+
30
+ def test_extract_sheet_id_invalid_domain():
31
+ url = (
32
+ "https://example.com/spreadsheets/d/"
33
+ "ABC123/edit"
34
+ )
35
+
36
+ with pytest.raises(InvalidGoogleUrl):
37
+ extract_sheet_id(url)
38
+
39
+
40
+ def test_extract_sheet_id_invalid_path():
41
+ url = "https://docs.google.com/spreadsheets"
42
+
43
+ with pytest.raises(InvalidGoogleUrl):
44
+ extract_sheet_id(url)
45
+
46
+
47
+ def test_is_google_sheet_url_valid_edit():
48
+ url = (
49
+ "https://docs.google.com/spreadsheets/d/"
50
+ "ABC123/edit"
51
+ )
52
+
53
+ assert is_google_sheet_url(url) is True
54
+
55
+
56
+ def test_is_google_sheet_url_valid_export():
57
+ url = (
58
+ "https://docs.google.com/spreadsheets/d/"
59
+ "ABC123/export?format=csv"
60
+ )
61
+
62
+ assert is_google_sheet_url(url) is True
63
+
64
+
65
+ def test_is_google_sheet_url_invalid_domain():
66
+ url = (
67
+ "https://example.com/spreadsheets/d/"
68
+ "ABC123/edit"
69
+ )
70
+
71
+ assert is_google_sheet_url(url) is False
72
+
73
+
74
+ def test_is_google_sheet_url_invalid_path():
75
+ url = "https://docs.google.com/spreadsheets"
76
+
77
+ assert is_google_sheet_url(url) is False
@@ -0,0 +1,10 @@
1
+ from google_data_utils.validators import extract_drive_file_id
2
+
3
+
4
+ def test_extract_drive_file_id():
5
+ url = (
6
+ "https://drive.google.com/file/d/"
7
+ "ABC123/view?usp=sharing"
8
+ )
9
+
10
+ assert extract_drive_file_id(url) == "ABC123"