bdcdata 0.0.7__tar.gz → 0.0.8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,8 @@
1
+ **/__pycache__/
2
+ .python-version
3
+ .env
4
+ .DS_Store
5
+ cache/
6
+ .env
7
+ *.log
8
+ debug1.py
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bdcdata
3
- Version: 0.0.7
3
+ Version: 0.0.8
4
4
  Summary: A tool to work with BDC data from the FCC.
5
5
  Project-URL: Homepage, https://github.com/npappin-wsu/bdc
6
6
  Project-URL: Issues, https://github.com/npappin-wsu/bdc/issues
@@ -22,7 +22,7 @@ username = os.getenv("BDC_USERNAME")
22
22
 
23
23
  # Configure logging
24
24
  logging.basicConfig(
25
- filename="bdc.log", filemode="w", format="%(asctime)s - %(levelname)s - %(message)s"
25
+ filename="bdc.log", filemode="a", format="%(asctime)s - %(levelname)s - %(message)s"
26
26
  )
27
27
  logger = logging.getLogger(__name__)
28
28
  logger.addHandler(logging.StreamHandler())
@@ -0,0 +1,194 @@
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ """
4
+ Created on Feb 10 2025.
5
+
6
+ @author: npappin-wsu
7
+ @license: MIT
8
+
9
+ Updated on May 14 2025.
10
+ """
11
+
12
+ from . import logger, session, metadata, bdcCache
13
+ import pandas as pd
14
+ import json, zipfile, io
15
+ from pprint import pprint
16
+ from .helpers import isEmpty
17
+
18
+ # from . import config
19
+
20
+
21
+ class availability:
22
+ """
23
+ A class to retrieve broadband availability data for specified states, technologies, and release dates.
24
+ """
25
+
26
+ def fixed(
27
+ states: int | str | list = "53",
28
+ technology: int | str | list = "50",
29
+ release: str | list = "2024-06-30",
30
+ cache: bool = False,
31
+ ) -> pd.DataFrame:
32
+ """
33
+ Retrieves broadband availability data for specified states, technologies, and release dates.
34
+
35
+ Args:
36
+ states (int | str | list, optional): State FIPS code(s) to filter by.
37
+ Can be a single value, a list of values, or "all" to include all states. Defaults to "53".
38
+ technology (int | str | list, optional): Technology code(s) to filter by.
39
+ Can be a single value, a list of values, "all" to include all technologies,
40
+ "fixed" for fixed technologies, or "mobile" for mobile technologies. Defaults to "50".
41
+ release (str | list, optional): Release date(s) to filter by. Can be a single value or a list of values. Defaults to "2024-06-30".
42
+ cache (bool, optional): Whether to use caching for downloaded files. Defaults to False.
43
+
44
+ Raises:
45
+ Exception: If one or more parameters are empty.
46
+ Exception: If availability data retrieval fails for a specific release.
47
+ Exception: If availability data retrieval fails for a specific state, technology, or release.
48
+
49
+ Returns:
50
+ pd.DataFrame: A DataFrame containing the filtered broadband availability data.
51
+ """
52
+ logger.info("Collecting availability...")
53
+ logger.debug(f"State: {states}")
54
+ logger.debug(f"Technology: {technology}")
55
+ logger.debug(f"Release: {release}")
56
+
57
+ # TODO: Add empty detection here
58
+ if isEmpty(states) or isEmpty(technology) or isEmpty(release):
59
+ raise Exception("One or more parameters are empty.")
60
+
61
+ # Normalization code
62
+ if type(states) is not list:
63
+ states = [states]
64
+ if type(technology) is not list:
65
+ technology = [technology]
66
+ if type(release) is not list:
67
+ release = [release]
68
+ technology = [str(t) for t in technology]
69
+ states = [str(s) for s in states]
70
+
71
+ # Retrieve availability data
72
+ availability = dict()
73
+ for r in release:
74
+ response = session.get(
75
+ f"https://broadbandmap.fcc.gov/api/public/map/downloads/listAvailabilityData/{r}"
76
+ )
77
+ if response.status_code != 200:
78
+ logger.error(f"Failed to retrieve availability data for {r}.")
79
+ raise Exception(f"Failed to retrieve availability data for {r}.")
80
+ # TODO: adding dtype hints here I think would be helpful.
81
+ availability[r] = pd.DataFrame.from_dict(response.json()["data"])
82
+ if "all" in states:
83
+ states = availability[r].state_fips.drop_duplicates().dropna().tolist()
84
+ if "all" in technology:
85
+ technology = (
86
+ availability[r].technology_code.drop_duplicates().dropna().tolist()
87
+ )
88
+ elif "fixed" in technology:
89
+ technology = (
90
+ availability[r][
91
+ (
92
+ (availability[r].subcategory == "Location Coverage")
93
+ & (availability[r].provider_id.isnull())
94
+ )
95
+ ]
96
+ .technology_code.drop_duplicates()
97
+ .dropna()
98
+ .tolist()
99
+ )
100
+ technology = [t for t in technology if int(t) < 100]
101
+ elif "mobile" in technology:
102
+ technology = (
103
+ availability[r][
104
+ (
105
+ (availability[r].subcategory == "Location Coverage")
106
+ & (availability[r].provider_id.isnull())
107
+ )
108
+ ]
109
+ .technology_code.drop_duplicates()
110
+ .dropna()
111
+ .tolist()
112
+ )
113
+ technology = [t for t in technology if int(t) >= 100]
114
+
115
+ df = pd.DataFrame()
116
+ columnHints = {
117
+ "frn": str,
118
+ "provider_id": "UInt32",
119
+ "brand_name": str,
120
+ "location_id": "UInt32",
121
+ "technology": "UInt16",
122
+ "max_advertised_download_speed": "UInt32",
123
+ "max_advertised_upload_speed": "UInt32",
124
+ "low_latency": "boolean",
125
+ "business_residental_code": "category",
126
+ "state_usps": "category",
127
+ "block_geoid": str,
128
+ "h3_res8_id": str,
129
+ }
130
+
131
+ if len(release) * len(states) * len(technology) > 100:
132
+ logger.warning(
133
+ f"Retrieving {len(release) * len(states) * len(technology)} records. This may take a while... or crash."
134
+ )
135
+
136
+ for r in release:
137
+ rlocal = availability[r]
138
+ items = rlocal[
139
+ (rlocal.category == "State")
140
+ & (rlocal.state_fips.isin(states))
141
+ & (rlocal.technology_code.isin(technology))
142
+ ].to_dict("records")
143
+ for item in items:
144
+ logger.debug(
145
+ f"State: {item['state_name']}({item})), Technology: {item['technology_code']}, Release: {r}"
146
+ )
147
+ # FUCK I DONT LIKE THIS
148
+ if cache and bdcCache.check(item["file_name"]):
149
+ data = bdcCache.get(item["file_name"])
150
+ logger.debug("Cache hit!")
151
+ pass
152
+ else:
153
+ response = session.get(
154
+ f"https://broadbandmap.fcc.gov/api/public/map/downloads/downloadFile/availability/{item['file_id']}"
155
+ )
156
+ if response.status_code == 200 and cache:
157
+ bdcCache.save(item["file_name"], response.content)
158
+ data = response.content
159
+ if response.status_code != 200:
160
+ logger.error(
161
+ f"Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
162
+ )
163
+ raise Exception(
164
+ f"Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
165
+ )
166
+ else:
167
+ zip = zipfile.ZipFile(io.BytesIO(data))
168
+ localdf = pd.read_csv(
169
+ zip.open(zip.filelist[0].filename),
170
+ dtype=columnHints,
171
+ dtype_backend="pyarrow",
172
+ )
173
+ df = pd.concat([df, localdf], ignore_index=True)
174
+ logger.info(
175
+ f"df memory size (hinted): {df.memory_usage(deep=True).sum()/1000000} MB"
176
+ )
177
+ logger.info(f"df shape: {df.shape}")
178
+ logger.debug(f"State: {states}, Technology: {technology}, Release: {release}")
179
+ return df
180
+
181
+
182
+ def echo(message):
183
+ logger.info(message)
184
+ print(message)
185
+ pass
186
+
187
+
188
+ def main():
189
+ logger.info("Starting the application...")
190
+ # Your code here
191
+
192
+
193
+ if __name__ == "__main__":
194
+ main()
@@ -32,7 +32,7 @@ def get_metadata():
32
32
  r = session.get("https://broadbandmap.fcc.gov/api/public/map/listAsOfDates")
33
33
  logger.debug(r.json())
34
34
  parsed = json.loads(r.text)
35
- logger.debug(parsed)
35
+ logger.info(parsed)
36
36
  logger.debug(parsed["data"])
37
37
  logger.info("Metadata collected.")
38
38
  types = set([item["data_type"] for item in parsed["data"]])
bdcdata-0.0.7/.gitignore DELETED
@@ -1,4 +0,0 @@
1
- **/__pycache__/
2
- .python-version
3
- .env
4
- .DS_Store
@@ -1,108 +0,0 @@
1
- #!/usr/bin/env python3
2
- # -*- coding: utf-8 -*-
3
- """
4
- Created on Feb 10 2025.
5
-
6
- @author: npappin-wsu
7
- @license: MIT
8
-
9
- Updated on Feb 11 2025.
10
- """
11
-
12
- from . import logger, session, metadata, bdcCache
13
- import pandas as pd
14
- import json, zipfile, io
15
- from pprint import pprint
16
- from .helpers import isEmpty
17
-
18
- # from . import config
19
-
20
-
21
- class availability:
22
-
23
- def state(
24
- states: int | str | list = "53",
25
- technology: int | str | list = "50",
26
- release: str | list = "2024-06-30",
27
- cache=False,
28
- ) -> pd.DataFrame:
29
- logger.info("Collecting availability...")
30
- logger.debug(f"State: {states}")
31
- logger.debug(f"Technology: {technology}")
32
- logger.debug(f"Release: {release}")
33
- # TODO: Add empty detection here
34
- if isEmpty(states) or isEmpty(technology) or isEmpty(release):
35
- raise Exception("One or more parameters are empty.")
36
- if type(states) is not list:
37
- states = [states]
38
- if type(technology) is not list:
39
- technology = [technology]
40
- if type(release) is not list:
41
- release = [release]
42
- # TODO: Add normalization code here
43
-
44
- # Retrieve availability data
45
- availability = dict()
46
- for r in release:
47
- response = session.get(
48
- f"https://broadbandmap.fcc.gov/api/public/map/downloads/listAvailabilityData/{r}"
49
- )
50
- if response.status_code != 200:
51
- logger.error(f"Failed to retrieve availability data for {r}.")
52
- raise Exception(f"Failed to retrieve availability data for {r}.")
53
- # TODO: adding dtype hints here I think would be helpful.
54
- availability[r] = pd.DataFrame.from_dict(response.json()["data"])
55
- df = pd.DataFrame()
56
- for r in release:
57
- rlocal = availability[r]
58
- items = rlocal[
59
- (rlocal.category == "State")
60
- & (rlocal.state_fips.isin(states))
61
- & (rlocal.technology_code.isin(technology))
62
- ].to_dict("records")
63
- for item in items:
64
- logger.debug(
65
- f"State: {item['state_name']}, Technology: {item['technology_code']}, Release: {r}"
66
- )
67
- # FUCK I DONT LIKE THIS
68
- if cache and bdcCache.check(item["file_name"]):
69
- data = bdcCache.get(item["file_name"])
70
- logger.debug("Cache hit!")
71
- pass
72
- else:
73
- response = session.get(
74
- f"https://broadbandmap.fcc.gov/api/public/map/downloads/downloadFile/availability/{item['file_id']}"
75
- )
76
- if response.status_code == 200 and cache:
77
- bdcCache.save(item["file_name"], response.content)
78
- data = response.content
79
- if response.status_code != 200:
80
- logger.error(
81
- f"Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
82
- )
83
- raise Exception(
84
- f"Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
85
- )
86
- else:
87
- zip = zipfile.ZipFile(io.BytesIO(data))
88
- localdf = pd.read_csv(
89
- zip.open(zip.filelist[0].filename)
90
- ) # , dtype=columnHints)
91
- df = pd.concat([df, localdf], ignore_index=True)
92
- logger.debug(f"State: {states}, Technology: {technology}, Release: {release}")
93
- return df
94
-
95
-
96
- def echo(message):
97
- logger.info(message)
98
- print(message)
99
- pass
100
-
101
-
102
- def main():
103
- logger.info("Starting the application...")
104
- # Your code here
105
-
106
-
107
- if __name__ == "__main__":
108
- main()
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes