bdcdata 0.0.9__tar.gz → 0.0.11__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bdcdata
3
- Version: 0.0.9
3
+ Version: 0.0.11
4
4
  Summary: A tool to work with BDC data from the FCC.
5
5
  Project-URL: Homepage, https://github.com/npappin-wsu/bdc
6
6
  Project-URL: Issues, https://github.com/npappin-wsu/bdc/issues
@@ -38,4 +38,7 @@ from .helpers import get_metadata, bdcCache
38
38
 
39
39
  metadata = get_metadata()
40
40
 
41
- from .bdc import *
41
+ from . import availability
42
+ from . import fabric
43
+ from . import challenge
44
+ from . import funding
@@ -0,0 +1,181 @@
1
+ from . import logger, session, metadata, bdcCache
2
+ import pandas as pd
3
+ import json, zipfile, io
4
+ from pprint import pprint
5
+ from .helpers import isEmpty
6
+
7
+ def fixed(
8
+ states: int | str | list = "53",
9
+ technology: int | str | list = "50",
10
+ release: str | list = "2024-06-30",
11
+ cache: bool = False,
12
+ ) -> pd.DataFrame:
13
+ """
14
+ Retrieves broadband availability data for specified states, technologies, and release dates.
15
+
16
+ Args:
17
+ states (int | str | list, optional): State FIPS code(s) to filter by.
18
+ Can be a single value, a list of values, or "all" to include all states. Defaults to "53".
19
+ technology (int | str | list, optional): Technology code(s) to filter by.
20
+ Can be a single value, a list of values, "all" to include all technologies,
21
+ "fixed" for fixed technologies, or "mobile" for mobile technologies. Defaults to "50".
22
+ release (str | list, optional): Release date(s) to filter by. Can be a single value or a list of values. Defaults to "2024-06-30".
23
+ cache (bool, optional): Whether to use caching for downloaded files. Defaults to False.
24
+
25
+ Raises:
26
+ Exception: If one or more parameters are empty.
27
+ Exception: If availability data retrieval fails for a specific release.
28
+ Exception: If availability data retrieval fails for a specific state, technology, or release.
29
+
30
+ Returns:
31
+ pd.DataFrame: A DataFrame containing the filtered broadband availability data.
32
+ """
33
+ logger.info("Collecting availability...")
34
+ logger.debug(f"State: {states}")
35
+ logger.debug(f"Technology: {technology}")
36
+ logger.debug(f"Release: {release}")
37
+
38
+ # TODO: Add empty detection here
39
+ if isEmpty(states) or isEmpty(technology) or isEmpty(release):
40
+ raise Exception("One or more parameters are empty.")
41
+
42
+ # Normalization code
43
+ if type(states) is not list:
44
+ states = [states]
45
+ if type(technology) is not list:
46
+ technology = [technology]
47
+ if type(release) is not list:
48
+ release = [release]
49
+ technology = [str(t) for t in technology]
50
+ states = [str(s) for s in states]
51
+
52
+ # Retrieve availability data
53
+ availability = dict()
54
+ for r in release:
55
+ response = session.get(
56
+ f"https://broadbandmap.fcc.gov/api/public/map/downloads/listAvailabilityData/{r}"
57
+ )
58
+ if response.status_code != 200:
59
+ logger.error(f"Failed to retrieve availability data for {r}.")
60
+ raise Exception(f"Failed to retrieve availability data for {r}.")
61
+ # TODO: adding dtype hints here I think would be helpful.
62
+ availability[r] = pd.DataFrame.from_dict(response.json()["data"])
63
+ if "all" in states:
64
+ states = availability[r].state_fips.drop_duplicates().dropna().tolist()
65
+ if "all" in technology or "fixed" in technology:
66
+ technology = (
67
+ availability[r][
68
+ (
69
+ (availability[r].subcategory == "Location Coverage")
70
+ & (availability[r].provider_id.isnull())
71
+ )
72
+ ]
73
+ .technology_code.drop_duplicates()
74
+ .dropna()
75
+ .tolist()
76
+ )
77
+ technology = [t for t in technology if int(t) < 100]
78
+ # TODO: Implement all release handling here.
79
+ # if "all" in release:
80
+
81
+
82
+ # elif "fixed" in technology:
83
+ # technology = (
84
+ # availability[r][
85
+ # (
86
+ # (availability[r].subcategory == "Location Coverage")
87
+ # & (availability[r].provider_id.isnull())
88
+ # )
89
+ # ]
90
+ # .technology_code.drop_duplicates()
91
+ # .dropna()
92
+ # .tolist()
93
+ # )
94
+ # technology = [t for t in technology if int(t) < 100]
95
+ # elif "mobile" in technology:
96
+ # technology = (
97
+ # availability[r][
98
+ # (
99
+ # (availability[r].subcategory == "Location Coverage")
100
+ # & (availability[r].provider_id.isnull())
101
+ # )
102
+ # ]
103
+ # .technology_code.drop_duplicates()
104
+ # .dropna()
105
+ # .tolist()
106
+ # )
107
+ # technology = [t for t in technology if int(t) >= 100]
108
+
109
+ df = pd.DataFrame()
110
+ columnHints = {
111
+ "frn": str,
112
+ "provider_id": "UInt32",
113
+ "brand_name": str,
114
+ "location_id": "UInt32",
115
+ "technology": "UInt16",
116
+ "max_advertised_download_speed": "UInt32",
117
+ "max_advertised_upload_speed": "UInt32",
118
+ "low_latency": "boolean",
119
+ "business_residental_code": "category",
120
+ "state_usps": "category",
121
+ "block_geoid": str,
122
+ "h3_res8_id": str,
123
+ }
124
+
125
+ if len(release) * len(states) * len(technology) > 100:
126
+ logger.warning(
127
+ f"Retrieving {len(release) * len(states) * len(technology)} records. This may take a while... or crash."
128
+ )
129
+
130
+ for r in release:
131
+ rlocal = availability[r]
132
+ items = rlocal[
133
+ (rlocal.category == "State")
134
+ & (rlocal.state_fips.isin(states))
135
+ & (rlocal.technology_code.isin(technology))
136
+ ].to_dict("records")
137
+ for item in items:
138
+ logger.debug(
139
+ f"State: {item['state_name']}({item})), Technology: {item['technology_code']}, Release: {r}"
140
+ )
141
+ # FUCK I DONT LIKE THIS
142
+ if cache and bdcCache.check(item["file_name"]):
143
+ data = bdcCache.get(item["file_name"])
144
+ logger.debug("Cache hit!")
145
+ pass
146
+ else:
147
+ response = session.get(
148
+ f"https://broadbandmap.fcc.gov/api/public/map/downloads/downloadFile/availability/{item['file_id']}"
149
+ )
150
+ if response.status_code == 200 and cache:
151
+ bdcCache.save(item["file_name"], response.content)
152
+ data = response.content
153
+ if response.status_code == 422:
154
+ logger.error(
155
+ f"(Error Code: 422) Unprocessable Entity for {item['state_name']} and {item['technology_code']} in {r}."
156
+ )
157
+ raise Exception(
158
+ f"(Error Code: 422) Unprocessable Entity for {item['state_name']} and {item['technology_code']} in {r}."
159
+ )
160
+ elif response.status_code != 200:
161
+ logger.error(
162
+ f"(Error Code: {response.status_code}) Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
163
+ )
164
+ raise Exception(
165
+ f"(Error Code: {response.status_code}) Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
166
+ )
167
+ else:
168
+ zip = zipfile.ZipFile(io.BytesIO(data))
169
+ localdf = pd.read_csv(
170
+ zip.open(zip.filelist[0].filename),
171
+ dtype=columnHints,
172
+ dtype_backend="pyarrow",
173
+ )
174
+ localdf['release'] = r
175
+ df = pd.concat([df, localdf], ignore_index=True)
176
+ # logger.info(
177
+ # f"df memory size (hinted): {df.memory_usage(deep=True).sum()/1000000} MB"
178
+ # )
179
+ # logger.info(f"df shape: {df.shape}")
180
+ logger.debug(f"State: {states}, Technology: {technology}, Release: {release}")
181
+ return df
@@ -0,0 +1,41 @@
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ """
4
+ Created on Feb 10 2025.
5
+
6
+ @author: npappin-wsu
7
+ @license: MIT
8
+
9
+ Updated on May 14 2025.
10
+ """
11
+
12
+ from . import logger, session, metadata, bdcCache
13
+ import pandas as pd
14
+ import json, zipfile, io
15
+ from pprint import pprint
16
+ from .helpers import isEmpty
17
+
18
+ # from . import config
19
+
20
+
21
+ # class availability:
22
+ # """
23
+ # A class to retrieve broadband availability data for specified states, technologies, and release dates.
24
+ # """
25
+
26
+
27
+
28
+
29
+ def echo(message):
30
+ logger.info(message)
31
+ print(message)
32
+ pass
33
+
34
+
35
+ def main():
36
+ logger.info("Starting the application...")
37
+ # Your code here
38
+
39
+
40
+ if __name__ == "__main__":
41
+ main()
@@ -0,0 +1,5 @@
1
+ from . import logger, session, metadata, bdcCache
2
+ import pandas as pd
3
+ import json, zipfile, io
4
+ from pprint import pprint
5
+ from .helpers import isEmpty
@@ -0,0 +1,19 @@
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+
4
+ import pathlib
5
+ import pandas as pd
6
+
7
+ def process(
8
+ files: list[pathlib.Path]
9
+ ) -> bool:
10
+ print("Processing data...")
11
+ return True
12
+
13
+ def load(
14
+ release: int | str
15
+ ) -> pd.DataFrame:
16
+ print("Loading data...")
17
+ if pathlib.Path('cache', ).exists():
18
+ df = pd.read_csv("data/processed/fabric.csv")
19
+ return df
@@ -0,0 +1,5 @@
1
+ from . import logger, session, metadata, bdcCache
2
+ import pandas as pd
3
+ import json, zipfile, io
4
+ from pprint import pprint
5
+ from .helpers import isEmpty
@@ -1,194 +0,0 @@
1
- #!/usr/bin/env python3
2
- # -*- coding: utf-8 -*-
3
- """
4
- Created on Feb 10 2025.
5
-
6
- @author: npappin-wsu
7
- @license: MIT
8
-
9
- Updated on May 14 2025.
10
- """
11
-
12
- from . import logger, session, metadata, bdcCache
13
- import pandas as pd
14
- import json, zipfile, io
15
- from pprint import pprint
16
- from .helpers import isEmpty
17
-
18
- # from . import config
19
-
20
-
21
- class availability:
22
- """
23
- A class to retrieve broadband availability data for specified states, technologies, and release dates.
24
- """
25
-
26
- def fixed(
27
- states: int | str | list = "53",
28
- technology: int | str | list = "50",
29
- release: str | list = "2024-06-30",
30
- cache: bool = False,
31
- ) -> pd.DataFrame:
32
- """
33
- Retrieves broadband availability data for specified states, technologies, and release dates.
34
-
35
- Args:
36
- states (int | str | list, optional): State FIPS code(s) to filter by.
37
- Can be a single value, a list of values, or "all" to include all states. Defaults to "53".
38
- technology (int | str | list, optional): Technology code(s) to filter by.
39
- Can be a single value, a list of values, "all" to include all technologies,
40
- "fixed" for fixed technologies, or "mobile" for mobile technologies. Defaults to "50".
41
- release (str | list, optional): Release date(s) to filter by. Can be a single value or a list of values. Defaults to "2024-06-30".
42
- cache (bool, optional): Whether to use caching for downloaded files. Defaults to False.
43
-
44
- Raises:
45
- Exception: If one or more parameters are empty.
46
- Exception: If availability data retrieval fails for a specific release.
47
- Exception: If availability data retrieval fails for a specific state, technology, or release.
48
-
49
- Returns:
50
- pd.DataFrame: A DataFrame containing the filtered broadband availability data.
51
- """
52
- logger.info("Collecting availability...")
53
- logger.debug(f"State: {states}")
54
- logger.debug(f"Technology: {technology}")
55
- logger.debug(f"Release: {release}")
56
-
57
- # TODO: Add empty detection here
58
- if isEmpty(states) or isEmpty(technology) or isEmpty(release):
59
- raise Exception("One or more parameters are empty.")
60
-
61
- # Normalization code
62
- if type(states) is not list:
63
- states = [states]
64
- if type(technology) is not list:
65
- technology = [technology]
66
- if type(release) is not list:
67
- release = [release]
68
- technology = [str(t) for t in technology]
69
- states = [str(s) for s in states]
70
-
71
- # Retrieve availability data
72
- availability = dict()
73
- for r in release:
74
- response = session.get(
75
- f"https://broadbandmap.fcc.gov/api/public/map/downloads/listAvailabilityData/{r}"
76
- )
77
- if response.status_code != 200:
78
- logger.error(f"Failed to retrieve availability data for {r}.")
79
- raise Exception(f"Failed to retrieve availability data for {r}.")
80
- # TODO: adding dtype hints here I think would be helpful.
81
- availability[r] = pd.DataFrame.from_dict(response.json()["data"])
82
- if "all" in states:
83
- states = availability[r].state_fips.drop_duplicates().dropna().tolist()
84
- if "all" in technology:
85
- technology = (
86
- availability[r].technology_code.drop_duplicates().dropna().tolist()
87
- )
88
- elif "fixed" in technology:
89
- technology = (
90
- availability[r][
91
- (
92
- (availability[r].subcategory == "Location Coverage")
93
- & (availability[r].provider_id.isnull())
94
- )
95
- ]
96
- .technology_code.drop_duplicates()
97
- .dropna()
98
- .tolist()
99
- )
100
- technology = [t for t in technology if int(t) < 100]
101
- elif "mobile" in technology:
102
- technology = (
103
- availability[r][
104
- (
105
- (availability[r].subcategory == "Location Coverage")
106
- & (availability[r].provider_id.isnull())
107
- )
108
- ]
109
- .technology_code.drop_duplicates()
110
- .dropna()
111
- .tolist()
112
- )
113
- technology = [t for t in technology if int(t) >= 100]
114
-
115
- df = pd.DataFrame()
116
- columnHints = {
117
- "frn": str,
118
- "provider_id": "UInt32",
119
- "brand_name": str,
120
- "location_id": "UInt32",
121
- "technology": "UInt16",
122
- "max_advertised_download_speed": "UInt32",
123
- "max_advertised_upload_speed": "UInt32",
124
- "low_latency": "boolean",
125
- "business_residental_code": "category",
126
- "state_usps": "category",
127
- "block_geoid": str,
128
- "h3_res8_id": str,
129
- }
130
-
131
- if len(release) * len(states) * len(technology) > 100:
132
- logger.warning(
133
- f"Retrieving {len(release) * len(states) * len(technology)} records. This may take a while... or crash."
134
- )
135
-
136
- for r in release:
137
- rlocal = availability[r]
138
- items = rlocal[
139
- (rlocal.category == "State")
140
- & (rlocal.state_fips.isin(states))
141
- & (rlocal.technology_code.isin(technology))
142
- ].to_dict("records")
143
- for item in items:
144
- logger.debug(
145
- f"State: {item['state_name']}({item})), Technology: {item['technology_code']}, Release: {r}"
146
- )
147
- # FUCK I DONT LIKE THIS
148
- if cache and bdcCache.check(item["file_name"]):
149
- data = bdcCache.get(item["file_name"])
150
- logger.debug("Cache hit!")
151
- pass
152
- else:
153
- response = session.get(
154
- f"https://broadbandmap.fcc.gov/api/public/map/downloads/downloadFile/availability/{item['file_id']}"
155
- )
156
- if response.status_code == 200 and cache:
157
- bdcCache.save(item["file_name"], response.content)
158
- data = response.content
159
- if response.status_code != 200:
160
- logger.error(
161
- f"Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
162
- )
163
- raise Exception(
164
- f"Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
165
- )
166
- else:
167
- zip = zipfile.ZipFile(io.BytesIO(data))
168
- localdf = pd.read_csv(
169
- zip.open(zip.filelist[0].filename),
170
- dtype=columnHints,
171
- dtype_backend="pyarrow",
172
- )
173
- df = pd.concat([df, localdf], ignore_index=True)
174
- # logger.info(
175
- # f"df memory size (hinted): {df.memory_usage(deep=True).sum()/1000000} MB"
176
- # )
177
- # logger.info(f"df shape: {df.shape}")
178
- logger.debug(f"State: {states}, Technology: {technology}, Release: {release}")
179
- return df
180
-
181
-
182
- def echo(message):
183
- logger.info(message)
184
- print(message)
185
- pass
186
-
187
-
188
- def main():
189
- logger.info("Starting the application...")
190
- # Your code here
191
-
192
-
193
- if __name__ == "__main__":
194
- main()
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes