bdcdata 0.0.7__tar.gz → 0.0.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bdcdata-0.0.8/.gitignore +8 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/PKG-INFO +1 -1
- bdcdata-0.0.8/broadband-map-data-downloads.pdf +0 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/src/bdcdata/__init__.py +1 -1
- bdcdata-0.0.8/src/bdcdata/bdc.py +194 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/src/bdcdata/helpers.py +1 -1
- bdcdata-0.0.7/.gitignore +0 -4
- bdcdata-0.0.7/src/bdcdata/bdc.py +0 -108
- {bdcdata-0.0.7 → bdcdata-0.0.8}/.github/workflows/python-publish.yml +0 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/LICENSE.txt +0 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/README.md +0 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/bdc-public-data-api-specifications.pdf +0 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/docs/.env.sample +0 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/docs/index.md +0 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/hatch.toml +0 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/pyproject.toml +0 -0
- {bdcdata-0.0.7 → bdcdata-0.0.8}/requirements.txt +0 -0
bdcdata-0.0.8/.gitignore
ADDED
|
Binary file
|
|
@@ -22,7 +22,7 @@ username = os.getenv("BDC_USERNAME")
|
|
|
22
22
|
|
|
23
23
|
# Configure logging
|
|
24
24
|
logging.basicConfig(
|
|
25
|
-
filename="bdc.log", filemode="
|
|
25
|
+
filename="bdc.log", filemode="a", format="%(asctime)s - %(levelname)s - %(message)s"
|
|
26
26
|
)
|
|
27
27
|
logger = logging.getLogger(__name__)
|
|
28
28
|
logger.addHandler(logging.StreamHandler())
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""
|
|
4
|
+
Created on Feb 10 2025.
|
|
5
|
+
|
|
6
|
+
@author: npappin-wsu
|
|
7
|
+
@license: MIT
|
|
8
|
+
|
|
9
|
+
Updated on May 14 2025.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from . import logger, session, metadata, bdcCache
|
|
13
|
+
import pandas as pd
|
|
14
|
+
import json, zipfile, io
|
|
15
|
+
from pprint import pprint
|
|
16
|
+
from .helpers import isEmpty
|
|
17
|
+
|
|
18
|
+
# from . import config
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class availability:
|
|
22
|
+
"""
|
|
23
|
+
A class to retrieve broadband availability data for specified states, technologies, and release dates.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
def fixed(
|
|
27
|
+
states: int | str | list = "53",
|
|
28
|
+
technology: int | str | list = "50",
|
|
29
|
+
release: str | list = "2024-06-30",
|
|
30
|
+
cache: bool = False,
|
|
31
|
+
) -> pd.DataFrame:
|
|
32
|
+
"""
|
|
33
|
+
Retrieves broadband availability data for specified states, technologies, and release dates.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
states (int | str | list, optional): State FIPS code(s) to filter by.
|
|
37
|
+
Can be a single value, a list of values, or "all" to include all states. Defaults to "53".
|
|
38
|
+
technology (int | str | list, optional): Technology code(s) to filter by.
|
|
39
|
+
Can be a single value, a list of values, "all" to include all technologies,
|
|
40
|
+
"fixed" for fixed technologies, or "mobile" for mobile technologies. Defaults to "50".
|
|
41
|
+
release (str | list, optional): Release date(s) to filter by. Can be a single value or a list of values. Defaults to "2024-06-30".
|
|
42
|
+
cache (bool, optional): Whether to use caching for downloaded files. Defaults to False.
|
|
43
|
+
|
|
44
|
+
Raises:
|
|
45
|
+
Exception: If one or more parameters are empty.
|
|
46
|
+
Exception: If availability data retrieval fails for a specific release.
|
|
47
|
+
Exception: If availability data retrieval fails for a specific state, technology, or release.
|
|
48
|
+
|
|
49
|
+
Returns:
|
|
50
|
+
pd.DataFrame: A DataFrame containing the filtered broadband availability data.
|
|
51
|
+
"""
|
|
52
|
+
logger.info("Collecting availability...")
|
|
53
|
+
logger.debug(f"State: {states}")
|
|
54
|
+
logger.debug(f"Technology: {technology}")
|
|
55
|
+
logger.debug(f"Release: {release}")
|
|
56
|
+
|
|
57
|
+
# TODO: Add empty detection here
|
|
58
|
+
if isEmpty(states) or isEmpty(technology) or isEmpty(release):
|
|
59
|
+
raise Exception("One or more parameters are empty.")
|
|
60
|
+
|
|
61
|
+
# Normalization code
|
|
62
|
+
if type(states) is not list:
|
|
63
|
+
states = [states]
|
|
64
|
+
if type(technology) is not list:
|
|
65
|
+
technology = [technology]
|
|
66
|
+
if type(release) is not list:
|
|
67
|
+
release = [release]
|
|
68
|
+
technology = [str(t) for t in technology]
|
|
69
|
+
states = [str(s) for s in states]
|
|
70
|
+
|
|
71
|
+
# Retrieve availability data
|
|
72
|
+
availability = dict()
|
|
73
|
+
for r in release:
|
|
74
|
+
response = session.get(
|
|
75
|
+
f"https://broadbandmap.fcc.gov/api/public/map/downloads/listAvailabilityData/{r}"
|
|
76
|
+
)
|
|
77
|
+
if response.status_code != 200:
|
|
78
|
+
logger.error(f"Failed to retrieve availability data for {r}.")
|
|
79
|
+
raise Exception(f"Failed to retrieve availability data for {r}.")
|
|
80
|
+
# TODO: adding dtype hints here I think would be helpful.
|
|
81
|
+
availability[r] = pd.DataFrame.from_dict(response.json()["data"])
|
|
82
|
+
if "all" in states:
|
|
83
|
+
states = availability[r].state_fips.drop_duplicates().dropna().tolist()
|
|
84
|
+
if "all" in technology:
|
|
85
|
+
technology = (
|
|
86
|
+
availability[r].technology_code.drop_duplicates().dropna().tolist()
|
|
87
|
+
)
|
|
88
|
+
elif "fixed" in technology:
|
|
89
|
+
technology = (
|
|
90
|
+
availability[r][
|
|
91
|
+
(
|
|
92
|
+
(availability[r].subcategory == "Location Coverage")
|
|
93
|
+
& (availability[r].provider_id.isnull())
|
|
94
|
+
)
|
|
95
|
+
]
|
|
96
|
+
.technology_code.drop_duplicates()
|
|
97
|
+
.dropna()
|
|
98
|
+
.tolist()
|
|
99
|
+
)
|
|
100
|
+
technology = [t for t in technology if int(t) < 100]
|
|
101
|
+
elif "mobile" in technology:
|
|
102
|
+
technology = (
|
|
103
|
+
availability[r][
|
|
104
|
+
(
|
|
105
|
+
(availability[r].subcategory == "Location Coverage")
|
|
106
|
+
& (availability[r].provider_id.isnull())
|
|
107
|
+
)
|
|
108
|
+
]
|
|
109
|
+
.technology_code.drop_duplicates()
|
|
110
|
+
.dropna()
|
|
111
|
+
.tolist()
|
|
112
|
+
)
|
|
113
|
+
technology = [t for t in technology if int(t) >= 100]
|
|
114
|
+
|
|
115
|
+
df = pd.DataFrame()
|
|
116
|
+
columnHints = {
|
|
117
|
+
"frn": str,
|
|
118
|
+
"provider_id": "UInt32",
|
|
119
|
+
"brand_name": str,
|
|
120
|
+
"location_id": "UInt32",
|
|
121
|
+
"technology": "UInt16",
|
|
122
|
+
"max_advertised_download_speed": "UInt32",
|
|
123
|
+
"max_advertised_upload_speed": "UInt32",
|
|
124
|
+
"low_latency": "boolean",
|
|
125
|
+
"business_residental_code": "category",
|
|
126
|
+
"state_usps": "category",
|
|
127
|
+
"block_geoid": str,
|
|
128
|
+
"h3_res8_id": str,
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
if len(release) * len(states) * len(technology) > 100:
|
|
132
|
+
logger.warning(
|
|
133
|
+
f"Retrieving {len(release) * len(states) * len(technology)} records. This may take a while... or crash."
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
for r in release:
|
|
137
|
+
rlocal = availability[r]
|
|
138
|
+
items = rlocal[
|
|
139
|
+
(rlocal.category == "State")
|
|
140
|
+
& (rlocal.state_fips.isin(states))
|
|
141
|
+
& (rlocal.technology_code.isin(technology))
|
|
142
|
+
].to_dict("records")
|
|
143
|
+
for item in items:
|
|
144
|
+
logger.debug(
|
|
145
|
+
f"State: {item['state_name']}({item})), Technology: {item['technology_code']}, Release: {r}"
|
|
146
|
+
)
|
|
147
|
+
# FUCK I DONT LIKE THIS
|
|
148
|
+
if cache and bdcCache.check(item["file_name"]):
|
|
149
|
+
data = bdcCache.get(item["file_name"])
|
|
150
|
+
logger.debug("Cache hit!")
|
|
151
|
+
pass
|
|
152
|
+
else:
|
|
153
|
+
response = session.get(
|
|
154
|
+
f"https://broadbandmap.fcc.gov/api/public/map/downloads/downloadFile/availability/{item['file_id']}"
|
|
155
|
+
)
|
|
156
|
+
if response.status_code == 200 and cache:
|
|
157
|
+
bdcCache.save(item["file_name"], response.content)
|
|
158
|
+
data = response.content
|
|
159
|
+
if response.status_code != 200:
|
|
160
|
+
logger.error(
|
|
161
|
+
f"Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
|
|
162
|
+
)
|
|
163
|
+
raise Exception(
|
|
164
|
+
f"Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
|
|
165
|
+
)
|
|
166
|
+
else:
|
|
167
|
+
zip = zipfile.ZipFile(io.BytesIO(data))
|
|
168
|
+
localdf = pd.read_csv(
|
|
169
|
+
zip.open(zip.filelist[0].filename),
|
|
170
|
+
dtype=columnHints,
|
|
171
|
+
dtype_backend="pyarrow",
|
|
172
|
+
)
|
|
173
|
+
df = pd.concat([df, localdf], ignore_index=True)
|
|
174
|
+
logger.info(
|
|
175
|
+
f"df memory size (hinted): {df.memory_usage(deep=True).sum()/1000000} MB"
|
|
176
|
+
)
|
|
177
|
+
logger.info(f"df shape: {df.shape}")
|
|
178
|
+
logger.debug(f"State: {states}, Technology: {technology}, Release: {release}")
|
|
179
|
+
return df
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def echo(message):
|
|
183
|
+
logger.info(message)
|
|
184
|
+
print(message)
|
|
185
|
+
pass
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def main():
|
|
189
|
+
logger.info("Starting the application...")
|
|
190
|
+
# Your code here
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
if __name__ == "__main__":
|
|
194
|
+
main()
|
|
@@ -32,7 +32,7 @@ def get_metadata():
|
|
|
32
32
|
r = session.get("https://broadbandmap.fcc.gov/api/public/map/listAsOfDates")
|
|
33
33
|
logger.debug(r.json())
|
|
34
34
|
parsed = json.loads(r.text)
|
|
35
|
-
logger.
|
|
35
|
+
logger.info(parsed)
|
|
36
36
|
logger.debug(parsed["data"])
|
|
37
37
|
logger.info("Metadata collected.")
|
|
38
38
|
types = set([item["data_type"] for item in parsed["data"]])
|
bdcdata-0.0.7/.gitignore
DELETED
bdcdata-0.0.7/src/bdcdata/bdc.py
DELETED
|
@@ -1,108 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env python3
|
|
2
|
-
# -*- coding: utf-8 -*-
|
|
3
|
-
"""
|
|
4
|
-
Created on Feb 10 2025.
|
|
5
|
-
|
|
6
|
-
@author: npappin-wsu
|
|
7
|
-
@license: MIT
|
|
8
|
-
|
|
9
|
-
Updated on Feb 11 2025.
|
|
10
|
-
"""
|
|
11
|
-
|
|
12
|
-
from . import logger, session, metadata, bdcCache
|
|
13
|
-
import pandas as pd
|
|
14
|
-
import json, zipfile, io
|
|
15
|
-
from pprint import pprint
|
|
16
|
-
from .helpers import isEmpty
|
|
17
|
-
|
|
18
|
-
# from . import config
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
class availability:
|
|
22
|
-
|
|
23
|
-
def state(
|
|
24
|
-
states: int | str | list = "53",
|
|
25
|
-
technology: int | str | list = "50",
|
|
26
|
-
release: str | list = "2024-06-30",
|
|
27
|
-
cache=False,
|
|
28
|
-
) -> pd.DataFrame:
|
|
29
|
-
logger.info("Collecting availability...")
|
|
30
|
-
logger.debug(f"State: {states}")
|
|
31
|
-
logger.debug(f"Technology: {technology}")
|
|
32
|
-
logger.debug(f"Release: {release}")
|
|
33
|
-
# TODO: Add empty detection here
|
|
34
|
-
if isEmpty(states) or isEmpty(technology) or isEmpty(release):
|
|
35
|
-
raise Exception("One or more parameters are empty.")
|
|
36
|
-
if type(states) is not list:
|
|
37
|
-
states = [states]
|
|
38
|
-
if type(technology) is not list:
|
|
39
|
-
technology = [technology]
|
|
40
|
-
if type(release) is not list:
|
|
41
|
-
release = [release]
|
|
42
|
-
# TODO: Add normalization code here
|
|
43
|
-
|
|
44
|
-
# Retrieve availability data
|
|
45
|
-
availability = dict()
|
|
46
|
-
for r in release:
|
|
47
|
-
response = session.get(
|
|
48
|
-
f"https://broadbandmap.fcc.gov/api/public/map/downloads/listAvailabilityData/{r}"
|
|
49
|
-
)
|
|
50
|
-
if response.status_code != 200:
|
|
51
|
-
logger.error(f"Failed to retrieve availability data for {r}.")
|
|
52
|
-
raise Exception(f"Failed to retrieve availability data for {r}.")
|
|
53
|
-
# TODO: adding dtype hints here I think would be helpful.
|
|
54
|
-
availability[r] = pd.DataFrame.from_dict(response.json()["data"])
|
|
55
|
-
df = pd.DataFrame()
|
|
56
|
-
for r in release:
|
|
57
|
-
rlocal = availability[r]
|
|
58
|
-
items = rlocal[
|
|
59
|
-
(rlocal.category == "State")
|
|
60
|
-
& (rlocal.state_fips.isin(states))
|
|
61
|
-
& (rlocal.technology_code.isin(technology))
|
|
62
|
-
].to_dict("records")
|
|
63
|
-
for item in items:
|
|
64
|
-
logger.debug(
|
|
65
|
-
f"State: {item['state_name']}, Technology: {item['technology_code']}, Release: {r}"
|
|
66
|
-
)
|
|
67
|
-
# FUCK I DONT LIKE THIS
|
|
68
|
-
if cache and bdcCache.check(item["file_name"]):
|
|
69
|
-
data = bdcCache.get(item["file_name"])
|
|
70
|
-
logger.debug("Cache hit!")
|
|
71
|
-
pass
|
|
72
|
-
else:
|
|
73
|
-
response = session.get(
|
|
74
|
-
f"https://broadbandmap.fcc.gov/api/public/map/downloads/downloadFile/availability/{item['file_id']}"
|
|
75
|
-
)
|
|
76
|
-
if response.status_code == 200 and cache:
|
|
77
|
-
bdcCache.save(item["file_name"], response.content)
|
|
78
|
-
data = response.content
|
|
79
|
-
if response.status_code != 200:
|
|
80
|
-
logger.error(
|
|
81
|
-
f"Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
|
|
82
|
-
)
|
|
83
|
-
raise Exception(
|
|
84
|
-
f"Failed to retrieve availability data for {item['state_name']} and {item['technology_code']} in {r}."
|
|
85
|
-
)
|
|
86
|
-
else:
|
|
87
|
-
zip = zipfile.ZipFile(io.BytesIO(data))
|
|
88
|
-
localdf = pd.read_csv(
|
|
89
|
-
zip.open(zip.filelist[0].filename)
|
|
90
|
-
) # , dtype=columnHints)
|
|
91
|
-
df = pd.concat([df, localdf], ignore_index=True)
|
|
92
|
-
logger.debug(f"State: {states}, Technology: {technology}, Release: {release}")
|
|
93
|
-
return df
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
def echo(message):
|
|
97
|
-
logger.info(message)
|
|
98
|
-
print(message)
|
|
99
|
-
pass
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
def main():
|
|
103
|
-
logger.info("Starting the application...")
|
|
104
|
-
# Your code here
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
if __name__ == "__main__":
|
|
108
|
-
main()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|