bdcdata 0.0.11__tar.gz → 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bdcdata-2.0.0/.github/workflows/ci.yml +50 -0
- bdcdata-2.0.0/.github/workflows/publish.yml +37 -0
- bdcdata-2.0.0/.gitignore +24 -0
- bdcdata-2.0.0/CHANGELOG.md +103 -0
- {bdcdata-0.0.11 → bdcdata-2.0.0}/LICENSE.txt +2 -2
- bdcdata-2.0.0/PKG-INFO +112 -0
- bdcdata-2.0.0/README.md +75 -0
- bdcdata-2.0.0/docs/availability.md +201 -0
- bdcdata-2.0.0/docs/challenges.md +156 -0
- bdcdata-2.0.0/docs/credentials.md +146 -0
- bdcdata-2.0.0/docs/funding.md +156 -0
- bdcdata-2.0.0/docs/index.md +87 -0
- bdcdata-2.0.0/docs/migrating-from-v1.md +224 -0
- bdcdata-2.0.0/docs/quickstart.md +183 -0
- bdcdata-2.0.0/mkdocs.yml +48 -0
- bdcdata-2.0.0/pyproject.toml +100 -0
- bdcdata-2.0.0/src/bdcdata/__init__.py +91 -0
- bdcdata-2.0.0/src/bdcdata/_cache.py +128 -0
- bdcdata-2.0.0/src/bdcdata/_client.py +318 -0
- bdcdata-2.0.0/src/bdcdata/_fetch.py +173 -0
- bdcdata-2.0.0/src/bdcdata/_normalize.py +356 -0
- bdcdata-2.0.0/src/bdcdata/_readers.py +179 -0
- bdcdata-2.0.0/src/bdcdata/_schemas.py +211 -0
- bdcdata-2.0.0/src/bdcdata/availability.py +429 -0
- bdcdata-2.0.0/src/bdcdata/catalog.py +275 -0
- bdcdata-2.0.0/src/bdcdata/challenges.py +297 -0
- bdcdata-2.0.0/src/bdcdata/config.py +119 -0
- bdcdata-2.0.0/src/bdcdata/credentials.py +185 -0
- bdcdata-2.0.0/src/bdcdata/exceptions.py +101 -0
- bdcdata-2.0.0/src/bdcdata/funding.py +370 -0
- bdcdata-2.0.0/src/bdcdata/lookups.py +216 -0
- bdcdata-2.0.0/src/bdcdata/py.typed +0 -0
- bdcdata-2.0.0/tests/conftest.py +65 -0
- bdcdata-2.0.0/tests/helpers.py +114 -0
- bdcdata-2.0.0/tests/integration/test_live.py +219 -0
- bdcdata-2.0.0/tests/test_availability.py +290 -0
- bdcdata-2.0.0/tests/test_cache.py +171 -0
- bdcdata-2.0.0/tests/test_catalog.py +235 -0
- bdcdata-2.0.0/tests/test_challenges.py +153 -0
- bdcdata-2.0.0/tests/test_client.py +239 -0
- bdcdata-2.0.0/tests/test_doctests.py +25 -0
- bdcdata-2.0.0/tests/test_funding.py +282 -0
- bdcdata-2.0.0/tests/test_import_purity.py +207 -0
- bdcdata-2.0.0/tests/test_lookups.py +85 -0
- bdcdata-2.0.0/tests/test_normalize.py +206 -0
- bdcdata-2.0.0/tests/test_readers.py +133 -0
- bdcdata-0.0.11/.github/workflows/python-publish.yml +0 -70
- bdcdata-0.0.11/.gitignore +0 -8
- bdcdata-0.0.11/PKG-INFO +0 -25
- bdcdata-0.0.11/README.md +0 -5
- bdcdata-0.0.11/bdc-public-data-api-specifications.pdf +0 -0
- bdcdata-0.0.11/broadband-map-data-downloads.pdf +0 -0
- bdcdata-0.0.11/docs/.env.sample +0 -2
- bdcdata-0.0.11/docs/index.md +0 -12
- bdcdata-0.0.11/hatch.toml +0 -2
- bdcdata-0.0.11/pyproject.toml +0 -36
- bdcdata-0.0.11/requirements.txt +0 -5
- bdcdata-0.0.11/src/bdcdata/__init__.py +0 -44
- bdcdata-0.0.11/src/bdcdata/availability.py +0 -181
- bdcdata-0.0.11/src/bdcdata/bdc.py +0 -41
- bdcdata-0.0.11/src/bdcdata/challenge.py +0 -5
- bdcdata-0.0.11/src/bdcdata/fabric.py +0 -19
- bdcdata-0.0.11/src/bdcdata/funding.py +0 -5
- bdcdata-0.0.11/src/bdcdata/helpers.py +0 -72
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
workflow_dispatch:
|
|
8
|
+
|
|
9
|
+
concurrency:
|
|
10
|
+
group: ci-${{ github.ref }}
|
|
11
|
+
cancel-in-progress: true
|
|
12
|
+
|
|
13
|
+
jobs:
|
|
14
|
+
lint:
|
|
15
|
+
runs-on: ubuntu-latest
|
|
16
|
+
steps:
|
|
17
|
+
- uses: actions/checkout@v4
|
|
18
|
+
- uses: actions/setup-python@v5
|
|
19
|
+
with:
|
|
20
|
+
python-version: "3.12"
|
|
21
|
+
- run: pip install ruff mypy pandas-stubs types-requests
|
|
22
|
+
- run: ruff check .
|
|
23
|
+
- run: ruff format --check .
|
|
24
|
+
- run: pip install -e .
|
|
25
|
+
- run: mypy
|
|
26
|
+
|
|
27
|
+
test:
|
|
28
|
+
runs-on: ${{ matrix.os }}
|
|
29
|
+
strategy:
|
|
30
|
+
fail-fast: false
|
|
31
|
+
matrix:
|
|
32
|
+
os: [ubuntu-latest]
|
|
33
|
+
python-version: ["3.10", "3.11", "3.12", "3.13"]
|
|
34
|
+
include:
|
|
35
|
+
- os: macos-latest
|
|
36
|
+
python-version: "3.12"
|
|
37
|
+
- os: windows-latest
|
|
38
|
+
python-version: "3.12"
|
|
39
|
+
steps:
|
|
40
|
+
- uses: actions/checkout@v4
|
|
41
|
+
with:
|
|
42
|
+
# hatch-vcs needs tags to compute a version
|
|
43
|
+
fetch-depth: 0
|
|
44
|
+
- uses: actions/setup-python@v5
|
|
45
|
+
with:
|
|
46
|
+
python-version: ${{ matrix.python-version }}
|
|
47
|
+
- run: pip install -e ".[mobile,dotenv,progress]" pytest responses
|
|
48
|
+
# The offline suite must pass with no credentials configured. Credentials
|
|
49
|
+
# are deliberately absent here -- that is the point of the test.
|
|
50
|
+
- run: pytest -m "not integration" -v
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
build:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
steps:
|
|
12
|
+
- uses: actions/checkout@v4
|
|
13
|
+
with:
|
|
14
|
+
fetch-depth: 0
|
|
15
|
+
- uses: actions/setup-python@v5
|
|
16
|
+
with:
|
|
17
|
+
python-version: "3.12"
|
|
18
|
+
- run: pip install build
|
|
19
|
+
- run: python -m build
|
|
20
|
+
- uses: actions/upload-artifact@v4
|
|
21
|
+
with:
|
|
22
|
+
name: dist
|
|
23
|
+
path: dist/
|
|
24
|
+
|
|
25
|
+
publish:
|
|
26
|
+
needs: build
|
|
27
|
+
runs-on: ubuntu-latest
|
|
28
|
+
environment: pypi
|
|
29
|
+
permissions:
|
|
30
|
+
# Required for PyPI Trusted Publishing -- no API token stored in the repo.
|
|
31
|
+
id-token: write
|
|
32
|
+
steps:
|
|
33
|
+
- uses: actions/download-artifact@v4
|
|
34
|
+
with:
|
|
35
|
+
name: dist
|
|
36
|
+
path: dist/
|
|
37
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
bdcdata-2.0.0/.gitignore
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*.egg-info/
|
|
5
|
+
build/
|
|
6
|
+
dist/
|
|
7
|
+
.venv/
|
|
8
|
+
venv/
|
|
9
|
+
|
|
10
|
+
# Tooling
|
|
11
|
+
.pytest_cache/
|
|
12
|
+
.mypy_cache/
|
|
13
|
+
.ruff_cache/
|
|
14
|
+
.coverage
|
|
15
|
+
htmlcov/
|
|
16
|
+
|
|
17
|
+
# Credentials -- never commit these
|
|
18
|
+
.env
|
|
19
|
+
|
|
20
|
+
# Default bdcdata download cache (cwd-relative, see bdcdata.set_cache)
|
|
21
|
+
bdc_cache/
|
|
22
|
+
|
|
23
|
+
# OS
|
|
24
|
+
.DS_Store
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented here.
|
|
4
|
+
|
|
5
|
+
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
6
|
+
and this project adheres to [Semantic Versioning](https://semver.org/).
|
|
7
|
+
|
|
8
|
+
## [2.0.0] — unreleased
|
|
9
|
+
|
|
10
|
+
A ground-up rewrite. The submodule layout (`bdcdata.availability.fixed()`) is
|
|
11
|
+
unchanged, but arguments, credential handling, and return types all changed.
|
|
12
|
+
See [docs/migrating-from-v1.md](docs/migrating-from-v1.md).
|
|
13
|
+
|
|
14
|
+
### Added
|
|
15
|
+
|
|
16
|
+
- **Challenge data.** `challenges.fabric()`, `challenges.fixed()`,
|
|
17
|
+
`challenges.mobile()`, `challenges.verification()`, `challenges.audit()`, and
|
|
18
|
+
`challenges.get()` for categories added after this release. These were
|
|
19
|
+
import-only stubs in 1.x.
|
|
20
|
+
- **Funding data.** `funding.programs()`, `funding.projects()`,
|
|
21
|
+
`funding.unserved_unfunded()`, `funding.funded_locations()`,
|
|
22
|
+
`funding.projects_in_geography()`, `funding.readmes()`, and
|
|
23
|
+
`funding.download_readme()`. Also stubs in 1.x.
|
|
24
|
+
- **Mobile availability.** `availability.mobile()` reads the H3 coverage
|
|
25
|
+
attribute table without requiring geopandas, via the optional
|
|
26
|
+
`bdcdata[mobile]` extra.
|
|
27
|
+
- **Served/unserved.** `availability.served_unserved()`, covering the export
|
|
28
|
+
added in the 2026-08-11 specification revision.
|
|
29
|
+
- **Summary tables.** `availability.provider_list()`,
|
|
30
|
+
`availability.provider_summary()`, and
|
|
31
|
+
`availability.summary_by_geography()`.
|
|
32
|
+
- **A browsable catalog.** `bdcdata.catalog` wraps `listAsOfDates`,
|
|
33
|
+
`listAvailabilityData`, `listChallengeData`, `listFundingData`,
|
|
34
|
+
`listReadmeFiles`, and `listGeographyData`.
|
|
35
|
+
- **Reference tables.** `bdcdata.lookups.states()`,
|
|
36
|
+
`lookups.technologies()`, and `lookups.challenge_categories()`.
|
|
37
|
+
- **Friendly identifiers.** `state=` accepts FIPS codes, USPS abbreviations,
|
|
38
|
+
and full names; `technology=` accepts codes, slugs, descriptions, aliases
|
|
39
|
+
(`"dsl"`, `"fttp"`, `"lte"`), and groups (`"wired"`, `"satellite"`,
|
|
40
|
+
`"terrestrial"`, `"wireless"`).
|
|
41
|
+
- **Typed exceptions.** `BdcError` and its subclasses, replacing bare
|
|
42
|
+
`Exception`. Every message says what to do next.
|
|
43
|
+
- **Credential helpers.** `set_credentials()`, `load_dotenv()`,
|
|
44
|
+
`have_credentials()`, `check_credentials()`.
|
|
45
|
+
- **Cache controls.** `set_cache()`, `cache_info()`, `clear_cache()`.
|
|
46
|
+
- Type hints throughout, with a `py.typed` marker.
|
|
47
|
+
|
|
48
|
+
### Changed
|
|
49
|
+
|
|
50
|
+
- **`import bdcdata` has no side effects.** No network call, no `.env` read, no
|
|
51
|
+
logging configuration. Credentials resolve on first request.
|
|
52
|
+
- **Logging no longer hijacks the root logger.** A `NullHandler` on the
|
|
53
|
+
`bdcdata` logger, nothing more. No `basicConfig()`, no `StreamHandler`, no
|
|
54
|
+
`bdc.log` file.
|
|
55
|
+
- **Base URL is now `https://bdc.fcc.gov`**, matching the current FCC
|
|
56
|
+
specification. Override with `set_base_url()`.
|
|
57
|
+
- **`states=` is now `state=`** (singular), and accepts far more spellings.
|
|
58
|
+
- **`release` defaults to `"latest"`** instead of a hardcoded date, and
|
|
59
|
+
resolves separately for availability and challenge data.
|
|
60
|
+
- **Identifier columns are strings**: `location_id`, `block_geoid`, `frn`,
|
|
61
|
+
`h3_res8_id`, `h3_res9_id`, `state_fips`.
|
|
62
|
+
- **Cache is off by default**, lives in `./bdc_cache`, is keyed by request URL
|
|
63
|
+
rather than filename, and is configured globally via `set_cache()`.
|
|
64
|
+
- **`release` and `state_fips` columns are added** to downloaded frames, so
|
|
65
|
+
multi-state and multi-release pulls stay separable.
|
|
66
|
+
- Validation happens before any network request.
|
|
67
|
+
- Multi-file downloads concatenate once instead of inside the loop.
|
|
68
|
+
|
|
69
|
+
### Fixed
|
|
70
|
+
|
|
71
|
+
- **`business_residential_code` dtype hint.** 1.x misspelled it
|
|
72
|
+
(`business_residental_code`), so the hint silently never applied.
|
|
73
|
+
- **`location_id` overflow.** 1.x typed it `UInt32` (max 4,294,967,295) while
|
|
74
|
+
the FCC specification defines it as `String{13}`. Now a string.
|
|
75
|
+
- **Challenge downloads were unreachable.** 1.x hardcoded `availability` as the
|
|
76
|
+
`data_type` path segment, so no challenge file could be fetched.
|
|
77
|
+
- **Archive member selection.** 1.x read `zip.filelist[0]` blindly; sidecar
|
|
78
|
+
entries (`__MACOSX`, readmes) are now skipped and CSVs preferred.
|
|
79
|
+
|
|
80
|
+
### Removed
|
|
81
|
+
|
|
82
|
+
- **`fabric` module.** Its two functions did nothing — one printed a message,
|
|
83
|
+
the other read a path nothing wrote. Fabric is licensed data on a different
|
|
84
|
+
access path, so it is out of scope rather than present and broken.
|
|
85
|
+
- `bdcdata.echo()` and the `main()` entry point.
|
|
86
|
+
- Module-level `apiKey`, `username`, `session`, and `metadata` globals.
|
|
87
|
+
- The `requests-ratelimiter` dependency, replaced by a built-in limiter that
|
|
88
|
+
also handles retries and `Retry-After`.
|
|
89
|
+
|
|
90
|
+
### Dependencies
|
|
91
|
+
|
|
92
|
+
- Core: `requests`, `pandas>=2`, `pyarrow`.
|
|
93
|
+
- Optional: `bdcdata[mobile]` (pyogrio), `bdcdata[dotenv]` (python-dotenv),
|
|
94
|
+
`bdcdata[progress]` (tqdm).
|
|
95
|
+
- Requires Python 3.10+.
|
|
96
|
+
|
|
97
|
+
---
|
|
98
|
+
|
|
99
|
+
## [1.x]
|
|
100
|
+
|
|
101
|
+
See the [pre-2.0 history](https://github.com/npappin/bdcdata/commits/main).
|
|
102
|
+
Supported fixed availability by state and technology; challenge, funding, and
|
|
103
|
+
fabric modules were stubs.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
MIT License
|
|
2
2
|
|
|
3
|
-
Copyright (c)
|
|
3
|
+
Copyright (c) 2026 W. Nick Pappin
|
|
4
4
|
|
|
5
5
|
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
6
|
of this software and associated documentation files (the "Software"), to deal
|
|
@@ -18,4 +18,4 @@ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
|
18
18
|
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
19
|
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
20
|
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
-
SOFTWARE.
|
|
21
|
+
SOFTWARE.
|
bdcdata-2.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: bdcdata
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Work with FCC Broadband Data Collection (BDC) data in pandas.
|
|
5
|
+
Project-URL: Homepage, https://github.com/npappin/bdcdata
|
|
6
|
+
Project-URL: Documentation, https://github.com/npappin/bdcdata/tree/main/docs
|
|
7
|
+
Project-URL: Issues, https://github.com/npappin/bdcdata/issues
|
|
8
|
+
Author-email: "W. Nick Pappin" <nick.pappin@wsu.edu>
|
|
9
|
+
Maintainer-email: "W. Nick Pappin" <npappin@gmail.com>
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE.txt
|
|
12
|
+
Keywords: bdc,broadband,fcc,national broadband map,pandas
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
23
|
+
Classifier: Typing :: Typed
|
|
24
|
+
Requires-Python: >=3.10
|
|
25
|
+
Requires-Dist: pandas>=2
|
|
26
|
+
Requires-Dist: pyarrow>=12
|
|
27
|
+
Requires-Dist: python-dotenv
|
|
28
|
+
Requires-Dist: requests>=2.28
|
|
29
|
+
Provides-Extra: all
|
|
30
|
+
Requires-Dist: pyogrio>=0.7; extra == 'all'
|
|
31
|
+
Requires-Dist: tqdm>=4.60; extra == 'all'
|
|
32
|
+
Provides-Extra: mobile
|
|
33
|
+
Requires-Dist: pyogrio>=0.7; extra == 'mobile'
|
|
34
|
+
Provides-Extra: progress
|
|
35
|
+
Requires-Dist: tqdm>=4.60; extra == 'progress'
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
|
|
38
|
+
# bdcdata
|
|
39
|
+
|
|
40
|
+
Work with FCC **Broadband Data Collection** (BDC) data in pandas.
|
|
41
|
+
|
|
42
|
+
`bdcdata` handles the parts of the National Broadband Map that are tedious to
|
|
43
|
+
get right — authentication, the 10-calls-per-minute rate limit, finding the
|
|
44
|
+
right file among thousands, unzipping it, and applying the column types from
|
|
45
|
+
the FCC's published specification — and hands you a DataFrame.
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
import bdcdata
|
|
49
|
+
|
|
50
|
+
bdcdata.set_credentials(username="you@example.com", token="...")
|
|
51
|
+
|
|
52
|
+
df = bdcdata.availability.fixed(state="WA", technology="fiber")
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
You can write `state="WA"`, `state="Washington"`, or `state=53`, and
|
|
56
|
+
`technology="fiber"`, `technology="fttp"`, or `technology=50`. They all mean
|
|
57
|
+
the same thing.
|
|
58
|
+
|
|
59
|
+
## Install
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
pip install bdcdata
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Reading the mobile H3 coverage files means opening a shapefile, which needs an
|
|
66
|
+
extra:
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
pip install 'bdcdata[mobile]'
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Credentials
|
|
73
|
+
|
|
74
|
+
Every BDC endpoint requires an FCC username and API token, including the
|
|
75
|
+
metadata endpoints.
|
|
76
|
+
|
|
77
|
+
1. Log in at <https://broadbandmap.fcc.gov/login> with your FCC User
|
|
78
|
+
Registration account.
|
|
79
|
+
2. Click your username in the top right, then **Manage API Access**.
|
|
80
|
+
3. Click **Generate**, accept the terms, and copy the token.
|
|
81
|
+
|
|
82
|
+
Your username is the email address on the account. Then pick whichever of
|
|
83
|
+
these suits you:
|
|
84
|
+
|
|
85
|
+
```python
|
|
86
|
+
bdcdata.set_credentials(username="you@example.com", token="...")
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
export BDC_USERNAME=you@example.com
|
|
91
|
+
export BDC_API_KEY=...
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
# .env file in your working directory, with BDC_USERNAME and BDC_API_KEY
|
|
96
|
+
bdcdata.load_dotenv()
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Check them with `bdcdata.check_credentials()`.
|
|
100
|
+
|
|
101
|
+
## Documentation
|
|
102
|
+
|
|
103
|
+
See the [`docs/`](docs/) directory.
|
|
104
|
+
|
|
105
|
+
## Upgrading from 1.x
|
|
106
|
+
|
|
107
|
+
**2.0 is a rewrite and the API changed.** See
|
|
108
|
+
[docs/migrating-from-v1.md](docs/migrating-from-v1.md).
|
|
109
|
+
|
|
110
|
+
## License
|
|
111
|
+
|
|
112
|
+
MIT
|
bdcdata-2.0.0/README.md
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# bdcdata
|
|
2
|
+
|
|
3
|
+
Work with FCC **Broadband Data Collection** (BDC) data in pandas.
|
|
4
|
+
|
|
5
|
+
`bdcdata` handles the parts of the National Broadband Map that are tedious to
|
|
6
|
+
get right — authentication, the 10-calls-per-minute rate limit, finding the
|
|
7
|
+
right file among thousands, unzipping it, and applying the column types from
|
|
8
|
+
the FCC's published specification — and hands you a DataFrame.
|
|
9
|
+
|
|
10
|
+
```python
|
|
11
|
+
import bdcdata
|
|
12
|
+
|
|
13
|
+
bdcdata.set_credentials(username="you@example.com", token="...")
|
|
14
|
+
|
|
15
|
+
df = bdcdata.availability.fixed(state="WA", technology="fiber")
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
You can write `state="WA"`, `state="Washington"`, or `state=53`, and
|
|
19
|
+
`technology="fiber"`, `technology="fttp"`, or `technology=50`. They all mean
|
|
20
|
+
the same thing.
|
|
21
|
+
|
|
22
|
+
## Install
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install bdcdata
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Reading the mobile H3 coverage files means opening a shapefile, which needs an
|
|
29
|
+
extra:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
pip install 'bdcdata[mobile]'
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## Credentials
|
|
36
|
+
|
|
37
|
+
Every BDC endpoint requires an FCC username and API token, including the
|
|
38
|
+
metadata endpoints.
|
|
39
|
+
|
|
40
|
+
1. Log in at <https://broadbandmap.fcc.gov/login> with your FCC User
|
|
41
|
+
Registration account.
|
|
42
|
+
2. Click your username in the top right, then **Manage API Access**.
|
|
43
|
+
3. Click **Generate**, accept the terms, and copy the token.
|
|
44
|
+
|
|
45
|
+
Your username is the email address on the account. Then pick whichever of
|
|
46
|
+
these suits you:
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
bdcdata.set_credentials(username="you@example.com", token="...")
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
export BDC_USERNAME=you@example.com
|
|
54
|
+
export BDC_API_KEY=...
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
```python
|
|
58
|
+
# .env file in your working directory, with BDC_USERNAME and BDC_API_KEY
|
|
59
|
+
bdcdata.load_dotenv()
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Check them with `bdcdata.check_credentials()`.
|
|
63
|
+
|
|
64
|
+
## Documentation
|
|
65
|
+
|
|
66
|
+
See the [`docs/`](docs/) directory.
|
|
67
|
+
|
|
68
|
+
## Upgrading from 1.x
|
|
69
|
+
|
|
70
|
+
**2.0 is a rewrite and the API changed.** See
|
|
71
|
+
[docs/migrating-from-v1.md](docs/migrating-from-v1.md).
|
|
72
|
+
|
|
73
|
+
## License
|
|
74
|
+
|
|
75
|
+
MIT
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
# Availability
|
|
2
|
+
|
|
3
|
+
Who reports offering broadband service, and where.
|
|
4
|
+
|
|
5
|
+
```python
|
|
6
|
+
import bdcdata
|
|
7
|
+
|
|
8
|
+
df = bdcdata.availability.fixed(state="WA", technology="fiber")
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## `fixed()` — service by location
|
|
12
|
+
|
|
13
|
+
The core dataset. One row per location, per provider, per technology: what each
|
|
14
|
+
provider reports offering at each Broadband Serviceable Location.
|
|
15
|
+
|
|
16
|
+
```python
|
|
17
|
+
bdcdata.availability.fixed(state, technology="all", release="latest")
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
| Column | Type | Notes |
|
|
21
|
+
|---|---|---|
|
|
22
|
+
| `frn` | string | 10-digit FCC Registration Number, leading zeros kept |
|
|
23
|
+
| `provider_id` | int | Join to `provider_list()` |
|
|
24
|
+
| `brand_name` | string | The name consumers see |
|
|
25
|
+
| `location_id` | string | Fabric location ID — **string**, see below |
|
|
26
|
+
| `technology` | int | See `lookups.technologies("fixed")` |
|
|
27
|
+
| `max_advertised_download_speed` | int | Mbps |
|
|
28
|
+
| `max_advertised_upload_speed` | int | Mbps |
|
|
29
|
+
| `low_latency` | bool | ≤100 ms round trip at the 95th percentile |
|
|
30
|
+
| `business_residential_code` | string | `B`, `R`, or `X` (both) |
|
|
31
|
+
| `state_usps` | string | Two-letter abbreviation |
|
|
32
|
+
| `block_geoid` | string | 15-digit census block, leading zeros kept |
|
|
33
|
+
| `h3_res8_id` | string | H3 resolution-8 cell |
|
|
34
|
+
| `release` | string | Added by bdcdata |
|
|
35
|
+
| `state_fips` | string | Added by bdcdata |
|
|
36
|
+
|
|
37
|
+
`location_id` and `block_geoid` are strings on purpose. A `block_geoid` like
|
|
38
|
+
`011010106033002` loses its meaning as a number, and a 13-digit `location_id`
|
|
39
|
+
exceeds what a 32-bit integer can hold. Join on them as text.
|
|
40
|
+
|
|
41
|
+
### Examples
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
# One state, one technology
|
|
45
|
+
bdcdata.availability.fixed(state="WA", technology="fiber")
|
|
46
|
+
|
|
47
|
+
# Several states
|
|
48
|
+
bdcdata.availability.fixed(state=["WA", "OR", "ID"], technology="fiber")
|
|
49
|
+
|
|
50
|
+
# All wired technologies (copper, cable, fiber)
|
|
51
|
+
bdcdata.availability.fixed(state="WA", technology="wired")
|
|
52
|
+
|
|
53
|
+
# Everything for one state — large
|
|
54
|
+
bdcdata.availability.fixed(state="WA", technology="all")
|
|
55
|
+
|
|
56
|
+
# A past release
|
|
57
|
+
bdcdata.availability.fixed(state="WA", technology="fiber", release="2023-12-31")
|
|
58
|
+
|
|
59
|
+
# Compare two releases in one frame
|
|
60
|
+
bdcdata.availability.fixed(state="WA", technology="fiber", release=["2023-12-31", "2024-06-30"])
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
The `release` column is what makes the last one useful:
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
df.groupby("release")["location_id"].nunique()
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
### Technology groups
|
|
70
|
+
|
|
71
|
+
Instead of listing codes, you can name a group:
|
|
72
|
+
|
|
73
|
+
| Group | Codes | Meaning |
|
|
74
|
+
|---|---|---|
|
|
75
|
+
| `"all"` | every fixed code | |
|
|
76
|
+
| `"wired"` | 10, 40, 50 | Copper, cable, fiber — the FCC's own definition |
|
|
77
|
+
| `"terrestrial"` | everything but 60, 61 | All non-satellite |
|
|
78
|
+
| `"satellite"` | 60, 61 | Geostationary and non-geostationary |
|
|
79
|
+
| `"wireless"` | 70, 71, 72 | Fixed wireless |
|
|
80
|
+
|
|
81
|
+
`"wired"` and `"terrestrial"` match the definitions the FCC uses in the
|
|
82
|
+
served/unserved file, so the groupings line up across datasets.
|
|
83
|
+
|
|
84
|
+
## `served_unserved()` — the 100/20 question
|
|
85
|
+
|
|
86
|
+
For **every** location in the Fabric — not just those with service — whether
|
|
87
|
+
any provider reported at least 100 Mbps down / 20 Mbps up.
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
df = bdcdata.availability.served_unserved(state="WA")
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
| Column | Type |
|
|
94
|
+
|---|---|
|
|
95
|
+
| `location_id` | string |
|
|
96
|
+
| `block_geoid` | string |
|
|
97
|
+
| `h3_res8_id` | string |
|
|
98
|
+
| `any_dl100_ul20` | bool |
|
|
99
|
+
| `wired_dl100_ul20` | bool |
|
|
100
|
+
| `terrestrial_dl100_ul20` | bool |
|
|
101
|
+
|
|
102
|
+
The three flags are real booleans, so they aggregate directly:
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
# Unserved locations by census block
|
|
106
|
+
unserved = df[~df["any_dl100_ul20"]]
|
|
107
|
+
unserved.groupby("block_geoid").size().sort_values(ascending=False)
|
|
108
|
+
|
|
109
|
+
# Share served, statewide
|
|
110
|
+
df["any_dl100_ul20"].mean()
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
This export was added in the August 2026 revision of the download
|
|
114
|
+
specification, so it may not exist for older releases. If it doesn't, you get
|
|
115
|
+
an empty DataFrame and a warning saying why.
|
|
116
|
+
|
|
117
|
+
## `mobile()` — mobile coverage by H3 cell
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
df = bdcdata.availability.mobile(state="WA", technology="5g")
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
Needs `pip install 'bdcdata[mobile]'`.
|
|
124
|
+
|
|
125
|
+
The FCC publishes these as GIS files (shapefile or GeoPackage), not CSV.
|
|
126
|
+
bdcdata reads the **attribute table only** and skips the geometry, so you get
|
|
127
|
+
one row per H3 resolution-9 cell without geopandas in your dependency tree.
|
|
128
|
+
|
|
129
|
+
| Column | Type | Notes |
|
|
130
|
+
|---|---|---|
|
|
131
|
+
| `technology` | int | 300 = 3G, 400 = 4G LTE, 500 = 5G-NR |
|
|
132
|
+
| `mindown` | float | Minimum modeled download, Mbps |
|
|
133
|
+
| `minup` | float | Minimum modeled upload, Mbps |
|
|
134
|
+
| `environmnt` | int | 0 = outdoor stationary only, 1 = also in-vehicle |
|
|
135
|
+
| `h3_res9_id` | string | H3 resolution-9 cell |
|
|
136
|
+
|
|
137
|
+
`environmnt` is spelled that way in the FCC's files. It's kept as published
|
|
138
|
+
rather than silently corrected, so what you see matches the specification.
|
|
139
|
+
|
|
140
|
+
To map it, join `h3_res9_id` to hexagon geometry with the
|
|
141
|
+
[h3](https://pypi.org/project/h3/) package:
|
|
142
|
+
|
|
143
|
+
```python
|
|
144
|
+
import h3
|
|
145
|
+
|
|
146
|
+
df["boundary"] = df["h3_res9_id"].map(lambda cell: h3.cell_to_boundary(cell))
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
**Mobile technology codes are a separate namespace from fixed ones.** Code `0`
|
|
150
|
+
means "Other" for fixed and "Mobile Voice" for mobile. Passing `"fiber"` to
|
|
151
|
+
`mobile()` is an error, not a silent empty result.
|
|
152
|
+
|
|
153
|
+
## Summary tables
|
|
154
|
+
|
|
155
|
+
Smaller aggregates, when you don't need location-level detail.
|
|
156
|
+
|
|
157
|
+
```python
|
|
158
|
+
# Providers that submitted data — join target for provider_id
|
|
159
|
+
bdcdata.availability.provider_list()
|
|
160
|
+
|
|
161
|
+
# Per-provider totals
|
|
162
|
+
bdcdata.availability.provider_summary(kind="fixed") # location and unit counts
|
|
163
|
+
bdcdata.availability.provider_summary(kind="mobile") # covered area in sq km
|
|
164
|
+
|
|
165
|
+
# Coverage percentages by geography, across all providers
|
|
166
|
+
bdcdata.availability.summary_by_geography(kind="fixed", geography="place")
|
|
167
|
+
bdcdata.availability.summary_by_geography(kind="fixed", geography="other")
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
`geography="place"` is census places; `geography="other"` is everything else
|
|
171
|
+
(state, county, congressional district, tribal area, CBSA). The FCC split these
|
|
172
|
+
into separate exports in June 2024.
|
|
173
|
+
|
|
174
|
+
## Size
|
|
175
|
+
|
|
176
|
+
An availability pull can be very large. A single state's fiber file is tens of
|
|
177
|
+
megabytes; `state="all", technology="all"` is many gigabytes and will likely
|
|
178
|
+
exhaust memory.
|
|
179
|
+
|
|
180
|
+
bdcdata sums the catalog's `record_count` before downloading and warns you when
|
|
181
|
+
a request is about to load millions of rows. To see what you're asking for
|
|
182
|
+
first:
|
|
183
|
+
|
|
184
|
+
```python
|
|
185
|
+
files = bdcdata.catalog.availability_files(
|
|
186
|
+
release="latest", category="State", subcategory="Location Coverage"
|
|
187
|
+
)
|
|
188
|
+
files["record_count"].sum()
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
If you need everything, loop a state at a time and write each to Parquet rather
|
|
192
|
+
than holding it all in memory:
|
|
193
|
+
|
|
194
|
+
```python
|
|
195
|
+
for state in bdcdata.lookups.states()["usps"]:
|
|
196
|
+
df = bdcdata.availability.fixed(state=state, technology="fiber")
|
|
197
|
+
df.to_parquet(f"fiber_{state}.parquet")
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
With `bdcdata.set_cache(True)`, a loop like that is resumable — re-running
|
|
201
|
+
skips anything already downloaded.
|