ncua-data-analysis 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ncua_data_analysis-0.1.0/LICENSE +21 -0
- ncua_data_analysis-0.1.0/PKG-INFO +9 -0
- ncua_data_analysis-0.1.0/README.md +99 -0
- ncua_data_analysis-0.1.0/pyproject.toml +20 -0
- ncua_data_analysis-0.1.0/setup.cfg +4 -0
- ncua_data_analysis-0.1.0/src/ncua_data/__init__.py +2 -0
- ncua_data_analysis-0.1.0/src/ncua_data/__main__.py +2 -0
- ncua_data_analysis-0.1.0/src/ncua_data/build.py +172 -0
- ncua_data_analysis-0.1.0/src/ncua_data/cli.py +50 -0
- ncua_data_analysis-0.1.0/src/ncua_data/dictionary.py +93 -0
- ncua_data_analysis-0.1.0/src/ncua_data/download.py +45 -0
- ncua_data_analysis-0.1.0/src/ncua_data/ingest.py +88 -0
- ncua_data_analysis-0.1.0/src/ncua_data/mcp_data/dictionary.json +1297 -0
- ncua_data_analysis-0.1.0/src/ncua_data/mcp_server.py +405 -0
- ncua_data_analysis-0.1.0/src/ncua_data/reconcile.py +97 -0
- ncua_data_analysis-0.1.0/src/ncua_data/spec.py +163 -0
- ncua_data_analysis-0.1.0/src/ncua_data_analysis.egg-info/PKG-INFO +9 -0
- ncua_data_analysis-0.1.0/src/ncua_data_analysis.egg-info/SOURCES.txt +21 -0
- ncua_data_analysis-0.1.0/src/ncua_data_analysis.egg-info/dependency_links.txt +1 -0
- ncua_data_analysis-0.1.0/src/ncua_data_analysis.egg-info/entry_points.txt +3 -0
- ncua_data_analysis-0.1.0/src/ncua_data_analysis.egg-info/requires.txt +2 -0
- ncua_data_analysis-0.1.0/src/ncua_data_analysis.egg-info/top_level.txt +1 -0
- ncua_data_analysis-0.1.0/tests/test_spec_and_metrics.py +53 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Bruno Novarini
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ncua-data-analysis
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Clean, documented NCUA call report data for credit union analysis: lending, deposits and operations
|
|
5
|
+
Requires-Python: >=3.10
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Requires-Dist: duckdb>=1.0
|
|
8
|
+
Requires-Dist: mcp<2,>=1.2
|
|
9
|
+
Dynamic: license-file
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
# ncua-data-analysis
|
|
2
|
+
|
|
3
|
+
Clean, documented, quarterly credit union data from NCUA call reports, 2018 to today: lending, deposit mix, earnings, staffing and efficiency.
|
|
4
|
+
|
|
5
|
+
NCUA publishes every federally insured credit union's quarterly 5300 call report as free bulk files. They are hard to use: 3,300+ account columns split across 17 wide tables, form changes that move accounts around, year-to-date income, and a data dictionary written as form instructions. This project turns that into four tidy tables you can query with DuckDB, pandas or anything that reads Parquet.
|
|
6
|
+
|
|
7
|
+
Status: v0. 115 curated fields and 52 computed metrics, 34 quarters (2018 Q1 to 2026 Q2), checked against NCUA's own published totals. See [what is not done yet](#not-done-yet).
|
|
8
|
+
|
|
9
|
+
## Tables
|
|
10
|
+
|
|
11
|
+
| Table | Grain | What it is |
|
|
12
|
+
|---|---|---|
|
|
13
|
+
| `dim_credit_union` | credit union x quarter | Name, location, charter type, peer group, low-income and MDI flags, and `is_federally_insured`. |
|
|
14
|
+
| `fact_call_report_curated` | credit union x quarter | 115 curated account values with plain names: balance sheet, shares and capital, deposit composition (share drafts, regular, money market, certificates, IRA, non-member), loan balances by type, originations, delinquency, charge-offs, income statement, employees and branches. |
|
|
15
|
+
| `metrics` | credit union x quarter | 52 computed metrics: delinquency, loan-to-share, ROA, NIM, loan and deposit mix, growth, efficiency ratio, members and assets per employee, per-branch figures, de-cumulated quarterly income. |
|
|
16
|
+
| `dictionary` | column | Description, unit, NCUA account codes, and first and last quarter each field has data. |
|
|
17
|
+
|
|
18
|
+
Join on `quarter` + `cu_number`. Dollar fields are dollars. Fields ending in `_ytd` are year to date and reset each January (the Q4 value is the full year); `metrics` annualizes them.
|
|
19
|
+
|
|
20
|
+
```sql
|
|
21
|
+
-- Which large credit unions grew auto lending fastest last year?
|
|
22
|
+
SELECT d.name, d.state, m.auto_loan_growth_yoy, f.loans_new_vehicle + f.loans_used_vehicle AS auto_loans
|
|
23
|
+
FROM fact_call_report_curated f
|
|
24
|
+
JOIN dim_credit_union d USING (quarter, cu_number)
|
|
25
|
+
JOIN metrics m USING (quarter, cu_number)
|
|
26
|
+
WHERE f.quarter = '2026-06' AND d.is_federally_insured AND d.peer_group = 6
|
|
27
|
+
ORDER BY m.auto_loan_growth_yoy DESC LIMIT 10;
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
## Run it
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
pip install -e .
|
|
34
|
+
ncua-data all # download 34 quarters (~270 MB), build tables, reconcile
|
|
35
|
+
python -m unittest discover -s tests
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Output lands in `data/out/` as Parquet. Raw ZIPs are never committed. Links are scraped from NCUA's [quarterly data page](https://ncua.gov/analysis/credit-union-corporate-call-report-data/quarterly-data).
|
|
39
|
+
|
|
40
|
+
## Numbers you can trust
|
|
41
|
+
|
|
42
|
+
For June 2026 and December 2025 the totals reconcile to the figures in NCUA's Quarterly Credit Union Data Summary: credit union count, members, loans by type, shares, net worth ratio, delinquency, income and expense. 60 of 67 checks match across six year-ends (2021 to June 2026). The 7 that do not are historical deposit lines that differ by under $0.5B (under 0.1%), for example Dec 2025 money market $367.5B versus $368.0B published. All June 2026 lines match. The likely cause is restated history in NCUA's table, which has not been confirmed. Full table: [docs/RECONCILIATION.md](docs/RECONCILIATION.md).
|
|
43
|
+
|
|
44
|
+
Things to know before using it:
|
|
45
|
+
|
|
46
|
+
- **Filter to `is_federally_insured`.** NCUA's raw files include about 85 state-chartered credit unions it does not insure. Its published totals leave them out.
|
|
47
|
+
- **The form changed in 2022 and 2023.** NCUA redesigned the call report and adopted CECL, so some accounts disappear and new ones appear. Fields that span the change map both account codes (for example the allowance is Acct_719 before CECL and Acct_AS0048 after). Fields that only exist after the change say so in the dictionary.
|
|
48
|
+
- **Net worth ratio.** NCUA's published ratio excludes the CECL transition provision from 2023 on. This dataset carries that provision (`cecl_transition_provision`) and `metrics.net_worth_ratio_ex_cecl` matches NCUA.
|
|
49
|
+
- **Employees are estimated.** `employees_fte_estimate` is full-time plus half of part-time. It is not an NCUA definition and is not reconciled to a published figure.
|
|
50
|
+
- **Net charge-off ratio, ROA and NIM** use NCUA's own average balances, which are not public. Metrics here use a four-quarter average and land within a few basis points of NCUA's published values, not on them.
|
|
51
|
+
- **Mergers.** Credit union counts fell from 5,375 to 4,214 since 2018. A merged credit union's history stays under its old charter number, so per-institution growth across a merger is not meaningful.
|
|
52
|
+
- Reports are self-reported and occasionally reposted as "Revised". The ZIPs are used as NCUA publishes them today.
|
|
53
|
+
|
|
54
|
+
## Not done yet
|
|
55
|
+
|
|
56
|
+
- Commercial-loan delinquency by type (NCUA moved these to new account codes in 2022).
|
|
57
|
+
- More fields. The target is 150 to 200; 115 are in and verified.
|
|
58
|
+
- Years before 2018 (the download links use two other naming patterns).
|
|
59
|
+
- A static analytics site and natural-language querying. The dictionary is built to be the semantic layer for that.
|
|
60
|
+
|
|
61
|
+
## MCP server (v1)
|
|
62
|
+
|
|
63
|
+
A local MCP server lets an AI assistant query this dataset in plain language. It runs over stdio, reads the release Parquet files with DuckDB (downloaded once to `~/.cache/ncua-data-analysis`), and builds its tool descriptions from the dictionary table.
|
|
64
|
+
|
|
65
|
+
```json
|
|
66
|
+
{
|
|
67
|
+
"mcpServers": {
|
|
68
|
+
"ncua-data": {
|
|
69
|
+
"command": "uvx",
|
|
70
|
+
"args": ["--from", "git+https://github.com/bnovarini/ncua-data-analysis", "ncua-data-mcp"]
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
**Hosted, nothing to install:** `https://ncua-data-analysis.fly.dev/mcp` (streamable HTTP, read-only, rate limited to 60 requests a minute per client). Add it as a remote MCP server in any client that supports one, for example `{"mcpServers": {"ncua-data": {"url": "https://ncua-data-analysis.fly.dev/mcp"}}}`.
|
|
77
|
+
|
|
78
|
+
[](https://cursor.com/en/install-mcp?name=ncua-data&config=eyJ1cmwiOiJodHRwczovL25jdWEtZGF0YS1hbmFseXNpcy5mbHkuZGV2L21jcCJ9) [](https://vscode.dev/redirect/mcp/install?name=ncua-data&config=%7B%22type%22%3A%22http%22%2C%22url%22%3A%22https%3A%2F%2Fncua-data-analysis.fly.dev%2Fmcp%22%7D) [](https://insiders.vscode.dev/redirect/mcp/install?name=ncua-data&config=%7B%22type%22%3A%22http%22%2C%22url%22%3A%22https%3A%2F%2Fncua-data-analysis.fly.dev%2Fmcp%22%7D)
|
|
79
|
+
|
|
80
|
+
Claude Code: `claude mcp add --transport http ncua-data https://ncua-data-analysis.fly.dev/mcp`
|
|
81
|
+
|
|
82
|
+
Tools: `list_fields`, `find_credit_union`, `credit_union_profile`, `metric_series` (one credit union or an aggregate across all), `peer_compare` (by asset group, state or charter), and `query_metrics` (filters, ordering and limits; no raw SQL). Set `NCUA_DATA_DIR` to use a folder of already-downloaded files. Status: first working version, tested over stdio with a real MCP client; not yet listed in the MCP registry.
|
|
83
|
+
|
|
84
|
+
## Download
|
|
85
|
+
|
|
86
|
+
Ready-made Parquet files are attached to the [v0.1 release](https://github.com/bnovarini/ncua-data-analysis/releases/tag/v0.1). GitHub caps release files at 25 MB, so `fact_call_report_curated` and `metrics` come in three parts by year (2018-2020, 2021-2023, 2024-2026) with identical columns:
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
import duckdb
|
|
90
|
+
base = "https://github.com/bnovarini/ncua-data-analysis/releases/download/v0.1/"
|
|
91
|
+
parts = [base + f"metrics_{y}.parquet" for y in ("2018_2020", "2021_2023", "2024_2026")]
|
|
92
|
+
duckdb.sql(f"SELECT quarter, count(*) FROM read_parquet({parts}) GROUP BY 1 ORDER BY 1").show()
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Data source and license
|
|
96
|
+
|
|
97
|
+
Data: National Credit Union Administration, 5300 Call Report Quarterly Data. NCUA does not state a license on the download page. As a US federal agency's work it is assumed to be public domain, but that assumption has not been confirmed. Credit NCUA when you use it.
|
|
98
|
+
|
|
99
|
+
Code: MIT, see LICENSE.
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "ncua-data-analysis"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Clean, documented NCUA call report data for credit union analysis: lending, deposits and operations"
|
|
5
|
+
requires-python = ">=3.10"
|
|
6
|
+
dependencies = ["duckdb>=1.0", "mcp>=1.2,<2"]
|
|
7
|
+
|
|
8
|
+
[project.scripts]
|
|
9
|
+
ncua-data = "ncua_data.cli:main"
|
|
10
|
+
ncua-data-mcp = "ncua_data.mcp_server:main"
|
|
11
|
+
|
|
12
|
+
[build-system]
|
|
13
|
+
requires = ["setuptools>=61"]
|
|
14
|
+
build-backend = "setuptools.build_meta"
|
|
15
|
+
|
|
16
|
+
[tool.setuptools.packages.find]
|
|
17
|
+
where = ["src"]
|
|
18
|
+
|
|
19
|
+
[tool.setuptools.package-data]
|
|
20
|
+
ncua_data = ["mcp_data/*.json"]
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""Build the curated tables (dim, fact, metrics, dictionary) from long Parquet."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
import duckdb
|
|
7
|
+
|
|
8
|
+
from .spec import FIELDS
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def fact_sql(long_glob: str) -> str:
|
|
12
|
+
cols = []
|
|
13
|
+
for f in FIELDS:
|
|
14
|
+
codes = ", ".join(f"'{c}'" for c in f.codes)
|
|
15
|
+
agg = f"sum(l.value) FILTER (WHERE l.acct IN ({codes}))"
|
|
16
|
+
# No row in the long table means the credit union reported zero (or left it blank).
|
|
17
|
+
expr = f"coalesce({agg}, 0)"
|
|
18
|
+
if f.first_quarter:
|
|
19
|
+
expr = f"CASE WHEN q.quarter >= '{f.first_quarter}' THEN {expr} END"
|
|
20
|
+
cols.append(f"{expr} AS {f.name}")
|
|
21
|
+
return f"""
|
|
22
|
+
SELECT q.quarter, q.cu_number, {', '.join(cols)}
|
|
23
|
+
FROM quarter_cus q
|
|
24
|
+
LEFT JOIN (SELECT * FROM read_parquet('{long_glob}')) l
|
|
25
|
+
ON l.quarter = q.quarter AND l.cu_number = q.cu_number
|
|
26
|
+
GROUP BY q.quarter, q.cu_number"""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def build(data_dir: Path, out_dir: Path) -> duckdb.DuckDBPyConnection:
|
|
30
|
+
con = duckdb.connect()
|
|
31
|
+
con.execute("SET memory_limit='1400MB'; SET threads=2")
|
|
32
|
+
long_glob = str(data_dir / "long" / "*_long.parquet")
|
|
33
|
+
foicu_glob = str(data_dir / "long" / "*_foicu.parquet")
|
|
34
|
+
con.execute(
|
|
35
|
+
f"""CREATE TABLE quarter_cus AS
|
|
36
|
+
SELECT quarter, CU_NUMBER::INTEGER AS cu_number
|
|
37
|
+
FROM read_parquet('{foicu_glob}', union_by_name=true)"""
|
|
38
|
+
)
|
|
39
|
+
# One quarter at a time keeps memory low enough for a 2 GB machine.
|
|
40
|
+
first = True
|
|
41
|
+
for lf in sorted((data_dir / "long").glob("*_long.parquet")):
|
|
42
|
+
q = lf.name.split("_")[0]
|
|
43
|
+
sql = fact_sql(str(lf)).replace("FROM quarter_cus q", f"FROM (SELECT * FROM quarter_cus WHERE quarter = '{q}') q")
|
|
44
|
+
if first:
|
|
45
|
+
con.execute(f"CREATE TABLE fact_call_report_curated AS {sql}")
|
|
46
|
+
first = False
|
|
47
|
+
else:
|
|
48
|
+
con.execute(f"INSERT INTO fact_call_report_curated {sql}")
|
|
49
|
+
return con
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
DIM_SQL = """
|
|
53
|
+
CREATE TABLE dim_credit_union AS
|
|
54
|
+
SELECT quarter,
|
|
55
|
+
CU_NUMBER::INTEGER AS cu_number,
|
|
56
|
+
try_cast(RSSD AS BIGINT) AS rssd,
|
|
57
|
+
trim(CU_NAME) AS name,
|
|
58
|
+
trim(CITY) AS city,
|
|
59
|
+
trim(STATE) AS state,
|
|
60
|
+
trim(CharterState) AS charter_state,
|
|
61
|
+
trim(ZIP_CODE) AS zip_code,
|
|
62
|
+
try_cast(COUNTY_CODE AS INTEGER) AS county_code,
|
|
63
|
+
CASE CU_TYPE WHEN '1' THEN 'federal' WHEN '2' THEN 'state_federally_insured'
|
|
64
|
+
WHEN '3' THEN 'state_not_federally_insured' END AS charter_type,
|
|
65
|
+
CU_TYPE <> '3' AS is_federally_insured,
|
|
66
|
+
try_cast(Peer_Group AS INTEGER) AS peer_group,
|
|
67
|
+
CASE try_cast(Peer_Group AS INTEGER)
|
|
68
|
+
WHEN 1 THEN 'under $2M' WHEN 2 THEN '$2M-$10M' WHEN 3 THEN '$10M-$50M'
|
|
69
|
+
WHEN 4 THEN '$50M-$100M' WHEN 5 THEN '$100M-$500M' WHEN 6 THEN '$500M+' END AS peer_group_label,
|
|
70
|
+
lower(IsMDI) IN ('true', '1') AS is_minority_depository,
|
|
71
|
+
try_cast(LIMITED_INC AS INTEGER) = 1 AS is_low_income,
|
|
72
|
+
try_cast(YEAR_OPENED AS INTEGER) AS year_opened,
|
|
73
|
+
trim(TOM_CODE) AS field_of_membership_code,
|
|
74
|
+
trim(REGION) AS ncua_region
|
|
75
|
+
FROM read_parquet('{foicu}', union_by_name=true)
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
# Metrics. Flow fields are year-to-date, so annualise by 4 / quarter number.
|
|
79
|
+
METRICS_SQL = """
|
|
80
|
+
CREATE TABLE metrics AS
|
|
81
|
+
WITH f AS (
|
|
82
|
+
SELECT f.*, d.is_federally_insured,
|
|
83
|
+
CAST(substr(f.quarter, 6, 2) AS INTEGER) / 3 AS qn,
|
|
84
|
+
CAST(substr(f.quarter, 1, 4) AS INTEGER) AS yr
|
|
85
|
+
FROM fact_call_report_curated f JOIN dim_credit_union d USING (quarter, cu_number)
|
|
86
|
+
), g AS (
|
|
87
|
+
SELECT f.*, 4.0 / qn AS ann,
|
|
88
|
+
lag(total_assets, 4) OVER w AS assets_1y_ago,
|
|
89
|
+
lag(loans_and_leases_total, 4) OVER w AS loans_1y_ago,
|
|
90
|
+
lag(total_shares_and_deposits, 4) OVER w AS shares_1y_ago,
|
|
91
|
+
lag(members, 4) OVER w AS members_1y_ago,
|
|
92
|
+
lag(loans_new_vehicle + loans_used_vehicle, 4) OVER w AS auto_1y_ago,
|
|
93
|
+
lag(loans_first_lien_residential, 4) OVER w AS first_lien_1y_ago,
|
|
94
|
+
CASE WHEN count(*) OVER w4 = 4 THEN avg(total_assets) OVER w4 END AS avg_assets_4q,
|
|
95
|
+
CASE WHEN count(*) OVER w4 = 4 THEN avg(loans_and_leases_total) OVER w4 END AS avg_loans_4q,
|
|
96
|
+
lag(net_income_ytd, 1) OVER w AS ni_prev, lag(interest_income_ytd, 1) OVER w AS ii_prev,
|
|
97
|
+
lag(interest_expense_ytd, 1) OVER w AS ie_prev, lag(non_interest_expense_ytd, 1) OVER w AS nie_prev,
|
|
98
|
+
lag(provision_for_loan_losses_ytd, 1) OVER w AS prov_prev
|
|
99
|
+
FROM f WINDOW w AS (PARTITION BY cu_number ORDER BY quarter),
|
|
100
|
+
w4 AS (PARTITION BY cu_number ORDER BY quarter ROWS BETWEEN 3 PRECEDING AND CURRENT ROW)
|
|
101
|
+
)
|
|
102
|
+
SELECT quarter, cu_number,
|
|
103
|
+
-- size and growth
|
|
104
|
+
CASE WHEN assets_1y_ago > 0 THEN total_assets / assets_1y_ago - 1 END AS asset_growth_yoy,
|
|
105
|
+
CASE WHEN loans_1y_ago > 0 THEN loans_and_leases_total / loans_1y_ago - 1 END AS loan_growth_yoy,
|
|
106
|
+
CASE WHEN shares_1y_ago > 0 THEN total_shares_and_deposits / shares_1y_ago - 1 END AS share_growth_yoy,
|
|
107
|
+
CASE WHEN members_1y_ago > 0 THEN members / members_1y_ago - 1 END AS member_growth_yoy,
|
|
108
|
+
CASE WHEN auto_1y_ago > 0 THEN (loans_new_vehicle + loans_used_vehicle) / auto_1y_ago - 1 END AS auto_loan_growth_yoy,
|
|
109
|
+
CASE WHEN first_lien_1y_ago > 0 THEN loans_first_lien_residential / first_lien_1y_ago - 1 END AS first_lien_growth_yoy,
|
|
110
|
+
-- balance sheet ratios
|
|
111
|
+
loans_and_leases_total / nullif(total_shares_and_deposits, 0) AS loan_to_share,
|
|
112
|
+
loans_and_leases_total / nullif(total_assets, 0) AS loans_to_assets,
|
|
113
|
+
net_worth / nullif(total_assets, 0) AS net_worth_to_assets,
|
|
114
|
+
(net_worth - coalesce(cecl_transition_provision, 0)) / nullif(total_assets, 0) AS net_worth_ratio_ex_cecl,
|
|
115
|
+
allowance_for_credit_losses / nullif(loans_and_leases_total, 0) AS allowance_to_loans,
|
|
116
|
+
loans_and_leases_total / nullif(members, 0) AS loans_per_member,
|
|
117
|
+
total_shares_and_deposits / nullif(members, 0) AS shares_per_member,
|
|
118
|
+
-- loan mix (share of total loans)
|
|
119
|
+
(loans_new_vehicle + loans_used_vehicle) / nullif(loans_and_leases_total, 0) AS mix_auto,
|
|
120
|
+
(loans_first_lien_residential + loans_junior_lien_residential + loans_other_real_estate) / nullif(loans_and_leases_total, 0) AS mix_residential_real_estate,
|
|
121
|
+
loans_credit_card / nullif(loans_and_leases_total, 0) AS mix_credit_card,
|
|
122
|
+
loans_commercial_total / nullif(loans_and_leases_total, 0) AS mix_commercial,
|
|
123
|
+
-- share and deposit mix (share of total shares and deposits)
|
|
124
|
+
shares_share_drafts / nullif(total_shares_and_deposits, 0) AS deposit_mix_share_drafts,
|
|
125
|
+
shares_regular / nullif(total_shares_and_deposits, 0) AS deposit_mix_regular,
|
|
126
|
+
shares_money_market / nullif(total_shares_and_deposits, 0) AS deposit_mix_money_market,
|
|
127
|
+
shares_certificates / nullif(total_shares_and_deposits, 0) AS deposit_mix_certificates,
|
|
128
|
+
shares_ira_keogh / nullif(total_shares_and_deposits, 0) AS deposit_mix_ira_keogh,
|
|
129
|
+
deposits_non_member / nullif(total_shares_and_deposits, 0) AS deposit_mix_non_member,
|
|
130
|
+
total_shares_and_deposits / nullif(accounts_share_drafts + accounts_certificates + accounts_money_market + accounts_ira_keogh + accounts_non_member, 0) AS avg_balance_per_listed_account,
|
|
131
|
+
-- staffing and branches
|
|
132
|
+
employees_full_time + employees_part_time / 2.0 AS employees_fte_estimate,
|
|
133
|
+
members / nullif(employees_full_time + employees_part_time / 2.0, 0) AS members_per_fte,
|
|
134
|
+
total_assets / nullif(employees_full_time + employees_part_time / 2.0, 0) AS assets_per_fte,
|
|
135
|
+
employee_compensation_ytd * ann / nullif(employees_full_time + employees_part_time / 2.0, 0) AS compensation_per_fte,
|
|
136
|
+
non_interest_expense_ytd * ann / nullif(employees_full_time + employees_part_time / 2.0, 0) AS operating_expense_per_fte,
|
|
137
|
+
total_assets / nullif(branches, 0) AS assets_per_branch,
|
|
138
|
+
members / nullif(branches, 0) AS members_per_branch,
|
|
139
|
+
employee_compensation_ytd / nullif(non_interest_expense_ytd, 0) AS compensation_share_of_opex,
|
|
140
|
+
-- asset quality
|
|
141
|
+
delinquent_2m_plus / nullif(loans_and_leases_total, 0) AS delinquency_rate,
|
|
142
|
+
(delinquent_new_vehicle + delinquent_used_vehicle) / nullif(loans_new_vehicle + loans_used_vehicle, 0) AS auto_delinquency_rate,
|
|
143
|
+
delinquent_credit_card / nullif(loans_credit_card, 0) AS credit_card_delinquency_rate,
|
|
144
|
+
(chargeoffs_ytd - recoveries_ytd) * ann / nullif(loans_and_leases_total, 0) AS net_chargeoff_rate,
|
|
145
|
+
(chargeoffs_ytd - recoveries_ytd) * ann / nullif(avg_loans_4q, 0) AS net_chargeoff_rate_avg_loans_4q,
|
|
146
|
+
provision_for_loan_losses_ytd * ann / nullif(loans_and_leases_total, 0) AS provision_to_loans,
|
|
147
|
+
-- earnings (annualised from year-to-date)
|
|
148
|
+
net_income_ytd * ann / nullif(total_assets, 0) AS roa_year_end_assets,
|
|
149
|
+
net_income_ytd * ann / nullif(avg_assets_4q, 0) AS roa_avg_assets_4q,
|
|
150
|
+
(interest_income_ytd - interest_expense_ytd) * ann / nullif(total_assets, 0) AS nim_year_end_assets,
|
|
151
|
+
(interest_income_ytd - interest_expense_ytd) * ann / nullif(avg_assets_4q, 0) AS nim_avg_assets_4q,
|
|
152
|
+
interest_on_loans_ytd * ann / nullif(loans_and_leases_total, 0) AS loan_yield,
|
|
153
|
+
interest_expense_ytd * ann / nullif(total_shares_and_deposits, 0) AS cost_of_shares,
|
|
154
|
+
non_interest_expense_ytd / nullif((interest_income_ytd - interest_expense_ytd) + non_interest_income_ytd, 0) AS efficiency_ratio,
|
|
155
|
+
non_interest_expense_ytd * ann / nullif(total_assets, 0) AS opex_to_assets,
|
|
156
|
+
fee_income_ytd / nullif(non_interest_income_ytd, 0) AS fee_share_of_non_interest_income,
|
|
157
|
+
-- de-cumulated quarterly flows
|
|
158
|
+
CASE WHEN qn = 1 THEN net_income_ytd ELSE net_income_ytd - ni_prev END AS net_income_quarter,
|
|
159
|
+
CASE WHEN qn = 1 THEN interest_income_ytd ELSE interest_income_ytd - ii_prev END AS interest_income_quarter,
|
|
160
|
+
CASE WHEN qn = 1 THEN interest_expense_ytd ELSE interest_expense_ytd - ie_prev END AS interest_expense_quarter,
|
|
161
|
+
CASE WHEN qn = 1 THEN non_interest_expense_ytd ELSE non_interest_expense_ytd - nie_prev END AS non_interest_expense_quarter,
|
|
162
|
+
CASE WHEN qn = 1 THEN provision_for_loan_losses_ytd ELSE provision_for_loan_losses_ytd - prov_prev END AS provision_quarter
|
|
163
|
+
FROM g
|
|
164
|
+
"""
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def build_all(data_dir: Path) -> duckdb.DuckDBPyConnection:
|
|
168
|
+
con = build(data_dir, data_dir)
|
|
169
|
+
foicu = str(data_dir / "long" / "*_foicu.parquet")
|
|
170
|
+
con.execute(DIM_SQL.format(foicu=foicu))
|
|
171
|
+
con.execute(METRICS_SQL)
|
|
172
|
+
return con
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Command line: download, build, reconcile."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from . import download as dl
|
|
9
|
+
from .build import build_all
|
|
10
|
+
from .dictionary import build_dictionary
|
|
11
|
+
from .ingest import ingest_quarter
|
|
12
|
+
from .reconcile import run as reconcile
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def main(argv=None) -> int:
|
|
16
|
+
ap = argparse.ArgumentParser(prog="ncua-data")
|
|
17
|
+
ap.add_argument("--data", type=Path, default=Path("data"))
|
|
18
|
+
ap.add_argument("--since", type=int, default=2018, help="first year to download")
|
|
19
|
+
ap.add_argument("command", choices=["download", "ingest", "build", "reconcile", "all"])
|
|
20
|
+
a = ap.parse_args(argv)
|
|
21
|
+
cmds = ["download", "ingest", "build", "reconcile"] if a.command == "all" else [a.command]
|
|
22
|
+
if "download" in cmds:
|
|
23
|
+
dl.download(dl.list_quarters(a.since), a.data / "raw")
|
|
24
|
+
if "ingest" in cmds:
|
|
25
|
+
for z in sorted((a.data / "raw").glob("*.zip")):
|
|
26
|
+
q = z.stem.replace("call-report-data-", "")
|
|
27
|
+
if not (a.data / "long" / f"{q}_dict.parquet").exists():
|
|
28
|
+
print(ingest_quarter(z, a.data / "work", a.data / "long"))
|
|
29
|
+
if "build" in cmds or "reconcile" in cmds:
|
|
30
|
+
con = build_all(a.data)
|
|
31
|
+
if "build" in cmds:
|
|
32
|
+
build_dictionary(con)
|
|
33
|
+
out = a.data / "out"
|
|
34
|
+
out.mkdir(exist_ok=True)
|
|
35
|
+
for t in ["dim_credit_union", "fact_call_report_curated", "metrics", "dictionary"]:
|
|
36
|
+
con.execute(f"COPY {t} TO '{out}/{t}.parquet' (FORMAT PARQUET, COMPRESSION ZSTD)")
|
|
37
|
+
print("wrote", sorted(p.name for p in out.glob("*.parquet")))
|
|
38
|
+
if "reconcile" in cmds:
|
|
39
|
+
res = reconcile(con)
|
|
40
|
+
(a.data / "out").mkdir(exist_ok=True)
|
|
41
|
+
(a.data / "out" / "reconciliation.json").write_text(json.dumps(res, indent=2))
|
|
42
|
+
bad = [r for r in res if not r["ok"]]
|
|
43
|
+
print(f"{len(res) - len(bad)}/{len(res)} checks within tolerance")
|
|
44
|
+
for r in bad:
|
|
45
|
+
print("MISMATCH", r["quarter"], r["check"], r["dataset"], "vs published", r["published"])
|
|
46
|
+
return 0
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
if __name__ == "__main__":
|
|
50
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Build the data dictionary table: one row per column in every published table."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from .spec import FIELDS
|
|
5
|
+
|
|
6
|
+
METRIC_DOCS = {
|
|
7
|
+
"asset_growth_yoy": ("Total assets versus the same quarter one year earlier.", "ratio"),
|
|
8
|
+
"loan_growth_yoy": ("Total loans versus the same quarter one year earlier.", "ratio"),
|
|
9
|
+
"share_growth_yoy": ("Total shares and deposits versus one year earlier.", "ratio"),
|
|
10
|
+
"member_growth_yoy": ("Members versus one year earlier.", "ratio"),
|
|
11
|
+
"auto_loan_growth_yoy": ("New plus used vehicle loans versus one year earlier.", "ratio"),
|
|
12
|
+
"first_lien_growth_yoy": ("First-lien 1-4 family loans versus one year earlier.", "ratio"),
|
|
13
|
+
"loan_to_share": ("Loans divided by total shares and deposits. NCUA's headline liquidity ratio.", "ratio"),
|
|
14
|
+
"loans_to_assets": ("Loans divided by total assets.", "ratio"),
|
|
15
|
+
"net_worth_to_assets": ("Net worth divided by total assets (1.0 = 100%).", "ratio"),
|
|
16
|
+
"allowance_to_loans": ("Allowance for credit losses divided by total loans.", "ratio"),
|
|
17
|
+
"net_worth_ratio_ex_cecl": ("Net worth minus the CECL transition provision, over assets. This is how NCUA publishes its net worth ratio from 2023 on.", "ratio"),
|
|
18
|
+
"deposit_mix_share_drafts": ("Share draft (checking) accounts as a share of total shares and deposits.", "ratio"),
|
|
19
|
+
"deposit_mix_regular": ("Regular shares as a share of total shares and deposits.", "ratio"),
|
|
20
|
+
"deposit_mix_money_market": ("Money market shares as a share of total shares and deposits.", "ratio"),
|
|
21
|
+
"deposit_mix_certificates": ("Share certificates as a share of total shares and deposits.", "ratio"),
|
|
22
|
+
"deposit_mix_ira_keogh": ("IRA and Keogh accounts as a share of total shares and deposits.", "ratio"),
|
|
23
|
+
"deposit_mix_non_member": ("Non-member deposits as a share of total shares and deposits.", "ratio"),
|
|
24
|
+
"avg_balance_per_listed_account": ("Total shares and deposits over the count of share draft, certificate, money market, IRA and non-member accounts. Rough; regular share account counts are not collected.", "dollars"),
|
|
25
|
+
"employees_fte_estimate": ("Full-time employees plus half of part-time employees. An estimate, not an NCUA definition.", "count"),
|
|
26
|
+
"members_per_fte": ("Members per estimated full-time-equivalent employee.", "count"),
|
|
27
|
+
"assets_per_fte": ("Assets per estimated full-time-equivalent employee.", "dollars"),
|
|
28
|
+
"compensation_per_fte": ("Annualized employee compensation and benefits per estimated FTE.", "dollars"),
|
|
29
|
+
"operating_expense_per_fte": ("Annualized non-interest expense per estimated FTE.", "dollars"),
|
|
30
|
+
"assets_per_branch": ("Assets per branch.", "dollars"),
|
|
31
|
+
"members_per_branch": ("Members per branch.", "count"),
|
|
32
|
+
"compensation_share_of_opex": ("Employee compensation and benefits as a share of non-interest expense.", "ratio"),
|
|
33
|
+
"loans_per_member": ("Average loan dollars per member.", "dollars"),
|
|
34
|
+
"shares_per_member": ("Average shares and deposits per member.", "dollars"),
|
|
35
|
+
"mix_auto": ("New plus used vehicle loans as a share of total loans.", "ratio"),
|
|
36
|
+
"mix_residential_real_estate": ("First lien, junior lien and other real estate loans as a share of total loans.", "ratio"),
|
|
37
|
+
"mix_credit_card": ("Credit card loans as a share of total loans.", "ratio"),
|
|
38
|
+
"mix_commercial": ("Commercial loans as a share of total loans.", "ratio"),
|
|
39
|
+
"delinquency_rate": ("Loans delinquent two or more months divided by total loans. Matches NCUA's published rate.", "ratio"),
|
|
40
|
+
"auto_delinquency_rate": ("Reportable delinquent vehicle loans divided by vehicle loans.", "ratio"),
|
|
41
|
+
"credit_card_delinquency_rate": ("Delinquent credit card loans divided by credit card loans.", "ratio"),
|
|
42
|
+
"net_chargeoff_rate": ("(Charge-offs minus recoveries), annualized from year to date, over year-end loans.", "ratio"),
|
|
43
|
+
"net_chargeoff_rate_avg_loans_4q": ("Same, over the average of the last four quarter-end loan balances. Closest to NCUA's published ratio but not identical.", "ratio"),
|
|
44
|
+
"provision_to_loans": ("Provision (credit loss expense) annualized over year-end loans.", "ratio"),
|
|
45
|
+
"roa_year_end_assets": ("Net income annualized over year-end assets.", "ratio"),
|
|
46
|
+
"roa_avg_assets_4q": ("Net income annualized over the average of the last four quarter-end assets. NCUA's own denominator is not public, so this can differ by a few basis points.", "ratio"),
|
|
47
|
+
"nim_year_end_assets": ("(Interest income minus interest expense) annualized over year-end assets.", "ratio"),
|
|
48
|
+
"nim_avg_assets_4q": ("Same, over four-quarter average assets. NCUA publishes 3.49% for June 2026; this gives about 3.5%.", "ratio"),
|
|
49
|
+
"loan_yield": ("Interest on loans annualized over year-end loans.", "ratio"),
|
|
50
|
+
"cost_of_shares": ("Total interest expense annualized over shares and deposits.", "ratio"),
|
|
51
|
+
"efficiency_ratio": ("Non-interest expense divided by (net interest income plus non-interest income).", "ratio"),
|
|
52
|
+
"opex_to_assets": ("Non-interest expense annualized over assets.", "ratio"),
|
|
53
|
+
"fee_share_of_non_interest_income": ("Fee income divided by non-interest income.", "ratio"),
|
|
54
|
+
"net_income_quarter": ("Net income for the single quarter (year-to-date minus prior quarter).", "dollars"),
|
|
55
|
+
"interest_income_quarter": ("Interest income for the single quarter.", "dollars"),
|
|
56
|
+
"interest_expense_quarter": ("Interest expense for the single quarter.", "dollars"),
|
|
57
|
+
"non_interest_expense_quarter": ("Non-interest expense for the single quarter.", "dollars"),
|
|
58
|
+
"provision_quarter": ("Provision for loan losses for the single quarter.", "dollars"),
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
DIM_DOCS = {
|
|
62
|
+
"quarter": "Report quarter, YYYY-MM of the quarter end (03, 06, 09, 12).",
|
|
63
|
+
"cu_number": "NCUA charter number. Stable for a credit union over time; disappears when it merges or closes.",
|
|
64
|
+
"rssd": "Federal Reserve RSSD identifier.",
|
|
65
|
+
"name": "Credit union name.", "city": "Mailing city.", "state": "Mailing state.",
|
|
66
|
+
"charter_state": "State of charter (state-chartered only).", "zip_code": "Mailing ZIP.",
|
|
67
|
+
"county_code": "County code.",
|
|
68
|
+
"charter_type": "federal, state_federally_insured, or state_not_federally_insured.",
|
|
69
|
+
"is_federally_insured": "False for the ~85 state-chartered credit unions NCUA does not insure. NCUA's published totals exclude them.",
|
|
70
|
+
"peer_group": "NCUA asset peer group 1-6.", "peer_group_label": "Peer group asset range.",
|
|
71
|
+
"is_minority_depository": "Minority depository institution flag.",
|
|
72
|
+
"is_low_income": "Low-income designation.", "year_opened": "Year the credit union was organized.",
|
|
73
|
+
"field_of_membership_code": "Type-of-membership code.", "ncua_region": "NCUA region code.",
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def build_dictionary(con) -> None:
|
|
78
|
+
rows = []
|
|
79
|
+
for col, desc in DIM_DOCS.items():
|
|
80
|
+
rows.append(("dim_credit_union", col, desc, "attribute", None, None, "FOICU.txt", None, None))
|
|
81
|
+
span = {}
|
|
82
|
+
for f in FIELDS:
|
|
83
|
+
r = con.execute(
|
|
84
|
+
f"SELECT min(quarter), max(quarter) FROM fact_call_report_curated WHERE {f.name} IS NOT NULL AND {f.name} <> 0"
|
|
85
|
+
).fetchone()
|
|
86
|
+
span[f.name] = r
|
|
87
|
+
rows.append(("fact_call_report_curated", f.name, f.description, f.kind, f.category,
|
|
88
|
+
", ".join(c.replace("ACCT_", "Acct_") for c in f.codes), "FS220* tables", r[0], r[1]))
|
|
89
|
+
for name, (desc, unit) in METRIC_DOCS.items():
|
|
90
|
+
rows.append(("metrics", name, desc, unit, "derived", None, "computed from fact_call_report_curated", None, None))
|
|
91
|
+
con.execute("""CREATE OR REPLACE TABLE dictionary(table_name VARCHAR, column_name VARCHAR, description VARCHAR,
|
|
92
|
+
kind VARCHAR, category VARCHAR, ncua_account_codes VARCHAR, source VARCHAR, first_quarter_nonzero VARCHAR, last_quarter_nonzero VARCHAR)""")
|
|
93
|
+
con.executemany("INSERT INTO dictionary VALUES (?,?,?,?,?,?,?,?,?)", rows)
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Find and download NCUA quarterly call report ZIPs.
|
|
2
|
+
|
|
3
|
+
Links are scraped from NCUA's quarterly data page rather than built from a
|
|
4
|
+
pattern, because the file naming has changed over the years.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
import sys
|
|
10
|
+
import urllib.request
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
BASE = "https://ncua.gov"
|
|
14
|
+
PAGE = BASE + "/analysis/credit-union-corporate-call-report-data/quarterly-data"
|
|
15
|
+
UA = {"User-Agent": "ncua-data-analysis (open source research project)"}
|
|
16
|
+
|
|
17
|
+
_LINK = re.compile(r'href="(/files/publications/analysis/call-report-data-(\d{4})-(\d{2})\.zip)"', re.I)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _get(url: str) -> bytes:
|
|
21
|
+
req = urllib.request.Request(url, headers=UA)
|
|
22
|
+
with urllib.request.urlopen(req, timeout=120) as r:
|
|
23
|
+
return r.read()
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def list_quarters(min_year: int = 2018) -> dict[str, str]:
|
|
27
|
+
"""Return {'2026-06': url} for every quarter on the page since min_year."""
|
|
28
|
+
html = _get(PAGE).decode("utf-8", "replace")
|
|
29
|
+
out: dict[str, str] = {}
|
|
30
|
+
for path, y, m in _LINK.findall(html):
|
|
31
|
+
if int(y) >= min_year:
|
|
32
|
+
out[f"{y}-{m}"] = BASE + path
|
|
33
|
+
return dict(sorted(out.items()))
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def download(quarters: dict[str, str], dest: Path) -> list[Path]:
|
|
37
|
+
dest.mkdir(parents=True, exist_ok=True)
|
|
38
|
+
paths = []
|
|
39
|
+
for q, url in quarters.items():
|
|
40
|
+
p = dest / f"call-report-data-{q}.zip"
|
|
41
|
+
if not p.exists() or p.stat().st_size == 0:
|
|
42
|
+
print(f"downloading {q}", file=sys.stderr)
|
|
43
|
+
p.write_bytes(_get(url))
|
|
44
|
+
paths.append(p)
|
|
45
|
+
return paths
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Load raw NCUA quarterly ZIPs into long-format Parquet files.
|
|
2
|
+
|
|
3
|
+
Every account value becomes one row (quarter, cu_number, acct, value), keeping
|
|
4
|
+
non-zero values only. That makes the multi-year account churn easy to handle
|
|
5
|
+
and keeps the intermediate store small.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import csv
|
|
10
|
+
import re
|
|
11
|
+
import zipfile
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
import duckdb
|
|
15
|
+
|
|
16
|
+
ACCT_TABLE = re.compile(r"^fs220[a-z]?\.txt$", re.I)
|
|
17
|
+
ID_COLS = ("CU_NUMBER", "CYCLE_DATE", "JOIN_NUMBER", "UPDATE_DATE")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def extract(zip_path: Path, dest: Path) -> Path:
|
|
21
|
+
out = dest / zip_path.stem.replace("call-report-data-", "")
|
|
22
|
+
if not out.exists():
|
|
23
|
+
out.mkdir(parents=True)
|
|
24
|
+
with zipfile.ZipFile(zip_path) as z:
|
|
25
|
+
z.extractall(out)
|
|
26
|
+
# NCUA files are Windows-1252 (stray 0x92 apostrophes appear in free text
|
|
27
|
+
# fields). Re-encode to UTF-8 so every reader agrees.
|
|
28
|
+
for f in out.iterdir():
|
|
29
|
+
if f.suffix.lower() == ".txt":
|
|
30
|
+
f.write_bytes(f.read_bytes().decode("cp1252", errors="replace").encode("utf-8"))
|
|
31
|
+
return out
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _find(folder: Path, name: str) -> Path | None:
|
|
35
|
+
for p in folder.iterdir():
|
|
36
|
+
if p.name.lower() == name.lower():
|
|
37
|
+
return p
|
|
38
|
+
return None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _unique_header(path: Path) -> list[str]:
|
|
42
|
+
"""Header row with case-insensitive duplicates suffixed (some files repeat a code)."""
|
|
43
|
+
with open(path, encoding="utf-8", newline="") as fh:
|
|
44
|
+
header = next(csv.reader(fh))
|
|
45
|
+
seen: dict[str, int] = {}
|
|
46
|
+
out = []
|
|
47
|
+
for h in header:
|
|
48
|
+
k = h.upper()
|
|
49
|
+
seen[k] = seen.get(k, 0) + 1
|
|
50
|
+
out.append(h if seen[k] == 1 else f"{h}__dup{seen[k]}")
|
|
51
|
+
return out
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def ingest_quarter(zip_path: Path, work: Path, out_dir: Path) -> dict:
|
|
55
|
+
"""Write long.parquet, foicu.parquet and dict.parquet for one quarter."""
|
|
56
|
+
quarter = zip_path.stem.replace("call-report-data-", "")
|
|
57
|
+
folder = extract(zip_path, work)
|
|
58
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
59
|
+
con = duckdb.connect()
|
|
60
|
+
parts = []
|
|
61
|
+
for p in sorted(folder.iterdir()):
|
|
62
|
+
if ACCT_TABLE.match(p.name):
|
|
63
|
+
names = _unique_header(p)
|
|
64
|
+
parts.append(
|
|
65
|
+
f"""SELECT CU_NUMBER::INTEGER AS cu_number, upper(acct) AS acct,
|
|
66
|
+
try_cast(val AS DOUBLE) AS value, '{p.stem.upper()}' AS source_table
|
|
67
|
+
FROM (UNPIVOT (SELECT * FROM read_csv('{p}', all_varchar=true, header=true,
|
|
68
|
+
names={names!r}, ignore_errors=true))
|
|
69
|
+
ON COLUMNS('(?i)^acct_')
|
|
70
|
+
INTO NAME acct VALUE val)
|
|
71
|
+
WHERE try_cast(val AS DOUBLE) <> 0"""
|
|
72
|
+
)
|
|
73
|
+
con.execute(
|
|
74
|
+
f"COPY (SELECT '{quarter}' AS quarter, * FROM ({' UNION ALL '.join(parts)})) "
|
|
75
|
+
f"TO '{out_dir}/{quarter}_long.parquet' (FORMAT PARQUET, COMPRESSION ZSTD)"
|
|
76
|
+
)
|
|
77
|
+
foicu = _find(folder, "FOICU.txt")
|
|
78
|
+
con.execute(
|
|
79
|
+
f"COPY (SELECT '{quarter}' AS quarter, * FROM read_csv('{foicu}', all_varchar=true, "
|
|
80
|
+
f"ignore_errors=true)) TO '{out_dir}/{quarter}_foicu.parquet' (FORMAT PARQUET)"
|
|
81
|
+
)
|
|
82
|
+
ad = _find(folder, "AcctDesc.txt")
|
|
83
|
+
con.execute(
|
|
84
|
+
f"COPY (SELECT '{quarter}' AS quarter, * FROM read_csv('{ad}', all_varchar=true, "
|
|
85
|
+
f"ignore_errors=true, strict_mode=false)) TO '{out_dir}/{quarter}_dict.parquet' (FORMAT PARQUET)"
|
|
86
|
+
)
|
|
87
|
+
n = con.execute(f"SELECT count(*), count(DISTINCT cu_number), count(DISTINCT acct) FROM '{out_dir}/{quarter}_long.parquet'").fetchone()
|
|
88
|
+
return {"quarter": quarter, "rows": n[0], "cus": n[1], "accts": n[2]}
|