ncua-data-analysis 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ncua_data/__init__.py +2 -0
- ncua_data/__main__.py +2 -0
- ncua_data/build.py +172 -0
- ncua_data/cli.py +50 -0
- ncua_data/dictionary.py +93 -0
- ncua_data/download.py +45 -0
- ncua_data/ingest.py +88 -0
- ncua_data/mcp_data/dictionary.json +1297 -0
- ncua_data/mcp_server.py +405 -0
- ncua_data/reconcile.py +97 -0
- ncua_data/spec.py +163 -0
- ncua_data_analysis-0.1.0.dist-info/METADATA +9 -0
- ncua_data_analysis-0.1.0.dist-info/RECORD +17 -0
- ncua_data_analysis-0.1.0.dist-info/WHEEL +5 -0
- ncua_data_analysis-0.1.0.dist-info/entry_points.txt +3 -0
- ncua_data_analysis-0.1.0.dist-info/licenses/LICENSE +21 -0
- ncua_data_analysis-0.1.0.dist-info/top_level.txt +1 -0
ncua_data/__init__.py
ADDED
ncua_data/__main__.py
ADDED
ncua_data/build.py
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""Build the curated tables (dim, fact, metrics, dictionary) from long Parquet."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
import duckdb
|
|
7
|
+
|
|
8
|
+
from .spec import FIELDS
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def fact_sql(long_glob: str) -> str:
|
|
12
|
+
cols = []
|
|
13
|
+
for f in FIELDS:
|
|
14
|
+
codes = ", ".join(f"'{c}'" for c in f.codes)
|
|
15
|
+
agg = f"sum(l.value) FILTER (WHERE l.acct IN ({codes}))"
|
|
16
|
+
# No row in the long table means the credit union reported zero (or left it blank).
|
|
17
|
+
expr = f"coalesce({agg}, 0)"
|
|
18
|
+
if f.first_quarter:
|
|
19
|
+
expr = f"CASE WHEN q.quarter >= '{f.first_quarter}' THEN {expr} END"
|
|
20
|
+
cols.append(f"{expr} AS {f.name}")
|
|
21
|
+
return f"""
|
|
22
|
+
SELECT q.quarter, q.cu_number, {', '.join(cols)}
|
|
23
|
+
FROM quarter_cus q
|
|
24
|
+
LEFT JOIN (SELECT * FROM read_parquet('{long_glob}')) l
|
|
25
|
+
ON l.quarter = q.quarter AND l.cu_number = q.cu_number
|
|
26
|
+
GROUP BY q.quarter, q.cu_number"""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def build(data_dir: Path, out_dir: Path) -> duckdb.DuckDBPyConnection:
|
|
30
|
+
con = duckdb.connect()
|
|
31
|
+
con.execute("SET memory_limit='1400MB'; SET threads=2")
|
|
32
|
+
long_glob = str(data_dir / "long" / "*_long.parquet")
|
|
33
|
+
foicu_glob = str(data_dir / "long" / "*_foicu.parquet")
|
|
34
|
+
con.execute(
|
|
35
|
+
f"""CREATE TABLE quarter_cus AS
|
|
36
|
+
SELECT quarter, CU_NUMBER::INTEGER AS cu_number
|
|
37
|
+
FROM read_parquet('{foicu_glob}', union_by_name=true)"""
|
|
38
|
+
)
|
|
39
|
+
# One quarter at a time keeps memory low enough for a 2 GB machine.
|
|
40
|
+
first = True
|
|
41
|
+
for lf in sorted((data_dir / "long").glob("*_long.parquet")):
|
|
42
|
+
q = lf.name.split("_")[0]
|
|
43
|
+
sql = fact_sql(str(lf)).replace("FROM quarter_cus q", f"FROM (SELECT * FROM quarter_cus WHERE quarter = '{q}') q")
|
|
44
|
+
if first:
|
|
45
|
+
con.execute(f"CREATE TABLE fact_call_report_curated AS {sql}")
|
|
46
|
+
first = False
|
|
47
|
+
else:
|
|
48
|
+
con.execute(f"INSERT INTO fact_call_report_curated {sql}")
|
|
49
|
+
return con
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
DIM_SQL = """
|
|
53
|
+
CREATE TABLE dim_credit_union AS
|
|
54
|
+
SELECT quarter,
|
|
55
|
+
CU_NUMBER::INTEGER AS cu_number,
|
|
56
|
+
try_cast(RSSD AS BIGINT) AS rssd,
|
|
57
|
+
trim(CU_NAME) AS name,
|
|
58
|
+
trim(CITY) AS city,
|
|
59
|
+
trim(STATE) AS state,
|
|
60
|
+
trim(CharterState) AS charter_state,
|
|
61
|
+
trim(ZIP_CODE) AS zip_code,
|
|
62
|
+
try_cast(COUNTY_CODE AS INTEGER) AS county_code,
|
|
63
|
+
CASE CU_TYPE WHEN '1' THEN 'federal' WHEN '2' THEN 'state_federally_insured'
|
|
64
|
+
WHEN '3' THEN 'state_not_federally_insured' END AS charter_type,
|
|
65
|
+
CU_TYPE <> '3' AS is_federally_insured,
|
|
66
|
+
try_cast(Peer_Group AS INTEGER) AS peer_group,
|
|
67
|
+
CASE try_cast(Peer_Group AS INTEGER)
|
|
68
|
+
WHEN 1 THEN 'under $2M' WHEN 2 THEN '$2M-$10M' WHEN 3 THEN '$10M-$50M'
|
|
69
|
+
WHEN 4 THEN '$50M-$100M' WHEN 5 THEN '$100M-$500M' WHEN 6 THEN '$500M+' END AS peer_group_label,
|
|
70
|
+
lower(IsMDI) IN ('true', '1') AS is_minority_depository,
|
|
71
|
+
try_cast(LIMITED_INC AS INTEGER) = 1 AS is_low_income,
|
|
72
|
+
try_cast(YEAR_OPENED AS INTEGER) AS year_opened,
|
|
73
|
+
trim(TOM_CODE) AS field_of_membership_code,
|
|
74
|
+
trim(REGION) AS ncua_region
|
|
75
|
+
FROM read_parquet('{foicu}', union_by_name=true)
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
# Metrics. Flow fields are year-to-date, so annualise by 4 / quarter number.
|
|
79
|
+
METRICS_SQL = """
|
|
80
|
+
CREATE TABLE metrics AS
|
|
81
|
+
WITH f AS (
|
|
82
|
+
SELECT f.*, d.is_federally_insured,
|
|
83
|
+
CAST(substr(f.quarter, 6, 2) AS INTEGER) / 3 AS qn,
|
|
84
|
+
CAST(substr(f.quarter, 1, 4) AS INTEGER) AS yr
|
|
85
|
+
FROM fact_call_report_curated f JOIN dim_credit_union d USING (quarter, cu_number)
|
|
86
|
+
), g AS (
|
|
87
|
+
SELECT f.*, 4.0 / qn AS ann,
|
|
88
|
+
lag(total_assets, 4) OVER w AS assets_1y_ago,
|
|
89
|
+
lag(loans_and_leases_total, 4) OVER w AS loans_1y_ago,
|
|
90
|
+
lag(total_shares_and_deposits, 4) OVER w AS shares_1y_ago,
|
|
91
|
+
lag(members, 4) OVER w AS members_1y_ago,
|
|
92
|
+
lag(loans_new_vehicle + loans_used_vehicle, 4) OVER w AS auto_1y_ago,
|
|
93
|
+
lag(loans_first_lien_residential, 4) OVER w AS first_lien_1y_ago,
|
|
94
|
+
CASE WHEN count(*) OVER w4 = 4 THEN avg(total_assets) OVER w4 END AS avg_assets_4q,
|
|
95
|
+
CASE WHEN count(*) OVER w4 = 4 THEN avg(loans_and_leases_total) OVER w4 END AS avg_loans_4q,
|
|
96
|
+
lag(net_income_ytd, 1) OVER w AS ni_prev, lag(interest_income_ytd, 1) OVER w AS ii_prev,
|
|
97
|
+
lag(interest_expense_ytd, 1) OVER w AS ie_prev, lag(non_interest_expense_ytd, 1) OVER w AS nie_prev,
|
|
98
|
+
lag(provision_for_loan_losses_ytd, 1) OVER w AS prov_prev
|
|
99
|
+
FROM f WINDOW w AS (PARTITION BY cu_number ORDER BY quarter),
|
|
100
|
+
w4 AS (PARTITION BY cu_number ORDER BY quarter ROWS BETWEEN 3 PRECEDING AND CURRENT ROW)
|
|
101
|
+
)
|
|
102
|
+
SELECT quarter, cu_number,
|
|
103
|
+
-- size and growth
|
|
104
|
+
CASE WHEN assets_1y_ago > 0 THEN total_assets / assets_1y_ago - 1 END AS asset_growth_yoy,
|
|
105
|
+
CASE WHEN loans_1y_ago > 0 THEN loans_and_leases_total / loans_1y_ago - 1 END AS loan_growth_yoy,
|
|
106
|
+
CASE WHEN shares_1y_ago > 0 THEN total_shares_and_deposits / shares_1y_ago - 1 END AS share_growth_yoy,
|
|
107
|
+
CASE WHEN members_1y_ago > 0 THEN members / members_1y_ago - 1 END AS member_growth_yoy,
|
|
108
|
+
CASE WHEN auto_1y_ago > 0 THEN (loans_new_vehicle + loans_used_vehicle) / auto_1y_ago - 1 END AS auto_loan_growth_yoy,
|
|
109
|
+
CASE WHEN first_lien_1y_ago > 0 THEN loans_first_lien_residential / first_lien_1y_ago - 1 END AS first_lien_growth_yoy,
|
|
110
|
+
-- balance sheet ratios
|
|
111
|
+
loans_and_leases_total / nullif(total_shares_and_deposits, 0) AS loan_to_share,
|
|
112
|
+
loans_and_leases_total / nullif(total_assets, 0) AS loans_to_assets,
|
|
113
|
+
net_worth / nullif(total_assets, 0) AS net_worth_to_assets,
|
|
114
|
+
(net_worth - coalesce(cecl_transition_provision, 0)) / nullif(total_assets, 0) AS net_worth_ratio_ex_cecl,
|
|
115
|
+
allowance_for_credit_losses / nullif(loans_and_leases_total, 0) AS allowance_to_loans,
|
|
116
|
+
loans_and_leases_total / nullif(members, 0) AS loans_per_member,
|
|
117
|
+
total_shares_and_deposits / nullif(members, 0) AS shares_per_member,
|
|
118
|
+
-- loan mix (share of total loans)
|
|
119
|
+
(loans_new_vehicle + loans_used_vehicle) / nullif(loans_and_leases_total, 0) AS mix_auto,
|
|
120
|
+
(loans_first_lien_residential + loans_junior_lien_residential + loans_other_real_estate) / nullif(loans_and_leases_total, 0) AS mix_residential_real_estate,
|
|
121
|
+
loans_credit_card / nullif(loans_and_leases_total, 0) AS mix_credit_card,
|
|
122
|
+
loans_commercial_total / nullif(loans_and_leases_total, 0) AS mix_commercial,
|
|
123
|
+
-- share and deposit mix (share of total shares and deposits)
|
|
124
|
+
shares_share_drafts / nullif(total_shares_and_deposits, 0) AS deposit_mix_share_drafts,
|
|
125
|
+
shares_regular / nullif(total_shares_and_deposits, 0) AS deposit_mix_regular,
|
|
126
|
+
shares_money_market / nullif(total_shares_and_deposits, 0) AS deposit_mix_money_market,
|
|
127
|
+
shares_certificates / nullif(total_shares_and_deposits, 0) AS deposit_mix_certificates,
|
|
128
|
+
shares_ira_keogh / nullif(total_shares_and_deposits, 0) AS deposit_mix_ira_keogh,
|
|
129
|
+
deposits_non_member / nullif(total_shares_and_deposits, 0) AS deposit_mix_non_member,
|
|
130
|
+
total_shares_and_deposits / nullif(accounts_share_drafts + accounts_certificates + accounts_money_market + accounts_ira_keogh + accounts_non_member, 0) AS avg_balance_per_listed_account,
|
|
131
|
+
-- staffing and branches
|
|
132
|
+
employees_full_time + employees_part_time / 2.0 AS employees_fte_estimate,
|
|
133
|
+
members / nullif(employees_full_time + employees_part_time / 2.0, 0) AS members_per_fte,
|
|
134
|
+
total_assets / nullif(employees_full_time + employees_part_time / 2.0, 0) AS assets_per_fte,
|
|
135
|
+
employee_compensation_ytd * ann / nullif(employees_full_time + employees_part_time / 2.0, 0) AS compensation_per_fte,
|
|
136
|
+
non_interest_expense_ytd * ann / nullif(employees_full_time + employees_part_time / 2.0, 0) AS operating_expense_per_fte,
|
|
137
|
+
total_assets / nullif(branches, 0) AS assets_per_branch,
|
|
138
|
+
members / nullif(branches, 0) AS members_per_branch,
|
|
139
|
+
employee_compensation_ytd / nullif(non_interest_expense_ytd, 0) AS compensation_share_of_opex,
|
|
140
|
+
-- asset quality
|
|
141
|
+
delinquent_2m_plus / nullif(loans_and_leases_total, 0) AS delinquency_rate,
|
|
142
|
+
(delinquent_new_vehicle + delinquent_used_vehicle) / nullif(loans_new_vehicle + loans_used_vehicle, 0) AS auto_delinquency_rate,
|
|
143
|
+
delinquent_credit_card / nullif(loans_credit_card, 0) AS credit_card_delinquency_rate,
|
|
144
|
+
(chargeoffs_ytd - recoveries_ytd) * ann / nullif(loans_and_leases_total, 0) AS net_chargeoff_rate,
|
|
145
|
+
(chargeoffs_ytd - recoveries_ytd) * ann / nullif(avg_loans_4q, 0) AS net_chargeoff_rate_avg_loans_4q,
|
|
146
|
+
provision_for_loan_losses_ytd * ann / nullif(loans_and_leases_total, 0) AS provision_to_loans,
|
|
147
|
+
-- earnings (annualised from year-to-date)
|
|
148
|
+
net_income_ytd * ann / nullif(total_assets, 0) AS roa_year_end_assets,
|
|
149
|
+
net_income_ytd * ann / nullif(avg_assets_4q, 0) AS roa_avg_assets_4q,
|
|
150
|
+
(interest_income_ytd - interest_expense_ytd) * ann / nullif(total_assets, 0) AS nim_year_end_assets,
|
|
151
|
+
(interest_income_ytd - interest_expense_ytd) * ann / nullif(avg_assets_4q, 0) AS nim_avg_assets_4q,
|
|
152
|
+
interest_on_loans_ytd * ann / nullif(loans_and_leases_total, 0) AS loan_yield,
|
|
153
|
+
interest_expense_ytd * ann / nullif(total_shares_and_deposits, 0) AS cost_of_shares,
|
|
154
|
+
non_interest_expense_ytd / nullif((interest_income_ytd - interest_expense_ytd) + non_interest_income_ytd, 0) AS efficiency_ratio,
|
|
155
|
+
non_interest_expense_ytd * ann / nullif(total_assets, 0) AS opex_to_assets,
|
|
156
|
+
fee_income_ytd / nullif(non_interest_income_ytd, 0) AS fee_share_of_non_interest_income,
|
|
157
|
+
-- de-cumulated quarterly flows
|
|
158
|
+
CASE WHEN qn = 1 THEN net_income_ytd ELSE net_income_ytd - ni_prev END AS net_income_quarter,
|
|
159
|
+
CASE WHEN qn = 1 THEN interest_income_ytd ELSE interest_income_ytd - ii_prev END AS interest_income_quarter,
|
|
160
|
+
CASE WHEN qn = 1 THEN interest_expense_ytd ELSE interest_expense_ytd - ie_prev END AS interest_expense_quarter,
|
|
161
|
+
CASE WHEN qn = 1 THEN non_interest_expense_ytd ELSE non_interest_expense_ytd - nie_prev END AS non_interest_expense_quarter,
|
|
162
|
+
CASE WHEN qn = 1 THEN provision_for_loan_losses_ytd ELSE provision_for_loan_losses_ytd - prov_prev END AS provision_quarter
|
|
163
|
+
FROM g
|
|
164
|
+
"""
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def build_all(data_dir: Path) -> duckdb.DuckDBPyConnection:
|
|
168
|
+
con = build(data_dir, data_dir)
|
|
169
|
+
foicu = str(data_dir / "long" / "*_foicu.parquet")
|
|
170
|
+
con.execute(DIM_SQL.format(foicu=foicu))
|
|
171
|
+
con.execute(METRICS_SQL)
|
|
172
|
+
return con
|
ncua_data/cli.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Command line: download, build, reconcile."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from . import download as dl
|
|
9
|
+
from .build import build_all
|
|
10
|
+
from .dictionary import build_dictionary
|
|
11
|
+
from .ingest import ingest_quarter
|
|
12
|
+
from .reconcile import run as reconcile
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def main(argv=None) -> int:
|
|
16
|
+
ap = argparse.ArgumentParser(prog="ncua-data")
|
|
17
|
+
ap.add_argument("--data", type=Path, default=Path("data"))
|
|
18
|
+
ap.add_argument("--since", type=int, default=2018, help="first year to download")
|
|
19
|
+
ap.add_argument("command", choices=["download", "ingest", "build", "reconcile", "all"])
|
|
20
|
+
a = ap.parse_args(argv)
|
|
21
|
+
cmds = ["download", "ingest", "build", "reconcile"] if a.command == "all" else [a.command]
|
|
22
|
+
if "download" in cmds:
|
|
23
|
+
dl.download(dl.list_quarters(a.since), a.data / "raw")
|
|
24
|
+
if "ingest" in cmds:
|
|
25
|
+
for z in sorted((a.data / "raw").glob("*.zip")):
|
|
26
|
+
q = z.stem.replace("call-report-data-", "")
|
|
27
|
+
if not (a.data / "long" / f"{q}_dict.parquet").exists():
|
|
28
|
+
print(ingest_quarter(z, a.data / "work", a.data / "long"))
|
|
29
|
+
if "build" in cmds or "reconcile" in cmds:
|
|
30
|
+
con = build_all(a.data)
|
|
31
|
+
if "build" in cmds:
|
|
32
|
+
build_dictionary(con)
|
|
33
|
+
out = a.data / "out"
|
|
34
|
+
out.mkdir(exist_ok=True)
|
|
35
|
+
for t in ["dim_credit_union", "fact_call_report_curated", "metrics", "dictionary"]:
|
|
36
|
+
con.execute(f"COPY {t} TO '{out}/{t}.parquet' (FORMAT PARQUET, COMPRESSION ZSTD)")
|
|
37
|
+
print("wrote", sorted(p.name for p in out.glob("*.parquet")))
|
|
38
|
+
if "reconcile" in cmds:
|
|
39
|
+
res = reconcile(con)
|
|
40
|
+
(a.data / "out").mkdir(exist_ok=True)
|
|
41
|
+
(a.data / "out" / "reconciliation.json").write_text(json.dumps(res, indent=2))
|
|
42
|
+
bad = [r for r in res if not r["ok"]]
|
|
43
|
+
print(f"{len(res) - len(bad)}/{len(res)} checks within tolerance")
|
|
44
|
+
for r in bad:
|
|
45
|
+
print("MISMATCH", r["quarter"], r["check"], r["dataset"], "vs published", r["published"])
|
|
46
|
+
return 0
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
if __name__ == "__main__":
|
|
50
|
+
raise SystemExit(main())
|
ncua_data/dictionary.py
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Build the data dictionary table: one row per column in every published table."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from .spec import FIELDS
|
|
5
|
+
|
|
6
|
+
METRIC_DOCS = {
|
|
7
|
+
"asset_growth_yoy": ("Total assets versus the same quarter one year earlier.", "ratio"),
|
|
8
|
+
"loan_growth_yoy": ("Total loans versus the same quarter one year earlier.", "ratio"),
|
|
9
|
+
"share_growth_yoy": ("Total shares and deposits versus one year earlier.", "ratio"),
|
|
10
|
+
"member_growth_yoy": ("Members versus one year earlier.", "ratio"),
|
|
11
|
+
"auto_loan_growth_yoy": ("New plus used vehicle loans versus one year earlier.", "ratio"),
|
|
12
|
+
"first_lien_growth_yoy": ("First-lien 1-4 family loans versus one year earlier.", "ratio"),
|
|
13
|
+
"loan_to_share": ("Loans divided by total shares and deposits. NCUA's headline liquidity ratio.", "ratio"),
|
|
14
|
+
"loans_to_assets": ("Loans divided by total assets.", "ratio"),
|
|
15
|
+
"net_worth_to_assets": ("Net worth divided by total assets (1.0 = 100%).", "ratio"),
|
|
16
|
+
"allowance_to_loans": ("Allowance for credit losses divided by total loans.", "ratio"),
|
|
17
|
+
"net_worth_ratio_ex_cecl": ("Net worth minus the CECL transition provision, over assets. This is how NCUA publishes its net worth ratio from 2023 on.", "ratio"),
|
|
18
|
+
"deposit_mix_share_drafts": ("Share draft (checking) accounts as a share of total shares and deposits.", "ratio"),
|
|
19
|
+
"deposit_mix_regular": ("Regular shares as a share of total shares and deposits.", "ratio"),
|
|
20
|
+
"deposit_mix_money_market": ("Money market shares as a share of total shares and deposits.", "ratio"),
|
|
21
|
+
"deposit_mix_certificates": ("Share certificates as a share of total shares and deposits.", "ratio"),
|
|
22
|
+
"deposit_mix_ira_keogh": ("IRA and Keogh accounts as a share of total shares and deposits.", "ratio"),
|
|
23
|
+
"deposit_mix_non_member": ("Non-member deposits as a share of total shares and deposits.", "ratio"),
|
|
24
|
+
"avg_balance_per_listed_account": ("Total shares and deposits over the count of share draft, certificate, money market, IRA and non-member accounts. Rough; regular share account counts are not collected.", "dollars"),
|
|
25
|
+
"employees_fte_estimate": ("Full-time employees plus half of part-time employees. An estimate, not an NCUA definition.", "count"),
|
|
26
|
+
"members_per_fte": ("Members per estimated full-time-equivalent employee.", "count"),
|
|
27
|
+
"assets_per_fte": ("Assets per estimated full-time-equivalent employee.", "dollars"),
|
|
28
|
+
"compensation_per_fte": ("Annualized employee compensation and benefits per estimated FTE.", "dollars"),
|
|
29
|
+
"operating_expense_per_fte": ("Annualized non-interest expense per estimated FTE.", "dollars"),
|
|
30
|
+
"assets_per_branch": ("Assets per branch.", "dollars"),
|
|
31
|
+
"members_per_branch": ("Members per branch.", "count"),
|
|
32
|
+
"compensation_share_of_opex": ("Employee compensation and benefits as a share of non-interest expense.", "ratio"),
|
|
33
|
+
"loans_per_member": ("Average loan dollars per member.", "dollars"),
|
|
34
|
+
"shares_per_member": ("Average shares and deposits per member.", "dollars"),
|
|
35
|
+
"mix_auto": ("New plus used vehicle loans as a share of total loans.", "ratio"),
|
|
36
|
+
"mix_residential_real_estate": ("First lien, junior lien and other real estate loans as a share of total loans.", "ratio"),
|
|
37
|
+
"mix_credit_card": ("Credit card loans as a share of total loans.", "ratio"),
|
|
38
|
+
"mix_commercial": ("Commercial loans as a share of total loans.", "ratio"),
|
|
39
|
+
"delinquency_rate": ("Loans delinquent two or more months divided by total loans. Matches NCUA's published rate.", "ratio"),
|
|
40
|
+
"auto_delinquency_rate": ("Reportable delinquent vehicle loans divided by vehicle loans.", "ratio"),
|
|
41
|
+
"credit_card_delinquency_rate": ("Delinquent credit card loans divided by credit card loans.", "ratio"),
|
|
42
|
+
"net_chargeoff_rate": ("(Charge-offs minus recoveries), annualized from year to date, over year-end loans.", "ratio"),
|
|
43
|
+
"net_chargeoff_rate_avg_loans_4q": ("Same, over the average of the last four quarter-end loan balances. Closest to NCUA's published ratio but not identical.", "ratio"),
|
|
44
|
+
"provision_to_loans": ("Provision (credit loss expense) annualized over year-end loans.", "ratio"),
|
|
45
|
+
"roa_year_end_assets": ("Net income annualized over year-end assets.", "ratio"),
|
|
46
|
+
"roa_avg_assets_4q": ("Net income annualized over the average of the last four quarter-end assets. NCUA's own denominator is not public, so this can differ by a few basis points.", "ratio"),
|
|
47
|
+
"nim_year_end_assets": ("(Interest income minus interest expense) annualized over year-end assets.", "ratio"),
|
|
48
|
+
"nim_avg_assets_4q": ("Same, over four-quarter average assets. NCUA publishes 3.49% for June 2026; this gives about 3.5%.", "ratio"),
|
|
49
|
+
"loan_yield": ("Interest on loans annualized over year-end loans.", "ratio"),
|
|
50
|
+
"cost_of_shares": ("Total interest expense annualized over shares and deposits.", "ratio"),
|
|
51
|
+
"efficiency_ratio": ("Non-interest expense divided by (net interest income plus non-interest income).", "ratio"),
|
|
52
|
+
"opex_to_assets": ("Non-interest expense annualized over assets.", "ratio"),
|
|
53
|
+
"fee_share_of_non_interest_income": ("Fee income divided by non-interest income.", "ratio"),
|
|
54
|
+
"net_income_quarter": ("Net income for the single quarter (year-to-date minus prior quarter).", "dollars"),
|
|
55
|
+
"interest_income_quarter": ("Interest income for the single quarter.", "dollars"),
|
|
56
|
+
"interest_expense_quarter": ("Interest expense for the single quarter.", "dollars"),
|
|
57
|
+
"non_interest_expense_quarter": ("Non-interest expense for the single quarter.", "dollars"),
|
|
58
|
+
"provision_quarter": ("Provision for loan losses for the single quarter.", "dollars"),
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
DIM_DOCS = {
|
|
62
|
+
"quarter": "Report quarter, YYYY-MM of the quarter end (03, 06, 09, 12).",
|
|
63
|
+
"cu_number": "NCUA charter number. Stable for a credit union over time; disappears when it merges or closes.",
|
|
64
|
+
"rssd": "Federal Reserve RSSD identifier.",
|
|
65
|
+
"name": "Credit union name.", "city": "Mailing city.", "state": "Mailing state.",
|
|
66
|
+
"charter_state": "State of charter (state-chartered only).", "zip_code": "Mailing ZIP.",
|
|
67
|
+
"county_code": "County code.",
|
|
68
|
+
"charter_type": "federal, state_federally_insured, or state_not_federally_insured.",
|
|
69
|
+
"is_federally_insured": "False for the ~85 state-chartered credit unions NCUA does not insure. NCUA's published totals exclude them.",
|
|
70
|
+
"peer_group": "NCUA asset peer group 1-6.", "peer_group_label": "Peer group asset range.",
|
|
71
|
+
"is_minority_depository": "Minority depository institution flag.",
|
|
72
|
+
"is_low_income": "Low-income designation.", "year_opened": "Year the credit union was organized.",
|
|
73
|
+
"field_of_membership_code": "Type-of-membership code.", "ncua_region": "NCUA region code.",
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def build_dictionary(con) -> None:
|
|
78
|
+
rows = []
|
|
79
|
+
for col, desc in DIM_DOCS.items():
|
|
80
|
+
rows.append(("dim_credit_union", col, desc, "attribute", None, None, "FOICU.txt", None, None))
|
|
81
|
+
span = {}
|
|
82
|
+
for f in FIELDS:
|
|
83
|
+
r = con.execute(
|
|
84
|
+
f"SELECT min(quarter), max(quarter) FROM fact_call_report_curated WHERE {f.name} IS NOT NULL AND {f.name} <> 0"
|
|
85
|
+
).fetchone()
|
|
86
|
+
span[f.name] = r
|
|
87
|
+
rows.append(("fact_call_report_curated", f.name, f.description, f.kind, f.category,
|
|
88
|
+
", ".join(c.replace("ACCT_", "Acct_") for c in f.codes), "FS220* tables", r[0], r[1]))
|
|
89
|
+
for name, (desc, unit) in METRIC_DOCS.items():
|
|
90
|
+
rows.append(("metrics", name, desc, unit, "derived", None, "computed from fact_call_report_curated", None, None))
|
|
91
|
+
con.execute("""CREATE OR REPLACE TABLE dictionary(table_name VARCHAR, column_name VARCHAR, description VARCHAR,
|
|
92
|
+
kind VARCHAR, category VARCHAR, ncua_account_codes VARCHAR, source VARCHAR, first_quarter_nonzero VARCHAR, last_quarter_nonzero VARCHAR)""")
|
|
93
|
+
con.executemany("INSERT INTO dictionary VALUES (?,?,?,?,?,?,?,?,?)", rows)
|
ncua_data/download.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Find and download NCUA quarterly call report ZIPs.
|
|
2
|
+
|
|
3
|
+
Links are scraped from NCUA's quarterly data page rather than built from a
|
|
4
|
+
pattern, because the file naming has changed over the years.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
import sys
|
|
10
|
+
import urllib.request
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
BASE = "https://ncua.gov"
|
|
14
|
+
PAGE = BASE + "/analysis/credit-union-corporate-call-report-data/quarterly-data"
|
|
15
|
+
UA = {"User-Agent": "ncua-data-analysis (open source research project)"}
|
|
16
|
+
|
|
17
|
+
_LINK = re.compile(r'href="(/files/publications/analysis/call-report-data-(\d{4})-(\d{2})\.zip)"', re.I)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _get(url: str) -> bytes:
|
|
21
|
+
req = urllib.request.Request(url, headers=UA)
|
|
22
|
+
with urllib.request.urlopen(req, timeout=120) as r:
|
|
23
|
+
return r.read()
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def list_quarters(min_year: int = 2018) -> dict[str, str]:
|
|
27
|
+
"""Return {'2026-06': url} for every quarter on the page since min_year."""
|
|
28
|
+
html = _get(PAGE).decode("utf-8", "replace")
|
|
29
|
+
out: dict[str, str] = {}
|
|
30
|
+
for path, y, m in _LINK.findall(html):
|
|
31
|
+
if int(y) >= min_year:
|
|
32
|
+
out[f"{y}-{m}"] = BASE + path
|
|
33
|
+
return dict(sorted(out.items()))
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def download(quarters: dict[str, str], dest: Path) -> list[Path]:
|
|
37
|
+
dest.mkdir(parents=True, exist_ok=True)
|
|
38
|
+
paths = []
|
|
39
|
+
for q, url in quarters.items():
|
|
40
|
+
p = dest / f"call-report-data-{q}.zip"
|
|
41
|
+
if not p.exists() or p.stat().st_size == 0:
|
|
42
|
+
print(f"downloading {q}", file=sys.stderr)
|
|
43
|
+
p.write_bytes(_get(url))
|
|
44
|
+
paths.append(p)
|
|
45
|
+
return paths
|
ncua_data/ingest.py
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Load raw NCUA quarterly ZIPs into long-format Parquet files.
|
|
2
|
+
|
|
3
|
+
Every account value becomes one row (quarter, cu_number, acct, value), keeping
|
|
4
|
+
non-zero values only. That makes the multi-year account churn easy to handle
|
|
5
|
+
and keeps the intermediate store small.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import csv
|
|
10
|
+
import re
|
|
11
|
+
import zipfile
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
import duckdb
|
|
15
|
+
|
|
16
|
+
ACCT_TABLE = re.compile(r"^fs220[a-z]?\.txt$", re.I)
|
|
17
|
+
ID_COLS = ("CU_NUMBER", "CYCLE_DATE", "JOIN_NUMBER", "UPDATE_DATE")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def extract(zip_path: Path, dest: Path) -> Path:
|
|
21
|
+
out = dest / zip_path.stem.replace("call-report-data-", "")
|
|
22
|
+
if not out.exists():
|
|
23
|
+
out.mkdir(parents=True)
|
|
24
|
+
with zipfile.ZipFile(zip_path) as z:
|
|
25
|
+
z.extractall(out)
|
|
26
|
+
# NCUA files are Windows-1252 (stray 0x92 apostrophes appear in free text
|
|
27
|
+
# fields). Re-encode to UTF-8 so every reader agrees.
|
|
28
|
+
for f in out.iterdir():
|
|
29
|
+
if f.suffix.lower() == ".txt":
|
|
30
|
+
f.write_bytes(f.read_bytes().decode("cp1252", errors="replace").encode("utf-8"))
|
|
31
|
+
return out
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _find(folder: Path, name: str) -> Path | None:
|
|
35
|
+
for p in folder.iterdir():
|
|
36
|
+
if p.name.lower() == name.lower():
|
|
37
|
+
return p
|
|
38
|
+
return None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _unique_header(path: Path) -> list[str]:
|
|
42
|
+
"""Header row with case-insensitive duplicates suffixed (some files repeat a code)."""
|
|
43
|
+
with open(path, encoding="utf-8", newline="") as fh:
|
|
44
|
+
header = next(csv.reader(fh))
|
|
45
|
+
seen: dict[str, int] = {}
|
|
46
|
+
out = []
|
|
47
|
+
for h in header:
|
|
48
|
+
k = h.upper()
|
|
49
|
+
seen[k] = seen.get(k, 0) + 1
|
|
50
|
+
out.append(h if seen[k] == 1 else f"{h}__dup{seen[k]}")
|
|
51
|
+
return out
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def ingest_quarter(zip_path: Path, work: Path, out_dir: Path) -> dict:
|
|
55
|
+
"""Write long.parquet, foicu.parquet and dict.parquet for one quarter."""
|
|
56
|
+
quarter = zip_path.stem.replace("call-report-data-", "")
|
|
57
|
+
folder = extract(zip_path, work)
|
|
58
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
59
|
+
con = duckdb.connect()
|
|
60
|
+
parts = []
|
|
61
|
+
for p in sorted(folder.iterdir()):
|
|
62
|
+
if ACCT_TABLE.match(p.name):
|
|
63
|
+
names = _unique_header(p)
|
|
64
|
+
parts.append(
|
|
65
|
+
f"""SELECT CU_NUMBER::INTEGER AS cu_number, upper(acct) AS acct,
|
|
66
|
+
try_cast(val AS DOUBLE) AS value, '{p.stem.upper()}' AS source_table
|
|
67
|
+
FROM (UNPIVOT (SELECT * FROM read_csv('{p}', all_varchar=true, header=true,
|
|
68
|
+
names={names!r}, ignore_errors=true))
|
|
69
|
+
ON COLUMNS('(?i)^acct_')
|
|
70
|
+
INTO NAME acct VALUE val)
|
|
71
|
+
WHERE try_cast(val AS DOUBLE) <> 0"""
|
|
72
|
+
)
|
|
73
|
+
con.execute(
|
|
74
|
+
f"COPY (SELECT '{quarter}' AS quarter, * FROM ({' UNION ALL '.join(parts)})) "
|
|
75
|
+
f"TO '{out_dir}/{quarter}_long.parquet' (FORMAT PARQUET, COMPRESSION ZSTD)"
|
|
76
|
+
)
|
|
77
|
+
foicu = _find(folder, "FOICU.txt")
|
|
78
|
+
con.execute(
|
|
79
|
+
f"COPY (SELECT '{quarter}' AS quarter, * FROM read_csv('{foicu}', all_varchar=true, "
|
|
80
|
+
f"ignore_errors=true)) TO '{out_dir}/{quarter}_foicu.parquet' (FORMAT PARQUET)"
|
|
81
|
+
)
|
|
82
|
+
ad = _find(folder, "AcctDesc.txt")
|
|
83
|
+
con.execute(
|
|
84
|
+
f"COPY (SELECT '{quarter}' AS quarter, * FROM read_csv('{ad}', all_varchar=true, "
|
|
85
|
+
f"ignore_errors=true, strict_mode=false)) TO '{out_dir}/{quarter}_dict.parquet' (FORMAT PARQUET)"
|
|
86
|
+
)
|
|
87
|
+
n = con.execute(f"SELECT count(*), count(DISTINCT cu_number), count(DISTINCT acct) FROM '{out_dir}/{quarter}_long.parquet'").fetchone()
|
|
88
|
+
return {"quarter": quarter, "rows": n[0], "cus": n[1], "accts": n[2]}
|