civic-data 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- civic_data/__init__.py +14 -0
- civic_data/adapters/__init__.py +7 -0
- civic_data/adapters/arcgis.py +128 -0
- civic_data/adapters/base.py +71 -0
- civic_data/adapters/socrata.py +28 -0
- civic_data/adapters/validation.py +135 -0
- civic_data/bulk.py +55 -0
- civic_data/cli.py +533 -0
- civic_data/consumer/__init__.py +6 -0
- civic_data/consumer/bootstrap.py +178 -0
- civic_data/consumer/bundle.py +213 -0
- civic_data/consumer/lake.py +87 -0
- civic_data/consumer/registry.py +193 -0
- civic_data/consumer/verbs.py +401 -0
- civic_data/datadict/__init__.py +12 -0
- civic_data/datadict/audit.py +105 -0
- civic_data/datadict/config.py +538 -0
- civic_data/datadict/generator.py +504 -0
- civic_data/datadict/inventory.py +46 -0
- civic_data/datadict/lineage.py +142 -0
- civic_data/datadict/lineage_config.py +756 -0
- civic_data/db.py +40 -0
- civic_data/handoff/__init__.py +7 -0
- civic_data/handoff/dump_local.py +126 -0
- civic_data/handoff/export_parquet.py +360 -0
- civic_data/ingest/__init__.py +13 -0
- civic_data/ingest/abc.py +225 -0
- civic_data/ingest/crime.py +340 -0
- civic_data/ingest/crime_geocode.py +178 -0
- civic_data/ingest/food_service.py +123 -0
- civic_data/ingest/foreclosures.py +153 -0
- civic_data/ingest/geography.py +232 -0
- civic_data/ingest/inspections.py +126 -0
- civic_data/ingest/landbank.py +146 -0
- civic_data/ingest/lien_orders.py +160 -0
- civic_data/ingest/lojic_reference.py +137 -0
- civic_data/ingest/permits.py +144 -0
- civic_data/ingest/population.py +125 -0
- civic_data/ingest/requests_311.py +292 -0
- civic_data/ingest/str_licenses.py +165 -0
- civic_data/ingest/zip_boundaries.py +101 -0
- civic_data/quality.py +212 -0
- civic_data-0.1.0.dist-info/METADATA +129 -0
- civic_data-0.1.0.dist-info/RECORD +49 -0
- civic_data-0.1.0.dist-info/WHEEL +5 -0
- civic_data-0.1.0.dist-info/entry_points.txt +2 -0
- civic_data-0.1.0.dist-info/licenses/LICENSE +21 -0
- civic_data-0.1.0.dist-info/licenses/LICENSE-data +37 -0
- civic_data-0.1.0.dist-info/top_level.txt +1 -0
civic_data/__init__.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""civic-data — the extraction / normalization / documentation spine.
|
|
2
|
+
|
|
3
|
+
One canonical home for the data-layer core: source adapters, medallion ingest
|
|
4
|
+
(bronze → silver), and a generated data dictionary with a transformation-lineage
|
|
5
|
+
layer. The product's value is not the data (the city gives that away raw) — it is
|
|
6
|
+
the *documented, reproducible transformation lineage*: what we did, why, and what
|
|
7
|
+
is lost. That lineage is generated from the live DB + code, never hand-maintained,
|
|
8
|
+
so it cannot drift from reality.
|
|
9
|
+
|
|
10
|
+
civic-graph (the private flagship) was the origin of this code and is now a
|
|
11
|
+
downstream consumer / worked example built on top.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""Source adapters — the extraction spine.
|
|
2
|
+
|
|
3
|
+
`arcgis` is the working portal adapter (Louisville / LOJIC, and any ArcGIS Hub
|
|
4
|
+
portal). `socrata` is a dormant placeholder that activates at city #2
|
|
5
|
+
(Cincinnati / Chicago). `validation` is the ZIP/geography gatekeeper shared by
|
|
6
|
+
every ingest path. `base` is the adapter contract.
|
|
7
|
+
"""
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""ArcGIS Feature Service adapter.
|
|
2
|
+
|
|
3
|
+
Louisville's open data portal runs on ArcGIS Hub. Most datasets are
|
|
4
|
+
exposed as ArcGIS Feature Services with a REST query endpoint.
|
|
5
|
+
|
|
6
|
+
Feature Service query pattern:
|
|
7
|
+
GET {service_url}/query?where=1=1&outFields=*&f=json&resultOffset=0&resultRecordCount=2000
|
|
8
|
+
|
|
9
|
+
Pagination: resultOffset increments by resultRecordCount until
|
|
10
|
+
exceededTransferLimit is false.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import time
|
|
14
|
+
import httpx
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def feature_count(service_url: str, timeout: float = 30.0) -> int | None:
|
|
19
|
+
"""Source row count via ArcGIS returnCountOnly (one tiny call). None on error.
|
|
20
|
+
|
|
21
|
+
Captured at ingest as source-pull metadata (the reproducibility substrate): the
|
|
22
|
+
source_rowcount at pull time, to compare against what we actually fetched/kept.
|
|
23
|
+
"""
|
|
24
|
+
try:
|
|
25
|
+
resp = httpx.get(f"{service_url}/query",
|
|
26
|
+
params={"where": "1=1", "returnCountOnly": "true", "f": "json"},
|
|
27
|
+
timeout=timeout)
|
|
28
|
+
resp.raise_for_status()
|
|
29
|
+
return resp.json().get("count")
|
|
30
|
+
except Exception:
|
|
31
|
+
return None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _query_page(client, url, params, max_retries: int = 4):
|
|
35
|
+
"""Fetch one page, retrying transient failures. Returns parsed JSON with a 'features' key.
|
|
36
|
+
|
|
37
|
+
Raises RuntimeError if the server never returns a valid page — so a transient error
|
|
38
|
+
(HTTP 5xx, an ArcGIS error payload, a truncated body) can NEVER be mistaken for
|
|
39
|
+
end-of-data and silently truncate a large pull. Only a *valid* response with an empty
|
|
40
|
+
'features' array or exceededTransferLimit=False ends pagination (handled by the caller).
|
|
41
|
+
"""
|
|
42
|
+
last_err = None
|
|
43
|
+
for attempt in range(max_retries):
|
|
44
|
+
try:
|
|
45
|
+
resp = client.get(url, params=params)
|
|
46
|
+
resp.raise_for_status()
|
|
47
|
+
data = resp.json()
|
|
48
|
+
if "error" in data:
|
|
49
|
+
last_err = f"ArcGIS error payload: {data['error']}"
|
|
50
|
+
elif "features" not in data:
|
|
51
|
+
last_err = f"response missing 'features' (keys={list(data)[:6]})"
|
|
52
|
+
else:
|
|
53
|
+
return data
|
|
54
|
+
except (httpx.HTTPError, ValueError) as e: # ValueError also covers JSON decode errors
|
|
55
|
+
last_err = repr(e)
|
|
56
|
+
time.sleep(1.0 * (attempt + 1)) # linear backoff between retries
|
|
57
|
+
raise RuntimeError(
|
|
58
|
+
f"ArcGIS page failed after {max_retries} attempts "
|
|
59
|
+
f"(resultOffset={params.get('resultOffset')}): {last_err}")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def fetch_feature_service(
|
|
63
|
+
service_url: str,
|
|
64
|
+
where: str = "1=1",
|
|
65
|
+
out_fields: str = "*",
|
|
66
|
+
page_size: int = 2000,
|
|
67
|
+
max_records: int | None = None,
|
|
68
|
+
delay: float = 0.5,
|
|
69
|
+
) -> list[dict[str, Any]]:
|
|
70
|
+
"""Fetch all records from an ArcGIS Feature Service.
|
|
71
|
+
|
|
72
|
+
Args:
|
|
73
|
+
service_url: The Feature Service layer URL
|
|
74
|
+
(e.g., https://services1.arcgis.com/.../FeatureServer/0)
|
|
75
|
+
where: SQL WHERE clause for filtering
|
|
76
|
+
out_fields: Comma-separated field names or "*" for all
|
|
77
|
+
page_size: Records per request (max varies by service, 2000 is safe)
|
|
78
|
+
max_records: Optional cap on total records fetched
|
|
79
|
+
delay: Seconds between paginated requests (polite scraping)
|
|
80
|
+
|
|
81
|
+
Returns:
|
|
82
|
+
List of feature attribute dicts (geometry stripped to lat/lng if present)
|
|
83
|
+
|
|
84
|
+
Pagination terminates only on a *valid* empty page or exceededTransferLimit=False;
|
|
85
|
+
transient per-page failures are retried (see _query_page), never treated as done.
|
|
86
|
+
"""
|
|
87
|
+
all_records = []
|
|
88
|
+
offset = 0
|
|
89
|
+
url = f"{service_url}/query"
|
|
90
|
+
|
|
91
|
+
with httpx.Client(timeout=60.0) as client:
|
|
92
|
+
while True:
|
|
93
|
+
params = {
|
|
94
|
+
"where": where,
|
|
95
|
+
"outFields": out_fields,
|
|
96
|
+
"f": "json",
|
|
97
|
+
"resultOffset": offset,
|
|
98
|
+
"resultRecordCount": page_size,
|
|
99
|
+
"outSR": 4326, # WGS84
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
data = _query_page(client, url, params)
|
|
103
|
+
features = data.get("features", [])
|
|
104
|
+
if not features:
|
|
105
|
+
break
|
|
106
|
+
|
|
107
|
+
for f in features:
|
|
108
|
+
record = f.get("attributes", {})
|
|
109
|
+
# Extract geometry as flat lat/lng if present
|
|
110
|
+
geom = f.get("geometry")
|
|
111
|
+
if geom:
|
|
112
|
+
record["_longitude"] = geom.get("x")
|
|
113
|
+
record["_latitude"] = geom.get("y")
|
|
114
|
+
all_records.append(record)
|
|
115
|
+
|
|
116
|
+
if max_records and len(all_records) >= max_records:
|
|
117
|
+
all_records = all_records[:max_records]
|
|
118
|
+
break
|
|
119
|
+
|
|
120
|
+
if not data.get("exceededTransferLimit", False):
|
|
121
|
+
break
|
|
122
|
+
|
|
123
|
+
# Increment by actual records returned, not page_size.
|
|
124
|
+
# Some services cap below our page_size (e.g., 1000 max).
|
|
125
|
+
offset += len(features)
|
|
126
|
+
time.sleep(delay)
|
|
127
|
+
|
|
128
|
+
return all_records
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""Base adapter interface for civic data sources."""
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class IngestResult:
|
|
10
|
+
"""Result of a data ingestion run."""
|
|
11
|
+
adapter: str
|
|
12
|
+
source_dataset: str
|
|
13
|
+
record_count: int
|
|
14
|
+
status: str # "ok" or "error"
|
|
15
|
+
error_message: str | None = None
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class CivicDataAdapter(ABC):
|
|
19
|
+
"""Abstract base class for all civic data adapters.
|
|
20
|
+
|
|
21
|
+
Each adapter handles one data domain (crime, 311, property, etc.)
|
|
22
|
+
and knows how to:
|
|
23
|
+
1. Fetch raw data from the source API
|
|
24
|
+
2. Store raw records in bronze layer (JSONB, append-only)
|
|
25
|
+
3. Transform raw records into silver layer (typed, geocoded)
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
@abstractmethod
|
|
30
|
+
def adapter_name(self) -> str:
|
|
31
|
+
"""Unique identifier for this adapter (e.g., 'crime', 'service_requests')."""
|
|
32
|
+
...
|
|
33
|
+
|
|
34
|
+
@abstractmethod
|
|
35
|
+
def fetch_raw(self, dataset_id: str, **kwargs) -> list[dict[str, Any]]:
|
|
36
|
+
"""Fetch raw records from the source API.
|
|
37
|
+
|
|
38
|
+
Returns a list of raw JSON-serializable dicts.
|
|
39
|
+
"""
|
|
40
|
+
...
|
|
41
|
+
|
|
42
|
+
@abstractmethod
|
|
43
|
+
def load_bronze(self, records: list[dict[str, Any]], dataset_id: str) -> int:
|
|
44
|
+
"""Store raw records in the bronze layer. Returns count of rows inserted."""
|
|
45
|
+
...
|
|
46
|
+
|
|
47
|
+
@abstractmethod
|
|
48
|
+
def promote_to_silver(self, dataset_id: str) -> int:
|
|
49
|
+
"""Transform bronze records into silver layer. Returns count of rows promoted."""
|
|
50
|
+
...
|
|
51
|
+
|
|
52
|
+
def ingest(self, dataset_id: str, **kwargs) -> IngestResult:
|
|
53
|
+
"""Full ingestion pipeline: fetch → bronze → silver."""
|
|
54
|
+
try:
|
|
55
|
+
raw = self.fetch_raw(dataset_id, **kwargs)
|
|
56
|
+
bronze_count = self.load_bronze(raw, dataset_id)
|
|
57
|
+
silver_count = self.promote_to_silver(dataset_id)
|
|
58
|
+
return IngestResult(
|
|
59
|
+
adapter=self.adapter_name,
|
|
60
|
+
source_dataset=dataset_id,
|
|
61
|
+
record_count=silver_count,
|
|
62
|
+
status="ok",
|
|
63
|
+
)
|
|
64
|
+
except Exception as e:
|
|
65
|
+
return IngestResult(
|
|
66
|
+
adapter=self.adapter_name,
|
|
67
|
+
source_dataset=dataset_id,
|
|
68
|
+
record_count=0,
|
|
69
|
+
status="error",
|
|
70
|
+
error_message=str(e),
|
|
71
|
+
)
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Socrata Open Data adapter — DORMANT (activates at city #2).
|
|
2
|
+
|
|
3
|
+
STATUS: placeholder, not implemented. v1 is Louisville-only (crime + 311) on
|
|
4
|
+
ArcGIS Hub; see `adapters/arcgis.py` for the working pattern. Socrata is the
|
|
5
|
+
multi-city unlock — Cincinnati, Chicago, and ~100 other US cities publish on the
|
|
6
|
+
Socrata platform (SODA API). When city #2 lands, implement `fetch_dataset`
|
|
7
|
+
against the SODA endpoint, mirroring the shape `arcgis.fetch_feature_service`
|
|
8
|
+
returns (a list of flat attribute dicts) so the downstream ingest/validation
|
|
9
|
+
spine is unchanged.
|
|
10
|
+
|
|
11
|
+
Design note (from the kickoff): this file was listed in the seed manifest as
|
|
12
|
+
inherited from civic-graph, but civic-graph never actually authored a Socrata
|
|
13
|
+
adapter — only the ArcGIS one exists there. So this is a fresh, honest stub, not
|
|
14
|
+
a lift. Multi-city is the thesis, not the v1 scope; do not build it out until a
|
|
15
|
+
real second city is in scope (guardrail: no future-proofing).
|
|
16
|
+
|
|
17
|
+
SODA query pattern (for reference when implemented):
|
|
18
|
+
GET https://{domain}/resource/{dataset_id}.json?$limit=50000&$offset=0
|
|
19
|
+
Pagination via $limit/$offset; optional $$app_token header to raise rate limits.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def fetch_dataset(*args, **kwargs):
|
|
24
|
+
raise NotImplementedError(
|
|
25
|
+
"The Socrata adapter is dormant. v1 is Louisville-only (ArcGIS). "
|
|
26
|
+
"Implement this against the SODA API when city #2 (Cincinnati/Chicago) is in scope, "
|
|
27
|
+
"returning the same flat-dict shape as adapters.arcgis.fetch_feature_service."
|
|
28
|
+
)
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""ZIP code validation for Louisville data quality.
|
|
2
|
+
|
|
3
|
+
Validates ZIP codes against silver.zip_codes (41 Jefferson County ZIPs).
|
|
4
|
+
Records with non-Louisville ZIPs are quarantined, not dropped.
|
|
5
|
+
|
|
6
|
+
Return value contract for validate_zip():
|
|
7
|
+
- valid ZIP string → insert with this normalized 5-digit ZIP
|
|
8
|
+
- None → unknown/missing ZIP, insert with SQL NULL
|
|
9
|
+
- QUARANTINED → record was quarantined, caller should skip
|
|
10
|
+
|
|
11
|
+
The canonical ZIP set (silver.zip_codes) is this project's master/reference data
|
|
12
|
+
(MDM). Every event table validates against it; it is the one canonical geography.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import re
|
|
17
|
+
|
|
18
|
+
# Louisville / Jefferson County has 41 unique ZIP code boundaries
|
|
19
|
+
# sourced from Louisville's ArcGIS portal (government data, not derived).
|
|
20
|
+
# ArcGIS feature count reports 42 but one is a duplicate geometry —
|
|
21
|
+
# 41 distinct ZIPCODEs confirmed against source.
|
|
22
|
+
EXPECTED_ZIP_COUNT = 41
|
|
23
|
+
|
|
24
|
+
# Sentinel returned when a record has been quarantined (caller should skip)
|
|
25
|
+
QUARANTINED = object()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def load_valid_zips(conn) -> frozenset:
|
|
29
|
+
"""Load and verify the Louisville ZIP whitelist from silver.zip_codes.
|
|
30
|
+
|
|
31
|
+
Returns an immutable set of valid ZIP strings.
|
|
32
|
+
Raises RuntimeError if the whitelist is missing or incomplete.
|
|
33
|
+
"""
|
|
34
|
+
with conn.cursor() as cur:
|
|
35
|
+
cur.execute("SELECT zip FROM silver.zip_codes ORDER BY zip")
|
|
36
|
+
zips = frozenset(row[0] for row in cur.fetchall())
|
|
37
|
+
if len(zips) < EXPECTED_ZIP_COUNT:
|
|
38
|
+
raise RuntimeError(
|
|
39
|
+
f"ZIP whitelist integrity check failed: expected {EXPECTED_ZIP_COUNT}, "
|
|
40
|
+
f"got {len(zips)}. Run `civic-data ingest-zips` first."
|
|
41
|
+
)
|
|
42
|
+
return zips
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def normalize_zip(zip_code):
|
|
46
|
+
"""Normalize a ZIP code to 5 digits.
|
|
47
|
+
|
|
48
|
+
Handles ZIP+4 variants: "40211-0000", "40211 0000", "402110000".
|
|
49
|
+
Returns None if input is empty/missing (unknown ZIP).
|
|
50
|
+
"""
|
|
51
|
+
if not zip_code:
|
|
52
|
+
return None
|
|
53
|
+
raw = str(zip_code).strip()
|
|
54
|
+
if not raw:
|
|
55
|
+
return None
|
|
56
|
+
|
|
57
|
+
# Extract leading 5 digits from common formats:
|
|
58
|
+
# "40211" -> "40211"
|
|
59
|
+
# "40211-0000" -> "40211"
|
|
60
|
+
# "40211 0000" -> "40211"
|
|
61
|
+
# "402110000" -> "40211"
|
|
62
|
+
# "40211-1769" -> "40211"
|
|
63
|
+
m = re.match(r'^(\d{5})[\s\-]?\d*$', raw)
|
|
64
|
+
if m:
|
|
65
|
+
return m.group(1)
|
|
66
|
+
|
|
67
|
+
# Didn't match — return as-is for quarantine (e.g., "UNKNOWN", "4020")
|
|
68
|
+
return raw
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def validate_zip(cur, zip_code, raw_record, source_table, valid_zips):
|
|
72
|
+
"""Validate a ZIP code against the Louisville whitelist.
|
|
73
|
+
|
|
74
|
+
Returns:
|
|
75
|
+
- normalized ZIP string if valid (insert with this ZIP)
|
|
76
|
+
- None if ZIP is missing/empty (insert with NULL — unknown, not invalid)
|
|
77
|
+
- QUARANTINED sentinel if the record was quarantined (caller should skip)
|
|
78
|
+
"""
|
|
79
|
+
if not zip_code or not str(zip_code).strip():
|
|
80
|
+
return None # unknown ZIP — not a validation failure
|
|
81
|
+
|
|
82
|
+
normalized = normalize_zip(zip_code)
|
|
83
|
+
|
|
84
|
+
if normalized in valid_zips:
|
|
85
|
+
return normalized
|
|
86
|
+
|
|
87
|
+
# Quarantine: ZIP exists but is not in Louisville
|
|
88
|
+
_quarantine_record(cur, source_table, str(zip_code).strip(), raw_record,
|
|
89
|
+
"ZIP not in Louisville boundary set")
|
|
90
|
+
return QUARANTINED
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def strip_pii_keys(record, keys):
|
|
94
|
+
"""Return a shallow copy of a raw source record with the named keys removed.
|
|
95
|
+
|
|
96
|
+
Enforces the no-PII-in-DB policy at the ingest boundary: person-name source fields are
|
|
97
|
+
stripped BEFORE the record is written to bronze raw or stored in the quarantine JSONB, so
|
|
98
|
+
individual names never persist anywhere. Addresses are NOT stripped (kept by policy).
|
|
99
|
+
Removing an absent key is a no-op (safe across source schema variants).
|
|
100
|
+
"""
|
|
101
|
+
if not record:
|
|
102
|
+
return record
|
|
103
|
+
return {k: v for k, v in record.items() if k not in keys}
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def derive_zip_from_point(cur, longitude, latitude):
|
|
107
|
+
"""Derive a Louisville ZIP from a point by spatial join against silver.zip_codes.
|
|
108
|
+
|
|
109
|
+
For point datasets that carry NO ZIP field (e.g. STR permits) — the geometry is the
|
|
110
|
+
only locator. Returns the containing ZIP string, or None when the point is missing or
|
|
111
|
+
falls outside the 41-ZIP boundary set.
|
|
112
|
+
|
|
113
|
+
Pairs with validate_zip()'s contract: a derived ZIP is always in the valid set (it came
|
|
114
|
+
FROM the set), so callers insert it directly; None means "real record, no usable
|
|
115
|
+
location" → insert with NULL zip (kept, not quarantined).
|
|
116
|
+
"""
|
|
117
|
+
if longitude is None or latitude is None:
|
|
118
|
+
return None
|
|
119
|
+
cur.execute(
|
|
120
|
+
"""SELECT zip FROM silver.zip_codes
|
|
121
|
+
WHERE ST_Contains(geom, ST_SetSRID(ST_MakePoint(%s, %s), 4326))
|
|
122
|
+
LIMIT 1""",
|
|
123
|
+
(longitude, latitude),
|
|
124
|
+
)
|
|
125
|
+
row = cur.fetchone()
|
|
126
|
+
return row[0] if row else None
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _quarantine_record(cur, source_table, zip_code, raw_record, reason):
|
|
130
|
+
"""Insert a rejected record into the quarantine table."""
|
|
131
|
+
cur.execute(
|
|
132
|
+
"INSERT INTO silver.quarantine (source_table, zip_code, record_data, reason) "
|
|
133
|
+
"VALUES (%s, %s, %s::jsonb, %s)",
|
|
134
|
+
(source_table, str(zip_code).strip(), json.dumps(raw_record, default=str), reason),
|
|
135
|
+
)
|
civic_data/bulk.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Bulk-insert helper — batched multi-row INSERT via psycopg2 execute_values.
|
|
2
|
+
|
|
3
|
+
Introduced for the LOJIC reference tables (~450K address points): the row-by-row
|
|
4
|
+
promote path is fine for typed event promotion, but far too slow at reference-table
|
|
5
|
+
scale. Reusable for any large silver load. Kept out of db.py so that module stays a
|
|
6
|
+
pure connection factory.
|
|
7
|
+
|
|
8
|
+
`template` describes ONE row's placeholder layout, so a caller can build PostGIS
|
|
9
|
+
geometry inline, e.g. "(%s, %s, ST_SetSRID(ST_MakePoint(%s, %s), 4326), %s)" — the
|
|
10
|
+
per-row tuple then supplies args in the order the %s appear.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from psycopg2.extras import execute_values
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def log_source_pull(conn, adapter, source_dataset, source_url, source_rowcount, record_count):
|
|
17
|
+
"""Record one completed source pull in bronze.sync_log (Constitution art. 1: every dataset
|
|
18
|
+
carries per-pull provenance). Domain ingests write their own running→ok ledger rows inline;
|
|
19
|
+
MDM/reference loads are quick single-shot pulls, so they log one 'ok' row after success.
|
|
20
|
+
|
|
21
|
+
Convention: MDM tables with no per-row source_dataset column ledger under their SILVER TABLE
|
|
22
|
+
NAME (e.g. 'zip_codes', 'council_districts') — the exporter's provenance fallback looks pulls
|
|
23
|
+
up by that name, keeping the mapping introspectable instead of hand-maintained."""
|
|
24
|
+
with conn.cursor() as cur:
|
|
25
|
+
cur.execute(
|
|
26
|
+
"""INSERT INTO bronze.sync_log
|
|
27
|
+
(adapter, source_dataset, source_url, source_rowcount, status,
|
|
28
|
+
completed_at, record_count)
|
|
29
|
+
VALUES (%s, %s, %s, %s, 'ok', NOW(), %s)""",
|
|
30
|
+
(adapter, source_dataset, source_url, source_rowcount, record_count),
|
|
31
|
+
)
|
|
32
|
+
conn.commit()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def bulk_insert(conn, table, columns, rows, template=None, page_size=5000) -> int:
|
|
36
|
+
"""Insert `rows` into `table` (a fully-qualified name) over `columns`, committing once.
|
|
37
|
+
|
|
38
|
+
`table`/`columns` are trusted, code-supplied identifiers (never user input) — the
|
|
39
|
+
only interpolation here. Row values are always parameterized by execute_values.
|
|
40
|
+
Returns the number of rows inserted.
|
|
41
|
+
"""
|
|
42
|
+
rows = list(rows)
|
|
43
|
+
if not rows:
|
|
44
|
+
return 0
|
|
45
|
+
col_list = ", ".join(columns)
|
|
46
|
+
with conn.cursor() as cur:
|
|
47
|
+
execute_values(
|
|
48
|
+
cur,
|
|
49
|
+
f"INSERT INTO {table} ({col_list}) VALUES %s",
|
|
50
|
+
rows,
|
|
51
|
+
template=template,
|
|
52
|
+
page_size=page_size,
|
|
53
|
+
)
|
|
54
|
+
conn.commit()
|
|
55
|
+
return len(rows)
|