civic-data 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. civic_data/__init__.py +14 -0
  2. civic_data/adapters/__init__.py +7 -0
  3. civic_data/adapters/arcgis.py +128 -0
  4. civic_data/adapters/base.py +71 -0
  5. civic_data/adapters/socrata.py +28 -0
  6. civic_data/adapters/validation.py +135 -0
  7. civic_data/bulk.py +55 -0
  8. civic_data/cli.py +533 -0
  9. civic_data/consumer/__init__.py +6 -0
  10. civic_data/consumer/bootstrap.py +178 -0
  11. civic_data/consumer/bundle.py +213 -0
  12. civic_data/consumer/lake.py +87 -0
  13. civic_data/consumer/registry.py +193 -0
  14. civic_data/consumer/verbs.py +401 -0
  15. civic_data/datadict/__init__.py +12 -0
  16. civic_data/datadict/audit.py +105 -0
  17. civic_data/datadict/config.py +538 -0
  18. civic_data/datadict/generator.py +504 -0
  19. civic_data/datadict/inventory.py +46 -0
  20. civic_data/datadict/lineage.py +142 -0
  21. civic_data/datadict/lineage_config.py +756 -0
  22. civic_data/db.py +40 -0
  23. civic_data/handoff/__init__.py +7 -0
  24. civic_data/handoff/dump_local.py +126 -0
  25. civic_data/handoff/export_parquet.py +360 -0
  26. civic_data/ingest/__init__.py +13 -0
  27. civic_data/ingest/abc.py +225 -0
  28. civic_data/ingest/crime.py +340 -0
  29. civic_data/ingest/crime_geocode.py +178 -0
  30. civic_data/ingest/food_service.py +123 -0
  31. civic_data/ingest/foreclosures.py +153 -0
  32. civic_data/ingest/geography.py +232 -0
  33. civic_data/ingest/inspections.py +126 -0
  34. civic_data/ingest/landbank.py +146 -0
  35. civic_data/ingest/lien_orders.py +160 -0
  36. civic_data/ingest/lojic_reference.py +137 -0
  37. civic_data/ingest/permits.py +144 -0
  38. civic_data/ingest/population.py +125 -0
  39. civic_data/ingest/requests_311.py +292 -0
  40. civic_data/ingest/str_licenses.py +165 -0
  41. civic_data/ingest/zip_boundaries.py +101 -0
  42. civic_data/quality.py +212 -0
  43. civic_data-0.1.0.dist-info/METADATA +129 -0
  44. civic_data-0.1.0.dist-info/RECORD +49 -0
  45. civic_data-0.1.0.dist-info/WHEEL +5 -0
  46. civic_data-0.1.0.dist-info/entry_points.txt +2 -0
  47. civic_data-0.1.0.dist-info/licenses/LICENSE +21 -0
  48. civic_data-0.1.0.dist-info/licenses/LICENSE-data +37 -0
  49. civic_data-0.1.0.dist-info/top_level.txt +1 -0
civic_data/__init__.py ADDED
@@ -0,0 +1,14 @@
1
+ """civic-data — the extraction / normalization / documentation spine.
2
+
3
+ One canonical home for the data-layer core: source adapters, medallion ingest
4
+ (bronze → silver), and a generated data dictionary with a transformation-lineage
5
+ layer. The product's value is not the data (the city gives that away raw) — it is
6
+ the *documented, reproducible transformation lineage*: what we did, why, and what
7
+ is lost. That lineage is generated from the live DB + code, never hand-maintained,
8
+ so it cannot drift from reality.
9
+
10
+ civic-graph (the private flagship) was the origin of this code and is now a
11
+ downstream consumer / worked example built on top.
12
+ """
13
+
14
+ __version__ = "0.1.0"
@@ -0,0 +1,7 @@
1
+ """Source adapters — the extraction spine.
2
+
3
+ `arcgis` is the working portal adapter (Louisville / LOJIC, and any ArcGIS Hub
4
+ portal). `socrata` is a dormant placeholder that activates at city #2
5
+ (Cincinnati / Chicago). `validation` is the ZIP/geography gatekeeper shared by
6
+ every ingest path. `base` is the adapter contract.
7
+ """
@@ -0,0 +1,128 @@
1
+ """ArcGIS Feature Service adapter.
2
+
3
+ Louisville's open data portal runs on ArcGIS Hub. Most datasets are
4
+ exposed as ArcGIS Feature Services with a REST query endpoint.
5
+
6
+ Feature Service query pattern:
7
+ GET {service_url}/query?where=1=1&outFields=*&f=json&resultOffset=0&resultRecordCount=2000
8
+
9
+ Pagination: resultOffset increments by resultRecordCount until
10
+ exceededTransferLimit is false.
11
+ """
12
+
13
+ import time
14
+ import httpx
15
+ from typing import Any
16
+
17
+
18
+ def feature_count(service_url: str, timeout: float = 30.0) -> int | None:
19
+ """Source row count via ArcGIS returnCountOnly (one tiny call). None on error.
20
+
21
+ Captured at ingest as source-pull metadata (the reproducibility substrate): the
22
+ source_rowcount at pull time, to compare against what we actually fetched/kept.
23
+ """
24
+ try:
25
+ resp = httpx.get(f"{service_url}/query",
26
+ params={"where": "1=1", "returnCountOnly": "true", "f": "json"},
27
+ timeout=timeout)
28
+ resp.raise_for_status()
29
+ return resp.json().get("count")
30
+ except Exception:
31
+ return None
32
+
33
+
34
+ def _query_page(client, url, params, max_retries: int = 4):
35
+ """Fetch one page, retrying transient failures. Returns parsed JSON with a 'features' key.
36
+
37
+ Raises RuntimeError if the server never returns a valid page — so a transient error
38
+ (HTTP 5xx, an ArcGIS error payload, a truncated body) can NEVER be mistaken for
39
+ end-of-data and silently truncate a large pull. Only a *valid* response with an empty
40
+ 'features' array or exceededTransferLimit=False ends pagination (handled by the caller).
41
+ """
42
+ last_err = None
43
+ for attempt in range(max_retries):
44
+ try:
45
+ resp = client.get(url, params=params)
46
+ resp.raise_for_status()
47
+ data = resp.json()
48
+ if "error" in data:
49
+ last_err = f"ArcGIS error payload: {data['error']}"
50
+ elif "features" not in data:
51
+ last_err = f"response missing 'features' (keys={list(data)[:6]})"
52
+ else:
53
+ return data
54
+ except (httpx.HTTPError, ValueError) as e: # ValueError also covers JSON decode errors
55
+ last_err = repr(e)
56
+ time.sleep(1.0 * (attempt + 1)) # linear backoff between retries
57
+ raise RuntimeError(
58
+ f"ArcGIS page failed after {max_retries} attempts "
59
+ f"(resultOffset={params.get('resultOffset')}): {last_err}")
60
+
61
+
62
+ def fetch_feature_service(
63
+ service_url: str,
64
+ where: str = "1=1",
65
+ out_fields: str = "*",
66
+ page_size: int = 2000,
67
+ max_records: int | None = None,
68
+ delay: float = 0.5,
69
+ ) -> list[dict[str, Any]]:
70
+ """Fetch all records from an ArcGIS Feature Service.
71
+
72
+ Args:
73
+ service_url: The Feature Service layer URL
74
+ (e.g., https://services1.arcgis.com/.../FeatureServer/0)
75
+ where: SQL WHERE clause for filtering
76
+ out_fields: Comma-separated field names or "*" for all
77
+ page_size: Records per request (max varies by service, 2000 is safe)
78
+ max_records: Optional cap on total records fetched
79
+ delay: Seconds between paginated requests (polite scraping)
80
+
81
+ Returns:
82
+ List of feature attribute dicts (geometry stripped to lat/lng if present)
83
+
84
+ Pagination terminates only on a *valid* empty page or exceededTransferLimit=False;
85
+ transient per-page failures are retried (see _query_page), never treated as done.
86
+ """
87
+ all_records = []
88
+ offset = 0
89
+ url = f"{service_url}/query"
90
+
91
+ with httpx.Client(timeout=60.0) as client:
92
+ while True:
93
+ params = {
94
+ "where": where,
95
+ "outFields": out_fields,
96
+ "f": "json",
97
+ "resultOffset": offset,
98
+ "resultRecordCount": page_size,
99
+ "outSR": 4326, # WGS84
100
+ }
101
+
102
+ data = _query_page(client, url, params)
103
+ features = data.get("features", [])
104
+ if not features:
105
+ break
106
+
107
+ for f in features:
108
+ record = f.get("attributes", {})
109
+ # Extract geometry as flat lat/lng if present
110
+ geom = f.get("geometry")
111
+ if geom:
112
+ record["_longitude"] = geom.get("x")
113
+ record["_latitude"] = geom.get("y")
114
+ all_records.append(record)
115
+
116
+ if max_records and len(all_records) >= max_records:
117
+ all_records = all_records[:max_records]
118
+ break
119
+
120
+ if not data.get("exceededTransferLimit", False):
121
+ break
122
+
123
+ # Increment by actual records returned, not page_size.
124
+ # Some services cap below our page_size (e.g., 1000 max).
125
+ offset += len(features)
126
+ time.sleep(delay)
127
+
128
+ return all_records
@@ -0,0 +1,71 @@
1
+ """Base adapter interface for civic data sources."""
2
+
3
+ from abc import ABC, abstractmethod
4
+ from dataclasses import dataclass
5
+ from typing import Any
6
+
7
+
8
+ @dataclass
9
+ class IngestResult:
10
+ """Result of a data ingestion run."""
11
+ adapter: str
12
+ source_dataset: str
13
+ record_count: int
14
+ status: str # "ok" or "error"
15
+ error_message: str | None = None
16
+
17
+
18
+ class CivicDataAdapter(ABC):
19
+ """Abstract base class for all civic data adapters.
20
+
21
+ Each adapter handles one data domain (crime, 311, property, etc.)
22
+ and knows how to:
23
+ 1. Fetch raw data from the source API
24
+ 2. Store raw records in bronze layer (JSONB, append-only)
25
+ 3. Transform raw records into silver layer (typed, geocoded)
26
+ """
27
+
28
+ @property
29
+ @abstractmethod
30
+ def adapter_name(self) -> str:
31
+ """Unique identifier for this adapter (e.g., 'crime', 'service_requests')."""
32
+ ...
33
+
34
+ @abstractmethod
35
+ def fetch_raw(self, dataset_id: str, **kwargs) -> list[dict[str, Any]]:
36
+ """Fetch raw records from the source API.
37
+
38
+ Returns a list of raw JSON-serializable dicts.
39
+ """
40
+ ...
41
+
42
+ @abstractmethod
43
+ def load_bronze(self, records: list[dict[str, Any]], dataset_id: str) -> int:
44
+ """Store raw records in the bronze layer. Returns count of rows inserted."""
45
+ ...
46
+
47
+ @abstractmethod
48
+ def promote_to_silver(self, dataset_id: str) -> int:
49
+ """Transform bronze records into silver layer. Returns count of rows promoted."""
50
+ ...
51
+
52
+ def ingest(self, dataset_id: str, **kwargs) -> IngestResult:
53
+ """Full ingestion pipeline: fetch → bronze → silver."""
54
+ try:
55
+ raw = self.fetch_raw(dataset_id, **kwargs)
56
+ bronze_count = self.load_bronze(raw, dataset_id)
57
+ silver_count = self.promote_to_silver(dataset_id)
58
+ return IngestResult(
59
+ adapter=self.adapter_name,
60
+ source_dataset=dataset_id,
61
+ record_count=silver_count,
62
+ status="ok",
63
+ )
64
+ except Exception as e:
65
+ return IngestResult(
66
+ adapter=self.adapter_name,
67
+ source_dataset=dataset_id,
68
+ record_count=0,
69
+ status="error",
70
+ error_message=str(e),
71
+ )
@@ -0,0 +1,28 @@
1
+ """Socrata Open Data adapter — DORMANT (activates at city #2).
2
+
3
+ STATUS: placeholder, not implemented. v1 is Louisville-only (crime + 311) on
4
+ ArcGIS Hub; see `adapters/arcgis.py` for the working pattern. Socrata is the
5
+ multi-city unlock — Cincinnati, Chicago, and ~100 other US cities publish on the
6
+ Socrata platform (SODA API). When city #2 lands, implement `fetch_dataset`
7
+ against the SODA endpoint, mirroring the shape `arcgis.fetch_feature_service`
8
+ returns (a list of flat attribute dicts) so the downstream ingest/validation
9
+ spine is unchanged.
10
+
11
+ Design note (from the kickoff): this file was listed in the seed manifest as
12
+ inherited from civic-graph, but civic-graph never actually authored a Socrata
13
+ adapter — only the ArcGIS one exists there. So this is a fresh, honest stub, not
14
+ a lift. Multi-city is the thesis, not the v1 scope; do not build it out until a
15
+ real second city is in scope (guardrail: no future-proofing).
16
+
17
+ SODA query pattern (for reference when implemented):
18
+ GET https://{domain}/resource/{dataset_id}.json?$limit=50000&$offset=0
19
+ Pagination via $limit/$offset; optional $$app_token header to raise rate limits.
20
+ """
21
+
22
+
23
+ def fetch_dataset(*args, **kwargs):
24
+ raise NotImplementedError(
25
+ "The Socrata adapter is dormant. v1 is Louisville-only (ArcGIS). "
26
+ "Implement this against the SODA API when city #2 (Cincinnati/Chicago) is in scope, "
27
+ "returning the same flat-dict shape as adapters.arcgis.fetch_feature_service."
28
+ )
@@ -0,0 +1,135 @@
1
+ """ZIP code validation for Louisville data quality.
2
+
3
+ Validates ZIP codes against silver.zip_codes (41 Jefferson County ZIPs).
4
+ Records with non-Louisville ZIPs are quarantined, not dropped.
5
+
6
+ Return value contract for validate_zip():
7
+ - valid ZIP string → insert with this normalized 5-digit ZIP
8
+ - None → unknown/missing ZIP, insert with SQL NULL
9
+ - QUARANTINED → record was quarantined, caller should skip
10
+
11
+ The canonical ZIP set (silver.zip_codes) is this project's master/reference data
12
+ (MDM). Every event table validates against it; it is the one canonical geography.
13
+ """
14
+
15
+ import json
16
+ import re
17
+
18
+ # Louisville / Jefferson County has 41 unique ZIP code boundaries
19
+ # sourced from Louisville's ArcGIS portal (government data, not derived).
20
+ # ArcGIS feature count reports 42 but one is a duplicate geometry —
21
+ # 41 distinct ZIPCODEs confirmed against source.
22
+ EXPECTED_ZIP_COUNT = 41
23
+
24
+ # Sentinel returned when a record has been quarantined (caller should skip)
25
+ QUARANTINED = object()
26
+
27
+
28
+ def load_valid_zips(conn) -> frozenset:
29
+ """Load and verify the Louisville ZIP whitelist from silver.zip_codes.
30
+
31
+ Returns an immutable set of valid ZIP strings.
32
+ Raises RuntimeError if the whitelist is missing or incomplete.
33
+ """
34
+ with conn.cursor() as cur:
35
+ cur.execute("SELECT zip FROM silver.zip_codes ORDER BY zip")
36
+ zips = frozenset(row[0] for row in cur.fetchall())
37
+ if len(zips) < EXPECTED_ZIP_COUNT:
38
+ raise RuntimeError(
39
+ f"ZIP whitelist integrity check failed: expected {EXPECTED_ZIP_COUNT}, "
40
+ f"got {len(zips)}. Run `civic-data ingest-zips` first."
41
+ )
42
+ return zips
43
+
44
+
45
+ def normalize_zip(zip_code):
46
+ """Normalize a ZIP code to 5 digits.
47
+
48
+ Handles ZIP+4 variants: "40211-0000", "40211 0000", "402110000".
49
+ Returns None if input is empty/missing (unknown ZIP).
50
+ """
51
+ if not zip_code:
52
+ return None
53
+ raw = str(zip_code).strip()
54
+ if not raw:
55
+ return None
56
+
57
+ # Extract leading 5 digits from common formats:
58
+ # "40211" -> "40211"
59
+ # "40211-0000" -> "40211"
60
+ # "40211 0000" -> "40211"
61
+ # "402110000" -> "40211"
62
+ # "40211-1769" -> "40211"
63
+ m = re.match(r'^(\d{5})[\s\-]?\d*$', raw)
64
+ if m:
65
+ return m.group(1)
66
+
67
+ # Didn't match — return as-is for quarantine (e.g., "UNKNOWN", "4020")
68
+ return raw
69
+
70
+
71
+ def validate_zip(cur, zip_code, raw_record, source_table, valid_zips):
72
+ """Validate a ZIP code against the Louisville whitelist.
73
+
74
+ Returns:
75
+ - normalized ZIP string if valid (insert with this ZIP)
76
+ - None if ZIP is missing/empty (insert with NULL — unknown, not invalid)
77
+ - QUARANTINED sentinel if the record was quarantined (caller should skip)
78
+ """
79
+ if not zip_code or not str(zip_code).strip():
80
+ return None # unknown ZIP — not a validation failure
81
+
82
+ normalized = normalize_zip(zip_code)
83
+
84
+ if normalized in valid_zips:
85
+ return normalized
86
+
87
+ # Quarantine: ZIP exists but is not in Louisville
88
+ _quarantine_record(cur, source_table, str(zip_code).strip(), raw_record,
89
+ "ZIP not in Louisville boundary set")
90
+ return QUARANTINED
91
+
92
+
93
+ def strip_pii_keys(record, keys):
94
+ """Return a shallow copy of a raw source record with the named keys removed.
95
+
96
+ Enforces the no-PII-in-DB policy at the ingest boundary: person-name source fields are
97
+ stripped BEFORE the record is written to bronze raw or stored in the quarantine JSONB, so
98
+ individual names never persist anywhere. Addresses are NOT stripped (kept by policy).
99
+ Removing an absent key is a no-op (safe across source schema variants).
100
+ """
101
+ if not record:
102
+ return record
103
+ return {k: v for k, v in record.items() if k not in keys}
104
+
105
+
106
+ def derive_zip_from_point(cur, longitude, latitude):
107
+ """Derive a Louisville ZIP from a point by spatial join against silver.zip_codes.
108
+
109
+ For point datasets that carry NO ZIP field (e.g. STR permits) — the geometry is the
110
+ only locator. Returns the containing ZIP string, or None when the point is missing or
111
+ falls outside the 41-ZIP boundary set.
112
+
113
+ Pairs with validate_zip()'s contract: a derived ZIP is always in the valid set (it came
114
+ FROM the set), so callers insert it directly; None means "real record, no usable
115
+ location" → insert with NULL zip (kept, not quarantined).
116
+ """
117
+ if longitude is None or latitude is None:
118
+ return None
119
+ cur.execute(
120
+ """SELECT zip FROM silver.zip_codes
121
+ WHERE ST_Contains(geom, ST_SetSRID(ST_MakePoint(%s, %s), 4326))
122
+ LIMIT 1""",
123
+ (longitude, latitude),
124
+ )
125
+ row = cur.fetchone()
126
+ return row[0] if row else None
127
+
128
+
129
+ def _quarantine_record(cur, source_table, zip_code, raw_record, reason):
130
+ """Insert a rejected record into the quarantine table."""
131
+ cur.execute(
132
+ "INSERT INTO silver.quarantine (source_table, zip_code, record_data, reason) "
133
+ "VALUES (%s, %s, %s::jsonb, %s)",
134
+ (source_table, str(zip_code).strip(), json.dumps(raw_record, default=str), reason),
135
+ )
civic_data/bulk.py ADDED
@@ -0,0 +1,55 @@
1
+ """Bulk-insert helper — batched multi-row INSERT via psycopg2 execute_values.
2
+
3
+ Introduced for the LOJIC reference tables (~450K address points): the row-by-row
4
+ promote path is fine for typed event promotion, but far too slow at reference-table
5
+ scale. Reusable for any large silver load. Kept out of db.py so that module stays a
6
+ pure connection factory.
7
+
8
+ `template` describes ONE row's placeholder layout, so a caller can build PostGIS
9
+ geometry inline, e.g. "(%s, %s, ST_SetSRID(ST_MakePoint(%s, %s), 4326), %s)" — the
10
+ per-row tuple then supplies args in the order the %s appear.
11
+ """
12
+
13
+ from psycopg2.extras import execute_values
14
+
15
+
16
+ def log_source_pull(conn, adapter, source_dataset, source_url, source_rowcount, record_count):
17
+ """Record one completed source pull in bronze.sync_log (Constitution art. 1: every dataset
18
+ carries per-pull provenance). Domain ingests write their own running→ok ledger rows inline;
19
+ MDM/reference loads are quick single-shot pulls, so they log one 'ok' row after success.
20
+
21
+ Convention: MDM tables with no per-row source_dataset column ledger under their SILVER TABLE
22
+ NAME (e.g. 'zip_codes', 'council_districts') — the exporter's provenance fallback looks pulls
23
+ up by that name, keeping the mapping introspectable instead of hand-maintained."""
24
+ with conn.cursor() as cur:
25
+ cur.execute(
26
+ """INSERT INTO bronze.sync_log
27
+ (adapter, source_dataset, source_url, source_rowcount, status,
28
+ completed_at, record_count)
29
+ VALUES (%s, %s, %s, %s, 'ok', NOW(), %s)""",
30
+ (adapter, source_dataset, source_url, source_rowcount, record_count),
31
+ )
32
+ conn.commit()
33
+
34
+
35
+ def bulk_insert(conn, table, columns, rows, template=None, page_size=5000) -> int:
36
+ """Insert `rows` into `table` (a fully-qualified name) over `columns`, committing once.
37
+
38
+ `table`/`columns` are trusted, code-supplied identifiers (never user input) — the
39
+ only interpolation here. Row values are always parameterized by execute_values.
40
+ Returns the number of rows inserted.
41
+ """
42
+ rows = list(rows)
43
+ if not rows:
44
+ return 0
45
+ col_list = ", ".join(columns)
46
+ with conn.cursor() as cur:
47
+ execute_values(
48
+ cur,
49
+ f"INSERT INTO {table} ({col_list}) VALUES %s",
50
+ rows,
51
+ template=template,
52
+ page_size=page_size,
53
+ )
54
+ conn.commit()
55
+ return len(rows)