geo-data-audit 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- geo_data_audit-0.1.0/PKG-INFO +105 -0
- geo_data_audit-0.1.0/README.md +83 -0
- geo_data_audit-0.1.0/geo_audit/__init__.py +3 -0
- geo_data_audit-0.1.0/geo_audit/cli.py +205 -0
- geo_data_audit-0.1.0/geo_audit/coordinates.py +115 -0
- geo_data_audit-0.1.0/geo_audit/discovery.py +40 -0
- geo_data_audit-0.1.0/geo_audit/findings.py +367 -0
- geo_data_audit-0.1.0/geo_audit/loaders.py +123 -0
- geo_data_audit-0.1.0/geo_audit/models.py +305 -0
- geo_data_audit-0.1.0/geo_audit/quality.py +17 -0
- geo_data_audit-0.1.0/geo_audit/reporting.py +507 -0
- geo_data_audit-0.1.0/geo_audit/schema.py +64 -0
- geo_data_audit-0.1.0/geo_audit/spatial.py +51 -0
- geo_data_audit-0.1.0/geo_data_audit.egg-info/PKG-INFO +105 -0
- geo_data_audit-0.1.0/geo_data_audit.egg-info/SOURCES.txt +20 -0
- geo_data_audit-0.1.0/geo_data_audit.egg-info/dependency_links.txt +1 -0
- geo_data_audit-0.1.0/geo_data_audit.egg-info/entry_points.txt +2 -0
- geo_data_audit-0.1.0/geo_data_audit.egg-info/requires.txt +13 -0
- geo_data_audit-0.1.0/geo_data_audit.egg-info/top_level.txt +4 -0
- geo_data_audit-0.1.0/pyproject.toml +60 -0
- geo_data_audit-0.1.0/setup.cfg +4 -0
- geo_data_audit-0.1.0/tests/test_geo_audit.py +671 -0
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: geo-data-audit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Audit heterogeneous geospatial datasets for common data-quality and spatial-consistency problems before they enter downstream geospatial ML pipelines.
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/souro26/geo-audit
|
|
7
|
+
Project-URL: Repository, https://github.com/souro26/geo-audit
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
Requires-Dist: pandas>=2.0
|
|
11
|
+
Requires-Dist: numpy>=1.24
|
|
12
|
+
Requires-Dist: geopandas>=0.14
|
|
13
|
+
Requires-Dist: shapely>=2.0
|
|
14
|
+
Requires-Dist: pyproj>=3.5
|
|
15
|
+
Requires-Dist: typer>=0.9
|
|
16
|
+
Requires-Dist: rich>=13.0
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: pytest>=7.4; extra == "dev"
|
|
19
|
+
Requires-Dist: pytest-cov>=4.1; extra == "dev"
|
|
20
|
+
Requires-Dist: ruff>=0.1.0; extra == "dev"
|
|
21
|
+
Requires-Dist: mypy>=1.5; extra == "dev"
|
|
22
|
+
|
|
23
|
+
# geo-audit
|
|
24
|
+
|
|
25
|
+
`geo-audit` is a command-line tool for auditing raw geospatial datasets before ingestion into spatial analysis, GIS workflows, or machine learning pipelines. It inspects CSV and GeoJSON files in a directory to identify coordinate anomalies, missing spatial metadata, attribute missingness, exact duplicate locations, and cross-dataset bounding box overlaps without modifying source data.
|
|
26
|
+
|
|
27
|
+
## Purpose
|
|
28
|
+
|
|
29
|
+
Geospatial data collected from field surveys, legacy archives, and third-party GIS exports frequently contains quality defects—such as out-of-range latitude/longitude coordinates, unprojected planar coordinates lacking CRS definitions, silent duplicate coordinates, and missing values across attribute columns. `geo-audit` runs static health checks across all datasets in a target directory to surface critical defects and risk levels in a single pass.
|
|
30
|
+
|
|
31
|
+
## Installation
|
|
32
|
+
|
|
33
|
+
Requires Python 3.10 or higher.
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
git clone https://github.com/souro26/geo-audit
|
|
37
|
+
cd geo-audit
|
|
38
|
+
pip install -e .
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Quick Start
|
|
42
|
+
|
|
43
|
+
Scan a directory of geospatial files for a human-readable terminal report:
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
geo-audit data/
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Generate machine-readable JSON output for automated pipelines:
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
geo-audit data/ --format json
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Save JSON directly to a file (suppresses terminal progress):
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
geo-audit data/ --format json --output report.json
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## Audit Capabilities
|
|
62
|
+
|
|
63
|
+
- **Geographic Coordinate Validation**: Auto-detects `latitude`/`longitude` columns (including variants like `lat`/`lon`/`lng`), coerces non-numeric entries, and flags values outside valid ranges ([-90, 90] and [-180, 180]).
|
|
64
|
+
- **Planar Coordinate Inspection**: Identifies planar coordinate pairs (`easting`/`northing`, `x`/`y`) and flags missing values. Planar coordinates are reported separately because their Coordinate Reference System (CRS) is unknown.
|
|
65
|
+
- **Duplicate Location Detection**: Identifies exact duplicate coordinate pairs across records within CSV datasets.
|
|
66
|
+
- **Attribute Missingness Profiling**: Checks column-level completeness. Attribute columns with missingness exceeding 5% are flagged (`WARN` at 5-20%, `HIGH` above 20%). Coordinate columns are excluded from column missingness to prevent duplicate reporting.
|
|
67
|
+
- **CRS and Bounding Box Analysis**: Extracts declared CRS from GeoJSON datasets and computes spatial bounding boxes for geographic datasets to evaluate spatial coverage and cross-dataset bounding box overlaps.
|
|
68
|
+
- **Cross-Dataset Schema Comparison**: Identifies normalized column name matches and data type discrepancies across datasets.
|
|
69
|
+
|
|
70
|
+
## Dataset Readiness Levels
|
|
71
|
+
|
|
72
|
+
Every scanned dataset is assigned an overall readiness status based on the highest severity finding detected:
|
|
73
|
+
|
|
74
|
+
- **READY**: No blocking or moderate issues detected by active checks.
|
|
75
|
+
- **REVIEW**: Moderate-severity findings detected (e.g., 5-20% missingness, duplicate locations, missing planar coordinates).
|
|
76
|
+
- **HIGH RISK**: High-severity findings detected (e.g., >20% missing or invalid geographic coordinates, high column missingness).
|
|
77
|
+
- **ERROR**: Unparseable files or critical dataset structural failure.
|
|
78
|
+
|
|
79
|
+
## Demo Data
|
|
80
|
+
|
|
81
|
+
The included `data/` directory contains synthetic test cases demonstrating common data issues:
|
|
82
|
+
|
|
83
|
+
| File | Description | Primary Findings |
|
|
84
|
+
|---|---|---|
|
|
85
|
+
| `exploration_samples.csv` | 12 sample points | Out-of-range coordinates, missing lat/lon, 41.7% missing notes, duplicate locations |
|
|
86
|
+
| `historical_surveys.csv` | 10 survey points | Clean dataset using `lat`/`lon` column headers |
|
|
87
|
+
| `geology.geojson` | 5 Point geometries | Clean GeoJSON dataset with EPSG:4326 |
|
|
88
|
+
| `planar_samples.csv` | 8 sample points | Easting/northing planar coordinates with 12.5% missing values |
|
|
89
|
+
|
|
90
|
+
Run the demo against the included data:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
geo-audit data/
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## Scope and Limitations
|
|
97
|
+
|
|
98
|
+
- **Supported Formats**: CSV (`.csv`) and GeoJSON (`.geojson`, `.json`).
|
|
99
|
+
- **Duplicate Detection**: Performs exact coordinate pair matching; near-duplicate spatial clustering is not performed.
|
|
100
|
+
- **Spatial Overlap**: Evaluates bounding box intersection; polygon geometry intersections are out of scope.
|
|
101
|
+
- **CRS Transformations**: Does not attempt automatic CRS inference or coordinate re-projection.
|
|
102
|
+
|
|
103
|
+
## License
|
|
104
|
+
|
|
105
|
+
MIT
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# geo-audit
|
|
2
|
+
|
|
3
|
+
`geo-audit` is a command-line tool for auditing raw geospatial datasets before ingestion into spatial analysis, GIS workflows, or machine learning pipelines. It inspects CSV and GeoJSON files in a directory to identify coordinate anomalies, missing spatial metadata, attribute missingness, exact duplicate locations, and cross-dataset bounding box overlaps without modifying source data.
|
|
4
|
+
|
|
5
|
+
## Purpose
|
|
6
|
+
|
|
7
|
+
Geospatial data collected from field surveys, legacy archives, and third-party GIS exports frequently contains quality defects—such as out-of-range latitude/longitude coordinates, unprojected planar coordinates lacking CRS definitions, silent duplicate coordinates, and missing values across attribute columns. `geo-audit` runs static health checks across all datasets in a target directory to surface critical defects and risk levels in a single pass.
|
|
8
|
+
|
|
9
|
+
## Installation
|
|
10
|
+
|
|
11
|
+
Requires Python 3.10 or higher.
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
git clone https://github.com/souro26/geo-audit
|
|
15
|
+
cd geo-audit
|
|
16
|
+
pip install -e .
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
## Quick Start
|
|
20
|
+
|
|
21
|
+
Scan a directory of geospatial files for a human-readable terminal report:
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
geo-audit data/
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Generate machine-readable JSON output for automated pipelines:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
geo-audit data/ --format json
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
Save JSON directly to a file (suppresses terminal progress):
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
geo-audit data/ --format json --output report.json
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Audit Capabilities
|
|
40
|
+
|
|
41
|
+
- **Geographic Coordinate Validation**: Auto-detects `latitude`/`longitude` columns (including variants like `lat`/`lon`/`lng`), coerces non-numeric entries, and flags values outside valid ranges ([-90, 90] and [-180, 180]).
|
|
42
|
+
- **Planar Coordinate Inspection**: Identifies planar coordinate pairs (`easting`/`northing`, `x`/`y`) and flags missing values. Planar coordinates are reported separately because their Coordinate Reference System (CRS) is unknown.
|
|
43
|
+
- **Duplicate Location Detection**: Identifies exact duplicate coordinate pairs across records within CSV datasets.
|
|
44
|
+
- **Attribute Missingness Profiling**: Checks column-level completeness. Attribute columns with missingness exceeding 5% are flagged (`WARN` at 5-20%, `HIGH` above 20%). Coordinate columns are excluded from column missingness to prevent duplicate reporting.
|
|
45
|
+
- **CRS and Bounding Box Analysis**: Extracts declared CRS from GeoJSON datasets and computes spatial bounding boxes for geographic datasets to evaluate spatial coverage and cross-dataset bounding box overlaps.
|
|
46
|
+
- **Cross-Dataset Schema Comparison**: Identifies normalized column name matches and data type discrepancies across datasets.
|
|
47
|
+
|
|
48
|
+
## Dataset Readiness Levels
|
|
49
|
+
|
|
50
|
+
Every scanned dataset is assigned an overall readiness status based on the highest severity finding detected:
|
|
51
|
+
|
|
52
|
+
- **READY**: No blocking or moderate issues detected by active checks.
|
|
53
|
+
- **REVIEW**: Moderate-severity findings detected (e.g., 5-20% missingness, duplicate locations, missing planar coordinates).
|
|
54
|
+
- **HIGH RISK**: High-severity findings detected (e.g., >20% missing or invalid geographic coordinates, high column missingness).
|
|
55
|
+
- **ERROR**: Unparseable files or critical dataset structural failure.
|
|
56
|
+
|
|
57
|
+
## Demo Data
|
|
58
|
+
|
|
59
|
+
The included `data/` directory contains synthetic test cases demonstrating common data issues:
|
|
60
|
+
|
|
61
|
+
| File | Description | Primary Findings |
|
|
62
|
+
|---|---|---|
|
|
63
|
+
| `exploration_samples.csv` | 12 sample points | Out-of-range coordinates, missing lat/lon, 41.7% missing notes, duplicate locations |
|
|
64
|
+
| `historical_surveys.csv` | 10 survey points | Clean dataset using `lat`/`lon` column headers |
|
|
65
|
+
| `geology.geojson` | 5 Point geometries | Clean GeoJSON dataset with EPSG:4326 |
|
|
66
|
+
| `planar_samples.csv` | 8 sample points | Easting/northing planar coordinates with 12.5% missing values |
|
|
67
|
+
|
|
68
|
+
Run the demo against the included data:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
geo-audit data/
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Scope and Limitations
|
|
75
|
+
|
|
76
|
+
- **Supported Formats**: CSV (`.csv`) and GeoJSON (`.geojson`, `.json`).
|
|
77
|
+
- **Duplicate Detection**: Performs exact coordinate pair matching; near-duplicate spatial clustering is not performed.
|
|
78
|
+
- **Spatial Overlap**: Evaluates bounding box intersection; polygon geometry intersections are out of scope.
|
|
79
|
+
- **CRS Transformations**: Does not attempt automatic CRS inference or coordinate re-projection.
|
|
80
|
+
|
|
81
|
+
## License
|
|
82
|
+
|
|
83
|
+
MIT
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"""Main CLI entry point."""
|
|
2
|
+
|
|
3
|
+
import sys
|
|
4
|
+
from datetime import datetime
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
import typer
|
|
8
|
+
from rich.console import Console
|
|
9
|
+
|
|
10
|
+
from geo_audit.coordinates import detect_coordinate_columns, detect_planar_coordinates
|
|
11
|
+
from geo_audit.discovery import discover_files, get_file_size
|
|
12
|
+
from geo_audit.findings import (
|
|
13
|
+
build_audit_summary,
|
|
14
|
+
compute_dataset_readiness,
|
|
15
|
+
generate_findings,
|
|
16
|
+
)
|
|
17
|
+
from geo_audit.loaders import (
|
|
18
|
+
calculate_bounding_box,
|
|
19
|
+
calculate_bounding_box_from_coords,
|
|
20
|
+
infer_geometry_types,
|
|
21
|
+
load_dataset,
|
|
22
|
+
profile_columns,
|
|
23
|
+
)
|
|
24
|
+
from geo_audit.models import (
|
|
25
|
+
AuditReport,
|
|
26
|
+
DatasetInfo,
|
|
27
|
+
FileFormat,
|
|
28
|
+
)
|
|
29
|
+
from geo_audit.quality import detect_exact_duplicates
|
|
30
|
+
from geo_audit.reporting import generate_json_report, generate_terminal_report
|
|
31
|
+
from geo_audit.spatial import compute_cross_dataset_overlaps
|
|
32
|
+
|
|
33
|
+
APP_HELP = (
|
|
34
|
+
"Audit heterogeneous geospatial datasets for data-quality "
|
|
35
|
+
"and spatial-consistency problems."
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
app = typer.Typer(
|
|
39
|
+
name="geo-audit",
|
|
40
|
+
help=APP_HELP,
|
|
41
|
+
add_completion=False,
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
# Separate console for progress (stderr) and report (stdout)
|
|
45
|
+
progress_console = Console(
|
|
46
|
+
stderr=True, force_terminal=True, color_system="standard", legacy_windows=False
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@app.command()
|
|
51
|
+
def audit(
|
|
52
|
+
input_dir: Path = typer.Argument(..., help="Directory containing geospatial datasets"),
|
|
53
|
+
output: Path | None = typer.Option(
|
|
54
|
+
None, "--output", "-o", help="Output file path (only with --format json)"
|
|
55
|
+
),
|
|
56
|
+
format: str = typer.Option("terminal", "--format", "-f", help="Output format: terminal, json"),
|
|
57
|
+
):
|
|
58
|
+
"""Audit geospatial datasets in a directory."""
|
|
59
|
+
if not input_dir.exists() or not input_dir.is_dir():
|
|
60
|
+
progress_console.print(f"[red]Error: {input_dir} is not a valid directory[/red]")
|
|
61
|
+
raise typer.Exit(1)
|
|
62
|
+
|
|
63
|
+
if output and format != "json":
|
|
64
|
+
progress_console.print("[red]Error: --output only works with --format json[/red]")
|
|
65
|
+
raise typer.Exit(1)
|
|
66
|
+
|
|
67
|
+
datasets = []
|
|
68
|
+
|
|
69
|
+
for file_path, file_format in discover_files(input_dir):
|
|
70
|
+
if format == "terminal":
|
|
71
|
+
progress_console.print(f"[dim]Processing {file_path.name}...[/dim]")
|
|
72
|
+
ds_info = process_dataset(file_path, file_format)
|
|
73
|
+
datasets.append(ds_info)
|
|
74
|
+
|
|
75
|
+
if not datasets:
|
|
76
|
+
progress_console.print("[yellow]No supported datasets found[/yellow]")
|
|
77
|
+
raise typer.Exit(0)
|
|
78
|
+
|
|
79
|
+
overlaps = compute_cross_dataset_overlaps(datasets)
|
|
80
|
+
|
|
81
|
+
# Generate findings and readiness
|
|
82
|
+
findings = generate_findings(datasets)
|
|
83
|
+
dataset_readiness = compute_dataset_readiness(datasets, findings)
|
|
84
|
+
summary = build_audit_summary(datasets, findings)
|
|
85
|
+
|
|
86
|
+
report = AuditReport(
|
|
87
|
+
datasets=datasets,
|
|
88
|
+
cross_dataset_overlaps=overlaps,
|
|
89
|
+
summary=summary.to_dict(),
|
|
90
|
+
generated_at=datetime.now().isoformat(),
|
|
91
|
+
findings=findings,
|
|
92
|
+
dataset_readiness=dataset_readiness,
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
if format == "json":
|
|
96
|
+
json_output = generate_json_report(report)
|
|
97
|
+
if output:
|
|
98
|
+
output.write_text(json_output)
|
|
99
|
+
# Don't print success message to stdout in JSON mode
|
|
100
|
+
else:
|
|
101
|
+
# Print JSON to stdout (only JSON, no progress messages)
|
|
102
|
+
sys.stdout.write(json_output + "\n")
|
|
103
|
+
else:
|
|
104
|
+
generate_terminal_report(report)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def process_dataset(file_path: Path, file_format: FileFormat) -> DatasetInfo:
|
|
108
|
+
"""Process a single dataset and return DatasetInfo."""
|
|
109
|
+
file_size = get_file_size(file_path)
|
|
110
|
+
data, parse_errors = load_dataset(file_path, file_format)
|
|
111
|
+
|
|
112
|
+
if data is None or (hasattr(data, 'empty') and data.empty):
|
|
113
|
+
return DatasetInfo(
|
|
114
|
+
filename=file_path.name,
|
|
115
|
+
format=file_format,
|
|
116
|
+
row_count=0,
|
|
117
|
+
columns=[],
|
|
118
|
+
column_profiles=[],
|
|
119
|
+
file_size_bytes=file_size,
|
|
120
|
+
parse_errors=parse_errors,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
if file_format == FileFormat.CSV:
|
|
124
|
+
columns = list(data.columns)
|
|
125
|
+
column_profiles = profile_columns(data)
|
|
126
|
+
row_count = len(data)
|
|
127
|
+
|
|
128
|
+
coord_info = detect_coordinate_columns(data)
|
|
129
|
+
planar_info = detect_planar_coordinates(data)
|
|
130
|
+
|
|
131
|
+
duplicate_count = 0
|
|
132
|
+
duplicate_pct = 0.0
|
|
133
|
+
if coord_info.lat_column and coord_info.lon_column:
|
|
134
|
+
duplicate_count, duplicate_pct = detect_exact_duplicates(
|
|
135
|
+
data, coord_info.lat_column, coord_info.lon_column
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
bounding_box = None
|
|
139
|
+
if coord_info.lat_column and coord_info.lon_column:
|
|
140
|
+
bounding_box = calculate_bounding_box_from_coords(
|
|
141
|
+
data, coord_info.lat_column, coord_info.lon_column
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
# Warnings kept for backward compatibility in DatasetInfo
|
|
145
|
+
warnings = []
|
|
146
|
+
if coord_info.lat_column and coord_info.lon_column:
|
|
147
|
+
if coord_info.missing_percentage > 5:
|
|
148
|
+
warnings.append(f"{coord_info.missing_percentage:.1f}% missing coordinates")
|
|
149
|
+
if coord_info.invalid_percentage > 0:
|
|
150
|
+
warnings.append(f"{coord_info.invalid_percentage:.1f}% invalid coordinates")
|
|
151
|
+
if duplicate_pct > 1:
|
|
152
|
+
warnings.append(f"{duplicate_pct:.1f}% exact duplicate locations")
|
|
153
|
+
if planar_info:
|
|
154
|
+
if planar_info.missing_percentage > 5:
|
|
155
|
+
msg = f"Planar coordinates ({planar_info.x_column}/{planar_info.y_column}): "
|
|
156
|
+
msg += f"{planar_info.missing_percentage:.1f}% missing"
|
|
157
|
+
warnings.append(msg)
|
|
158
|
+
|
|
159
|
+
return DatasetInfo(
|
|
160
|
+
filename=file_path.name,
|
|
161
|
+
format=file_format,
|
|
162
|
+
row_count=row_count,
|
|
163
|
+
columns=columns,
|
|
164
|
+
column_profiles=column_profiles,
|
|
165
|
+
file_size_bytes=file_size,
|
|
166
|
+
coordinate_info=coord_info,
|
|
167
|
+
planar_coordinate_info=planar_info,
|
|
168
|
+
bounding_box=bounding_box,
|
|
169
|
+
duplicate_count=duplicate_count,
|
|
170
|
+
duplicate_percentage=duplicate_pct,
|
|
171
|
+
parse_errors=parse_errors,
|
|
172
|
+
warnings=warnings,
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
else:
|
|
176
|
+
columns = [c for c in data.columns if c != 'geometry']
|
|
177
|
+
non_geom = data.drop(columns=['geometry']) if 'geometry' in data.columns else data
|
|
178
|
+
column_profiles = profile_columns(non_geom)
|
|
179
|
+
row_count = len(data)
|
|
180
|
+
|
|
181
|
+
geometry_types = infer_geometry_types(data)
|
|
182
|
+
crs = str(data.crs) if data.crs else None
|
|
183
|
+
bounding_box = calculate_bounding_box(data)
|
|
184
|
+
|
|
185
|
+
warnings = []
|
|
186
|
+
if crs is None:
|
|
187
|
+
warnings.append("No CRS declared in file (GeoJSON spec assumes WGS84)")
|
|
188
|
+
|
|
189
|
+
return DatasetInfo(
|
|
190
|
+
filename=file_path.name,
|
|
191
|
+
format=file_format,
|
|
192
|
+
row_count=row_count,
|
|
193
|
+
columns=columns,
|
|
194
|
+
column_profiles=column_profiles,
|
|
195
|
+
file_size_bytes=file_size,
|
|
196
|
+
geometry_types=geometry_types,
|
|
197
|
+
crs=crs,
|
|
198
|
+
bounding_box=bounding_box,
|
|
199
|
+
parse_errors=parse_errors,
|
|
200
|
+
warnings=warnings,
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
if __name__ == "__main__":
|
|
205
|
+
app()
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""Coordinate detection and validation."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from geo_audit.models import CoordinateInfo, PlanarCoordinateInfo
|
|
7
|
+
|
|
8
|
+
LAT_PATTERNS = [
|
|
9
|
+
"latitude", "lat", "lat_dd", "lat_ddm", "lat_dms",
|
|
10
|
+
"y_lat", "ycoord_lat", "gps_lat", "gps_latitude",
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
LON_PATTERNS = [
|
|
14
|
+
"longitude", "lon", "lng", "long",
|
|
15
|
+
"lon_dd", "lon_ddm", "lon_dms",
|
|
16
|
+
"x_lon", "xcoord_lon", "gps_lon", "gps_longitude",
|
|
17
|
+
]
|
|
18
|
+
|
|
19
|
+
PLANAR_PATTERNS = [
|
|
20
|
+
("x", "y"),
|
|
21
|
+
("x_coord", "y_coord"),
|
|
22
|
+
("xcoordinate", "ycoordinate"),
|
|
23
|
+
("easting", "northing"),
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def detect_coordinate_columns(df: pd.DataFrame) -> CoordinateInfo:
|
|
28
|
+
"""Detect latitude/longitude columns in a DataFrame.
|
|
29
|
+
|
|
30
|
+
Only explicit latitude/longitude column names are treated as geographic coordinates.
|
|
31
|
+
Generic x/y or easting/northing columns are NOT assumed to be lat/lon.
|
|
32
|
+
"""
|
|
33
|
+
cols_lower = {c.lower().strip(): c for c in df.columns}
|
|
34
|
+
info = CoordinateInfo()
|
|
35
|
+
|
|
36
|
+
lat_col = _find_column(cols_lower, LAT_PATTERNS)
|
|
37
|
+
lon_col = _find_column(cols_lower, LON_PATTERNS)
|
|
38
|
+
|
|
39
|
+
if lat_col and lon_col:
|
|
40
|
+
info.lat_column = lat_col
|
|
41
|
+
info.lon_column = lon_col
|
|
42
|
+
info.detection_method = "explicit_lat_lon"
|
|
43
|
+
info.confidence = 0.95
|
|
44
|
+
_validate_coordinates(df, info)
|
|
45
|
+
|
|
46
|
+
return info
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def detect_planar_coordinates(df: pd.DataFrame) -> PlanarCoordinateInfo | None:
|
|
50
|
+
"""Detect planar coordinate columns (x/y, easting/northing).
|
|
51
|
+
|
|
52
|
+
These are NOT validated as lat/lon since their CRS is unknown.
|
|
53
|
+
Returns info about the columns if found.
|
|
54
|
+
"""
|
|
55
|
+
cols_lower = {c.lower().strip(): c for c in df.columns}
|
|
56
|
+
|
|
57
|
+
for x_pat, y_pat in PLANAR_PATTERNS:
|
|
58
|
+
x_col = _find_column(cols_lower, [x_pat])
|
|
59
|
+
y_col = _find_column(cols_lower, [y_pat])
|
|
60
|
+
if x_col and y_col:
|
|
61
|
+
total = len(df)
|
|
62
|
+
missing = int(df[x_col].isna().sum() | df[y_col].isna().sum())
|
|
63
|
+
return PlanarCoordinateInfo(
|
|
64
|
+
x_column=x_col,
|
|
65
|
+
y_column=y_col,
|
|
66
|
+
detection_method="planar_heuristic",
|
|
67
|
+
count=total,
|
|
68
|
+
missing_count=missing,
|
|
69
|
+
)
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _find_column(cols_lower: dict[str, str], patterns: list[str]) -> str | None:
|
|
74
|
+
"""Find a column matching any of the patterns."""
|
|
75
|
+
for pat in patterns:
|
|
76
|
+
if pat in cols_lower:
|
|
77
|
+
return cols_lower[pat]
|
|
78
|
+
return None
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _validate_coordinates(df: pd.DataFrame, info: CoordinateInfo) -> None:
|
|
82
|
+
"""Validate coordinate values as lat/lon."""
|
|
83
|
+
lat_col = info.lat_column
|
|
84
|
+
lon_col = info.lon_column
|
|
85
|
+
|
|
86
|
+
if not lat_col or not lon_col:
|
|
87
|
+
return
|
|
88
|
+
|
|
89
|
+
total = len(df)
|
|
90
|
+
info.total_count = total
|
|
91
|
+
|
|
92
|
+
lat_series = pd.to_numeric(df[lat_col], errors="coerce")
|
|
93
|
+
lon_series = pd.to_numeric(df[lon_col], errors="coerce")
|
|
94
|
+
|
|
95
|
+
missing_mask = lat_series.isna() | lon_series.isna()
|
|
96
|
+
info.missing_count = int(missing_mask.sum())
|
|
97
|
+
|
|
98
|
+
valid_rows_lat = lat_series[~missing_mask]
|
|
99
|
+
valid_rows_lon = lon_series[~missing_mask]
|
|
100
|
+
if valid_rows_lat.empty:
|
|
101
|
+
info.valid_count = 0
|
|
102
|
+
info.invalid_count = 0
|
|
103
|
+
return
|
|
104
|
+
|
|
105
|
+
lat_valid = (valid_rows_lat >= -90) & (valid_rows_lat <= 90)
|
|
106
|
+
lon_valid = (valid_rows_lon >= -180) & (valid_rows_lon <= 180)
|
|
107
|
+
both_valid = lat_valid & lon_valid
|
|
108
|
+
|
|
109
|
+
info.valid_count = int(both_valid.sum())
|
|
110
|
+
info.invalid_count = int((~both_valid).sum())
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def validate_lat_lon(lat: float, lon: float) -> bool:
|
|
114
|
+
"""Check if lat/lon values are valid."""
|
|
115
|
+
return -90 <= lat <= 90 and -180 <= lon <= 180
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Dataset discovery module."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from geo_audit.models import FileFormat
|
|
7
|
+
|
|
8
|
+
SUPPORTED_EXTENSIONS = {
|
|
9
|
+
".csv": FileFormat.CSV,
|
|
10
|
+
".geojson": FileFormat.GEOJSON,
|
|
11
|
+
".json": FileFormat.GEOJSON,
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def discover_files(root: Path) -> Iterator[tuple[Path, FileFormat]]:
|
|
16
|
+
"""Recursively discover supported files in a directory."""
|
|
17
|
+
if not root.exists():
|
|
18
|
+
raise FileNotFoundError(f"Path does not exist: {root}")
|
|
19
|
+
if not root.is_dir():
|
|
20
|
+
raise NotADirectoryError(f"Path is not a directory: {root}")
|
|
21
|
+
|
|
22
|
+
for path in root.rglob("*"):
|
|
23
|
+
if path.is_file():
|
|
24
|
+
ext = path.suffix.lower()
|
|
25
|
+
if ext in SUPPORTED_EXTENSIONS:
|
|
26
|
+
yield path, SUPPORTED_EXTENSIONS[ext]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def get_file_format(path: Path) -> FileFormat:
|
|
30
|
+
"""Determine file format from extension."""
|
|
31
|
+
ext = path.suffix.lower()
|
|
32
|
+
return SUPPORTED_EXTENSIONS.get(ext, FileFormat.UNKNOWN)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def get_file_size(path: Path) -> int:
|
|
36
|
+
"""Get file size in bytes."""
|
|
37
|
+
try:
|
|
38
|
+
return path.stat().st_size
|
|
39
|
+
except OSError:
|
|
40
|
+
return 0
|