core-lens 0.1.dev184__tar.gz → 0.1.dev188__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/PKG-INFO +1 -1
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/SKILLS.md +38 -10
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/_version.py +2 -2
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/base/namespaces/stats.py +176 -24
- core_lens-0.1.dev188/src/core_lens/entities/farms.py +63 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/uv.lock +3 -3
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.github/ISSUE_TEMPLATE/blank-proposal.yaml +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.github/ISSUE_TEMPLATE/bug-report.yaml +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.github/ISSUE_TEMPLATE/feature-request.yaml +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.github/pull_request_template.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.github/workflows/ci.yml +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.github/workflows/gh-pages.yml +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.github/workflows/pre-release.yml +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.github/workflows/release.yml +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.gitignore +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.gitmessage +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.pre-commit-config.yaml +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/.python-version +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/CONTRIBUTING.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/LICENSE +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/README.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/README.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/bench_aoi.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/bench_entity.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/bench_export.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/bench_polars_utils.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/bench_result.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/bench_schema.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/bench_season.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/bench_spatial.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/bench_view.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/benchmarks/run_all.sh +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/Makefile +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/make.bat +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/concepts.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/conf.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/export.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/index.rst +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/intro.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/logging.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/plots.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/plugins.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/queries.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/quickstart.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/docs/source/stats.md +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/examples/demo_mws.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/examples/demo_tehsil.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/hooks/mypy.sh +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/hooks/no-parquet-outside-fixtures.sh +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/hooks/pytest.sh +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/pyproject.toml +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/__init__.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/__main__.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/aoi.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/base/__init__.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/base/entity.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/base/namespaces/__init__.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/base/namespaces/plot.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/base/result.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/base/view.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/entities/__init__.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/entities/mws.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/entities/tehsil.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/entities/waterbody.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/export/__init__.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/export/formats.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/py.typed +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/schema/__init__.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/schema/detection.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/schema/profile.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/utils/__init__.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/utils/paths.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/utils/polars_utils.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/utils/season.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/src/core_lens/utils/spatial.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/fixtures/generate_fixtures.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/conftest.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_aoi.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_entities.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_entity.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_export.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_main.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_plot.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_polars_utils.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_profile.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_result.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_schema_detection.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_schema_profile.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_season.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_season_config.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_spatial.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_stats.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/tests/unit/test_view.py +0 -0
- {core_lens-0.1.dev184 → core_lens-0.1.dev188}/usage.md +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: core-lens
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.dev188
|
|
4
4
|
Summary: Query, analyse, and visualise CoreStack's microwatershed and Earth science data through a clean, composable Python API.
|
|
5
5
|
Project-URL: Homepage, https://github.com/ApoorvaKashyap/core-lens
|
|
6
6
|
Project-URL: Issues, https://github.com/ApoorvaKashyap/core-lens/issues
|
|
@@ -57,7 +57,9 @@ aoi = AoI(DATA_ROOT, bbox=(76.0, 31.0, 78.0, 33.0))
|
|
|
57
57
|
aoi = AoI(DATA_ROOT, mws_id="13_551")
|
|
58
58
|
|
|
59
59
|
# Optional: Override default seasons (kharif, rabi, zaid)
|
|
60
|
-
custom_seasons = SeasonConfig(
|
|
60
|
+
custom_seasons = SeasonConfig(
|
|
61
|
+
kharif=("06-01", "10-15"), rabi=("10-16", "02-28"), zaid=("03-01", "05-31")
|
|
62
|
+
)
|
|
61
63
|
aoi_custom = AoI(DATA_ROOT, tehsil="Pangi", seasons=custom_seasons)
|
|
62
64
|
```
|
|
63
65
|
|
|
@@ -78,6 +80,7 @@ spatial_filtered = mws_view.spatial_filter(bbox=(76.5, 31.5, 77.5, 32.5))
|
|
|
78
80
|
# Temporal Filtering
|
|
79
81
|
# Note: You can filter by exact years, a range of years, or predefined seasons.
|
|
80
82
|
from core_lens.base.view import Season
|
|
83
|
+
|
|
81
84
|
temporal_view = mws_view.between(year=(2020, 2022), season=Season.KHARIF)
|
|
82
85
|
current_view = mws_view.between(season=Season.CURRENT)
|
|
83
86
|
```
|
|
@@ -98,9 +101,9 @@ annual_result = temporal_view.annual.materialise()
|
|
|
98
101
|
|
|
99
102
|
```python
|
|
100
103
|
# To access underlying data structures:
|
|
101
|
-
df = static_result.df()
|
|
102
|
-
lazy_df = static_result.lazy()
|
|
103
|
-
gdf = static_result.gdf()
|
|
104
|
+
df = static_result.df() # Polars DataFrame (Zero-copy)
|
|
105
|
+
lazy_df = static_result.lazy() # Polars LazyFrame
|
|
106
|
+
gdf = static_result.gdf() # GeoPandas GeoDataFrame (Heavy!)
|
|
104
107
|
```
|
|
105
108
|
|
|
106
109
|
## 5. Computation and Data Manipulation (Result API)
|
|
@@ -145,30 +148,53 @@ result.stats.describe(columns=["ndvi", "rainfall"])
|
|
|
145
148
|
|
|
146
149
|
# Correlation (pearson, spearman, kendall)
|
|
147
150
|
from core_lens.base.namespaces.stats import CorrelateMethod
|
|
148
|
-
|
|
151
|
+
|
|
152
|
+
result.stats.correlate(
|
|
153
|
+
columns=["ndvi", "rainfall"], method=CorrelateMethod.PEARSON, across="entity"
|
|
154
|
+
)
|
|
149
155
|
|
|
150
156
|
# Hypothesis Testing (t-test, mann-whitney, wilcoxon, ks, chi-square)
|
|
151
157
|
from core_lens.base.namespaces.stats import TestMethod
|
|
152
|
-
|
|
158
|
+
|
|
159
|
+
result.stats.test(
|
|
160
|
+
column="cropping_intensity",
|
|
161
|
+
groups="temperature_zone",
|
|
162
|
+
method=TestMethod.MANN_WHITNEY,
|
|
163
|
+
)
|
|
153
164
|
|
|
154
165
|
# Change Detection (absolute, percentage, trend)
|
|
155
166
|
from core_lens.base.namespaces.stats import ChangeMethod
|
|
156
|
-
|
|
167
|
+
|
|
168
|
+
result.stats.change(
|
|
169
|
+
column="tree_cover",
|
|
170
|
+
from_period=2018,
|
|
171
|
+
to_period=2023,
|
|
172
|
+
method=ChangeMethod.PERCENTAGE,
|
|
173
|
+
)
|
|
157
174
|
|
|
158
175
|
# Anomaly Detection
|
|
159
176
|
from core_lens.base.namespaces.stats import AnomalyTsMethod, AnomalyCrossMethod
|
|
177
|
+
|
|
160
178
|
# Mode 1: Cross-sectional (zscore, iqr, percentile, threshold)
|
|
161
|
-
result.stats.anomaly(
|
|
179
|
+
result.stats.anomaly(
|
|
180
|
+
column="ndvi",
|
|
181
|
+
mode="cross_sectional",
|
|
182
|
+
method=AnomalyCrossMethod.ZSCORE,
|
|
183
|
+
baseline=(2010, 2020),
|
|
184
|
+
)
|
|
162
185
|
# Mode 2: Time-series (stl, cusum, mad)
|
|
163
|
-
result.stats.anomaly(
|
|
186
|
+
result.stats.anomaly(
|
|
187
|
+
column="ndvi", mode="timeseries", method=AnomalyTsMethod.STL, baseline=(2010, 2018)
|
|
188
|
+
)
|
|
164
189
|
|
|
165
190
|
# Similarity Matching (euclidean, cosine, mahalanobis, manhattan)
|
|
166
191
|
from core_lens.base.namespaces.stats import SimilarityMethod
|
|
192
|
+
|
|
167
193
|
result.stats.similarity(
|
|
168
194
|
target="13_551",
|
|
169
195
|
columns={"rainfall": ("annual", {"year": 2018})},
|
|
170
196
|
method=SimilarityMethod.EUCLIDEAN,
|
|
171
|
-
top_n=10
|
|
197
|
+
top_n=10,
|
|
172
198
|
)
|
|
173
199
|
```
|
|
174
200
|
|
|
@@ -214,6 +240,7 @@ You can extend core-lens with custom entities by subclassing `BaseEntity`.
|
|
|
214
240
|
```python
|
|
215
241
|
from core_lens.base import BaseEntity
|
|
216
242
|
|
|
243
|
+
|
|
217
244
|
class CustomEntity(BaseEntity):
|
|
218
245
|
@property
|
|
219
246
|
def key_cols(self) -> list[str]:
|
|
@@ -231,6 +258,7 @@ class CustomEntity(BaseEntity):
|
|
|
231
258
|
def annual_path(self) -> str | None:
|
|
232
259
|
return "custom/annual.parquet"
|
|
233
260
|
|
|
261
|
+
|
|
234
262
|
AoI.register(CustomEntity)
|
|
235
263
|
```
|
|
236
264
|
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.1.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 1, '
|
|
21
|
+
__version__ = version = '0.1.dev188'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 1, 'dev188')
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|
|
@@ -11,6 +11,7 @@ import numpy as np
|
|
|
11
11
|
import polars as pl
|
|
12
12
|
|
|
13
13
|
from core_lens.utils.polars_utils import collect_lf, cached_read_schema
|
|
14
|
+
from core_lens.base.result import Result
|
|
14
15
|
|
|
15
16
|
if TYPE_CHECKING:
|
|
16
17
|
from core_lens.base.result import Result
|
|
@@ -157,7 +158,7 @@ class StatsNamespace:
|
|
|
157
158
|
Computed values always go in ``data``; method parameters go in ``metadata``.
|
|
158
159
|
"""
|
|
159
160
|
|
|
160
|
-
def __init__(self, result:
|
|
161
|
+
def __init__(self, result: Result) -> None:
|
|
161
162
|
"""Initialize StatsNamespace.
|
|
162
163
|
|
|
163
164
|
Args:
|
|
@@ -178,7 +179,7 @@ class StatsNamespace:
|
|
|
178
179
|
self,
|
|
179
180
|
columns: list[str] | None = None,
|
|
180
181
|
by: str = "column",
|
|
181
|
-
) ->
|
|
182
|
+
) -> Result:
|
|
182
183
|
r"""Per-column or per-entity descriptive statistics.
|
|
183
184
|
|
|
184
185
|
Uses polars' in-built methods for mean, std, min, max, quantiles etc.
|
|
@@ -238,26 +239,80 @@ class StatsNamespace:
|
|
|
238
239
|
columns: list[str],
|
|
239
240
|
method: CorrelateMethod = CorrelateMethod.PEARSON,
|
|
240
241
|
across: str = "entity",
|
|
242
|
+
group_by: str | None = None,
|
|
243
|
+
min_obs: int = 3,
|
|
241
244
|
) -> "Result":
|
|
242
|
-
"""
|
|
245
|
+
"""Compute pairwise correlations between columns.
|
|
246
|
+
|
|
247
|
+
Correlations are computed with ``scipy.stats`` and returned as a
|
|
248
|
+
:class:`~core_lens.base.result.Result` whose ``data`` is a Polars
|
|
249
|
+
``DataFrame``. Two modes are supported:
|
|
250
|
+
|
|
251
|
+
* **Pooled** (``group_by is None``): all rows are treated as one
|
|
252
|
+
observation set and a single correlation is computed per column pair.
|
|
253
|
+
* **Per-entity** (``group_by`` set): the result is grouped by the given
|
|
254
|
+
key column (e.g. ``"mws_id"``) and one correlation is computed per
|
|
255
|
+
entity across its remaining rows (typically its time series). This is
|
|
256
|
+
what enables questions such as "which MWS show the strongest
|
|
257
|
+
dependence of cropping intensity on annual rainfall?".
|
|
243
258
|
|
|
244
259
|
Args:
|
|
245
|
-
columns
|
|
246
|
-
|
|
247
|
-
|
|
260
|
+
columns: Column names to correlate. Must contain at least two
|
|
261
|
+
entries. In pooled mode every pair is computed; in per-entity
|
|
262
|
+
mode the first two columns are used as the pair.
|
|
263
|
+
method: The correlation coefficient to compute. One of
|
|
264
|
+
:class:`CorrelateMethod` (``PEARSON``, ``SPEARMAN``, ``KENDALL``).
|
|
265
|
+
across: ``"entity"`` or ``"time"``. Recorded in result metadata to
|
|
266
|
+
describe the intended axis of the relationship.
|
|
267
|
+
group_by: Optional key column to compute one correlation per group
|
|
268
|
+
(e.g. ``"mws_id"``). When ``None``, the pooled behaviour is
|
|
269
|
+
preserved for backward compatibility.
|
|
270
|
+
min_obs: Minimum number of non-null observations required per group
|
|
271
|
+
(per-entity mode) or overall (pooled mode) before a correlation
|
|
272
|
+
is computed. Groups below this threshold are skipped.
|
|
248
273
|
|
|
249
274
|
Returns:
|
|
250
|
-
Result: Result whose data
|
|
251
|
-
|
|
275
|
+
Result: A :class:`~core_lens.base.result.Result` whose ``data``
|
|
276
|
+
DataFrame has the following columns:
|
|
277
|
+
|
|
278
|
+
* Pooled mode: ``column_a | column_b | correlation | p_value``
|
|
279
|
+
* Per-entity mode: ``{group_by} | correlation | p_value | n_obs``
|
|
280
|
+
|
|
281
|
+
Geometry is dropped (``has_geometry=False``).
|
|
252
282
|
|
|
253
283
|
Raises:
|
|
254
|
-
CorrelationError: If fewer than
|
|
255
|
-
ValueError: If ``method`` is not
|
|
284
|
+
CorrelationError: If fewer than two columns are supplied.
|
|
285
|
+
ValueError: If ``method`` is not a valid :class:`CorrelateMethod`,
|
|
286
|
+
or if ``group_by`` is set but the column is not present in the
|
|
287
|
+
underlying data.
|
|
256
288
|
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
to compute the corresponding correlation coefficients and p-values.
|
|
289
|
+
Examples:
|
|
290
|
+
Pooled correlation across all MWS in the AoI::
|
|
260
291
|
|
|
292
|
+
result = aoi.mws.annual.stats.correlate(
|
|
293
|
+
columns=["dw_precipitation", "ci_cropping_intensity"],
|
|
294
|
+
method=CorrelateMethod.PEARSON,
|
|
295
|
+
)
|
|
296
|
+
print(result.data.collect())
|
|
297
|
+
|
|
298
|
+
Per-MWS correlation, ranked by strength of dependence::
|
|
299
|
+
|
|
300
|
+
top5 = (
|
|
301
|
+
aoi.mws.annual.stats
|
|
302
|
+
.correlate(
|
|
303
|
+
columns=["dw_precipitation", "ci_cropping_intensity"],
|
|
304
|
+
method=CorrelateMethod.PEARSON,
|
|
305
|
+
group_by="mws_id",
|
|
306
|
+
min_obs=3,
|
|
307
|
+
)
|
|
308
|
+
.data
|
|
309
|
+
.with_columns(pl.col("correlation").abs().alias("abs_corr"))
|
|
310
|
+
.sort("abs_corr", descending=True)
|
|
311
|
+
.select("mws_id", "correlation", "p_value", "n_obs")
|
|
312
|
+
.limit(5)
|
|
313
|
+
.collect()
|
|
314
|
+
)
|
|
315
|
+
print(top5)
|
|
261
316
|
"""
|
|
262
317
|
if len(columns) < 2:
|
|
263
318
|
raise CorrelationError(
|
|
@@ -265,27 +320,124 @@ class StatsNamespace:
|
|
|
265
320
|
)
|
|
266
321
|
if not isinstance(method, CorrelateMethod):
|
|
267
322
|
raise ValueError(
|
|
268
|
-
f"StatsNamespace.correlate: method must be a CorrelateMethod.
|
|
323
|
+
f"StatsNamespace.correlate: method must be a CorrelateMethod. "
|
|
324
|
+
f"Valid options: {[e.name for e in CorrelateMethod]}."
|
|
269
325
|
)
|
|
270
326
|
|
|
271
327
|
import scipy.stats as sp
|
|
272
328
|
|
|
273
|
-
|
|
329
|
+
lf = (
|
|
330
|
+
self._r.lazy()
|
|
331
|
+
) # LazyFrame — stays lazy until an explicit collect boundary below
|
|
332
|
+
|
|
333
|
+
def _t_pvals(corr_col: "pl.Expr", n_col: "pl.Expr") -> "pl.Expr":
|
|
334
|
+
r_safe = corr_col.clip(-0.9999999, 0.9999999)
|
|
335
|
+
return (corr_col * ((n_col - 2) / (1 - r_safe**2)).sqrt()).alias("t_stat")
|
|
336
|
+
|
|
337
|
+
# Per-entity mode
|
|
338
|
+
if group_by is not None:
|
|
339
|
+
# Schema check only — does NOT execute the query plan.
|
|
340
|
+
if group_by not in lf.collect_schema().names():
|
|
341
|
+
raise ValueError(
|
|
342
|
+
f"StatsNamespace.correlate: group_by column '{group_by}' "
|
|
343
|
+
f"not found in data. Available: {lf.collect_schema().names()}"
|
|
344
|
+
)
|
|
345
|
+
|
|
346
|
+
col_a, col_b = columns[0], columns[1]
|
|
347
|
+
|
|
348
|
+
if method in (CorrelateMethod.PEARSON, CorrelateMethod.SPEARMAN):
|
|
349
|
+
# Fully lazy path: rank (if Spearman) -> group_by -> agg -> filter.
|
|
350
|
+
# No collect until the very end, where scipy needs numpy anyway.
|
|
351
|
+
if method is CorrelateMethod.SPEARMAN:
|
|
352
|
+
src = lf.with_columns(
|
|
353
|
+
pl.col(col_a)
|
|
354
|
+
.rank(method="average")
|
|
355
|
+
.over(group_by)
|
|
356
|
+
.alias("__a"),
|
|
357
|
+
pl.col(col_b)
|
|
358
|
+
.rank(method="average")
|
|
359
|
+
.over(group_by)
|
|
360
|
+
.alias("__b"),
|
|
361
|
+
)
|
|
362
|
+
a_col, b_col = "__a", "__b"
|
|
363
|
+
else:
|
|
364
|
+
src, a_col, b_col = lf, col_a, col_b
|
|
365
|
+
|
|
366
|
+
lazy_result = (
|
|
367
|
+
src.group_by(group_by)
|
|
368
|
+
.agg(
|
|
369
|
+
pl.corr(pl.col(a_col), pl.col(b_col), method="pearson").alias(
|
|
370
|
+
"correlation"
|
|
371
|
+
),
|
|
372
|
+
pl.len().alias("n_obs"),
|
|
373
|
+
)
|
|
374
|
+
.filter(pl.col("correlation").is_finite())
|
|
375
|
+
.filter(pl.col("n_obs") >= min_obs)
|
|
376
|
+
.with_columns(_t_pvals(pl.col("correlation"), pl.col("n_obs")))
|
|
377
|
+
.filter(pl.col("t_stat").is_finite())
|
|
378
|
+
)
|
|
379
|
+
|
|
380
|
+
# --- single collect boundary: scipy.t.sf needs eager numpy ---
|
|
381
|
+
result = lazy_result.collect()
|
|
382
|
+
t_vals = result["t_stat"].to_numpy()
|
|
383
|
+
df_vals = result["n_obs"].to_numpy() - 2
|
|
384
|
+
p_vals = 2.0 * sp.t.sf(np.abs(t_vals), df_vals)
|
|
385
|
+
result = result.with_columns(pl.Series("p_value", p_vals))
|
|
386
|
+
|
|
387
|
+
data = result.select([group_by, "correlation", "p_value", "n_obs"])
|
|
388
|
+
|
|
389
|
+
else:
|
|
390
|
+
# Kendall: scipy has no batched/lazy equivalent, so this branch
|
|
391
|
+
# is inherently eager. Collect ONCE here — not scattered filters.
|
|
392
|
+
df = lf.collect()
|
|
393
|
+
rows: list[dict[str, Any]] = []
|
|
394
|
+
for sub_full in df.partition_by(group_by):
|
|
395
|
+
key = sub_full[group_by][0]
|
|
396
|
+
sub = sub_full.select([col_a, col_b]).drop_nulls()
|
|
397
|
+
a = sub[col_a].to_numpy().astype(float)
|
|
398
|
+
b = sub[col_b].to_numpy().astype(float)
|
|
399
|
+
n = len(a)
|
|
400
|
+
if n < min_obs or a.std() == 0 or b.std() == 0:
|
|
401
|
+
continue
|
|
402
|
+
corr, pval = sp.kendalltau(a, b)
|
|
403
|
+
rows.append(
|
|
404
|
+
{
|
|
405
|
+
group_by: key,
|
|
406
|
+
"correlation": float(cast(float, corr)),
|
|
407
|
+
"p_value": float(cast(float, pval)),
|
|
408
|
+
"n_obs": n,
|
|
409
|
+
}
|
|
410
|
+
)
|
|
411
|
+
data = pl.DataFrame(rows)
|
|
412
|
+
|
|
413
|
+
metadata: dict[str, Any] = {
|
|
414
|
+
"method": method.value,
|
|
415
|
+
"columns": [col_a, col_b],
|
|
416
|
+
"across": across,
|
|
417
|
+
"group_by": group_by,
|
|
418
|
+
"min_obs": min_obs,
|
|
419
|
+
"n_entities_computed": len(data),
|
|
420
|
+
}
|
|
421
|
+
return self._r._replace(data=data, has_geometry=False, metadata=metadata)
|
|
422
|
+
|
|
423
|
+
# Pooled mode — eager needed regardless (scipy.pearsonr/spearmanr/
|
|
424
|
+
# kendalltau all take raw numpy arrays). Collect ONCE here.
|
|
425
|
+
df = lf.collect()
|
|
274
426
|
n_obs = len(df)
|
|
275
|
-
rows: list[dict[str, Any]] = []
|
|
276
427
|
|
|
428
|
+
rows = []
|
|
277
429
|
for col_a, col_b in combinations(columns, 2):
|
|
278
430
|
sub = df.select([col_a, col_b]).drop_nulls()
|
|
279
431
|
a = sub[col_a].to_numpy().astype(float)
|
|
280
432
|
b = sub[col_b].to_numpy().astype(float)
|
|
281
|
-
|
|
433
|
+
if len(a) < min_obs or a.std() == 0 or b.std() == 0:
|
|
434
|
+
continue
|
|
282
435
|
if method is CorrelateMethod.PEARSON:
|
|
283
436
|
corr, pval = sp.pearsonr(a, b)
|
|
284
437
|
elif method is CorrelateMethod.SPEARMAN:
|
|
285
438
|
corr, pval = sp.spearmanr(a, b)
|
|
286
439
|
else:
|
|
287
440
|
corr, pval = sp.kendalltau(a, b)
|
|
288
|
-
|
|
289
441
|
rows.append(
|
|
290
442
|
{
|
|
291
443
|
"column_a": col_a,
|
|
@@ -296,8 +448,8 @@ class StatsNamespace:
|
|
|
296
448
|
)
|
|
297
449
|
|
|
298
450
|
data = pl.DataFrame(rows)
|
|
299
|
-
metadata
|
|
300
|
-
"method": method.value
|
|
451
|
+
metadata = {
|
|
452
|
+
"method": method.value,
|
|
301
453
|
"columns": columns,
|
|
302
454
|
"across": across,
|
|
303
455
|
"n_observations": n_obs,
|
|
@@ -312,7 +464,7 @@ class StatsNamespace:
|
|
|
312
464
|
against: float | None = None,
|
|
313
465
|
method: TestMethod | None = None,
|
|
314
466
|
significance_level: float = 0.05,
|
|
315
|
-
) ->
|
|
467
|
+
) -> Result:
|
|
316
468
|
"""Hypothesis test in three modes: group-based, period-based, single-sample.
|
|
317
469
|
|
|
318
470
|
Args:
|
|
@@ -470,7 +622,7 @@ class StatsNamespace:
|
|
|
470
622
|
from_period: int,
|
|
471
623
|
to_period: int,
|
|
472
624
|
method: ChangeMethod = ChangeMethod.ABSOLUTE,
|
|
473
|
-
) ->
|
|
625
|
+
) -> Result:
|
|
474
626
|
"""Change between two time periods per entity.
|
|
475
627
|
|
|
476
628
|
Args:
|
|
@@ -596,7 +748,7 @@ class StatsNamespace:
|
|
|
596
748
|
method: AnomalyCrossMethod | AnomalyTsMethod,
|
|
597
749
|
baseline: tuple[int, int] | None = None,
|
|
598
750
|
threshold: float = 2.0,
|
|
599
|
-
) ->
|
|
751
|
+
) -> Result:
|
|
600
752
|
"""Anomaly detection in cross-sectional or timeseries mode.
|
|
601
753
|
|
|
602
754
|
Args:
|
|
@@ -863,7 +1015,7 @@ class StatsNamespace:
|
|
|
863
1015
|
columns: dict[str, Any],
|
|
864
1016
|
method: SimilarityMethod = SimilarityMethod.EUCLIDEAN,
|
|
865
1017
|
top_n: int = 10,
|
|
866
|
-
) ->
|
|
1018
|
+
) -> Result:
|
|
867
1019
|
"""Find entities most similar to ``target`` across ``columns``.
|
|
868
1020
|
|
|
869
1021
|
Args:
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
from core_lens.base import BaseEntity
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class FarmEntity(BaseEntity):
|
|
5
|
+
"""Farms entity.
|
|
6
|
+
|
|
7
|
+
Backed by:
|
|
8
|
+
- Static: farms/static
|
|
9
|
+
- Annual: farms/annual
|
|
10
|
+
- SubAnnual: farms/sub_annual
|
|
11
|
+
|
|
12
|
+
All paths are relative to the AoI ``data_root``.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
@property
|
|
16
|
+
def key_cols(self) -> list[str]:
|
|
17
|
+
"""Get the property value.
|
|
18
|
+
|
|
19
|
+
Returns:
|
|
20
|
+
list[str]: The key columns for the entity.
|
|
21
|
+
|
|
22
|
+
"""
|
|
23
|
+
return ["farm_id"]
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def geometry_col(self) -> str:
|
|
27
|
+
"""Get the property value.
|
|
28
|
+
|
|
29
|
+
Returns:
|
|
30
|
+
str: The geometry column name.
|
|
31
|
+
|
|
32
|
+
"""
|
|
33
|
+
return "geometry"
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def static_path(self) -> str:
|
|
37
|
+
"""Get the property value.
|
|
38
|
+
|
|
39
|
+
Returns:
|
|
40
|
+
str: The relative path to the static data.
|
|
41
|
+
|
|
42
|
+
"""
|
|
43
|
+
return "farms/static.parquet"
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def annual_path(self) -> str:
|
|
47
|
+
"""Get the property value.
|
|
48
|
+
|
|
49
|
+
Returns:
|
|
50
|
+
str: The relative path to the annual data.
|
|
51
|
+
|
|
52
|
+
"""
|
|
53
|
+
return "farms/annual.parquet"
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def sub_annual_path(self) -> str:
|
|
57
|
+
"""Get the property value.
|
|
58
|
+
|
|
59
|
+
Returns:
|
|
60
|
+
str: The relative path to the monthly data.
|
|
61
|
+
|
|
62
|
+
"""
|
|
63
|
+
return "farms/sub_annual.parquet"
|
|
@@ -1409,14 +1409,14 @@ wheels = [
|
|
|
1409
1409
|
|
|
1410
1410
|
[[package]]
|
|
1411
1411
|
name = "gitpython"
|
|
1412
|
-
version = "3.1.
|
|
1412
|
+
version = "3.1.62"
|
|
1413
1413
|
source = { registry = "https://pypi.org/simple" }
|
|
1414
1414
|
dependencies = [
|
|
1415
1415
|
{ name = "gitdb" },
|
|
1416
1416
|
]
|
|
1417
|
-
sdist = { url = "https://files.pythonhosted.org/packages/
|
|
1417
|
+
sdist = { url = "https://files.pythonhosted.org/packages/e0/db/3ca813cbacb23ab6fe46ff38a9b5ef8e73e970c8051f2ce903aacafe0446/gitpython-3.1.62.tar.gz", hash = "sha256:1791de66309bc0c7cfca40bf8d2e3de7ca091cbf94e6051be1ad0722c61062af", size = 231728, upload-time = "2026-09-07T02:57:21.155Z" }
|
|
1418
1418
|
wheels = [
|
|
1419
|
-
{ url = "https://files.pythonhosted.org/packages/
|
|
1419
|
+
{ url = "https://files.pythonhosted.org/packages/d6/0b/29d7965215f8ef830a7ca1f42997fe13e5693d85e9edb18f938d063ef5f2/gitpython-3.1.62-py3-none-any.whl", hash = "sha256:7002251225e10e29d2e1f49e6532613fe5d5d9f0b6f1f02997a52b38fe56899e", size = 222753, upload-time = "2026-09-07T02:57:19.762Z" },
|
|
1420
1420
|
]
|
|
1421
1421
|
|
|
1422
1422
|
[[package]]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|