downsampler 0.3.1__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. {downsampler-0.3.1/src/downsampler.egg-info → downsampler-0.4.0}/PKG-INFO +1 -1
  2. {downsampler-0.3.1 → downsampler-0.4.0}/pyproject.toml +1 -1
  3. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/__init__.py +2 -0
  4. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/gaps.py +37 -0
  5. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/lttb.py +38 -14
  6. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/m4.py +13 -2
  7. {downsampler-0.3.1 → downsampler-0.4.0/src/downsampler.egg-info}/PKG-INFO +1 -1
  8. {downsampler-0.3.1 → downsampler-0.4.0}/tests/test_gaps.py +28 -0
  9. {downsampler-0.3.1 → downsampler-0.4.0}/tests/test_lttb.py +53 -5
  10. {downsampler-0.3.1 → downsampler-0.4.0}/tests/test_m4.py +3 -2
  11. {downsampler-0.3.1 → downsampler-0.4.0}/LICENSE +0 -0
  12. {downsampler-0.3.1 → downsampler-0.4.0}/README.md +0 -0
  13. {downsampler-0.3.1 → downsampler-0.4.0}/setup.cfg +0 -0
  14. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/aggregators.py +0 -0
  15. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/config.py +0 -0
  16. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/core.py +0 -0
  17. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/edges.py +0 -0
  18. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/fidelity/__init__.py +0 -0
  19. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/fidelity/comparison.py +0 -0
  20. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/fidelity/metrics.py +0 -0
  21. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/ranged.py +0 -0
  22. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler/utils.py +0 -0
  23. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler.egg-info/SOURCES.txt +0 -0
  24. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler.egg-info/dependency_links.txt +0 -0
  25. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler.egg-info/requires.txt +0 -0
  26. {downsampler-0.3.1 → downsampler-0.4.0}/src/downsampler.egg-info/top_level.txt +0 -0
  27. {downsampler-0.3.1 → downsampler-0.4.0}/tests/test_aggregators.py +0 -0
  28. {downsampler-0.3.1 → downsampler-0.4.0}/tests/test_core.py +0 -0
  29. {downsampler-0.3.1 → downsampler-0.4.0}/tests/test_edges.py +0 -0
  30. {downsampler-0.3.1 → downsampler-0.4.0}/tests/test_fidelity.py +0 -0
  31. {downsampler-0.3.1 → downsampler-0.4.0}/tests/test_ranged.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: downsampler
3
- Version: 0.3.1
3
+ Version: 0.4.0
4
4
  Summary: Timeseries DataFrame downsampling with LTTB, aggregation methods, gap handling, and fidelity testing
5
5
  Author-email: Eelco Doornbos <eelco.doornbos@knmi.nl>
6
6
  License-Expression: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "downsampler"
7
- version = "0.3.1"
7
+ version = "0.4.0"
8
8
  description = "Timeseries DataFrame downsampling with LTTB, aggregation methods, gap handling, and fidelity testing"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -47,6 +47,7 @@ from downsampler.gaps import (
47
47
  split_at_gaps,
48
48
  wrap_in_nans,
49
49
  mark_gaps_in_dataframe,
50
+ gap_marker_frame,
50
51
  interpolate_small_gaps,
51
52
  concatenate_with_gap_markers,
52
53
  )
@@ -87,6 +88,7 @@ __all__ = [
87
88
  "split_at_gaps",
88
89
  "wrap_in_nans",
89
90
  "mark_gaps_in_dataframe",
91
+ "gap_marker_frame",
90
92
  "interpolate_small_gaps",
91
93
  "concatenate_with_gap_markers",
92
94
  # LTTB
@@ -414,6 +414,43 @@ def _interpolate_column_small_gaps(
414
414
  return filled
415
415
 
416
416
 
417
+ def gap_marker_frame(
418
+ index: pd.DatetimeIndex,
419
+ columns: list[str],
420
+ ) -> pd.DataFrame:
421
+ """NaN marker rows bracketing a window that yielded no usable points.
422
+
423
+ Returned in place of an empty frame when the input covers a real time
424
+ window but every sample in it is unusable — e.g. a batch whose target
425
+ column is entirely NaN (fill-encoded), or one where every segment was
426
+ dropped as too short. Returning nothing there would be indistinguishable
427
+ from "this period was never processed": a plot drawn from the stored
428
+ output connects straight across the window, showing a confident straight
429
+ line (or, zoomed out, a quiet period) where the truth is missing data.
430
+
431
+ Marking both ends of the window rather than a single point makes the gap's
432
+ *extent* explicit, so consumers see where invalid data began and ended
433
+ instead of a single break of unknown width. A one-sample window yields one
434
+ marker row.
435
+
436
+ Args:
437
+ index: DatetimeIndex of the input that produced no output.
438
+ columns: Output columns the caller would otherwise have produced.
439
+
440
+ Returns:
441
+ DataFrame of all-NaN rows at the first and last input timestamps, or
442
+ an empty frame when ``index`` is empty (nothing is knowable then).
443
+ """
444
+ if len(index) == 0:
445
+ return pd.DataFrame(columns=columns)
446
+
447
+ marker_times = [index[0]] if len(index) == 1 else [index[0], index[-1]]
448
+ return pd.DataFrame(
449
+ {col: [np.nan] * len(marker_times) for col in columns},
450
+ index=pd.DatetimeIndex(marker_times, name=index.name),
451
+ )
452
+
453
+
417
454
  def concatenate_with_gap_markers(
418
455
  segments: list[pd.DataFrame],
419
456
  offset: pd.Timedelta = pd.Timedelta('0.1s'),
@@ -5,7 +5,12 @@ import numpy as np
5
5
  import lttbc
6
6
 
7
7
  from downsampler.config import DownsampleConfig, EdgeHandling
8
- from downsampler.gaps import split_at_gaps, interpolate_small_gaps, concatenate_with_gap_markers
8
+ from downsampler.gaps import (
9
+ split_at_gaps,
10
+ interpolate_small_gaps,
11
+ concatenate_with_gap_markers,
12
+ gap_marker_frame,
13
+ )
9
14
  from downsampler.edges import apply_edge_handling
10
15
  from downsampler.utils import parse_cadence, get_numeric_columns, compute_output_points, filter_columns
11
16
 
@@ -71,7 +76,11 @@ def downsample_lttb(
71
76
  # data exactly like missing-row gaps.
72
77
  df_valid = df_in.dropna(subset=[target_column])
73
78
  if df_valid.empty:
74
- return pd.DataFrame(columns=df_in.columns)
79
+ # The window is covered but wholly unusable — mark it rather than
80
+ # returning nothing, so the gap survives into the stored output.
81
+ return gap_marker_frame(
82
+ df_in.index, _output_columns(df_in, target_column, include_columns)
83
+ )
75
84
 
76
85
  # Interpolate small gaps so LTTB receives continuous input
77
86
  df_interp = interpolate_small_gaps(df_valid, gap_threshold, source_cadence)
@@ -94,11 +103,36 @@ def downsample_lttb(
94
103
  resampled_segments.append(resampled)
95
104
 
96
105
  if not resampled_segments:
97
- return pd.DataFrame(columns=df_in.columns)
106
+ return gap_marker_frame(
107
+ df_in.index, _output_columns(df_in, target_column, include_columns)
108
+ )
98
109
 
99
110
  return concatenate_with_gap_markers(resampled_segments)
100
111
 
101
112
 
113
+ def _output_columns(
114
+ df: pd.DataFrame,
115
+ target_column: str,
116
+ include_columns: list[str] | None = None,
117
+ ) -> list[str]:
118
+ """Columns an LTTB run emits: the target plus carried numeric columns.
119
+
120
+ Shared by the normal path and the all-gap marker path so that a batch
121
+ which produced only markers has the same schema as one that produced
122
+ data — a storage writer must not see the column set change with content.
123
+ """
124
+ keep = [target_column]
125
+ for col in df.columns:
126
+ if col in ('time', 'time_num', target_column):
127
+ continue
128
+ if include_columns is not None and col not in include_columns:
129
+ continue
130
+ if not pd.api.types.is_numeric_dtype(df[col]):
131
+ continue
132
+ keep.append(col)
133
+ return keep
134
+
135
+
102
136
  def _lttb_single_segment(
103
137
  df: pd.DataFrame,
104
138
  target_column: str,
@@ -160,17 +194,7 @@ def _lttb_single_segment(
160
194
  # original datetime index, no interpolation.
161
195
  selected = df_work[df_work['time_num'].isin(x_down)]
162
196
 
163
- keep = [target_column]
164
- for col in df.columns:
165
- if col in ('time', 'time_num', target_column):
166
- continue
167
- if include_columns is not None and col not in include_columns:
168
- continue
169
- if not pd.api.types.is_numeric_dtype(df[col]):
170
- continue
171
- keep.append(col)
172
-
173
- return selected[keep].copy()
197
+ return selected[_output_columns(df, target_column, include_columns)].copy()
174
198
 
175
199
 
176
200
  def downsample_lttb_with_config(
@@ -14,7 +14,11 @@ import pandas as pd
14
14
  import numpy as np
15
15
 
16
16
  from downsampler.config import DownsampleConfig
17
- from downsampler.gaps import split_at_gaps, concatenate_with_gap_markers
17
+ from downsampler.gaps import (
18
+ split_at_gaps,
19
+ concatenate_with_gap_markers,
20
+ gap_marker_frame,
21
+ )
18
22
  from downsampler.edges import apply_edge_handling
19
23
  from downsampler.utils import parse_cadence, get_numeric_columns, compute_output_points, filter_columns
20
24
 
@@ -92,7 +96,14 @@ def downsample_m4(
92
96
  resampled_segments.append(resampled)
93
97
 
94
98
  if not resampled_segments:
95
- return pd.DataFrame(columns=df_in.columns)
99
+ # Covered but unusable (every segment too short, or all-NaN input):
100
+ # mark the window so the gap survives into the stored output rather
101
+ # than reading as "never processed".
102
+ if include_columns is None:
103
+ columns = get_numeric_columns(df_in)
104
+ else:
105
+ columns = [c for c in include_columns if c in df_in.columns]
106
+ return gap_marker_frame(df_in.index, columns)
96
107
 
97
108
  return concatenate_with_gap_markers(resampled_segments)
98
109
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: downsampler
3
- Version: 0.3.1
3
+ Version: 0.4.0
4
4
  Summary: Timeseries DataFrame downsampling with LTTB, aggregation methods, gap handling, and fidelity testing
5
5
  Author-email: Eelco Doornbos <eelco.doornbos@knmi.nl>
6
6
  License-Expression: MIT
@@ -10,6 +10,7 @@ from downsampler.gaps import (
10
10
  split_at_gaps,
11
11
  iter_segments,
12
12
  wrap_in_nans,
13
+ gap_marker_frame,
13
14
  mark_gaps_in_dataframe,
14
15
  has_gaps,
15
16
  count_gaps,
@@ -392,3 +393,30 @@ class TestConcatenateWithGapMarkers:
392
393
 
393
394
  expected_marker_time = seg1.index[-1] + offset
394
395
  assert result.index[1] == expected_marker_time
396
+
397
+
398
+ class TestGapMarkerFrame:
399
+ """Marker rows standing in for a window that produced no usable output."""
400
+
401
+ def test_marks_both_ends(self):
402
+ idx = pd.date_range('2024-01-01', periods=1440, freq='1min')
403
+ result = gap_marker_frame(idx, ['a', 'b'])
404
+
405
+ assert list(result.columns) == ['a', 'b']
406
+ assert result.isna().all().all()
407
+ assert list(result.index) == [idx[0], idx[-1]]
408
+
409
+ def test_single_sample_window(self):
410
+ idx = pd.date_range('2024-01-01', periods=1, freq='1min')
411
+ assert len(gap_marker_frame(idx, ['a'])) == 1
412
+
413
+ def test_empty_index_yields_empty_frame(self):
414
+ """No input means no known window — nothing to assert about it."""
415
+ result = gap_marker_frame(pd.DatetimeIndex([]), ['a'])
416
+
417
+ assert len(result) == 0
418
+ assert list(result.columns) == ['a']
419
+
420
+ def test_preserves_index_name(self):
421
+ idx = pd.date_range('2024-01-01', periods=10, freq='1min', name='time')
422
+ assert gap_marker_frame(idx, ['a']).index.name == 'time'
@@ -81,8 +81,10 @@ class TestDownsampleLttb:
81
81
  min_points_per_segment=3
82
82
  )
83
83
 
84
- # Should return empty or minimal result
85
- assert len(result) == 0
84
+ # The segment is dropped as too short, but the window it covered is
85
+ # marked rather than silently vanishing.
86
+ assert result['value'].isna().all()
87
+ assert list(result.index) == [small_df.index[0], small_df.index[-1]]
86
88
 
87
89
 
88
90
  class TestLttbGapHandling:
@@ -211,15 +213,61 @@ class TestLttbNanEncodedGaps:
211
213
  assert pd.Timestamp('2024-01-01 00:59') < markers.index[0]
212
214
  assert markers.index[0] < pd.Timestamp('2024-01-01 01:45')
213
215
 
214
- def test_all_nan_target_returns_empty(self):
215
- """A frame whose target column is entirely NaN yields no output."""
216
+ def test_all_nan_target_marks_the_window(self):
217
+ """An all-NaN target marks its window instead of vanishing.
218
+
219
+ Returning nothing would make a covered-but-unusable period
220
+ indistinguishable from one that was never processed, and a plot drawn
221
+ from the stored output would connect straight across it.
222
+ """
216
223
  idx = pd.date_range('2024-01-01', periods=100, freq='1min')
217
224
  df = pd.DataFrame({'signal': np.nan, 'other': 1.0}, index=idx)
218
225
 
219
226
  result = downsample_lttb(df, target_column='signal', target_cadence='PT15M')
220
227
 
221
- assert len(result) == 0
222
228
  assert list(result.columns) == ['signal', 'other']
229
+ assert result.isna().all().all()
230
+ # Both ends marked, so the gap's extent is explicit.
231
+ assert list(result.index) == [idx[0], idx[-1]]
232
+
233
+ def test_all_nan_target_single_sample_marks_once(self):
234
+ """A one-sample window needs only one marker."""
235
+ idx = pd.date_range('2024-01-01', periods=1, freq='1min')
236
+ df = pd.DataFrame({'signal': [np.nan], 'other': [1.0]}, index=idx)
237
+
238
+ result = downsample_lttb(df, target_column='signal', target_cadence='PT15M')
239
+
240
+ assert len(result) == 1
241
+ assert result.index[0] == idx[0]
242
+
243
+ def test_empty_input_still_returns_empty(self):
244
+ """With no input there is no window to mark."""
245
+ df = pd.DataFrame(
246
+ {'signal': [], 'other': []},
247
+ index=pd.DatetimeIndex([], name='time'),
248
+ )
249
+
250
+ result = downsample_lttb(df, target_column='signal', target_cadence='PT15M')
251
+
252
+ assert len(result) == 0
253
+
254
+ def test_marked_window_matches_normal_batch_schema(self):
255
+ """A marker-only batch has the same columns as a batch with data.
256
+
257
+ A storage writer must not see the column set change with content.
258
+ """
259
+ idx = pd.date_range('2024-01-01', periods=100, freq='1min')
260
+ good = pd.DataFrame(
261
+ {'signal': np.sin(np.arange(100) / 5.0), 'other': 1.0, 'label': 'x'},
262
+ index=idx,
263
+ )
264
+ bad = good.copy()
265
+ bad['signal'] = np.nan
266
+
267
+ normal = downsample_lttb(good, target_column='signal', target_cadence='PT15M')
268
+ marked = downsample_lttb(bad, target_column='signal', target_cadence='PT15M')
269
+
270
+ assert list(marked.columns) == list(normal.columns)
223
271
 
224
272
  def test_include_column_own_nan_holes(self):
225
273
  """Include columns interpolate from their own valid samples only."""
@@ -154,8 +154,9 @@ class TestDownsampleM4:
154
154
  min_points_per_segment=3
155
155
  )
156
156
 
157
- # Should return empty result
158
- assert len(result) == 0
157
+ # The segment is dropped as too short, but its window is marked.
158
+ assert result['value'].isna().all()
159
+ assert len(result) == 2
159
160
 
160
161
  def test_output_has_irregular_timestamps(self, sine_df):
161
162
  """Test that M4 produces irregular timestamps (not aligned to grid)."""
File without changes
File without changes
File without changes