edmkit 0.0.4__tar.gz → 0.0.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: edmkit
3
- Version: 0.0.4
3
+ Version: 0.0.5
4
4
  Summary: Simple EDM (Empirical Dynamic Modeling) library
5
5
  Author: FUJISHIGE TEMMA
6
6
  Author-email: FUJISHIGE TEMMA <tenma.x0@gmail.com>
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "edmkit"
3
- version = "0.0.4"
3
+ version = "0.0.5"
4
4
  description = "Simple EDM (Empirical Dynamic Modeling) library"
5
5
  authors = [{ name = "FUJISHIGE TEMMA", email = "tenma.x0@gmail.com" }]
6
6
  readme = "README.md"
@@ -195,8 +195,11 @@ def select(
195
195
  ) -> tuple[int, int, float]:
196
196
  """Select best (E, tau) from scan results.
197
197
 
198
- Aggregates over the fold axis (axis=2) with nanmean, then
199
- finds the (E, tau) combination with the highest mean score.
198
+ Ranks each (E, tau) by ``mean - SE`` where SE is the standard error
199
+ of the mean across folds. This penalises combinations whose scores
200
+ vary widely across folds (unstable predictions) and those with fewer
201
+ valid folds (less certainty), favouring parameters we are *confident*
202
+ perform well.
200
203
 
201
204
  Parameters
202
205
  ----------
@@ -210,15 +213,31 @@ def select(
210
213
  Returns
211
214
  -------
212
215
  (best_E, best_tau, best_score)
216
+ ``best_score`` is the mean over folds (not the adjusted value)
217
+ so that it remains directly interpretable.
213
218
  """
214
- valid_counts = np.sum(~np.isnan(scores), axis=2)
215
- summed_scores = np.nansum(scores, axis=2)
219
+ K = np.sum(~np.isnan(scores), axis=2)
220
+ nan_out = np.full(scores.shape[:2], np.nan)
221
+
216
222
  mean_scores = np.divide(
217
- summed_scores,
218
- valid_counts,
219
- out=np.full(summed_scores.shape, np.nan, dtype=float),
220
- where=valid_counts > 0,
223
+ np.nansum(scores, axis=2),
224
+ K,
225
+ out=nan_out.copy(),
226
+ where=K > 0,
221
227
  )
222
- flat_idx = int(np.nanargmax(mean_scores))
223
- e_idx, t_idx = np.unravel_index(flat_idx, mean_scores.shape)
228
+
229
+ # SE = sqrt(var / K) = sqrt(sum_sq / (K * (K - 1)))
230
+ sum_sq = np.nansum((scores - mean_scores[:, :, None]) ** 2, axis=2)
231
+ se = np.sqrt(
232
+ np.divide(
233
+ sum_sq,
234
+ K * np.maximum(K - 1, 1),
235
+ out=np.zeros_like(nan_out),
236
+ where=K > 1,
237
+ )
238
+ )
239
+
240
+ adjusted = mean_scores - se
241
+ flat_idx = int(np.nanargmax(adjusted))
242
+ e_idx, t_idx = np.unravel_index(flat_idx, adjusted.shape)
224
243
  return E[e_idx], tau[t_idx], float(mean_scores[e_idx, t_idx])
@@ -103,6 +103,8 @@ def expanding_folds(
103
103
 
104
104
  if stride is None:
105
105
  stride = validation_size
106
+ if stride <= 0:
107
+ raise ValueError(f"stride must be positive, got {stride}")
106
108
 
107
109
  folds: list[Fold] = []
108
110
  validation_start = initial_train_size + gap
@@ -156,6 +158,8 @@ def sliding_folds(
156
158
 
157
159
  if stride is None:
158
160
  stride = validation_size
161
+ if stride <= 0:
162
+ raise ValueError(f"stride must be positive, got {stride}")
159
163
 
160
164
  folds: list[Fold] = []
161
165
  validation_start = train_size + gap
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes