metaforecast 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
File without changes
File without changes
@@ -0,0 +1,320 @@
1
+ from typing import Union, Tuple, List, Optional
2
+
3
+ import pandas as pd
4
+ import lightgbm as lgb
5
+ from mlforecast import MLForecast
6
+ from statsforecast import StatsForecast
7
+
8
+ from sklearn.multioutput import MultiOutputRegressor as MIMO
9
+
10
+ from metaforecast.utils.normalization import Normalizations
11
+ from metaforecast.ensembles.base import BaseADE
12
+
13
+ DForDFTuple = Union[pd.DataFrame, Tuple[pd.DataFrame, pd.DataFrame]]
14
+
15
+
16
+ class ADE(BaseADE):
17
+ LGB_PARS = {'verbosity': -1, 'n_jobs': 1, 'linear_tree': True}
18
+
19
+ def __init__(self,
20
+ freq: str,
21
+ trim_ratio: float,
22
+ meta_lags: List[int],
23
+ trim_by_uid: bool = True,
24
+ meta_model=MIMO(lgb.LGBMRegressor(**LGB_PARS))):
25
+ """
26
+ :param trim_ratio:
27
+ :param meta_model:
28
+ """
29
+ self.frequency = freq
30
+
31
+ super().__init__(window_size=self.WINDOW_SIZE_BY_FREQ[self.frequency],
32
+ trim_ratio=trim_ratio,
33
+ trim_by_uid=trim_by_uid,
34
+ meta_model=meta_model)
35
+
36
+ self.model_names = None
37
+
38
+ self.meta_lags = meta_lags
39
+ self.lag_names = [f'lag{i}' for i in self.meta_lags]
40
+
41
+ self.meta_mlf = MLForecast(
42
+ models=[],
43
+ freq=self.frequency,
44
+ lags=self.meta_lags
45
+ )
46
+
47
+ self.meta_df = None
48
+ self.raw_meta_data = None
49
+ self.insample_scores = None
50
+ self.use_window = False
51
+
52
+ def fit(self, insample_fcst: pd.DataFrame, **kwargs):
53
+ """
54
+
55
+ """
56
+
57
+ self._fit(insample_fcst)
58
+
59
+ def _fit(self, insample_fcst):
60
+ if self.model_names is None:
61
+ self.model_names = insample_fcst.columns.to_list()
62
+ self.model_names = [x for x in self.model_names if x not in self.METADATA + ['h']]
63
+
64
+ self._set_n_models()
65
+
66
+ in_sample_loss_df = self._get_insample_loss(insample_fcst)
67
+
68
+ self.insample_scores = self.evaluate_base_fcst(insample_fcst=insample_fcst,
69
+ use_window=self.use_window)
70
+
71
+ self.raw_meta_data = self.meta_mlf.preprocess(in_sample_loss_df)
72
+
73
+ self.meta_df = self._process_meta_data(self.raw_meta_data)
74
+
75
+ x, y = self.meta_df
76
+ # print(y.isna().mean())
77
+ if y.isna().any().any():
78
+ y = y.ffill().bfill()
79
+
80
+ self.meta_model.fit(x, y)
81
+
82
+ def predict(self, preds: pd.DataFrame, train: pd.DataFrame, h: int):
83
+
84
+ fcst = self._predict(preds=preds, train=train, h=h)
85
+ fcst.name = self.alias
86
+
87
+ return fcst
88
+
89
+ def update_weights(self, fcst: pd.DataFrame):
90
+ raise NotImplementedError
91
+
92
+ def _predict(self, preds: pd.DataFrame, train: pd.DataFrame, h: int):
93
+ # could use ade.mlf.make_future_dataframe(h=4)
94
+ df_ext = train.merge(preds, on=['unique_id', 'ds'], how='outer')
95
+ df_ext = df_ext[self.METADATA]
96
+ df_ext['y'] = df_ext['y'].fillna(value=-1)
97
+
98
+ meta_dataset = self.meta_mlf.preprocess(df_ext)
99
+
100
+ weights = self._weights_by_uid(meta_dataset, h=h)
101
+
102
+ fcst = preds.apply(lambda x: self._weighted_average(x, weights), axis=1)
103
+
104
+ return fcst
105
+
106
+ def _get_insample_loss(self, insample_fcst: pd.DataFrame):
107
+ in_sample_loss = []
108
+ in_sample_uid = insample_fcst.copy().groupby('unique_id')
109
+ for uid, uid_df in in_sample_uid:
110
+ for mod in self.model_names:
111
+ uid_df[mod] = uid_df[mod] - uid_df['y']
112
+
113
+ in_sample_loss.append(uid_df)
114
+
115
+ in_sample_loss_df = pd.concat(in_sample_loss)
116
+
117
+ # first h forward
118
+ # could average all horizons
119
+ if 'h' in in_sample_loss_df.columns:
120
+ in_sample_loss_df = in_sample_loss_df.query('h==1').drop(columns=['h'])
121
+
122
+ return in_sample_loss_df
123
+
124
+ def _process_meta_data(self,
125
+ meta_data: pd.DataFrame,
126
+ return_X_y: bool = True) -> DForDFTuple:
127
+
128
+ lag_locs = meta_data.columns.str.startswith('lag')
129
+ lag_cols = meta_data.columns[lag_locs].to_list()
130
+
131
+ if return_X_y:
132
+ X_meta, Y_meta = meta_data[lag_cols], meta_data[self.model_names]
133
+ return X_meta, Y_meta
134
+ else:
135
+ meta_df = meta_data[lag_cols + self.model_names]
136
+
137
+ return meta_df
138
+
139
+ def _weights_by_uid(self, df: pd.DataFrame, h: int):
140
+ top_overall = self._get_top_k(self.insample_scores.mean())
141
+ top_by_uid = self.insample_scores.apply(self._get_top_k, axis=1)
142
+
143
+ uid_weights = {}
144
+ for uid, meta_uid_df in df.groupby('unique_id'):
145
+ if h > 1:
146
+ lags = meta_uid_df.head(-(h - 1)).tail(1)[self.lag_names]
147
+ else:
148
+ lags = meta_uid_df.tail(1)[self.lag_names]
149
+
150
+ meta_pred = self.meta_model.predict(lags)
151
+ meta_pred = pd.DataFrame(meta_pred, columns=self.model_names)
152
+
153
+ weights = self._weights_from_errors(meta_pred)
154
+
155
+ if self.trim_by_uid:
156
+ poor_models = [x not in top_by_uid[uid] for x in weights.index]
157
+ else:
158
+ poor_models = [x not in top_overall for x in weights.index]
159
+
160
+ weights[poor_models] = 0
161
+ weights /= weights.sum()
162
+
163
+ uid_weights[uid] = weights
164
+
165
+ weights_df = pd.DataFrame(uid_weights).T
166
+ weights_df.index.name = 'unique_id'
167
+
168
+ return weights_df
169
+
170
+ def _reweight_by_redundancy(self):
171
+ raise NotImplementedError
172
+
173
+ @staticmethod
174
+ def _weights_from_errors(meta_predictions: pd.DataFrame) -> pd.Series:
175
+ e_hat = meta_predictions.abs()
176
+
177
+ W = e_hat.apply(
178
+ func=lambda x: Normalizations.normalize_and_proportion(-x),
179
+ axis=1)
180
+
181
+ weight_s = W.iloc[0]
182
+
183
+ return weight_s
184
+
185
+
186
+ class GlobalADE(ADE):
187
+ ## todo in inference lags are the same of all
188
+ # could include past errors...
189
+
190
+ def __init__(self,
191
+ freq: str,
192
+ trim_ratio: float,
193
+ meta_lags: List[int],
194
+ trim_by_uid: bool = True,
195
+ meta_model=lgb.LGBMRegressor(**ADE.LGB_PARS)):
196
+
197
+ super().__init__(freq=freq,
198
+ trim_ratio=trim_ratio,
199
+ meta_lags=meta_lags,
200
+ trim_by_uid=trim_by_uid,
201
+ meta_model=meta_model)
202
+
203
+ self.alias = 'GADE'
204
+
205
+ def _process_meta_data(self,
206
+ meta_data: pd.DataFrame,
207
+ return_X_y: bool = True) -> DForDFTuple:
208
+
209
+ lag_locs = meta_data.columns.str.startswith('lag')
210
+ lag_cols = meta_data.columns[lag_locs].to_list()
211
+
212
+ df_melt = meta_data.drop(columns='y').melt(['unique_id', 'ds'] + lag_cols)
213
+ df_melt['unique_id'] = df_melt.apply(lambda x: f'{x["unique_id"]}_{x["variable"]}', axis=1)
214
+ df_melt = df_melt.rename(columns={'value': 'error'})
215
+
216
+ if return_X_y:
217
+ X_meta, Y_meta = df_melt[lag_cols], df_melt['error']
218
+ return X_meta, Y_meta
219
+ else:
220
+ meta_df = df_melt[lag_cols + ['error']]
221
+
222
+ return meta_df
223
+
224
+ def _weights_by_uid(self, df: pd.DataFrame, h: int):
225
+ top_overall = self._get_top_k(self.insample_scores.mean())
226
+ top_by_uid = self.insample_scores.apply(self._get_top_k, axis=1)
227
+
228
+ uid_weights = {}
229
+ for uid, meta_uid_df in df.groupby('unique_id'):
230
+ lags = meta_uid_df.head(-(h - 1)).tail(1)[self.lag_names]
231
+
232
+ meta_pred = self.meta_model.predict(lags)
233
+ meta_pred = pd.DataFrame(meta_pred, columns=self.model_names)
234
+
235
+ weights = self._weights_from_errors(meta_pred)
236
+
237
+ if self.trim_by_uid:
238
+ poor_models = [x not in top_by_uid[uid] for x in weights.index]
239
+ else:
240
+ poor_models = [x not in top_overall for x in weights.index]
241
+
242
+ weights[poor_models] = 0
243
+ weights /= weights.sum()
244
+
245
+ uid_weights[uid] = weights
246
+
247
+ weights_df = pd.DataFrame(uid_weights).T
248
+ weights_df.index.name = 'unique_id'
249
+
250
+ return weights_df
251
+
252
+ def _reweight_by_redundancy(self):
253
+ raise NotImplementedError
254
+
255
+ def update_weights(self, fcst: pd.DataFrame):
256
+ raise NotImplementedError
257
+
258
+
259
+ class MLForecastADE(ADE):
260
+
261
+ def __init__(self,
262
+ mlf: MLForecast,
263
+ trim_ratio: float,
264
+ sf: Optional[StatsForecast] = None,
265
+ meta_model=MIMO(lgb.LGBMRegressor(**ADE.LGB_PARS))):
266
+ """
267
+ :param trim_ratio:
268
+ :param meta_model:
269
+ """
270
+ self.mlf = mlf
271
+ self.sf = sf
272
+ self.frequency = self.mlf.ts.freq
273
+
274
+ super().__init__(freq=self.frequency,
275
+ trim_ratio=trim_ratio,
276
+ meta_model=meta_model,
277
+ meta_lags=self.mlf.ts.lags)
278
+
279
+ def fit(self, **kwargs):
280
+ """
281
+
282
+ """
283
+
284
+ insample_fcst = self.mlf.fcst_fitted_values_
285
+
286
+ if self.sf is not None:
287
+ self.sf.forecast(fitted=True, h=1)
288
+ insample_fcst_sf = self.sf.forecast_fitted_values()
289
+
290
+ insample_fcst = insample_fcst.merge(insample_fcst_sf.drop(columns='y'),
291
+ on=self.METADATA_NO_T)
292
+
293
+ self._fit(insample_fcst)
294
+
295
+ def predict(self, train: pd.DataFrame, h: int, **kwargs):
296
+ base_fcst = self.mlf.predict(h=h)
297
+
298
+ if self.sf is not None:
299
+ base_fcst_sf = self.sf.predict(h=h)
300
+
301
+ base_fcst = base_fcst.merge(base_fcst_sf, on=self.METADATA_NO_T)
302
+
303
+ fcst = self._predict(preds=base_fcst, train=train, h=h)
304
+
305
+ return fcst
306
+
307
+ def update_weights(self, fcst: pd.DataFrame):
308
+ raise NotImplementedError
309
+
310
+ def _reweight_by_redundancy(self):
311
+ raise NotImplementedError
312
+
313
+ def update_estimates(self, df: pd.DataFrame):
314
+ """
315
+ Updating loss statistics for dynamic model selection
316
+
317
+ :param df: dataset with actual values and predictions, similar to insample predictions
318
+ """
319
+
320
+ raise NotImplementedError
@@ -0,0 +1,351 @@
1
+ from abc import ABC, abstractmethod
2
+
3
+ import numpy as np
4
+ from neuralforecast.losses.numpy import smape
5
+ import pandas as pd
6
+
7
+ from metaforecast.utils.normalization import Normalizations
8
+ from metaforecast.ensembles.expert_loss import (SquaredLoss,
9
+ PinballLoss,
10
+ PercentageLoss,
11
+ AbsoluteLoss,
12
+ LogLoss)
13
+
14
+ EXPERT_LOSS = {
15
+ 'square': SquaredLoss,
16
+ 'pinball': PinballLoss,
17
+ 'percentage': PercentageLoss,
18
+ 'absolute': AbsoluteLoss,
19
+ 'log': LogLoss,
20
+ }
21
+
22
+
23
+ class ForecastingEnsemble(ABC):
24
+ """ ForecastingEnsemble
25
+
26
+ """
27
+
28
+ METADATA = ['unique_id', 'ds', 'y']
29
+ METADATA_NO_T = ['unique_id', 'ds']
30
+
31
+ WINDOW_SIZE_BY_FREQ = {
32
+ 'H': 48,
33
+ 'D': 14,
34
+ 'W': 16,
35
+ 'M': 12,
36
+ 'ME': 12,
37
+ 'MS': 12,
38
+ 'Q': 4,
39
+ 'QS': 4,
40
+ 'Y': 6,
41
+ '': -1,
42
+ }
43
+
44
+ def __init__(self):
45
+ super().__init__()
46
+
47
+ self.models = []
48
+ self.model_names = None
49
+ self.tot_n_models = -1
50
+ self.n_models = -1
51
+ self.n_poor_models = -1
52
+ self.trim_ratio = 0
53
+ self.window_size = 0
54
+
55
+ @abstractmethod
56
+ def fit(self, **kwargs):
57
+ """ fit
58
+
59
+ Fits the ensemble combination rule
60
+
61
+ Parameters
62
+ ----------
63
+ kwargs: Not defined
64
+ Whatever input value the ensemble takes.
65
+
66
+ Returns
67
+ -------
68
+ ForecastingEnsemble
69
+ self, optional
70
+
71
+ """
72
+ raise NotImplementedError
73
+
74
+ @abstractmethod
75
+ def predict(self, **kwargs):
76
+ """ predict
77
+
78
+ Predicts the weights of ensemble combination rule
79
+
80
+ Parameters
81
+ ----------
82
+ kwargs: Not defined
83
+ Whatever input value the ensemble takes.
84
+
85
+ Returns
86
+ -------
87
+ Array-like
88
+ Weights of each model in the ensemble
89
+
90
+ """
91
+ raise NotImplementedError
92
+
93
+ def update_weights(self, fcst: pd.DataFrame):
94
+ """
95
+ Updating loss statistics for dynamic model selection
96
+
97
+ param fcst: dataset with actual values and predictions, similar to insample predictions
98
+
99
+ """
100
+
101
+ raise NotImplementedError
102
+
103
+ def evaluate_base_fcst(self, insample_fcst: pd.DataFrame, use_window: bool):
104
+
105
+ all_scores, window_scores = {}, {}
106
+ in_sample_loss_g = insample_fcst.groupby('unique_id')
107
+ for uid, uid_df in in_sample_loss_g:
108
+
109
+ uid_a_loss, uid_w_loss = {}, {}
110
+ for m in self.model_names:
111
+ uid_a_loss[m] = smape(y=uid_df['y'], y_hat=uid_df[m])
112
+ try:
113
+ uid_w_loss[m] = smape(y=uid_df.tail(self.window_size)['y'],
114
+ y_hat=uid_df.tail(self.window_size)[m])
115
+ except AssertionError:
116
+ uid_w_loss[m] = np.nan
117
+
118
+ all_scores[uid] = uid_a_loss
119
+ window_scores[uid] = uid_w_loss
120
+
121
+ all_scr_df = pd.DataFrame(all_scores).T
122
+ wdw_scr_df = pd.DataFrame(window_scores).T
123
+
124
+ if use_window:
125
+ return wdw_scr_df
126
+ else:
127
+ return all_scr_df
128
+
129
+ @abstractmethod
130
+ def _weights_by_uid(self, **kwargs):
131
+ raise NotImplementedError
132
+
133
+ def _set_n_models(self):
134
+ self.tot_n_models = len(self.model_names)
135
+
136
+ self.n_models = int(self.trim_ratio * self.tot_n_models)
137
+ if self.n_models < 1:
138
+ self.n_models = 1
139
+
140
+ self.n_poor_models = self.tot_n_models - self.n_models
141
+
142
+ def _get_top_k(self, scores: pd.Series):
143
+ """
144
+ :param scores: models LOSS scores (to minimize)
145
+ """
146
+ return scores.sort_values().index.tolist()[:self.n_models]
147
+
148
+ @staticmethod
149
+ def _weights_from_errors(scores: pd.Series) -> pd.Series:
150
+
151
+ weights = Normalizations.normalize_and_proportion(-scores)
152
+
153
+ return weights
154
+
155
+ @staticmethod
156
+ def _weighted_average(pred: pd.Series, weights: pd.DataFrame):
157
+ w = weights.loc[pred['unique_id']]
158
+
159
+ wa = (pred[w.index] * w).sum()
160
+
161
+ return wa
162
+
163
+
164
+ class Mixture(ForecastingEnsemble):
165
+
166
+ def __init__(self,
167
+ loss_type: str,
168
+ gradient: bool,
169
+ trim_ratio: float,
170
+ weight_by_uid: bool):
171
+ self.alias = 'Mixture'
172
+
173
+ super().__init__()
174
+
175
+ assert loss_type in EXPERT_LOSS.keys()
176
+
177
+ self.gradient = gradient
178
+ self.loss_type = loss_type
179
+ self.trim_ratio = trim_ratio
180
+ self.weight_by_uid = weight_by_uid
181
+
182
+ self.eta = None
183
+ self.regret = None
184
+ self.weights = None
185
+ self.ensemble_fcst = None
186
+
187
+ self.uid_weights = {}
188
+ self.uid_coefficient = {}
189
+
190
+ def fit(self, insample_fcst: pd.DataFrame):
191
+
192
+ if self.model_names is None:
193
+ self.model_names = insample_fcst.columns.to_list()
194
+ self.model_names = [x for x in self.model_names if x not in self.METADATA + ['h']]
195
+
196
+ self._initialize_params(insample_fcst)
197
+ self._set_n_models()
198
+
199
+ if self.weight_by_uid:
200
+ self._fit_by_uid(insample_fcst)
201
+ else:
202
+ self._fit_all(insample_fcst)
203
+
204
+ def _fit_by_uid(self, insample_fcst: pd.DataFrame):
205
+ grouped_fcst = insample_fcst.groupby('unique_id')
206
+
207
+ for uid, fcst_uid in grouped_fcst:
208
+
209
+ y = fcst_uid['y'].values
210
+
211
+ fcst_uid = fcst_uid.reset_index(drop=True)
212
+ fcst_uid = fcst_uid.drop(columns=self.METADATA)
213
+ if 'h' in fcst_uid.columns:
214
+ fcst_uid = fcst_uid.drop(columns='h')
215
+
216
+ self._initialize_params(fcst_uid)
217
+
218
+ self._update_mixture(fcst_uid, y)
219
+
220
+ self.weights = pd.DataFrame(self.weights, columns=self.model_names)
221
+
222
+ self.uid_weights[uid] = self.weights.iloc[-1]
223
+ self.uid_coefficient[uid] = self._weights_from_regret()
224
+
225
+ def _fit_all(self, insample_fcst: pd.DataFrame):
226
+ # todo add sort back
227
+ # just commenting so the estimation match the r bridge
228
+ insample_fcst_ = insample_fcst.sort_values('ds')
229
+
230
+ uid_list = insample_fcst_['unique_id'].unique().tolist()
231
+
232
+ y = insample_fcst_['y'].values
233
+
234
+ fcst = insample_fcst_.reset_index(drop=True)
235
+ fcst = fcst.drop(columns=self.METADATA)
236
+ if 'h' in fcst.columns:
237
+ fcst = fcst.drop(columns='h')
238
+
239
+ self._update_mixture(fcst, y)
240
+ self.weights = pd.DataFrame(self.weights, columns=self.model_names)
241
+
242
+ for uid in uid_list:
243
+ self.uid_weights[uid] = self.weights.iloc[-1]
244
+ self.uid_coefficient[uid] = self._weights_from_regret()
245
+
246
+ def predict(self, fcst: pd.DataFrame, **kwargs):
247
+ weights = pd.DataFrame(self.uid_weights).T
248
+ # weights = pd.DataFrame(self.uid_coefficient).T
249
+
250
+ if self.trim_ratio < 1:
251
+ weights = self._weights_by_uid(weights)
252
+
253
+ fcst_c = fcst.apply(lambda x: self._weighted_average(x, weights), axis=1)
254
+ fcst_c.name = self.alias
255
+
256
+ return fcst_c
257
+
258
+ def _calc_loss(self, fcst: pd.Series, y: float, fcst_c: float):
259
+ if self.gradient:
260
+ loss = EXPERT_LOSS[self.loss_type].gradient(fcst=fcst, y=y, fcst_c=fcst_c)
261
+ else:
262
+ loss = EXPERT_LOSS[self.loss_type].loss(fcst=fcst, y=y)
263
+
264
+ return loss
265
+
266
+ def update_weights(self, fcst: pd.DataFrame):
267
+ raise NotImplementedError
268
+
269
+ def _initialize_params(self, fcst: pd.DataFrame):
270
+ raise NotImplementedError
271
+
272
+ def _weights_from_regret(self, **kwargs):
273
+ raise NotImplementedError
274
+
275
+ def _update_mixture(self, fcst: pd.DataFrame, y: np.ndarray, **kwargs):
276
+ raise NotImplementedError
277
+
278
+ def _weights_by_uid(self, weights: pd.DataFrame):
279
+ neg_w = -weights
280
+
281
+ top_overall = self._get_top_k(-weights.mean())
282
+ top_by_uid = neg_w.apply(self._get_top_k, axis=1)
283
+
284
+ uid_weights = {}
285
+ for uid, w in weights.iterrows():
286
+ if self.weight_by_uid:
287
+ poor_models = [x not in top_by_uid[uid] for x in w.index]
288
+ else:
289
+ poor_models = [x not in top_overall for x in w.index]
290
+
291
+ w[poor_models] = 0
292
+ w /= w.sum()
293
+
294
+ uid_weights[uid] = w
295
+
296
+ weights_df = pd.DataFrame(uid_weights).T
297
+ weights_df.sum()
298
+ weights_df.index.name = 'unique_id'
299
+
300
+ return weights_df
301
+
302
+ @staticmethod
303
+ def _calc_ensemble_fcst(fcst: pd.Series, weight: pd.Series):
304
+
305
+ # form the mixture and the prediction
306
+ mixture = weight / np.sum(weight)
307
+
308
+ fcst_c = np.sum(fcst * mixture)
309
+
310
+ return mixture, fcst_c
311
+
312
+
313
+ class BaseADE(ForecastingEnsemble):
314
+ """
315
+ ADE
316
+
317
+ Arbitrated Dynamic Ensemble
318
+
319
+ """
320
+
321
+ def __init__(self,
322
+ window_size: int,
323
+ trim_ratio: float,
324
+ trim_by_uid: bool,
325
+ meta_model):
326
+ """
327
+ : param window_size: No of recent observations used to trim ensemble
328
+ :param trim_ratio:
329
+ :param meta_model:
330
+ """
331
+
332
+ super().__init__()
333
+
334
+ self.window_size = window_size
335
+ self.trim_ratio = trim_ratio
336
+ self.trim_by_uid = trim_by_uid
337
+ self.meta_model = meta_model
338
+
339
+ self.alias = 'ADE'
340
+
341
+ def fit(self, **kwargs):
342
+ raise NotImplementedError
343
+
344
+ def predict(self, **kwargs):
345
+ raise NotImplementedError
346
+
347
+ def update_weights(self, **kwargs):
348
+ raise NotImplementedError
349
+
350
+ def _weights_by_uid(self, **kwargs):
351
+ raise NotImplementedError