metaforecast 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- metaforecast/__init__.py +0 -0
- metaforecast/ensembles/__init__.py +0 -0
- metaforecast/ensembles/ade.py +320 -0
- metaforecast/ensembles/base.py +351 -0
- metaforecast/ensembles/expert_loss.py +86 -0
- metaforecast/ensembles/mlewa.py +71 -0
- metaforecast/ensembles/mlpol.py +75 -0
- metaforecast/ensembles/static.py +58 -0
- metaforecast/ensembles/windowing.py +97 -0
- metaforecast/longhorizon/__init__.py +0 -0
- metaforecast/longhorizon/ftn.py +162 -0
- metaforecast/synth/__init__.py +0 -0
- metaforecast/synth/generators/__init__.py +0 -0
- metaforecast/synth/generators/_base.py +181 -0
- metaforecast/synth/generators/dba.py +56 -0
- metaforecast/synth/generators/jittering.py +24 -0
- metaforecast/synth/generators/kernelsynth.py +118 -0
- metaforecast/synth/generators/mbb.py +82 -0
- metaforecast/synth/generators/scaling.py +21 -0
- metaforecast/synth/generators/tsmixup.py +78 -0
- metaforecast/synth/generators/warping_mag.py +44 -0
- metaforecast/synth/generators/warping_time.py +43 -0
- metaforecast/utils/__init__.py +0 -0
- metaforecast/utils/barycenters.py +67 -0
- metaforecast/utils/data.py +23 -0
- metaforecast/utils/log.py +16 -0
- metaforecast/utils/normalization.py +25 -0
- metaforecast/utils/windows.py +78 -0
- metaforecast-0.1.0.dist-info/METADATA +15 -0
- metaforecast-0.1.0.dist-info/RECORD +31 -0
- metaforecast-0.1.0.dist-info/WHEEL +4 -0
metaforecast/__init__.py
ADDED
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
from typing import Union, Tuple, List, Optional
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import lightgbm as lgb
|
|
5
|
+
from mlforecast import MLForecast
|
|
6
|
+
from statsforecast import StatsForecast
|
|
7
|
+
|
|
8
|
+
from sklearn.multioutput import MultiOutputRegressor as MIMO
|
|
9
|
+
|
|
10
|
+
from metaforecast.utils.normalization import Normalizations
|
|
11
|
+
from metaforecast.ensembles.base import BaseADE
|
|
12
|
+
|
|
13
|
+
DForDFTuple = Union[pd.DataFrame, Tuple[pd.DataFrame, pd.DataFrame]]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ADE(BaseADE):
|
|
17
|
+
LGB_PARS = {'verbosity': -1, 'n_jobs': 1, 'linear_tree': True}
|
|
18
|
+
|
|
19
|
+
def __init__(self,
|
|
20
|
+
freq: str,
|
|
21
|
+
trim_ratio: float,
|
|
22
|
+
meta_lags: List[int],
|
|
23
|
+
trim_by_uid: bool = True,
|
|
24
|
+
meta_model=MIMO(lgb.LGBMRegressor(**LGB_PARS))):
|
|
25
|
+
"""
|
|
26
|
+
:param trim_ratio:
|
|
27
|
+
:param meta_model:
|
|
28
|
+
"""
|
|
29
|
+
self.frequency = freq
|
|
30
|
+
|
|
31
|
+
super().__init__(window_size=self.WINDOW_SIZE_BY_FREQ[self.frequency],
|
|
32
|
+
trim_ratio=trim_ratio,
|
|
33
|
+
trim_by_uid=trim_by_uid,
|
|
34
|
+
meta_model=meta_model)
|
|
35
|
+
|
|
36
|
+
self.model_names = None
|
|
37
|
+
|
|
38
|
+
self.meta_lags = meta_lags
|
|
39
|
+
self.lag_names = [f'lag{i}' for i in self.meta_lags]
|
|
40
|
+
|
|
41
|
+
self.meta_mlf = MLForecast(
|
|
42
|
+
models=[],
|
|
43
|
+
freq=self.frequency,
|
|
44
|
+
lags=self.meta_lags
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
self.meta_df = None
|
|
48
|
+
self.raw_meta_data = None
|
|
49
|
+
self.insample_scores = None
|
|
50
|
+
self.use_window = False
|
|
51
|
+
|
|
52
|
+
def fit(self, insample_fcst: pd.DataFrame, **kwargs):
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
self._fit(insample_fcst)
|
|
58
|
+
|
|
59
|
+
def _fit(self, insample_fcst):
|
|
60
|
+
if self.model_names is None:
|
|
61
|
+
self.model_names = insample_fcst.columns.to_list()
|
|
62
|
+
self.model_names = [x for x in self.model_names if x not in self.METADATA + ['h']]
|
|
63
|
+
|
|
64
|
+
self._set_n_models()
|
|
65
|
+
|
|
66
|
+
in_sample_loss_df = self._get_insample_loss(insample_fcst)
|
|
67
|
+
|
|
68
|
+
self.insample_scores = self.evaluate_base_fcst(insample_fcst=insample_fcst,
|
|
69
|
+
use_window=self.use_window)
|
|
70
|
+
|
|
71
|
+
self.raw_meta_data = self.meta_mlf.preprocess(in_sample_loss_df)
|
|
72
|
+
|
|
73
|
+
self.meta_df = self._process_meta_data(self.raw_meta_data)
|
|
74
|
+
|
|
75
|
+
x, y = self.meta_df
|
|
76
|
+
# print(y.isna().mean())
|
|
77
|
+
if y.isna().any().any():
|
|
78
|
+
y = y.ffill().bfill()
|
|
79
|
+
|
|
80
|
+
self.meta_model.fit(x, y)
|
|
81
|
+
|
|
82
|
+
def predict(self, preds: pd.DataFrame, train: pd.DataFrame, h: int):
|
|
83
|
+
|
|
84
|
+
fcst = self._predict(preds=preds, train=train, h=h)
|
|
85
|
+
fcst.name = self.alias
|
|
86
|
+
|
|
87
|
+
return fcst
|
|
88
|
+
|
|
89
|
+
def update_weights(self, fcst: pd.DataFrame):
|
|
90
|
+
raise NotImplementedError
|
|
91
|
+
|
|
92
|
+
def _predict(self, preds: pd.DataFrame, train: pd.DataFrame, h: int):
|
|
93
|
+
# could use ade.mlf.make_future_dataframe(h=4)
|
|
94
|
+
df_ext = train.merge(preds, on=['unique_id', 'ds'], how='outer')
|
|
95
|
+
df_ext = df_ext[self.METADATA]
|
|
96
|
+
df_ext['y'] = df_ext['y'].fillna(value=-1)
|
|
97
|
+
|
|
98
|
+
meta_dataset = self.meta_mlf.preprocess(df_ext)
|
|
99
|
+
|
|
100
|
+
weights = self._weights_by_uid(meta_dataset, h=h)
|
|
101
|
+
|
|
102
|
+
fcst = preds.apply(lambda x: self._weighted_average(x, weights), axis=1)
|
|
103
|
+
|
|
104
|
+
return fcst
|
|
105
|
+
|
|
106
|
+
def _get_insample_loss(self, insample_fcst: pd.DataFrame):
|
|
107
|
+
in_sample_loss = []
|
|
108
|
+
in_sample_uid = insample_fcst.copy().groupby('unique_id')
|
|
109
|
+
for uid, uid_df in in_sample_uid:
|
|
110
|
+
for mod in self.model_names:
|
|
111
|
+
uid_df[mod] = uid_df[mod] - uid_df['y']
|
|
112
|
+
|
|
113
|
+
in_sample_loss.append(uid_df)
|
|
114
|
+
|
|
115
|
+
in_sample_loss_df = pd.concat(in_sample_loss)
|
|
116
|
+
|
|
117
|
+
# first h forward
|
|
118
|
+
# could average all horizons
|
|
119
|
+
if 'h' in in_sample_loss_df.columns:
|
|
120
|
+
in_sample_loss_df = in_sample_loss_df.query('h==1').drop(columns=['h'])
|
|
121
|
+
|
|
122
|
+
return in_sample_loss_df
|
|
123
|
+
|
|
124
|
+
def _process_meta_data(self,
|
|
125
|
+
meta_data: pd.DataFrame,
|
|
126
|
+
return_X_y: bool = True) -> DForDFTuple:
|
|
127
|
+
|
|
128
|
+
lag_locs = meta_data.columns.str.startswith('lag')
|
|
129
|
+
lag_cols = meta_data.columns[lag_locs].to_list()
|
|
130
|
+
|
|
131
|
+
if return_X_y:
|
|
132
|
+
X_meta, Y_meta = meta_data[lag_cols], meta_data[self.model_names]
|
|
133
|
+
return X_meta, Y_meta
|
|
134
|
+
else:
|
|
135
|
+
meta_df = meta_data[lag_cols + self.model_names]
|
|
136
|
+
|
|
137
|
+
return meta_df
|
|
138
|
+
|
|
139
|
+
def _weights_by_uid(self, df: pd.DataFrame, h: int):
|
|
140
|
+
top_overall = self._get_top_k(self.insample_scores.mean())
|
|
141
|
+
top_by_uid = self.insample_scores.apply(self._get_top_k, axis=1)
|
|
142
|
+
|
|
143
|
+
uid_weights = {}
|
|
144
|
+
for uid, meta_uid_df in df.groupby('unique_id'):
|
|
145
|
+
if h > 1:
|
|
146
|
+
lags = meta_uid_df.head(-(h - 1)).tail(1)[self.lag_names]
|
|
147
|
+
else:
|
|
148
|
+
lags = meta_uid_df.tail(1)[self.lag_names]
|
|
149
|
+
|
|
150
|
+
meta_pred = self.meta_model.predict(lags)
|
|
151
|
+
meta_pred = pd.DataFrame(meta_pred, columns=self.model_names)
|
|
152
|
+
|
|
153
|
+
weights = self._weights_from_errors(meta_pred)
|
|
154
|
+
|
|
155
|
+
if self.trim_by_uid:
|
|
156
|
+
poor_models = [x not in top_by_uid[uid] for x in weights.index]
|
|
157
|
+
else:
|
|
158
|
+
poor_models = [x not in top_overall for x in weights.index]
|
|
159
|
+
|
|
160
|
+
weights[poor_models] = 0
|
|
161
|
+
weights /= weights.sum()
|
|
162
|
+
|
|
163
|
+
uid_weights[uid] = weights
|
|
164
|
+
|
|
165
|
+
weights_df = pd.DataFrame(uid_weights).T
|
|
166
|
+
weights_df.index.name = 'unique_id'
|
|
167
|
+
|
|
168
|
+
return weights_df
|
|
169
|
+
|
|
170
|
+
def _reweight_by_redundancy(self):
|
|
171
|
+
raise NotImplementedError
|
|
172
|
+
|
|
173
|
+
@staticmethod
|
|
174
|
+
def _weights_from_errors(meta_predictions: pd.DataFrame) -> pd.Series:
|
|
175
|
+
e_hat = meta_predictions.abs()
|
|
176
|
+
|
|
177
|
+
W = e_hat.apply(
|
|
178
|
+
func=lambda x: Normalizations.normalize_and_proportion(-x),
|
|
179
|
+
axis=1)
|
|
180
|
+
|
|
181
|
+
weight_s = W.iloc[0]
|
|
182
|
+
|
|
183
|
+
return weight_s
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
class GlobalADE(ADE):
|
|
187
|
+
## todo in inference lags are the same of all
|
|
188
|
+
# could include past errors...
|
|
189
|
+
|
|
190
|
+
def __init__(self,
|
|
191
|
+
freq: str,
|
|
192
|
+
trim_ratio: float,
|
|
193
|
+
meta_lags: List[int],
|
|
194
|
+
trim_by_uid: bool = True,
|
|
195
|
+
meta_model=lgb.LGBMRegressor(**ADE.LGB_PARS)):
|
|
196
|
+
|
|
197
|
+
super().__init__(freq=freq,
|
|
198
|
+
trim_ratio=trim_ratio,
|
|
199
|
+
meta_lags=meta_lags,
|
|
200
|
+
trim_by_uid=trim_by_uid,
|
|
201
|
+
meta_model=meta_model)
|
|
202
|
+
|
|
203
|
+
self.alias = 'GADE'
|
|
204
|
+
|
|
205
|
+
def _process_meta_data(self,
|
|
206
|
+
meta_data: pd.DataFrame,
|
|
207
|
+
return_X_y: bool = True) -> DForDFTuple:
|
|
208
|
+
|
|
209
|
+
lag_locs = meta_data.columns.str.startswith('lag')
|
|
210
|
+
lag_cols = meta_data.columns[lag_locs].to_list()
|
|
211
|
+
|
|
212
|
+
df_melt = meta_data.drop(columns='y').melt(['unique_id', 'ds'] + lag_cols)
|
|
213
|
+
df_melt['unique_id'] = df_melt.apply(lambda x: f'{x["unique_id"]}_{x["variable"]}', axis=1)
|
|
214
|
+
df_melt = df_melt.rename(columns={'value': 'error'})
|
|
215
|
+
|
|
216
|
+
if return_X_y:
|
|
217
|
+
X_meta, Y_meta = df_melt[lag_cols], df_melt['error']
|
|
218
|
+
return X_meta, Y_meta
|
|
219
|
+
else:
|
|
220
|
+
meta_df = df_melt[lag_cols + ['error']]
|
|
221
|
+
|
|
222
|
+
return meta_df
|
|
223
|
+
|
|
224
|
+
def _weights_by_uid(self, df: pd.DataFrame, h: int):
|
|
225
|
+
top_overall = self._get_top_k(self.insample_scores.mean())
|
|
226
|
+
top_by_uid = self.insample_scores.apply(self._get_top_k, axis=1)
|
|
227
|
+
|
|
228
|
+
uid_weights = {}
|
|
229
|
+
for uid, meta_uid_df in df.groupby('unique_id'):
|
|
230
|
+
lags = meta_uid_df.head(-(h - 1)).tail(1)[self.lag_names]
|
|
231
|
+
|
|
232
|
+
meta_pred = self.meta_model.predict(lags)
|
|
233
|
+
meta_pred = pd.DataFrame(meta_pred, columns=self.model_names)
|
|
234
|
+
|
|
235
|
+
weights = self._weights_from_errors(meta_pred)
|
|
236
|
+
|
|
237
|
+
if self.trim_by_uid:
|
|
238
|
+
poor_models = [x not in top_by_uid[uid] for x in weights.index]
|
|
239
|
+
else:
|
|
240
|
+
poor_models = [x not in top_overall for x in weights.index]
|
|
241
|
+
|
|
242
|
+
weights[poor_models] = 0
|
|
243
|
+
weights /= weights.sum()
|
|
244
|
+
|
|
245
|
+
uid_weights[uid] = weights
|
|
246
|
+
|
|
247
|
+
weights_df = pd.DataFrame(uid_weights).T
|
|
248
|
+
weights_df.index.name = 'unique_id'
|
|
249
|
+
|
|
250
|
+
return weights_df
|
|
251
|
+
|
|
252
|
+
def _reweight_by_redundancy(self):
|
|
253
|
+
raise NotImplementedError
|
|
254
|
+
|
|
255
|
+
def update_weights(self, fcst: pd.DataFrame):
|
|
256
|
+
raise NotImplementedError
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
class MLForecastADE(ADE):
|
|
260
|
+
|
|
261
|
+
def __init__(self,
|
|
262
|
+
mlf: MLForecast,
|
|
263
|
+
trim_ratio: float,
|
|
264
|
+
sf: Optional[StatsForecast] = None,
|
|
265
|
+
meta_model=MIMO(lgb.LGBMRegressor(**ADE.LGB_PARS))):
|
|
266
|
+
"""
|
|
267
|
+
:param trim_ratio:
|
|
268
|
+
:param meta_model:
|
|
269
|
+
"""
|
|
270
|
+
self.mlf = mlf
|
|
271
|
+
self.sf = sf
|
|
272
|
+
self.frequency = self.mlf.ts.freq
|
|
273
|
+
|
|
274
|
+
super().__init__(freq=self.frequency,
|
|
275
|
+
trim_ratio=trim_ratio,
|
|
276
|
+
meta_model=meta_model,
|
|
277
|
+
meta_lags=self.mlf.ts.lags)
|
|
278
|
+
|
|
279
|
+
def fit(self, **kwargs):
|
|
280
|
+
"""
|
|
281
|
+
|
|
282
|
+
"""
|
|
283
|
+
|
|
284
|
+
insample_fcst = self.mlf.fcst_fitted_values_
|
|
285
|
+
|
|
286
|
+
if self.sf is not None:
|
|
287
|
+
self.sf.forecast(fitted=True, h=1)
|
|
288
|
+
insample_fcst_sf = self.sf.forecast_fitted_values()
|
|
289
|
+
|
|
290
|
+
insample_fcst = insample_fcst.merge(insample_fcst_sf.drop(columns='y'),
|
|
291
|
+
on=self.METADATA_NO_T)
|
|
292
|
+
|
|
293
|
+
self._fit(insample_fcst)
|
|
294
|
+
|
|
295
|
+
def predict(self, train: pd.DataFrame, h: int, **kwargs):
|
|
296
|
+
base_fcst = self.mlf.predict(h=h)
|
|
297
|
+
|
|
298
|
+
if self.sf is not None:
|
|
299
|
+
base_fcst_sf = self.sf.predict(h=h)
|
|
300
|
+
|
|
301
|
+
base_fcst = base_fcst.merge(base_fcst_sf, on=self.METADATA_NO_T)
|
|
302
|
+
|
|
303
|
+
fcst = self._predict(preds=base_fcst, train=train, h=h)
|
|
304
|
+
|
|
305
|
+
return fcst
|
|
306
|
+
|
|
307
|
+
def update_weights(self, fcst: pd.DataFrame):
|
|
308
|
+
raise NotImplementedError
|
|
309
|
+
|
|
310
|
+
def _reweight_by_redundancy(self):
|
|
311
|
+
raise NotImplementedError
|
|
312
|
+
|
|
313
|
+
def update_estimates(self, df: pd.DataFrame):
|
|
314
|
+
"""
|
|
315
|
+
Updating loss statistics for dynamic model selection
|
|
316
|
+
|
|
317
|
+
:param df: dataset with actual values and predictions, similar to insample predictions
|
|
318
|
+
"""
|
|
319
|
+
|
|
320
|
+
raise NotImplementedError
|
|
@@ -0,0 +1,351 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
from neuralforecast.losses.numpy import smape
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
from metaforecast.utils.normalization import Normalizations
|
|
8
|
+
from metaforecast.ensembles.expert_loss import (SquaredLoss,
|
|
9
|
+
PinballLoss,
|
|
10
|
+
PercentageLoss,
|
|
11
|
+
AbsoluteLoss,
|
|
12
|
+
LogLoss)
|
|
13
|
+
|
|
14
|
+
EXPERT_LOSS = {
|
|
15
|
+
'square': SquaredLoss,
|
|
16
|
+
'pinball': PinballLoss,
|
|
17
|
+
'percentage': PercentageLoss,
|
|
18
|
+
'absolute': AbsoluteLoss,
|
|
19
|
+
'log': LogLoss,
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ForecastingEnsemble(ABC):
|
|
24
|
+
""" ForecastingEnsemble
|
|
25
|
+
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
METADATA = ['unique_id', 'ds', 'y']
|
|
29
|
+
METADATA_NO_T = ['unique_id', 'ds']
|
|
30
|
+
|
|
31
|
+
WINDOW_SIZE_BY_FREQ = {
|
|
32
|
+
'H': 48,
|
|
33
|
+
'D': 14,
|
|
34
|
+
'W': 16,
|
|
35
|
+
'M': 12,
|
|
36
|
+
'ME': 12,
|
|
37
|
+
'MS': 12,
|
|
38
|
+
'Q': 4,
|
|
39
|
+
'QS': 4,
|
|
40
|
+
'Y': 6,
|
|
41
|
+
'': -1,
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
def __init__(self):
|
|
45
|
+
super().__init__()
|
|
46
|
+
|
|
47
|
+
self.models = []
|
|
48
|
+
self.model_names = None
|
|
49
|
+
self.tot_n_models = -1
|
|
50
|
+
self.n_models = -1
|
|
51
|
+
self.n_poor_models = -1
|
|
52
|
+
self.trim_ratio = 0
|
|
53
|
+
self.window_size = 0
|
|
54
|
+
|
|
55
|
+
@abstractmethod
|
|
56
|
+
def fit(self, **kwargs):
|
|
57
|
+
""" fit
|
|
58
|
+
|
|
59
|
+
Fits the ensemble combination rule
|
|
60
|
+
|
|
61
|
+
Parameters
|
|
62
|
+
----------
|
|
63
|
+
kwargs: Not defined
|
|
64
|
+
Whatever input value the ensemble takes.
|
|
65
|
+
|
|
66
|
+
Returns
|
|
67
|
+
-------
|
|
68
|
+
ForecastingEnsemble
|
|
69
|
+
self, optional
|
|
70
|
+
|
|
71
|
+
"""
|
|
72
|
+
raise NotImplementedError
|
|
73
|
+
|
|
74
|
+
@abstractmethod
|
|
75
|
+
def predict(self, **kwargs):
|
|
76
|
+
""" predict
|
|
77
|
+
|
|
78
|
+
Predicts the weights of ensemble combination rule
|
|
79
|
+
|
|
80
|
+
Parameters
|
|
81
|
+
----------
|
|
82
|
+
kwargs: Not defined
|
|
83
|
+
Whatever input value the ensemble takes.
|
|
84
|
+
|
|
85
|
+
Returns
|
|
86
|
+
-------
|
|
87
|
+
Array-like
|
|
88
|
+
Weights of each model in the ensemble
|
|
89
|
+
|
|
90
|
+
"""
|
|
91
|
+
raise NotImplementedError
|
|
92
|
+
|
|
93
|
+
def update_weights(self, fcst: pd.DataFrame):
|
|
94
|
+
"""
|
|
95
|
+
Updating loss statistics for dynamic model selection
|
|
96
|
+
|
|
97
|
+
param fcst: dataset with actual values and predictions, similar to insample predictions
|
|
98
|
+
|
|
99
|
+
"""
|
|
100
|
+
|
|
101
|
+
raise NotImplementedError
|
|
102
|
+
|
|
103
|
+
def evaluate_base_fcst(self, insample_fcst: pd.DataFrame, use_window: bool):
|
|
104
|
+
|
|
105
|
+
all_scores, window_scores = {}, {}
|
|
106
|
+
in_sample_loss_g = insample_fcst.groupby('unique_id')
|
|
107
|
+
for uid, uid_df in in_sample_loss_g:
|
|
108
|
+
|
|
109
|
+
uid_a_loss, uid_w_loss = {}, {}
|
|
110
|
+
for m in self.model_names:
|
|
111
|
+
uid_a_loss[m] = smape(y=uid_df['y'], y_hat=uid_df[m])
|
|
112
|
+
try:
|
|
113
|
+
uid_w_loss[m] = smape(y=uid_df.tail(self.window_size)['y'],
|
|
114
|
+
y_hat=uid_df.tail(self.window_size)[m])
|
|
115
|
+
except AssertionError:
|
|
116
|
+
uid_w_loss[m] = np.nan
|
|
117
|
+
|
|
118
|
+
all_scores[uid] = uid_a_loss
|
|
119
|
+
window_scores[uid] = uid_w_loss
|
|
120
|
+
|
|
121
|
+
all_scr_df = pd.DataFrame(all_scores).T
|
|
122
|
+
wdw_scr_df = pd.DataFrame(window_scores).T
|
|
123
|
+
|
|
124
|
+
if use_window:
|
|
125
|
+
return wdw_scr_df
|
|
126
|
+
else:
|
|
127
|
+
return all_scr_df
|
|
128
|
+
|
|
129
|
+
@abstractmethod
|
|
130
|
+
def _weights_by_uid(self, **kwargs):
|
|
131
|
+
raise NotImplementedError
|
|
132
|
+
|
|
133
|
+
def _set_n_models(self):
|
|
134
|
+
self.tot_n_models = len(self.model_names)
|
|
135
|
+
|
|
136
|
+
self.n_models = int(self.trim_ratio * self.tot_n_models)
|
|
137
|
+
if self.n_models < 1:
|
|
138
|
+
self.n_models = 1
|
|
139
|
+
|
|
140
|
+
self.n_poor_models = self.tot_n_models - self.n_models
|
|
141
|
+
|
|
142
|
+
def _get_top_k(self, scores: pd.Series):
|
|
143
|
+
"""
|
|
144
|
+
:param scores: models LOSS scores (to minimize)
|
|
145
|
+
"""
|
|
146
|
+
return scores.sort_values().index.tolist()[:self.n_models]
|
|
147
|
+
|
|
148
|
+
@staticmethod
|
|
149
|
+
def _weights_from_errors(scores: pd.Series) -> pd.Series:
|
|
150
|
+
|
|
151
|
+
weights = Normalizations.normalize_and_proportion(-scores)
|
|
152
|
+
|
|
153
|
+
return weights
|
|
154
|
+
|
|
155
|
+
@staticmethod
|
|
156
|
+
def _weighted_average(pred: pd.Series, weights: pd.DataFrame):
|
|
157
|
+
w = weights.loc[pred['unique_id']]
|
|
158
|
+
|
|
159
|
+
wa = (pred[w.index] * w).sum()
|
|
160
|
+
|
|
161
|
+
return wa
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
class Mixture(ForecastingEnsemble):
|
|
165
|
+
|
|
166
|
+
def __init__(self,
|
|
167
|
+
loss_type: str,
|
|
168
|
+
gradient: bool,
|
|
169
|
+
trim_ratio: float,
|
|
170
|
+
weight_by_uid: bool):
|
|
171
|
+
self.alias = 'Mixture'
|
|
172
|
+
|
|
173
|
+
super().__init__()
|
|
174
|
+
|
|
175
|
+
assert loss_type in EXPERT_LOSS.keys()
|
|
176
|
+
|
|
177
|
+
self.gradient = gradient
|
|
178
|
+
self.loss_type = loss_type
|
|
179
|
+
self.trim_ratio = trim_ratio
|
|
180
|
+
self.weight_by_uid = weight_by_uid
|
|
181
|
+
|
|
182
|
+
self.eta = None
|
|
183
|
+
self.regret = None
|
|
184
|
+
self.weights = None
|
|
185
|
+
self.ensemble_fcst = None
|
|
186
|
+
|
|
187
|
+
self.uid_weights = {}
|
|
188
|
+
self.uid_coefficient = {}
|
|
189
|
+
|
|
190
|
+
def fit(self, insample_fcst: pd.DataFrame):
|
|
191
|
+
|
|
192
|
+
if self.model_names is None:
|
|
193
|
+
self.model_names = insample_fcst.columns.to_list()
|
|
194
|
+
self.model_names = [x for x in self.model_names if x not in self.METADATA + ['h']]
|
|
195
|
+
|
|
196
|
+
self._initialize_params(insample_fcst)
|
|
197
|
+
self._set_n_models()
|
|
198
|
+
|
|
199
|
+
if self.weight_by_uid:
|
|
200
|
+
self._fit_by_uid(insample_fcst)
|
|
201
|
+
else:
|
|
202
|
+
self._fit_all(insample_fcst)
|
|
203
|
+
|
|
204
|
+
def _fit_by_uid(self, insample_fcst: pd.DataFrame):
|
|
205
|
+
grouped_fcst = insample_fcst.groupby('unique_id')
|
|
206
|
+
|
|
207
|
+
for uid, fcst_uid in grouped_fcst:
|
|
208
|
+
|
|
209
|
+
y = fcst_uid['y'].values
|
|
210
|
+
|
|
211
|
+
fcst_uid = fcst_uid.reset_index(drop=True)
|
|
212
|
+
fcst_uid = fcst_uid.drop(columns=self.METADATA)
|
|
213
|
+
if 'h' in fcst_uid.columns:
|
|
214
|
+
fcst_uid = fcst_uid.drop(columns='h')
|
|
215
|
+
|
|
216
|
+
self._initialize_params(fcst_uid)
|
|
217
|
+
|
|
218
|
+
self._update_mixture(fcst_uid, y)
|
|
219
|
+
|
|
220
|
+
self.weights = pd.DataFrame(self.weights, columns=self.model_names)
|
|
221
|
+
|
|
222
|
+
self.uid_weights[uid] = self.weights.iloc[-1]
|
|
223
|
+
self.uid_coefficient[uid] = self._weights_from_regret()
|
|
224
|
+
|
|
225
|
+
def _fit_all(self, insample_fcst: pd.DataFrame):
|
|
226
|
+
# todo add sort back
|
|
227
|
+
# just commenting so the estimation match the r bridge
|
|
228
|
+
insample_fcst_ = insample_fcst.sort_values('ds')
|
|
229
|
+
|
|
230
|
+
uid_list = insample_fcst_['unique_id'].unique().tolist()
|
|
231
|
+
|
|
232
|
+
y = insample_fcst_['y'].values
|
|
233
|
+
|
|
234
|
+
fcst = insample_fcst_.reset_index(drop=True)
|
|
235
|
+
fcst = fcst.drop(columns=self.METADATA)
|
|
236
|
+
if 'h' in fcst.columns:
|
|
237
|
+
fcst = fcst.drop(columns='h')
|
|
238
|
+
|
|
239
|
+
self._update_mixture(fcst, y)
|
|
240
|
+
self.weights = pd.DataFrame(self.weights, columns=self.model_names)
|
|
241
|
+
|
|
242
|
+
for uid in uid_list:
|
|
243
|
+
self.uid_weights[uid] = self.weights.iloc[-1]
|
|
244
|
+
self.uid_coefficient[uid] = self._weights_from_regret()
|
|
245
|
+
|
|
246
|
+
def predict(self, fcst: pd.DataFrame, **kwargs):
|
|
247
|
+
weights = pd.DataFrame(self.uid_weights).T
|
|
248
|
+
# weights = pd.DataFrame(self.uid_coefficient).T
|
|
249
|
+
|
|
250
|
+
if self.trim_ratio < 1:
|
|
251
|
+
weights = self._weights_by_uid(weights)
|
|
252
|
+
|
|
253
|
+
fcst_c = fcst.apply(lambda x: self._weighted_average(x, weights), axis=1)
|
|
254
|
+
fcst_c.name = self.alias
|
|
255
|
+
|
|
256
|
+
return fcst_c
|
|
257
|
+
|
|
258
|
+
def _calc_loss(self, fcst: pd.Series, y: float, fcst_c: float):
|
|
259
|
+
if self.gradient:
|
|
260
|
+
loss = EXPERT_LOSS[self.loss_type].gradient(fcst=fcst, y=y, fcst_c=fcst_c)
|
|
261
|
+
else:
|
|
262
|
+
loss = EXPERT_LOSS[self.loss_type].loss(fcst=fcst, y=y)
|
|
263
|
+
|
|
264
|
+
return loss
|
|
265
|
+
|
|
266
|
+
def update_weights(self, fcst: pd.DataFrame):
|
|
267
|
+
raise NotImplementedError
|
|
268
|
+
|
|
269
|
+
def _initialize_params(self, fcst: pd.DataFrame):
|
|
270
|
+
raise NotImplementedError
|
|
271
|
+
|
|
272
|
+
def _weights_from_regret(self, **kwargs):
|
|
273
|
+
raise NotImplementedError
|
|
274
|
+
|
|
275
|
+
def _update_mixture(self, fcst: pd.DataFrame, y: np.ndarray, **kwargs):
|
|
276
|
+
raise NotImplementedError
|
|
277
|
+
|
|
278
|
+
def _weights_by_uid(self, weights: pd.DataFrame):
|
|
279
|
+
neg_w = -weights
|
|
280
|
+
|
|
281
|
+
top_overall = self._get_top_k(-weights.mean())
|
|
282
|
+
top_by_uid = neg_w.apply(self._get_top_k, axis=1)
|
|
283
|
+
|
|
284
|
+
uid_weights = {}
|
|
285
|
+
for uid, w in weights.iterrows():
|
|
286
|
+
if self.weight_by_uid:
|
|
287
|
+
poor_models = [x not in top_by_uid[uid] for x in w.index]
|
|
288
|
+
else:
|
|
289
|
+
poor_models = [x not in top_overall for x in w.index]
|
|
290
|
+
|
|
291
|
+
w[poor_models] = 0
|
|
292
|
+
w /= w.sum()
|
|
293
|
+
|
|
294
|
+
uid_weights[uid] = w
|
|
295
|
+
|
|
296
|
+
weights_df = pd.DataFrame(uid_weights).T
|
|
297
|
+
weights_df.sum()
|
|
298
|
+
weights_df.index.name = 'unique_id'
|
|
299
|
+
|
|
300
|
+
return weights_df
|
|
301
|
+
|
|
302
|
+
@staticmethod
|
|
303
|
+
def _calc_ensemble_fcst(fcst: pd.Series, weight: pd.Series):
|
|
304
|
+
|
|
305
|
+
# form the mixture and the prediction
|
|
306
|
+
mixture = weight / np.sum(weight)
|
|
307
|
+
|
|
308
|
+
fcst_c = np.sum(fcst * mixture)
|
|
309
|
+
|
|
310
|
+
return mixture, fcst_c
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
class BaseADE(ForecastingEnsemble):
|
|
314
|
+
"""
|
|
315
|
+
ADE
|
|
316
|
+
|
|
317
|
+
Arbitrated Dynamic Ensemble
|
|
318
|
+
|
|
319
|
+
"""
|
|
320
|
+
|
|
321
|
+
def __init__(self,
|
|
322
|
+
window_size: int,
|
|
323
|
+
trim_ratio: float,
|
|
324
|
+
trim_by_uid: bool,
|
|
325
|
+
meta_model):
|
|
326
|
+
"""
|
|
327
|
+
: param window_size: No of recent observations used to trim ensemble
|
|
328
|
+
:param trim_ratio:
|
|
329
|
+
:param meta_model:
|
|
330
|
+
"""
|
|
331
|
+
|
|
332
|
+
super().__init__()
|
|
333
|
+
|
|
334
|
+
self.window_size = window_size
|
|
335
|
+
self.trim_ratio = trim_ratio
|
|
336
|
+
self.trim_by_uid = trim_by_uid
|
|
337
|
+
self.meta_model = meta_model
|
|
338
|
+
|
|
339
|
+
self.alias = 'ADE'
|
|
340
|
+
|
|
341
|
+
def fit(self, **kwargs):
|
|
342
|
+
raise NotImplementedError
|
|
343
|
+
|
|
344
|
+
def predict(self, **kwargs):
|
|
345
|
+
raise NotImplementedError
|
|
346
|
+
|
|
347
|
+
def update_weights(self, **kwargs):
|
|
348
|
+
raise NotImplementedError
|
|
349
|
+
|
|
350
|
+
def _weights_by_uid(self, **kwargs):
|
|
351
|
+
raise NotImplementedError
|