modnn 2.0.2__tar.gz → 3.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: modnn
3
- Version: 2.0.2
3
+ Version: 3.0.1
4
4
  Summary: Physics-Informed Modularized Neural Network for Building Energy Modeling
5
5
  Author-email: Zixin Jiang <zjiang19@syr.edu>
6
6
  License-Expression: MIT
@@ -5,10 +5,10 @@ def _paras(**kwargs):
5
5
  """
6
6
  para = {
7
7
  # Internal gain module
8
- "Int_in": 3, "Int_h": 9, "Int_out": 1,
8
+ "Int_in": 3, "Int_h": 12, "Int_out": 1,
9
9
 
10
10
  # External disturbance module
11
- "Ext_in": 5, "Ext_h": 18, "Ext_out": 1,
11
+ "Ext_in": 1, "Ext_h": 10, "Ext_out": 1,
12
12
 
13
13
  # Zone module
14
14
  "Zone_in": 1, "Zone_out": 1,
@@ -18,11 +18,18 @@ def _paras(**kwargs):
18
18
 
19
19
  # Look back window
20
20
  "window" : 1,
21
+ "diff_alpha": 0.3,
22
+ # LSTM baseline
23
+ "LSTM_h" : 24,
24
+
25
+ "en_policy_in":3, "en_policy_hidden":24, "en_policy_out":1,
26
+ "de_policy_in":6, "de_policy_hidden":24, "de_policy_out":1,
27
+ "policy_epochs":200, "policy_lr":0.01,
21
28
 
22
29
  # Training hyperparameters
23
30
  "lr": 0.01,
24
- "epochs": 50,
25
- "patience": 20, #Early stop
31
+ "epochs": 30,
32
+ "patience": 3, #Early stop
26
33
  }
27
34
 
28
35
  # Allow override from kwargs
@@ -36,7 +43,7 @@ def _envelops(**kwargs):
36
43
  Accepts keyword arguments to override default values.
37
44
  #TODO: Pre-trained envelop library
38
45
  """
39
- envelop = {"direct_coef": 0.1,
46
+ envelop = {"direct_coef": 0.1,
40
47
  "abs_wall_coef": 0.1,
41
48
  "abs_roof_coef": 0.1,
42
49
  "r_opaque_coef": 10,
@@ -44,9 +51,9 @@ def _envelops(**kwargs):
44
51
  "r_transparent_coef": 10,
45
52
  "c_zone": 1000,
46
53
  # Module assembly
47
- "n_wall": 4,
48
- "n_roof": 1,
49
- "n_window": 2,
54
+ "n_wall": 1,
55
+ "n_roof": 0,
56
+ "n_window": 1,
50
57
  }
51
58
  # Allow override from kwargs
52
59
  envelop.update(kwargs)
@@ -65,21 +72,28 @@ def _args(**kwargs):
65
72
  "para": _paras(**para_overrides),
66
73
  "envelop": _envelops(**envelop_overrides),
67
74
  # Paths and device
68
- "datapath": "/home/zjiang19/Documents/GitHub/Eplus_ModNN_Compare/dataset/Eplus/EPlus_train_AC_off_2month.csv", #"../Dataset/EPlus.csv",
69
- "device": "cuda:0",
70
- "save_name": "Mid",
75
+ #/home/zjiang19/Documents/GitHub/Eplus_ModNN_Compare/dataset/Eplus/EPlus_train_noAC.csv---EPlus_train_AC_off_2month
76
+ # "datapath": "/home/zjiang19/Documents/GitHub/Eplus_ModNN_Compare/dataset/Eplus/EPlus_train_AC_off_2month.csv", #"../Dataset/EPlus.csv",
77
+ # "datapath": "/home/zjiang19/Documents/GitHub/Eplus_ModNN_Compare/dataset/Eplus/EPlus_train_case1.csv",
78
+ # "datapath": "/home/zjiang19/Documents/GitHub/Physical-Incorporated-Neural-Network-BEM/update/403_new_dyn.csv",
79
+ "datapath": "/home/zjiang19/Documents/GitHub/ModNN-RL-403/dataset/Data_Process/data_coe_update.csv",
80
+ "device": "cuda:1",
81
+ "save_name": "Eplus",
71
82
 
72
83
  # Data settings
84
+ "use_data_cleaning": True,
85
+ "tolerance_hours": 1,
73
86
  "resolution": 15, # 15 minutes data
74
87
  "enLen": 48, # "Kind of warm-up"
75
88
  "deLen": 96, # Prediction horizon, 96 is for 24 hours
76
- "startday": 30, # Training data selection
77
- "trainday": 180, # Training data selection
89
+ "startday": 390, # Training data selection
90
+ "trainday": 90, # Training data selection
78
91
  "testday": 1, # Testing data selection
79
92
  "training_batch": 1024*1,
80
- "envelop_mdl": "physics", # We provide "physics" and "datadriven"
93
+ "multi_deLen": [4, 8, 16, 24, 32, 48],
94
+ "envelop_mdl": "datadriven", # We provide "physics" and "datadriven"
81
95
  # "physics" rely on heatbalance equation, "datadriven" is a blackbox
82
- "ext_mdl": "LSTM", # We provide LSTM and RNN module, for RNN, we can apply positive constraint easily
96
+ "ext_mdl": "RNN", # We provide LSTM and RNN module, for RNN, we can apply positive constraint easily
83
97
  # But LSTM has Hadamard product, making this constraint hard to integrate
84
98
  # However, disturbance variables always have similiar distribution, in other word, is this constraint really necessary?
85
99
  "plott": '-all', # all: If want to see how model response to max heating/cooling; else: only Tzone prediction
@@ -93,7 +107,10 @@ def _args(**kwargs):
93
107
  {
94
108
  "temp": None, # For example (50, 120) Temperature(°F)
95
109
  "flux": None # For example (-5000, 5000) Power(W)
96
- }
110
+ },
111
+ #Policy NN Args
112
+
113
+ "control_mode": "Both",
97
114
 
98
115
  }
99
116
 
@@ -0,0 +1,427 @@
1
+ from sklearn.preprocessing import MinMaxScaler
2
+ from torch.utils.data import DataLoader, Dataset, random_split
3
+ import pandas as pd
4
+ import numpy as np
5
+ import os
6
+ import torch
7
+ import pickle
8
+
9
+
10
+ def _get_ModNN_input(start_time, timestep_minutes, occupancy=None, hvac=None,
11
+ temp_amb=None, solar=None, temp_room=None, path_to_save=None):
12
+ """
13
+ Generate dataframe that can be used for modnn.
14
+
15
+ Args:
16
+ start_time (str): Start timestamp (e.g., '2023-07-01 00:00')
17
+ timestep_minutes (int): Time resolution (e.g., 15 for 15-minute steps)
18
+ occupancy (list or np.array): Occupancy schedule
19
+ hvac (list or np.array): HVAC power values (in W)
20
+ temp_amb (list or np.array): Ambient temperature [°F]
21
+ solar (list or np.array): Solar radiation [W/m²]
22
+ temp_room (list or np.array): Room temp [°F]
23
+
24
+ Returns:
25
+ pd.DataFrame: Formatted and time-indexed input DataFrame
26
+ """
27
+
28
+ n = len(hvac)
29
+ index = pd.date_range(start=start_time, periods=n, freq=f"{timestep_minutes}min")
30
+
31
+ def fill_or_default(arr, default):
32
+ return arr if arr is not None else np.full(n, default)
33
+
34
+ df = pd.DataFrame({
35
+ "temp_room": fill_or_default(temp_room, 72),
36
+ "temp_amb": fill_or_default(temp_amb, 85),
37
+ "solar": fill_or_default(solar, 0),
38
+ "occ": fill_or_default(occupancy, 0),
39
+ "phvac": hvac
40
+ }, index=index)
41
+ df=df.resample('15T').mean()
42
+ df.to_csv(path_to_save)
43
+
44
+ return df
45
+
46
+
47
+ class MyData(Dataset):
48
+ """
49
+ Generate sequence-to-sequence pairs for learning.
50
+ """
51
+ def __init__(self, X_seq, y_seq):
52
+ self.X = np.array(X_seq)
53
+ self.y = np.array(y_seq)[:, :, [0]] # Index 0 is Tzone
54
+
55
+ def __len__(self):
56
+ return len(self.X)
57
+
58
+ def __getitem__(self, index):
59
+ return torch.tensor(self.X[index], dtype=torch.float32), \
60
+ torch.tensor(self.y[index], dtype=torch.float32)
61
+
62
+
63
+ class DataCook:
64
+ """
65
+ Full data pipeline for modnn:
66
+ - Load CSV
67
+ - Add time-based features
68
+ - Normalize inputs
69
+ - Slice into training/testing sequences
70
+ - Create PyTorch DataLoaders
71
+ """
72
+
73
+ def __init__(self, args, df):
74
+ """
75
+ Args:
76
+ args (dict): Configuration arguments from config.py
77
+ """
78
+ self.args = args
79
+ self.user_defined_minmax = args["user_defined_minmax"]
80
+ self.scaler_save_name = args["scaler_save_name"]
81
+ self.scaler_load = args["scaler_load"]
82
+ self.df = df
83
+ self.processed_data = None
84
+ self.scalers = None
85
+ folder_name = "../Scaler/{}".format(self.args['save_name'])
86
+ self.scaler_path = os.path.join(folder_name, self.scaler_save_name)
87
+
88
+ def load_data(self):
89
+ """Load raw CSV data and preprocess it."""
90
+ if self.df is not None:
91
+ pass
92
+ else:
93
+ try:
94
+ self.df = pd.read_csv(self.args["datapath"], index_col=[0])
95
+ except:
96
+ print("Input error")
97
+
98
+ self._parse_time_index()
99
+ self._generate_time_features()
100
+ self._check_missing_features()
101
+ self._scale_features()
102
+
103
+ def load_scaler(self, path):
104
+ """Load scaler dictionary from pickle file."""
105
+ with open(path, "rb") as f:
106
+ self.scalers = pickle.load(f)
107
+
108
+ def _check_missing_features(self):
109
+ """Check dataset for required features and fill in defaults for missing ones."""
110
+ required_features = ["temp_room", "temp_amb", "solar", "occ", "phvac", "setpt_cool", "setpt_heat", "price"]
111
+ missing_features = [f for f in required_features if f not in self.df.columns]
112
+
113
+ if missing_features:
114
+ print(f"Missing features: {missing_features}")
115
+
116
+ # Generate setpt_cool and setpt_heat if missing
117
+ occupied = (self.df.index.hour <= 8) | (self.df.index.hour >= 17)
118
+ if "setpt_cool" not in self.df.columns:
119
+ self.df["setpt_cool"] = 75 # default occupied cooling setpoint
120
+ self.df.loc[~occupied, "setpt_cool"] += 8 # apply 8°F setback when unoccupied
121
+
122
+ if "setpt_heat" not in self.df.columns:
123
+ self.df["setpt_heat"] = 70 # default occupied heating setpoint
124
+ self.df.loc[~occupied, "setpt_heat"] -= 8 # apply 8°F setback when unoccupied
125
+
126
+ # Generate price if missing
127
+ if "price" not in self.df.columns:
128
+ self.df["price"] = 1 # normal price
129
+ peak_hours = (self.df.index.hour >= 17) & (self.df.index.hour < 20) # 5pm to 8pm
130
+ self.df.loc[peak_hours, "price"] = 5
131
+
132
+ print("Missing features filled by default values")
133
+
134
+
135
+ def _parse_time_index(self):
136
+ """Convert index to datetime format (auto-detect)."""
137
+ try:
138
+ self.df.index = pd.to_datetime(self.df.index, format="%m/%d/%Y %H:%M")
139
+ except:
140
+ try:
141
+ self.df.index = pd.to_datetime(self.df.index, format="%Y-%m-%d %H:%M:%S")
142
+ except:
143
+ raise ValueError("Unsupported datetime format in index.")
144
+
145
+ def _generate_time_features(self):
146
+ """Add normalized time-of-day features."""
147
+ if "day_sin" not in self.df.columns:
148
+ time_hours = self.df.index.hour + self.df.index.minute / 60
149
+ self.df["day_sin"] = np.sin(2 * np.pi * time_hours / 24) / 2 + 0.5
150
+ self.df["day_cos"] = np.cos(2 * np.pi * time_hours / 24) / 2 + 0.5
151
+
152
+ def _scale_features(self):
153
+ """Apply MinMax scaling to each feature and save scalers."""
154
+ features = ["temp_room", "temp_amb", "solar", "occ", "phvac", "setpt_cool", "setpt_heat", "price"]
155
+
156
+ # Load existing scaler if provided
157
+ if self.scaler_load:
158
+ self.load_scaler(self.scaler_path)
159
+ temp_scaler = self.scalers["temp"]
160
+ flux_scaler = self.scalers["flux"]
161
+ else:
162
+ # Or fit a new sacler
163
+ scalers = {}
164
+ # Temperature scaler
165
+ if self.user_defined_minmax["temp"]:
166
+ temp_min, temp_max = self.user_defined_minmax["temp"]
167
+ temp_scaler = MinMaxScaler(feature_range=(-1, 1))
168
+ temp_scaler.fit(np.array([[temp_min], [temp_max]]))
169
+ else:
170
+ temp_scaler = MinMaxScaler(feature_range=(-1, 1))
171
+ temp_scaler.fit(self.df[["temp_room", "temp_amb"]].values.flatten().reshape(-1, 1))
172
+ scalers["temp"] = temp_scaler
173
+
174
+ # Flux scaler
175
+ if self.user_defined_minmax["flux"]:
176
+ flux_min, flux_max = self.user_defined_minmax["flux"]
177
+ flux_scaler = MinMaxScaler(feature_range=(-1, 1))
178
+ flux_scaler.fit(np.array([[flux_min], [flux_max]]))
179
+ else:
180
+ flux_scaler = MinMaxScaler(feature_range=(-1, 1))
181
+ flux_scaler.fit(self.df[["phvac"]].values.flatten().reshape(-1, 1))
182
+ scalers["flux"] = flux_scaler
183
+
184
+ # Other feature scalers
185
+ for f in features:
186
+ if f in ["temp_room", "temp_amb", "phvac", "setpt_cool", "setpt_heat"]:
187
+ continue
188
+ else:
189
+ scalers[f] = MinMaxScaler(feature_range=(-1, 1))
190
+ scalers[f].fit(self.df[f].values.flatten().reshape(-1, 1))
191
+ self.scalers = scalers
192
+
193
+ # Save newly created scalers
194
+ folder_name = "../Scaler/{}".format(self.args['save_name'])
195
+ if not os.path.exists(folder_name):
196
+ os.makedirs(folder_name)
197
+ scaler_path = os.path.join(folder_name, self.scaler_save_name)
198
+ with open(scaler_path, "wb") as f:
199
+ pickle.dump(self.scalers, f)
200
+
201
+ # Transform using scalers
202
+ scaled_temp_room = self.scalers["temp"].transform(self.df[["temp_room"]].values)
203
+ scaled_temp_amb = self.scalers["temp"].transform(self.df[["temp_amb"]].values)
204
+ scaled_solar = self.scalers["solar"].transform(self.df[["solar"]].values)
205
+ scaled_phvac = self.scalers["flux"].transform(self.df[["phvac"]].values)
206
+ scaled_occ = self.scalers["occ"].fit_transform(self.df[["occ"]].values)
207
+ scaled_setpt_cool = self.scalers["temp"].transform(self.df[["setpt_cool"]].values)
208
+ scaled_setpt_heat = self.scalers["temp"].transform(self.df[["setpt_heat"]].values)
209
+ scaled_price = self.scalers["price"].fit_transform(self.df[["price"]].values)
210
+
211
+ # Combine
212
+ self.processed_data = np.hstack([
213
+ scaled_temp_room,
214
+ scaled_temp_amb,
215
+ scaled_solar,
216
+ self.df[["day_sin", "day_cos"]].values,
217
+ scaled_occ,
218
+ scaled_phvac,
219
+ scaled_setpt_cool,
220
+ scaled_setpt_heat,
221
+ scaled_price
222
+ ])
223
+
224
+ def prepare_data_splits(self):
225
+ """Split into training and testing datasets."""
226
+ res = int(1440 / self.args["resolution"])
227
+ start, train, test = self.args["startday"], self.args["trainday"], self.args["testday"]
228
+ en_len, de_len = self.args["enLen"], self.args["deLen"]
229
+
230
+ self.trainingdf = self.processed_data[res * start : res * (start + train)]
231
+ # offset an encoder, since the prediction needs to start from 12:00
232
+ self.testingdf = self.processed_data[res * (start + train) - en_len :
233
+ res * (start + train + test) + de_len]
234
+
235
+ self.test_raw_df = self.df.iloc[res * (start + train): res * (start + train + test + 1)]
236
+ self.test_start = self.test_raw_df.index[0].strftime("%m-%d")
237
+ self.test_end = self.test_raw_df.index[-1].strftime("%m-%d")
238
+
239
+ def clean_training_data(self):
240
+ """
241
+ Clean training data by removing periods where data remains constant for too long.
242
+ Split data into clean segments and regenerate training/validation datasets.
243
+
244
+ Args:
245
+ tolerance_hours (float): Hours of constant data to consider as breaking point
246
+ """
247
+ tolerance_hours = self.args["tolerance_hours"]
248
+ res = int(1440 / self.args["resolution"]) # timesteps per day
249
+ tolerance_steps = int(tolerance_hours * 60 / self.args["resolution"]) # convert hours to timesteps
250
+
251
+ # 1) Identify breaking points
252
+ breaking_points = []
253
+
254
+ # Check for constant periods in key features (temp_room is index 0)
255
+ temp_room_data = self.trainingdf[:, 0] # First column is temp_room
256
+
257
+ i = 0
258
+ while i < len(temp_room_data) - tolerance_steps:
259
+ # Check if data is constant for tolerance_steps
260
+ window = temp_room_data[i:i + tolerance_steps]
261
+ if np.all(np.abs(window - window[0]) < 1e-6): # Using small epsilon for floating point comparison
262
+ # Found constant period, mark as breaking point
263
+ breaking_points.append(i)
264
+ # Skip to end of constant period
265
+ j = i + tolerance_steps
266
+ while j < len(temp_room_data) and np.abs(temp_room_data[j] - temp_room_data[i]) < 1e-6:
267
+ j += 1
268
+ breaking_points.append(j)
269
+ i = j
270
+ else:
271
+ i += 1
272
+
273
+ # 2) Split data into clean segments
274
+ clean_segments = []
275
+ start_idx = 0
276
+
277
+ for i in range(0, len(breaking_points), 2):
278
+ if i < len(breaking_points):
279
+ # Add segment before breaking point
280
+ if breaking_points[i] > start_idx:
281
+ clean_segments.append(self.trainingdf[start_idx:breaking_points[i]])
282
+
283
+ # Update start index to after the breaking point
284
+ if i + 1 < len(breaking_points):
285
+ start_idx = breaking_points[i + 1]
286
+ else:
287
+ start_idx = breaking_points[i] + tolerance_steps
288
+
289
+ # Add final segment if exists
290
+ if start_idx < len(self.trainingdf):
291
+ clean_segments.append(self.trainingdf[start_idx:])
292
+
293
+ # Filter out segments that are too short for sequence generation
294
+ min_length = self.args["enLen"] + max(self.args.get("multi_deLen", [self.args["deLen"]]))
295
+ clean_segments = [seg for seg in clean_segments if len(seg) >= min_length]
296
+
297
+ print(f"Original training data length: {len(self.trainingdf)}")
298
+ print(f"Found {len(breaking_points) // 2} breaking points")
299
+ print(f"Split into {len(clean_segments)} clean segments")
300
+ print(f"Segment lengths: {[len(seg) for seg in clean_segments]}")
301
+
302
+ # 3) Generate training and validation datasets from clean segments
303
+ decoder_lengths = self.args.get("multi_deLen", [self.args["deLen"]])
304
+ self.TrainLoader = []
305
+ self.ValidLoader = []
306
+
307
+ for dlen in decoder_lengths:
308
+ all_train_datasets = []
309
+ all_valid_datasets = []
310
+
311
+ # Process each clean segment
312
+ for segment in clean_segments:
313
+ if len(segment) > self.args["enLen"] + dlen:
314
+ train_loader, valid_loader = self._create_dataloader(segment,
315
+ self.args["training_batch"],
316
+ shuffle=True,
317
+ split=0.3,
318
+ en_len=self.args["enLen"],
319
+ de_len=dlen)
320
+ if train_loader.dataset:
321
+ all_train_datasets.append(train_loader.dataset)
322
+ if valid_loader and valid_loader.dataset:
323
+ all_valid_datasets.append(valid_loader.dataset)
324
+
325
+ # Combine all segments
326
+ if all_train_datasets:
327
+ from torch.utils.data import ConcatDataset
328
+ combined_train_dataset = ConcatDataset(all_train_datasets)
329
+ combined_valid_dataset = ConcatDataset(all_valid_datasets) if all_valid_datasets else None
330
+
331
+ self.TrainLoader.append(DataLoader(combined_train_dataset,
332
+ batch_size=self.args["training_batch"],
333
+ shuffle=True))
334
+ if combined_valid_dataset:
335
+ self.ValidLoader.append(DataLoader(combined_valid_dataset,
336
+ batch_size=self.args["training_batch"],
337
+ shuffle=True))
338
+ else:
339
+ self.ValidLoader.append(None)
340
+
341
+ # Generate control loader from clean segments
342
+ all_control_datasets = []
343
+ for segment in clean_segments:
344
+ if len(segment) >= self.args["enLen"] + self.args["deLen"]:
345
+ control_loader = self._create_dataloader(segment,
346
+ self.args["training_batch"],
347
+ shuffle=True,
348
+ split=None,
349
+ en_len=self.args["enLen"],
350
+ de_len=self.args["deLen"])[0]
351
+ if control_loader.dataset:
352
+ all_control_datasets.append(control_loader.dataset)
353
+
354
+ if all_control_datasets:
355
+ from torch.utils.data import ConcatDataset
356
+ combined_control_dataset = ConcatDataset(all_control_datasets)
357
+ self.ControlLoader = DataLoader(combined_control_dataset,
358
+ batch_size=self.args["training_batch"],
359
+ shuffle=True)
360
+
361
+ # Generate TestLoader (keep same as original method - no cleaning applied to test data)
362
+ self.TestLoader = self._create_dataloader(self.testingdf,
363
+ batch_size=len(self.testingdf),
364
+ shuffle=False,
365
+ en_len=self.args["enLen"],
366
+ de_len=self.args["deLen"])[0]
367
+
368
+ def create_dataloaders(self):
369
+ """Generate PyTorch dataloaders with multiple decoder lengths."""
370
+ decoder_lengths = self.args.get("multi_deLen", [self.args["deLen"]]) # e.g., [8, 16, 24, 48, 96]
371
+ self.TrainLoader = []
372
+ self.ValidLoader = []
373
+
374
+ for dlen in decoder_lengths:
375
+ TrainLoader_, ValidLoader_ = self._create_dataloader(self.trainingdf,
376
+ self.args["training_batch"],
377
+ shuffle=True,
378
+ split=0.3,
379
+ en_len=self.args["enLen"],
380
+ de_len=dlen)
381
+ self.TrainLoader.append(TrainLoader_)
382
+ self.ValidLoader.append(ValidLoader_)
383
+
384
+ # For testing, keep it fixed with the original decoder length
385
+ self.TestLoader = self._create_dataloader(self.testingdf,
386
+ batch_size=len(self.testingdf),
387
+ shuffle=False,
388
+ en_len=self.args["enLen"],
389
+ de_len=self.args["deLen"])[0]
390
+
391
+ self.ControlLoader = self._create_dataloader(self.trainingdf,
392
+ self.args["training_batch"],
393
+ shuffle=True,
394
+ split=None,
395
+ en_len=self.args["enLen"],
396
+ de_len=self.args["deLen"])[0]
397
+
398
+ def _create_dataloader(self, data, batch_size, shuffle, split=None, en_len=48, de_len=96):
399
+ X, y = self._generate_sequences(data, en_len, de_len)
400
+ dataset = MyData(X, y)
401
+ if split:
402
+ train_size = int((1 - split) * len(dataset))
403
+ valid_size = len(dataset) - train_size
404
+ train_ds, valid_ds = random_split(dataset, [train_size, valid_size])
405
+ return DataLoader(train_ds, batch_size=batch_size, shuffle=shuffle), \
406
+ DataLoader(valid_ds, batch_size=batch_size, shuffle=shuffle)
407
+ return DataLoader(dataset, batch_size=batch_size, shuffle=shuffle), None
408
+
409
+ def _generate_sequences(self, data, en_len, de_len):
410
+ """Slice data into overlapping sequences of encoder+decoder length."""
411
+ X, y = [], []
412
+ for i in range(len(data) - (en_len + de_len)):
413
+ seq = data[i: i + en_len + de_len]
414
+ # Don't be surprise why X and Y are suing same index
415
+ # The offset was considered in model itself
416
+ X.append(seq)
417
+ y.append(seq)
418
+ return X, y
419
+
420
+ def cook(self):
421
+ """Run the full pipeline."""
422
+ self.load_data()
423
+ self.prepare_data_splits()
424
+ if self.args["use_data_cleaning"]:
425
+ self.clean_training_data()
426
+ else:
427
+ self.create_dataloaders()
@@ -23,16 +23,19 @@ class Baseline(nn.Module):
23
23
 
24
24
  def __init__(self, args):
25
25
  super().__init__()
26
+ self.LSTM_h=args["para"]["LSTM_h"]
26
27
  self.encoLen = args["enLen"]
27
28
  self.decoder = LSTM(input_size= 6,
28
- hidden_size=24,
29
+ hidden_size=self.LSTM_h,
29
30
  output_size=1)
30
31
  self.device = args['device']
31
32
 
32
33
  def forward(self, input_X):
33
34
  Decoder_X = input_X[:, self.encoLen:, [1, 2, 3, 4, 5, 6]]
34
35
  Current_X_zone = input_X[:, self.encoLen:self.encoLen + 1, [0]]
35
- outputs, _ = self.decoder(Decoder_X, (Current_X_zone.reshape(1, Decoder_X.shape[0], 1).repeat(1, 1, 24),
36
- Current_X_zone.reshape(1, Decoder_X.shape[0], 1).repeat(1, 1, 24)))
37
-
38
- return torch.cat((input_X[:, :self.encoLen, [0]], outputs), 1), torch.cat((input_X[:, :self.encoLen, [0]], outputs), 1),torch.cat((input_X[:, :self.encoLen, [0]], outputs), 1)
36
+ outputs, _ = self.decoder(Decoder_X, (Current_X_zone.reshape(1, Decoder_X.shape[0], 1).repeat(1, 1, self.LSTM_h),
37
+ Current_X_zone.reshape(1, Decoder_X.shape[0], 1).repeat(1, 1, self.LSTM_h)))
38
+ # repeat to match the util code
39
+ return (torch.cat((input_X[:, :self.encoLen, [0]], outputs), 1),
40
+ torch.cat((input_X[:, :self.encoLen, [0]], outputs), 1),
41
+ torch.cat((input_X[:, :self.encoLen, [0]], outputs), 1))