mirror of
https://github.com/webclinic017/drift.git
synced 2026-08-17 04:48:09 +00:00
feat(Forecasting): pytorch-forecasting scaffolding is now working, added narrow data format, fixed missing time column index name (#5)
* feat: Started implementing pytorch-forecasting. * feat(Forecasting): pytorch-forecasting scaffolding is now working, added narrow data format, fixed missing `time` column index name Co-authored-by: Daniel Szemerey <szemy2@gmail.com>
This commit is contained in:
co-authored by
Daniel Szemerey
parent
a59e705f02
commit
797791d2f4
@@ -0,0 +1,114 @@
|
||||
#%%
|
||||
import pytorch_lightning as pl
|
||||
from pytorch_lightning.callbacks import EarlyStopping, LearningRateMonitor
|
||||
|
||||
from pytorch_forecasting import TimeSeriesDataSet, TemporalFusionTransformer
|
||||
from pytorch_forecasting.data import GroupNormalizer
|
||||
from pytorch_forecasting.metrics import QuantileLoss
|
||||
|
||||
import sys
|
||||
sys.path.insert(0, '..')
|
||||
|
||||
from load_data import load_files
|
||||
|
||||
print("success")
|
||||
|
||||
#%%
|
||||
# load data
|
||||
data = load_files('data/', add_features=True, log_returns=False, narrow_format=True)
|
||||
# need to treat time as an independent column
|
||||
data = data.reset_index().rename({'index':'time'}, axis = 'columns')
|
||||
# we need a `time_idx` column for pytorch-forecasting, so we convert the time column to a time index by encoding the dates as consecutive days from the first date.
|
||||
data['time_idx'] = (data['time']-data['time'].min()).astype('timedelta64[D]').astype(int)+1
|
||||
# volume needs some love before we can use it
|
||||
data.drop(columns=['volume'], inplace=True)
|
||||
|
||||
data['month'] = data['month'].astype(str)
|
||||
data['day_month'] = data['day_month'].astype(str)
|
||||
data['day_week'] = data['day_week'].astype(str)
|
||||
|
||||
|
||||
|
||||
#%%
|
||||
# define dataset
|
||||
max_encoder_length = 36 # this is the look-back window, see https://github.com/jdb78/pytorch-forecasting/issues/448
|
||||
max_prediction_length = 6
|
||||
# training_cutoff = "YYYY-MM-DD" # day for cutoff
|
||||
|
||||
# training_cutoff = data["time_idx"].max() - max_prediction_length
|
||||
|
||||
training = TimeSeriesDataSet(
|
||||
data, # data[lambda x: x.date < training_cutoff],
|
||||
time_idx= 'time_idx',
|
||||
target= 'returns',
|
||||
# weight="weight",
|
||||
group_ids=[ 'ticker' ],
|
||||
min_encoder_length = max_encoder_length,##max_encoder_length//2,
|
||||
max_encoder_length = max_encoder_length,
|
||||
min_prediction_length = 1,
|
||||
max_prediction_length = 1,#max_prediction_length,
|
||||
static_categoricals=[ ],
|
||||
static_reals=[ ],
|
||||
time_varying_known_categoricals=[ 'day_month', 'day_week', 'month' ],
|
||||
time_varying_known_reals=[ 'vol_10', 'vol_20', 'vol_30', 'vol_60', 'mom_10', 'mom_20', 'mom_30', 'mom_60', 'mom_90'],
|
||||
time_varying_unknown_categoricals=[ ],
|
||||
time_varying_unknown_reals=[ 'returns' ],
|
||||
allow_missing_timesteps=True,
|
||||
# target_normalizer=GroupNormalizer(
|
||||
# groups=['ADA_returns', 'BTC_returns'], transformation="softplus")
|
||||
)
|
||||
|
||||
#%%
|
||||
print(training.index.time)
|
||||
print(training.index.time.max())
|
||||
|
||||
#%%
|
||||
# create validation and training dataset
|
||||
|
||||
validation = TimeSeriesDataSet.from_dataset(training, data, predict=True, stop_randomization=True)
|
||||
batch_size = 128
|
||||
train_dataloader = training.to_dataloader(train=True, batch_size=batch_size, num_workers=2)
|
||||
val_dataloader = validation.to_dataloader(train=False, batch_size=batch_size, num_workers=2)
|
||||
|
||||
# define trainer with early stopping
|
||||
early_stop_callback = EarlyStopping(monitor="val_loss", min_delta=1e-4, patience=1, verbose=False, mode="min")
|
||||
lr_logger = LearningRateMonitor()
|
||||
trainer = pl.Trainer(
|
||||
max_epochs=100,
|
||||
gpus=0,
|
||||
gradient_clip_val=0.1,
|
||||
limit_train_batches=30,
|
||||
callbacks=[lr_logger, early_stop_callback],
|
||||
)
|
||||
|
||||
# create the model
|
||||
tft = TemporalFusionTransformer.from_dataset(
|
||||
training,
|
||||
learning_rate=0.03,
|
||||
hidden_size=32,
|
||||
attention_head_size=1,
|
||||
dropout=0.1,
|
||||
hidden_continuous_size=16,
|
||||
output_size=7,
|
||||
loss=QuantileLoss(),
|
||||
log_interval=2,
|
||||
reduce_on_plateau_patience=4
|
||||
)
|
||||
print(f"Number of parameters in network: {tft.size()/1e3:.1f}k")
|
||||
|
||||
# find optimal learning rate (set limit_train_batches to 1.0 and log_interval = -1)
|
||||
res = trainer.tuner.lr_find(
|
||||
tft, train_dataloader=train_dataloader, val_dataloaders=val_dataloader, early_stop_threshold=1000.0, max_lr=0.3,
|
||||
)
|
||||
|
||||
print(f"suggested learning rate: {res.suggestion()}")
|
||||
fig = res.plot(show=True, suggest=True)
|
||||
fig.show()
|
||||
|
||||
# fit the model
|
||||
trainer.fit(
|
||||
tft, train_dataloader=train_dataloader, val_dataloaders=val_dataloader,
|
||||
)
|
||||
|
||||
|
||||
# %%
|
||||
@@ -0,0 +1,60 @@
|
||||
|
||||
import torch
|
||||
from torch import nn
|
||||
from torch.nn import functional as F
|
||||
from torch.utils.data import DataLoader, random_split
|
||||
import pytorch_lightning as pl
|
||||
|
||||
|
||||
class LitManualAutoEncoder(pl.LightningModule):
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.encoder = nn.Sequential(nn.Linear(28 * 28, 128), nn.ReLU(), nn.Linear(128, 3))
|
||||
self.decoder = nn.Sequential(nn.Linear(3, 128), nn.ReLU(), nn.Linear(128, 28 * 28))
|
||||
print("success")
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
# --------------------------
|
||||
# REPLACE WITH YOUR OWN
|
||||
opt_a = self.optimizers()
|
||||
|
||||
x, y = batch
|
||||
x = x.view(x.size(0), -1)
|
||||
z = self.encoder(x)
|
||||
x_hat = self.decoder(z)
|
||||
loss = F.mse_loss(x_hat, x)
|
||||
|
||||
# backward acts like normal backward
|
||||
self.manual_backward(loss, opt_a, retain_graph=True)
|
||||
self.manual_backward(loss, opt_a)
|
||||
opt_a.step()
|
||||
opt_a.zero_grad()
|
||||
|
||||
# --------------------------
|
||||
|
||||
def validation_step(self, batch, batch_idx):
|
||||
# --------------------------
|
||||
# REPLACE WITH YOUR OWN
|
||||
x, y = batch
|
||||
x = x.view(x.size(0), -1)
|
||||
z = self.encoder(x)
|
||||
x_hat = self.decoder(z)
|
||||
loss = F.mse_loss(x_hat, x)
|
||||
self.log('val_loss', loss)
|
||||
# --------------------------
|
||||
|
||||
def test_step(self, batch, batch_idx):
|
||||
# --------------------------
|
||||
# REPLACE WITH YOUR OWN
|
||||
x, y = batch
|
||||
x = x.view(x.size(0), -1)
|
||||
z = self.encoder(x)
|
||||
x_hat = self.decoder(z)
|
||||
loss = F.mse_loss(x_hat, x)
|
||||
self.log('test_loss', loss)
|
||||
# --------------------------
|
||||
|
||||
def configure_optimizers(self):
|
||||
optimizer = torch.optim.Adam(self.parameters(), lr=1e-3)
|
||||
return optimizer
|
||||
@@ -0,0 +1,48 @@
|
||||
from load_data import load_files
|
||||
import pandas as pd
|
||||
# from tensorflow import keras
|
||||
from utils.normalize import normalize
|
||||
# import tensorflow as tf
|
||||
from utils.visualize import visualize_loss
|
||||
|
||||
from torch.utils.data import DataLoader, random_split
|
||||
from model_lightning import LitManualAutoEncoder
|
||||
import pytorch_lightning as pl
|
||||
|
||||
#%%
|
||||
data = load_files('data/', False)
|
||||
data.reset_index(drop=True, inplace=True)
|
||||
data = data[[column for column in data.columns if not column.endswith('volume')]]
|
||||
|
||||
data.head()
|
||||
|
||||
#%%
|
||||
ticker_to_predict = 'ETH_returns'
|
||||
|
||||
learning_rate = 0.002
|
||||
batch_size = 64
|
||||
epochs = 100
|
||||
|
||||
split_fraction = 0.715
|
||||
train_split = int(split_fraction * int(data.shape[0]))
|
||||
|
||||
past = 10
|
||||
future = 1
|
||||
|
||||
start = past + future
|
||||
end = start + train_split
|
||||
|
||||
|
||||
# train = DataLoader(train, batch_size=32)
|
||||
# test = DataLoader(test, batch_size=32)
|
||||
# val = DataLoader(val, batch_size=32)
|
||||
|
||||
|
||||
# init model
|
||||
ae = LitManualAutoEncoder()
|
||||
|
||||
# Initialize a trainer
|
||||
trainer = pl.Trainer(gpus=1, max_epochs=3, progress_bar_refresh_rate=20)
|
||||
|
||||
# Train the model ⚡
|
||||
# trainer.fit(ae, train, val)
|
||||
@@ -0,0 +1,18 @@
|
||||
import os
|
||||
import pandas as pd
|
||||
|
||||
import torch
|
||||
from torch.utils.data import Dataset
|
||||
|
||||
|
||||
class TimeSeriesDataset(Dataset):
|
||||
def __init__(self, file_loader_hook):
|
||||
|
||||
df = file_loader_hook()
|
||||
|
||||
def __len__(self):
|
||||
return len(self.img_labels)
|
||||
|
||||
def __getitem__(self, idx):
|
||||
pass
|
||||
# return image, label
|
||||
Reference in New Issue
Block a user