Files
drift/training/walk_forward/process_transformations.py
T
Mark Aron Szulyovszky 3eb3ea94e3 Refactor(Training): new outcome types, representative pipeline steps, bet-sizing (#187)
* refactor(Training): added InferenceResult & TrainedModel types

* refactor(Pipeline): introduced TrainingOutcome, BetSizingWithMetaOutcome, etc.

* fix(Pipeline): getting it to compile

* refactor(WalkForward): separate preprocessing step

* feat(Pipeline): separate out transformations processing step

* refactor(Pipeline): use the Directional model terminology, put bet_sizing into pipeline instead of hiding it in a step

* refactor(WalkForward): moved functions to separate folder

* fix(WalkForward): use sparse array to store models, process transformations in parallel (lot faster)

* fix(Tests): and evaluation

* fix(Tests): for realz

* fix(Inference): preloading everything now, renamed primary models to directional models

* fix(BetSizing): was running transformations on the wrong data, oops

* fix(BetSizing): concatenated on the wrong axis accidentally

* fix(Reporting): able to use the new Stats type

* fix(BetSizing): renamed int column names

* fix(Portfolio): name the column properly

* fix(Reporting): rename the correct Series, lol

* fix(Inference): walk_forwad_inference() can deal with models not being aligned with the starting index

* fix(WalkForward): accidentally using the wrong index

* fix(WalkForward): use the correct indicies to fetch last model/transformations

* fix(CI): changed the name of the results
2022-01-29 06:41:40 +01:00

49 lines
2.2 KiB
Python

import pandas as pd
from training.types import TransformationsOverTime
from utils.helpers import get_first_valid_return_index
from tqdm import tqdm
from transformations.base import Transformation
from typing import Optional
from data_loader.types import ForwardReturnSeries, XDataFrame, ySeries
def walk_forward_process_transformations(
X: XDataFrame,
y: ySeries,
forward_returns: ForwardReturnSeries,
expanding_window: bool,
window_size: int,
retrain_every: int,
from_index: Optional[pd.Timestamp],
transformations: list[Transformation],
) -> TransformationsOverTime:
transformations_over_time = [pd.Series(index=y.index).rename(t.get_name()) for t in transformations]
first_nonzero_return = max(get_first_valid_return_index(forward_returns), get_first_valid_return_index(X.iloc[:,0]), get_first_valid_return_index(y))
train_from = first_nonzero_return + window_size + 1 if from_index is None else X.index.to_list().index(from_index)
train_till = len(y)
iterations_before_retrain = 0
for index in tqdm(range(train_from, train_till)):
train_window_start = X.index[first_nonzero_return] if expanding_window else X.index[index - window_size - 1]
if iterations_before_retrain <= 0 or pd.isna(transformations_over_time[0][index-1]):
train_window_end = X.index[index - 1]
X_expanding_window = X[train_window_start:train_window_end]
y_expanding_window = y[train_window_start:train_window_end]
current_transformations = [t.clone() for t in transformations]
for transformation_index, transformation in enumerate(current_transformations):
X_expanding_window = transformation.fit_transform(X_expanding_window, y_expanding_window)
iterations_before_retrain = retrain_every
for transformation_index, transformation in enumerate(current_transformations):
transformations_over_time[transformation_index][X.index[index]] = transformation
iterations_before_retrain -= 1
return transformations_over_time