2022-01-04 11:44:35 +01:00
from config.hashing import hash_data_config
2021-12-31 19:04:27 +01:00
from data_loader.load_data import load_data
2021-12-14 18:16:17 +01:00
import pandas as pd
2021-12-21 17:28:36 +01:00
from training.training import run_single_asset_trainig
2021-12-23 10:35:20 +01:00
from reporting.wandb import launch_wandb , send_report_to_wandb , register_config_with_wandb
2022-01-03 13:57:36 +01:00
from models.model_map import default_feature_selector_regression , default_feature_selector_classification
2021-12-26 12:15:11 +01:00
from utils.helpers import get_first_valid_return_index , weighted_average
2022-01-03 13:57:36 +01:00
from config.config import get_default_level_1_daily_config , get_default_level_2_daily_config , get_default_level_2_hourly_config
from config.preprocess import validate_config , get_model_name , preprocess_config
2021-12-27 21:59:22 +01:00
from feature_selection.feature_selection import select_features
from feature_selection.dim_reduction import reduce_dimensionality
2021-12-28 22:50:09 +01:00
import ray
ray . init ()
2021-12-14 18:16:17 +01:00
2022-01-05 18:35:52 +01:00
def run_pipeline ( project_name : str , with_wandb : bool , sweep : bool ):
wandb , model_config , training_config , data_config = setup_pipeline ( project_name , with_wandb , sweep )
results , all_predictions , all_probabilities = run_training ( project_name , wandb , sweep , model_config , training_config , data_config )
reporting ( results , all_predictions , all_probabilities , model_config , wandb , sweep , project_name )
2021-12-21 17:28:36 +01:00
def setup_pipeline ( project_name : str , with_wandb : bool , sweep : bool ):
2021-12-31 19:04:27 +01:00
model_config , training_config , data_config = get_default_level_2_daily_config ()
2021-12-20 17:49:11 +01:00
wandb = None
if with_wandb :
2021-12-21 17:28:36 +01:00
wandb = launch_wandb ( project_name = project_name , default_config = dict ( ** model_config , ** training_config , ** data_config ), sweep = sweep )
2021-12-23 10:35:20 +01:00
register_config_with_wandb ( wandb , model_config , training_config , data_config )
2022-01-03 13:57:36 +01:00
model_config , training_config , data_config = preprocess_config ( model_config , training_config , data_config )
2021-12-21 17:28:36 +01:00
2021-12-22 12:04:38 +01:00
2022-01-05 18:35:52 +01:00
return wandb , model_config , training_config , data_config
def run_training ( project_name : str , wandb , sweep : bool , model_config : dict , training_config : dict , data_config : dict ):
2021-12-20 17:49:11 +01:00
results = pd . DataFrame ()
2022-01-03 13:57:36 +01:00
all_predictions = pd . DataFrame ()
2022-01-04 11:44:35 +01:00
all_probabilities = pd . DataFrame ()
2021-12-22 12:04:38 +01:00
validate_config ( model_config , training_config , data_config )
2021-12-14 22:59:44 +01:00
2021-12-31 19:04:27 +01:00
for asset in data_config [ 'assets' ]:
print ( '-------- \n Predicting: ' , asset [ 1 ])
2021-12-15 21:11:12 +01:00
2021-12-20 17:49:11 +01:00
# 1. Load data
data_params = data_config . copy ()
data_params [ 'target_asset' ] = asset
2021-12-14 22:59:44 +01:00
2021-12-20 17:49:11 +01:00
X , y , target_returns = load_data ( ** data_params )
2021-12-27 21:59:22 +01:00
original_X = X . copy ()
2021-12-23 23:48:59 +01:00
first_valid_index = get_first_valid_return_index ( X . iloc [:, 0 ])
samples_to_train = len ( y ) - first_valid_index
2021-12-28 22:50:09 +01:00
if samples_to_train < training_config [ 'sliding_window_size_level1' ] * 3 :
2021-12-23 23:48:59 +01:00
print ( "Not enough samples to train" )
continue
2021-12-17 14:32:17 +01:00
2021-12-27 21:59:22 +01:00
# 2a. Dimensionality Reduction (optional)
if training_config [ 'dimensionality_reduction' ]:
X = reduce_dimensionality ( X , int ( len ( X . columns ) / 2 ))
# 2b. Feature Selection (optional)
if training_config [ 'feature_selection' ]:
print ( "Feature Selection started" )
# TODO: this needs to be done per model!
backup_model = default_feature_selector_regression if data_config [ 'method' ] == 'regression' else default_feature_selector_classification
2022-01-04 11:44:35 +01:00
X = select_features ( X = X , y = y , model = model_config [ 'level_1_models' ][ 0 ][ 1 ], n_features_to_select = training_config [ 'n_features_to_select' ], backup_model = backup_model , scaling = training_config [ 'scaler' ], data_config_hash = hash_data_config ( data_params ))
2021-12-27 21:59:22 +01:00
print ( "Feature Selection ended" )
# 3. Train Level-1 models
2022-01-04 11:44:35 +01:00
current_result , current_predictions , current_probabilities = run_single_asset_trainig (
2021-12-31 19:04:27 +01:00
ticker_to_predict = asset [ 1 ],
2021-12-27 21:59:22 +01:00
original_X = original_X ,
2021-12-20 17:49:11 +01:00
X = X ,
y = y ,
target_returns = target_returns ,
2021-12-21 09:23:06 +01:00
models = model_config [ 'level_1_models' ],
2021-12-20 17:49:11 +01:00
method = data_config [ 'method' ],
2021-12-28 22:50:09 +01:00
expanding_window = training_config [ 'expanding_window_level1' ],
sliding_window_size = training_config [ 'sliding_window_size_level1' ],
2021-12-20 17:49:11 +01:00
retrain_every = training_config [ 'retrain_every' ],
scaler = training_config [ 'scaler' ],
2021-12-23 23:48:59 +01:00
no_of_classes = data_config [ 'no_of_classes' ],
level = 1
2021-12-20 17:49:11 +01:00
)
results = pd . concat ([ results , current_result ], axis = 1 )
2021-12-27 21:59:22 +01:00
# With static models, because of the lag in the indicator, the first prediction is NA, so we fill it with zero.
all_predictions = pd . concat ([ all_predictions , current_predictions ], axis = 1 ) . fillna ( 0. )
2022-01-04 11:44:35 +01:00
all_probabilities = pd . concat ([ all_probabilities , current_probabilities ], axis = 1 ) . fillna ( 0. )
2021-12-17 14:32:17 +01:00
2021-12-27 21:59:22 +01:00
# 3. Train Level-2 (Ensemble) model (Optional)
if model_config [ 'level_2_model' ] is not None :
2022-01-04 11:44:35 +01:00
ensemble_X = pd . concat ([ all_predictions , all_probabilities ], axis = 1 )
2021-12-21 17:28:36 +01:00
if training_config [ 'include_original_data_in_ensemble' ]:
ensemble_X = pd . concat ([ ensemble_X , X ], axis = 1 )
2021-12-17 14:32:17 +01:00
2022-01-04 11:44:35 +01:00
ensemble_result , ensemble_preds , ensemble_probabilities = run_single_asset_trainig (
2021-12-31 19:04:27 +01:00
ticker_to_predict = asset [ 1 ],
2021-12-27 21:59:22 +01:00
original_X = ensemble_X ,
2021-12-21 17:28:36 +01:00
X = ensemble_X ,
y = y ,
target_returns = target_returns ,
2021-12-27 21:59:22 +01:00
models = [ model_config [ 'level_2_model' ]],
2021-12-21 17:28:36 +01:00
method = data_config [ 'method' ],
2021-12-28 22:50:09 +01:00
expanding_window = training_config [ 'expanding_window_level2' ],
sliding_window_size = training_config [ 'sliding_window_size_level2' ],
2021-12-21 17:28:36 +01:00
retrain_every = training_config [ 'retrain_every' ],
scaler = training_config [ 'scaler' ],
2021-12-23 23:48:59 +01:00
no_of_classes = data_config [ 'no_of_classes' ],
level = 2
2021-12-21 17:28:36 +01:00
)
2021-12-20 17:49:11 +01:00
2021-12-21 17:28:36 +01:00
results = pd . concat ([ results , ensemble_result ], axis = 1 )
all_predictions = pd . concat ([ all_predictions , ensemble_preds ], axis = 1 )
2022-01-04 11:44:35 +01:00
all_probabilities = pd . concat ([ all_probabilities , ensemble_probabilities ], axis = 1 ) . fillna ( 0. )
2021-12-22 12:04:38 +01:00
2022-01-05 18:35:52 +01:00
return results , all_predictions , all_probabilities
def reporting ( results : pd . DataFrame , all_predictions : pd . DataFrame , all_probabilities : pd . DataFrame , model_config : dict , wandb , sweep : bool , project_name : str ):
2021-12-20 17:49:11 +01:00
results . to_csv ( 'results.csv' )
2021-12-26 12:15:11 +01:00
level1_columns = results [[ column for column in results . columns if 'lvl1' in column ]]
level2_columns = results [[ column for column in results . columns if 'lvl2' in column ]]
2022-01-03 13:57:36 +01:00
2021-12-26 12:15:11 +01:00
# Only send the results of the final model to wandb
results_to_send = level2_columns if level2_columns . shape [ 1 ] > 0 else level1_columns
send_report_to_wandb ( results_to_send , wandb , project_name , get_model_name ( model_config ))
2021-12-20 17:49:11 +01:00
2022-01-03 13:57:36 +01:00
level1_predictions = all_predictions [[ column for column in all_predictions . columns if 'lvl1' in column ]]
level2_predictions = all_predictions [[ column for column in all_predictions . columns if 'lvl2' in column ]]
predictions_to_save = level2_predictions if level2_predictions . shape [ 1 ] > 0 else level1_predictions
predictions_to_save . to_csv ( 'predictions.csv' )
2021-12-23 17:06:22 +01:00
print ( " \n -------- \n " )
2021-12-26 12:15:11 +01:00
print ( "Benchmark buy-and-hold sharpe: " , round ( weighted_average ( results , 'no_of_samples' ) . loc [ 'benchmark_sharpe' ], 3 ))
2021-12-23 17:06:22 +01:00
print ( "Level-1: Number of samples evaluated: " , level1_columns . loc [ 'no_of_samples' ] . sum ())
2021-12-26 12:15:11 +01:00
print ( "Mean Sharpe ratio for Level-1 models: " , round ( weighted_average ( level1_columns , 'no_of_samples' ) . loc [ 'sharpe' ], 3 ))
print ( "Mean Probabilistic Sharpe ratio for Level-1 models: " , round ( weighted_average ( level1_columns , 'no_of_samples' ) . loc [ 'prob_sharpe' ] . mean (), 3 ))
2021-12-23 17:06:22 +01:00
2021-12-27 21:59:22 +01:00
if model_config [ 'level_2_model' ] is not None :
print ( "Level-2 (Ensemble): Number of samples evaluated: " , level2_columns . loc [ 'no_of_samples' ] . sum ())
print ( "Mean Sharpe ratio for Level-2 (Ensemble) models: " , round ( weighted_average ( level2_columns , 'no_of_samples' ) . loc [ 'sharpe' ] . mean (), 3 ))
print ( "Mean Probabilistic Sharpe ratio for Level-2 (Ensemble) models: " , round ( weighted_average ( level2_columns , 'no_of_samples' ) . loc [ 'prob_sharpe' ] . mean (), 3 ))
2021-12-17 14:32:17 +01:00
2021-12-21 17:28:36 +01:00
if sweep :
if wandb . run is not None :
wandb . finish ()
2021-12-20 14:01:14 +01:00
2021-12-20 17:49:11 +01:00
if __name__ == '__main__' :
2022-01-05 18:35:52 +01:00
run_pipeline ( project_name = 'price-prediction' , with_wandb = False , sweep = False )