From 6b26643ecef0901c3f2169c5e68d0550f804e640 Mon Sep 17 00:00:00 2001 From: Mark Aron Szulyovszky Date: Mon, 17 Jan 2022 11:43:51 +0100 Subject: [PATCH] feat(Transformations): replaced feature selection pre-processing step with online version (with cache) (#170) * feat(Transformations): removed feature-selection pre-processing step completely * fix(Core): removed unnecessary `original_X` * fix(Transformations): use the X_expanding_window to transform subsequent data * fix(RFE): should check for model correctly * fix(Config): only re-train the model every 40 timestamp * fix(MetaLabeling): pass in the correct X to meta-labeling step * fix(Transformation): PCA should at least keep as many features as sliding_window_size * feat(Transformations): cache transformations across the same asset * fix(Tests): missing preloaded_transformations arg * chore(Config): got rid of unnecessary 'classification_models' and 'regression_models' dictionary keys --- config/config.py | 14 +- config/preprocess.py | 6 +- data_loader/process_data.py | 23 --- feature_selection/dim_reduction.py | 8 - feature_selection/feature_selection.py | 40 ----- models/average.py | 1 - models/base.py | 3 +- models/model_map.py | 195 ++++++++++++------------- models/momentum.py | 2 +- models/naive.py | 2 +- models/neural.py | 2 +- models/sklearn.py | 8 +- models/statsmodels.py | 1 - models/xgboost.py | 2 +- run_feature_selection.py | 39 ----- run_pipeline.py | 7 +- sweep_ensemble.yaml | 8 +- sweep_meta_labeling.yaml | 4 +- sweep_primary_models.yaml | 4 +- tests/test_evaluation.py | 5 +- tests/test_walk_forward.py | 4 +- training/meta_labeling.py | 15 +- training/primary_model.py | 18 ++- training/training_steps.py | 8 +- training/walk_forward.py | 14 +- transformations/pca.py | 39 +++++ transformations/rfe.py | 46 ++++++ 27 files changed, 240 insertions(+), 278 deletions(-) delete mode 100644 feature_selection/dim_reduction.py delete mode 100644 feature_selection/feature_selection.py delete mode 100644 run_feature_selection.py create mode 100644 transformations/pca.py create mode 100644 transformations/rfe.py diff --git a/config/config.py b/config/config.py index 145c132..e339901 100644 --- a/config/config.py +++ b/config/config.py @@ -30,7 +30,7 @@ def get_dev_config() -> tuple[dict, dict, dict]: ) regression_models = ["Lasso"] - classification_models = ["LR_two_class"] + classification_models = ["LogisticRegression_two_class"] model_config = dict( primary_models = regression_models if data_config['method'] == 'regression' else classification_models, @@ -52,7 +52,7 @@ def get_default_ensemble_config() -> tuple[dict, dict, dict]: expanding_window_meta_labeling = True, sliding_window_size_primary = 380, sliding_window_size_meta_labeling = 240, - retrain_every = 20, + retrain_every = 40, scaler = 'minmax', # 'normalize' 'minmax' 'standardize' ) @@ -73,8 +73,8 @@ def get_default_ensemble_config() -> tuple[dict, dict, dict]: ) regression_models = ["Lasso", "KNN", "RFR"] - classification_models = ["LR_two_class", "LDA", "NB", "RFC", "XGB_two_class", "LGBM", "StaticMom"] - meta_labeling_models = ['LR_two_class', 'LGBM'] + classification_models = ["LogisticRegression_two_class", "LDA", "NB", "RFC", "XGB_two_class", "LGBM", "StaticMom"] + meta_labeling_models = ['LogisticRegression_two_class', 'LGBM'] ensemble_model = 'Average' model_config = dict( @@ -97,7 +97,7 @@ def get_lightweight_ensemble_config() -> tuple[dict, dict, dict]: expanding_window_meta_labeling = True, sliding_window_size_primary = 380, sliding_window_size_meta_labeling = 240, - retrain_every = 20, + retrain_every = 40, scaler = 'minmax', # 'normalize' 'minmax' 'standardize' ) @@ -118,8 +118,8 @@ def get_lightweight_ensemble_config() -> tuple[dict, dict, dict]: ) regression_models = ["Lasso", "KNN"] - classification_models = ['LR_two_class', 'SVC'] - meta_labeling_models = ['LR_two_class', 'LGBM'] + classification_models = ['LogisticRegression_two_class', 'SVC'] + meta_labeling_models = ['LogisticRegression_two_class', 'LGBM'] ensemble_model = 'Average' model_config = dict( diff --git a/config/preprocess.py b/config/preprocess.py index 38d1abb..bee987a 100644 --- a/config/preprocess.py +++ b/config/preprocess.py @@ -21,10 +21,10 @@ def __preprocess_feature_extractors_config(data_dict: dict) -> dict: return data_dict def __preprocess_model_config(model_config:dict, method:str) -> dict: - model_map, _, _, _, _ = get_model_map(model_config) - model_config['primary_models'] = [(model_name, model_map[method + '_models'][model_name]) for model_name in model_config['primary_models']] + model_map = get_model_map(model_config) + model_config['primary_models'] = [(model_name, model_map['primary_models'][model_name]) for model_name in model_config['primary_models']] if len(model_config['meta_labeling_models']) > 0: - model_config['meta_labeling_models'] = [(model_name, model_map[method + '_models'][model_name]) for model_name in model_config['meta_labeling_models']] + model_config['meta_labeling_models'] = [(model_name, model_map['primary_models'][model_name]) for model_name in model_config['meta_labeling_models']] if model_config['ensemble_model'] is not None: model_config['ensemble_model'] = (model_config['ensemble_model'], model_map['ensemble_models'][model_config['ensemble_model']]) diff --git a/data_loader/process_data.py b/data_loader/process_data.py index d2bbb5d..931d6b8 100644 --- a/data_loader/process_data.py +++ b/data_loader/process_data.py @@ -1,31 +1,8 @@ import pandas as pd -from operator import itemgetter from utils.helpers import has_enough_samples_to_train -from feature_selection.dim_reduction import reduce_dimensionality -from models.model_map import get_model_map -from feature_selection.feature_selection import select_features import warnings - - - - -def process_data(X:pd.DataFrame, y:pd.Series, configs: dict) -> tuple[pd.DataFrame,pd.DataFrame]: - - model_config, training_config, data_config = itemgetter('model_config', 'training_config', 'data_config')(configs) - _, _, _, default_feature_selector_regression, default_feature_selector_classification = get_model_map(model_config) - - original_X = X.copy() - - # 2b. Feature Selection - print("Feature Selection started") - # TODO: this needs to be done per model! - backup_model = default_feature_selector_regression if data_config['method'] == 'regression' else default_feature_selector_classification - X = select_features(X = X, y = y, model = model_config['primary_models'][0][1], n_features_to_select = training_config['n_features_to_select'], backup_model = backup_model, scaling = training_config['scaler']) - - return X, original_X - def check_data(X:pd.DataFrame, y:pd.Series, training_config:dict): """ Returns True if data is valid, else returns False.""" diff --git a/feature_selection/dim_reduction.py b/feature_selection/dim_reduction.py deleted file mode 100644 index a112036..0000000 --- a/feature_selection/dim_reduction.py +++ /dev/null @@ -1,8 +0,0 @@ -import pandas as pd -from sklearn.decomposition import PCA - -def reduce_dimensionality(X: pd.DataFrame, no_of_compoments: int) -> pd.DataFrame: - pca = PCA(n_components= no_of_compoments) - result = pd.DataFrame(pca.fit_transform(X), index= X.index) - result.columns = ['PCA_' + str(i) for i in range(1, no_of_compoments+1)] - return result diff --git a/feature_selection/feature_selection.py b/feature_selection/feature_selection.py deleted file mode 100644 index 55f5751..0000000 --- a/feature_selection/feature_selection.py +++ /dev/null @@ -1,40 +0,0 @@ -from sklearn.feature_selection import RFE -from sklearn.model_selection import TimeSeriesSplit -import pandas as pd -from models.base import Model -from models.sklearn import SKLearnModel -from utils.scaler import get_scaler -from utils.types import ScalerTypes -# from utils.hashing import hash_df, hash_series -# from diskcache import Cache -# cache = Cache(".cachedir/feature_selection") - -# def select_features(**kwargs) -> pd.DataFrame: -# hashed = kwargs['data_config_hash'] + kwargs['model'].get_name() + str(kwargs['n_features_to_select']) + kwargs['backup_model'].get_name() + kwargs['scaling'] -# if hashed in cache: -# return cache.get(hashed) -# else: -# return_value = __select_features(**kwargs) -# cache[hashed] = return_value -# return return_value - -def select_features(X: pd.DataFrame, y: pd.Series, model: Model, n_features_to_select: int, backup_model: SKLearnModel, scaling: ScalerTypes) -> pd.DataFrame: - ''' Select features using RFECV, returns a pd.DataFrame (X) with only the selected features.''' - if model.model_type != 'ml': return X - - # 2. Recursive feature selection - cv = TimeSeriesSplit(n_splits=5) - scaler = get_scaler(scaling) - X_scaled = scaler.fit_transform(X, y) - - feat_selector_model = model.model - if hasattr(feat_selector_model, 'feature_importances_') == False and hasattr(feat_selector_model, 'coef_') == False: - feat_selector_model = backup_model.model - -# selector = RFECV(feat_selector_model, cv = cv, step=5, min_features_to_select=min_features_to_select) - step = 0.05 - selector = RFE(feat_selector_model, n_features_to_select= n_features_to_select, step=step) - selector = selector.fit(X_scaled, y) - print("Kept %d features out of %d" % (selector.n_features_, X_scaled.shape[1])) - - return pd.DataFrame(X[X.columns[selector.support_]], index= X.index) diff --git a/models/average.py b/models/average.py index ac77642..d72bc5a 100644 --- a/models/average.py +++ b/models/average.py @@ -9,7 +9,6 @@ class StaticAverageModel(Model): data_transformation = 'original' only_column = 'model_' - feature_selection = 'off' model_type = 'static' predict_window_size = 'single_timestamp' diff --git a/models/base.py b/models/base.py index 5ed44ce..4ffd3b0 100644 --- a/models/base.py +++ b/models/base.py @@ -7,9 +7,8 @@ import numpy as np class Model(ABC): + method: Literal["regression", "classification"] data_transformation: Literal["transformed", "original"] - feature_selection: Literal["on", "off"] - # data_format: Literal["wide", "narrow"] only_column: Optional[str] model_type: Literal['ml', 'static'] predict_window_size: Literal['single_timestamp', 'window_size'] diff --git a/models/model_map.py b/models/model_map.py index 46246cd..485ebef 100644 --- a/models/model_map.py +++ b/models/model_map.py @@ -1,111 +1,100 @@ from models.sklearn import SKLearnModel -from sklearn.ensemble import RandomForestClassifier -from sklearn.ensemble import RandomForestRegressor +from sklearnex.ensemble import RandomForestClassifier +from sklearnex.ensemble import RandomForestRegressor -model_map = { - "regression_models": dict(), - "classification_models": dict(), - "ensemble_models": dict() -} +default_feature_selector_classification = SKLearnModel(RandomForestClassifier(n_jobs=-1, max_depth=20, random_state=1), 'classification') +default_feature_selector_regression = SKLearnModel(RandomForestRegressor(n_jobs=-1, max_depth=20, random_state=1), 'regression') def get_model_map(config:dict): - - if len(config['primary_models']) > 0 and isinstance(config['primary_models'][0], str): - print("Going to Load models") - combined_list = config['primary_models'] + config['meta_labeling_models'] + [config['ensemble_model']] - for model_name in combined_list: - if model_name == 'LR': - from sklearn.linear_model import LinearRegression - model_map['regression_models']['LR'] = SKLearnModel(LinearRegression(n_jobs=-1)) - elif model_name == 'Lasso': - from sklearn.linear_model import Lasso - model_map['regression_models']['Lasso'] = SKLearnModel(Lasso(alpha=100, random_state=1)) - elif model_name == 'Ridge': - from sklearn.linear_model import Ridge - model_map['regression_models']['Ridge'] = SKLearnModel(Ridge(alpha=0.1)) - elif model_name == 'BayesianRidge': - from sklearn.linear_model import BayesianRidge - model_map['regression_models']['BayesianRidge'] = SKLearnModel(BayesianRidge()) - elif model_name == 'KNN': - from sklearnex.neighbors import KNeighborsRegressor - model_map['regression_models']['KNN'] = SKLearnModel(KNeighborsRegressor(n_neighbors=25)) - elif model_name == 'AB': - from sklearn.ensemble import AdaBoostRegressor - model_map['regression_models']['AB'] = SKLearnModel(AdaBoostRegressor(random_state=1)) - elif model_name == 'MLP': - from sklearn.neural_network import MLPRegressor - model_map['regression_models']['MLP'] = SKLearnModel(MLPRegressor(hidden_layer_sizes=(100,20), max_iter=1000)) - elif model_name == 'RFR': - # from sklearn.ensemble import RandomForestRegressor - model_map['regression_models']['RFR'] = SKLearnModel(RandomForestRegressor(n_jobs=-1, max_depth=20, random_state=1)) - elif model_name == 'SVR': - from sklearnex.svm import SVR - model_map['regression_models']['SVR'] = SKLearnModel(SVR(kernel='rbf', C=1e3, gamma=0.1)) - elif model_name == 'StaticNaive': - from models.naive import StaticNaiveModel - model_map['regression_models']['StaticNaive'] = StaticNaiveModel() - elif model_name == 'DNN': - from models.neural import LightningNeuralNetModel - from models.pytorch.neural_nets import MultiLayerPerceptron - import torch.nn.functional as F - model_map['regression_models']['DNN'] = LightningNeuralNetModel( - MultiLayerPerceptron( - hidden_layers_ratio = [1.0], - probabilities = False, - loss_function = F.mse_loss), - max_epochs=15 - ) - - - elif model_name == 'LR_two_class': - from sklearn.linear_model import LogisticRegression - model_map['classification_models']['LR_two_class'] = SKLearnModel(LogisticRegression(C=10, random_state=1, solver='liblinear', max_iter=1000)) - elif model_name == 'LR_three_class': - from sklearnex.linear_model import LogisticRegression as LogisticRegression_EX - model_map['classification_models']['LR_three_class'] = SKLearnModel(LogisticRegression_EX(C=10, random_state=1, max_iter=1000, n_jobs=-1)) - elif model_name == 'LDA': - from sklearn.discriminant_analysis import LinearDiscriminantAnalysis - model_map['classification_models']['LDA'] = SKLearnModel(LinearDiscriminantAnalysis()) - elif model_name == 'KNN': - from sklearn.neighbors import KNeighborsClassifier - model_map['classification_models']['KNN'] = SKLearnModel(KNeighborsClassifier()) - elif model_name == 'CART': - from sklearn.tree import DecisionTreeClassifier - model_map['classification_models']['CART'] = SKLearnModel(DecisionTreeClassifier(max_depth=15, random_state=1)) - elif model_name == 'NB': - from sklearn.naive_bayes import GaussianNB - model_map['classification_models']['NB'] = SKLearnModel(GaussianNB()) - elif model_name == 'AB': - from sklearn.ensemble import AdaBoostClassifier - model_map['classification_models']['AB'] = SKLearnModel(AdaBoostClassifier(n_estimators=15)) - elif model_name == 'RFC': - # from sklearn.ensemble import RandomForestClassifier - model_map['classification_models']['RFC'] = SKLearnModel(RandomForestClassifier(n_jobs=-1, max_depth=20, random_state=1)) - elif model_name == 'SVC': - from sklearn.svm import SVC - model_map['classification_models']['SVC'] = SKLearnModel(SVC(kernel='rbf', C=1e3, probability=True, random_state=1)) - elif model_name == 'XGB_two_class': - from xgboost import XGBClassifier - from models.xgboost import XGBoostModel - model_map['classification_models']['XGB_two_class'] = XGBoostModel(XGBClassifier(n_jobs=-1, max_depth = 20, random_state=1, objective='binary:logistic', use_label_encoder= False, eval_metric='mlogloss')) - elif model_name == 'LGBM': - from lightgbm import LGBMClassifier - model_map['classification_models']['LGBM'] = SKLearnModel(LGBMClassifier(n_jobs=-1, max_depth=20, random_state=1)) - elif model_name == 'StaticMom': - from models.momentum import StaticMomentumModel - model_map['classification_models']['StaticMom'] = StaticMomentumModel(allow_short=True) - elif model_name == 'Average': - from models.average import StaticAverageModel - model_map['ensemble_models']['Average'] = StaticAverageModel() - - - model_names_classification = list(model_map["classification_models"].keys()) - model_names_regression = list(model_map["regression_models"].keys()) + model_map = { + "primary_models": dict(), + "ensemble_models": dict(), + } - default_feature_selector_classification = SKLearnModel(RandomForestClassifier(n_jobs=-1, max_depth=20, random_state=1)) - default_feature_selector_regression = SKLearnModel(RandomForestRegressor(n_jobs=-1, max_depth=20, random_state=1)) - + combined_list = config['primary_models'] + config['meta_labeling_models'] + [config['ensemble_model']] + for model_name in combined_list: + if model_name == 'LinearRegression': + from sklearn.linear_model import LinearRegression + model_map['primary_models']['LR'] = SKLearnModel(LinearRegression(n_jobs=-1), 'regression') + elif model_name == 'Lasso': + from sklearn.linear_model import Lasso + model_map['primary_models']['Lasso'] = SKLearnModel(Lasso(alpha=100, random_state=1), 'regression') + elif model_name == 'Ridge': + from sklearn.linear_model import Ridge + model_map['primary_models']['Ridge'] = SKLearnModel(Ridge(alpha=0.1), 'regression') + elif model_name == 'BayesianRidge': + from sklearn.linear_model import BayesianRidge + model_map['primary_models']['BayesianRidge'] = SKLearnModel(BayesianRidge(), 'regression') + elif model_name == 'KNN': + from sklearnex.neighbors import KNeighborsRegressor + model_map['primary_models']['KNN'] = SKLearnModel(KNeighborsRegressor(n_neighbors=25), 'regression') + elif model_name == 'AB': + from sklearn.ensemble import AdaBoostRegressor + model_map['primary_models']['AB'] = SKLearnModel(AdaBoostRegressor(random_state=1), 'regression') + elif model_name == 'MLP': + from sklearn.neural_network import MLPRegressor + model_map['primary_models']['MLP'] = SKLearnModel(MLPRegressor(hidden_layer_sizes=(100,20), max_iter=1000), 'regression') + elif model_name == 'RFR': + # from sklearn.ensemble import RandomForestRegressor + model_map['primary_models']['RFR'] = SKLearnModel(RandomForestRegressor(n_jobs=-1, max_depth=20, random_state=1), 'regression') + elif model_name == 'SVR': + from sklearnex.svm import SVR + model_map['primary_models']['SVR'] = SKLearnModel(SVR(kernel='rbf', C=1e3, gamma=0.1), 'regression') + elif model_name == 'StaticNaive': + from models.naive import StaticNaiveModel + model_map['primary_models']['StaticNaive'] = StaticNaiveModel() + elif model_name == 'DNN': + from models.neural import LightningNeuralNetModel + from models.pytorch.neural_nets import MultiLayerPerceptron + import torch.nn.functional as F + model_map['primary_models']['DNN'] = LightningNeuralNetModel( + MultiLayerPerceptron( + hidden_layers_ratio = [1.0], + probabilities = False, + loss_function = F.mse_loss), + max_epochs=15 + ) + + elif model_name == 'LogisticRegression_two_class': + from sklearn.linear_model import LogisticRegression + model_map['primary_models']['LogisticRegression_two_class'] = SKLearnModel(LogisticRegression(C=10, random_state=1, solver='liblinear', max_iter=1000), 'classification') + elif model_name == 'LogisticRegression_three_class': + from sklearnex.linear_model import LogisticRegression as LogisticRegression_EX + model_map['primary_models']['LogisticRegression_three_class'] = SKLearnModel(LogisticRegression_EX(C=10, random_state=1, max_iter=1000, n_jobs=-1), 'classification') + elif model_name == 'LDA': + from sklearn.discriminant_analysis import LinearDiscriminantAnalysis + model_map['primary_models']['LDA'] = SKLearnModel(LinearDiscriminantAnalysis(), 'classification') + elif model_name == 'KNN': + from sklearn.neighbors import KNeighborsClassifier + model_map['primary_models']['KNN'] = SKLearnModel(KNeighborsClassifier(), 'classification') + elif model_name == 'CART': + from sklearn.tree import DecisionTreeClassifier + model_map['primary_models']['CART'] = SKLearnModel(DecisionTreeClassifier(max_depth=15, random_state=1), 'classification') + elif model_name == 'NB': + from sklearn.naive_bayes import GaussianNB + model_map['primary_models']['NB'] = SKLearnModel(GaussianNB(), 'classification') + elif model_name == 'AB': + from sklearn.ensemble import AdaBoostClassifier + model_map['primary_models']['AB'] = SKLearnModel(AdaBoostClassifier(n_estimators=15), 'classification') + elif model_name == 'RFC': + model_map['primary_models']['RFC'] = SKLearnModel(RandomForestClassifier(n_jobs=-1, max_depth=20, random_state=1), 'classification') + elif model_name == 'SVC': + from sklearn.svm import SVC + model_map['primary_models']['SVC'] = SKLearnModel(SVC(kernel='rbf', C=1e3, probability=True, random_state=1), 'classification') + elif model_name == 'XGB_two_class': + from xgboost import XGBClassifier + from models.xgboost import XGBoostModel + model_map['primary_models']['XGB_two_class'] = XGBoostModel(XGBClassifier(n_jobs=-1, max_depth = 20, random_state=1, objective='binary:logistic', use_label_encoder= False, eval_metric='mlogloss')) + elif model_name == 'LGBM': + from lightgbm import LGBMClassifier + model_map['primary_models']['LGBM'] = SKLearnModel(LGBMClassifier(n_jobs=-1, max_depth=20, random_state=1), 'classification') + elif model_name == 'StaticMom': + from models.momentum import StaticMomentumModel + model_map['primary_models']['StaticMom'] = StaticMomentumModel(allow_short=True) + elif model_name == 'Average': + from models.average import StaticAverageModel + model_map['ensemble_models']['Average'] = StaticAverageModel() - return model_map, model_names_classification, model_names_regression, default_feature_selector_regression, default_feature_selector_classification + return model_map diff --git a/models/momentum.py b/models/momentum.py index a4f17f6..b93e93e 100644 --- a/models/momentum.py +++ b/models/momentum.py @@ -7,9 +7,9 @@ class StaticMomentumModel(Model): Model that uses only one feature: momentum. It's positive if momentum is greater than 0, otherwise it's negative. ''' + method = 'classification' data_transformation = 'original' only_column = 'mom' - feature_selection = 'off' model_type = 'static' predict_window_size = 'single_timestamp' diff --git a/models/naive.py b/models/naive.py index bbf6688..1efa317 100644 --- a/models/naive.py +++ b/models/naive.py @@ -7,9 +7,9 @@ class StaticNaiveModel(Model): Model that carries the last observation (from returns) to the next one, naively. ''' + method = 'regression' data_transformation = 'original' only_column = None - feature_selection = 'off' model_type = 'static' predict_window_size = 'single_timestamp' diff --git a/models/neural.py b/models/neural.py index 0114ebc..2c78809 100644 --- a/models/neural.py +++ b/models/neural.py @@ -7,9 +7,9 @@ import pytorch_lightning as pl class LightningNeuralNetModel(Model): + method = 'regression' data_transformation = 'transformed' only_column = None - feature_selection = 'off' model_type = 'ml' ''' Standard lightning methods ''' diff --git a/models/sklearn.py b/models/sklearn.py index 269fcf9..2da08e7 100644 --- a/models/sklearn.py +++ b/models/sklearn.py @@ -1,4 +1,5 @@ from __future__ import annotations +from typing import Literal from models.base import Model import numpy as np from sklearn.base import clone @@ -6,14 +7,15 @@ from sklearn.base import clone class SKLearnModel(Model): + method: Literal["regression", "classification"] data_transformation = 'transformed' only_column = None - feature_selection = 'on' model_type = 'ml' predict_window_size = 'single_timestamp' - def __init__(self, model): + def __init__(self, model, method: Literal['regression', 'classification']): self.model = model + self.method = method def fit(self, X: np.ndarray, y: np.ndarray) -> None: self.model.fit(X, y) @@ -24,7 +26,7 @@ class SKLearnModel(Model): return (pred, probability) def clone(self) -> SKLearnModel: - return SKLearnModel(clone(self.model)) + return SKLearnModel(clone(self.model), self.method) def get_name(self) -> str: return self.model.__class__.__name__ diff --git a/models/statsmodels.py b/models/statsmodels.py index 4ac8e20..d67086e 100644 --- a/models/statsmodels.py +++ b/models/statsmodels.py @@ -10,7 +10,6 @@ class StatsModel(Model): # This is work in progress data_transformation = 'transformed' only_column = None - feature_selection = 'on' model_type = 'ml' predict_window_size = 'single_timestamp' diff --git a/models/xgboost.py b/models/xgboost.py index d6e9189..a7c0ab9 100644 --- a/models/xgboost.py +++ b/models/xgboost.py @@ -6,9 +6,9 @@ from sklearn.base import clone class XGBoostModel(Model): + method = 'classification' data_transformation = 'transformed' only_column = None - feature_selection = 'on' model_type = 'ml' predict_window_size = 'single_timestamp' diff --git a/run_feature_selection.py b/run_feature_selection.py deleted file mode 100644 index 2f30c80..0000000 --- a/run_feature_selection.py +++ /dev/null @@ -1,39 +0,0 @@ -# #%% -# import pandas as pd -# import pandas_ta as ta -# from config.config import get_default_ensemble_config -# from config.preprocess import preprocess_config -# from data_loader.load_data import load_data - -# # %% -# model_config, training_config, data_config = get_default_ensemble_config() -# model_config, training_config, data_config = preprocess_config(model_config, training_config, data_config) - -# data_config['target_asset'] = data_config['assets'][0] -# X, y, target_returns = load_data(**data_config) -# # %% -# X.ta.donchian() - - -# # %% -# X.ta.ema() -# # %% -# X.ta.adjusted = "ADA_USD_returns" - -# # %% -# X.ta.sma(length=10) - -# # %% -# X -# # %% -# X.ta.categories - -# # %% -# ind_list = X.ta.indicators(as_list=True) - -# # %% -# ind_list -# # %% -# X.ta.ao('ADA_USD_returns', length=10) -# # %% -# ta.ao() \ No newline at end of file diff --git a/run_pipeline.py b/run_pipeline.py index 8b34c3f..9e2dd88 100644 --- a/run_pipeline.py +++ b/run_pipeline.py @@ -2,7 +2,7 @@ import pandas as pd from typing import Callable, Optional from data_loader.load_data import load_data -from data_loader.process_data import process_data, check_data +from data_loader.process_data import check_data from reporting.wandb import launch_wandb, register_config_with_wandb from reporting.reporting import report_results @@ -56,13 +56,12 @@ def __run_training(model_config:dict, training_config:dict, data_config:dict): # 1. Load data, check for validity and process data (feature selection, dimensionality reduction, etc.) X, y, target_returns = load_data(**configs['data_config']) if check_data(X, y, configs['training_config']) is False: continue - X, original_X = process_data(X, y, configs) # 2. Train a Primary model with optional metalabeling for each asset - training_step_primary, current_predictions = primary_step(X, y, original_X, asset, target_returns, configs, reporting) + training_step_primary, current_predictions = primary_step(X, y, asset, target_returns, configs, reporting) # 3. Train an Ensemble model with optional metalabeling for each asset - training_step_secondary = secondary_step(X, y, original_X, current_predictions, asset, target_returns, configs, reporting) + training_step_secondary = secondary_step(X, y, current_predictions, asset, target_returns, configs, reporting) # 4. Save the models reporting.all_assets.append(Reporting.Asset(ticker=asset[1], primary=training_step_primary, secondary=training_step_secondary)) diff --git a/sweep_ensemble.yaml b/sweep_ensemble.yaml index ccdaaf6..d1cf254 100644 --- a/sweep_ensemble.yaml +++ b/sweep_ensemble.yaml @@ -47,11 +47,11 @@ parameters: primary_models: distribution: categorical values: - - ["LDA", "LR_two_class", "KNN", "SVC", "CART", "NB", "AB", "RFC", "XGB_two_class", "LGBM", "StaticMom"] - - ["LR_two_class", "LDA", "NB", "RFC", "XGB_two_class", "LGBM", "StaticMom"] - - ["LR_two_class", "LDA", "LGBM", "RFC", "XGB_two_class"] + - ["LDA", "LogisticRegression_two_class", "KNN", "SVC", "CART", "NB", "AB", "RFC", "XGB_two_class", "LGBM", "StaticMom"] + - ["LogisticRegression_two_class", "LDA", "NB", "RFC", "XGB_two_class", "LGBM", "StaticMom"] + - ["LogisticRegression_two_class", "LDA", "LGBM", "RFC", "XGB_two_class"] meta_labeling_models: - value: ["LGBM", "LR_two_class"] + value: ["LGBM", "LogisticRegression_two_class"] own_features: value: ['date_days', 'level_2', 'lags_up_to_5'] other_features: diff --git a/sweep_meta_labeling.yaml b/sweep_meta_labeling.yaml index 8f1c35f..5d1b0a5 100644 --- a/sweep_meta_labeling.yaml +++ b/sweep_meta_labeling.yaml @@ -51,9 +51,9 @@ parameters: index_column: value: 'int' primary_models: - value: ["LR_two_class", "LDA", "KNN", "CART", "NB", "AB", "RFC", "StaticMom"] + value: ["LogisticRegression_two_class", "LDA", "KNN", "CART", "NB", "AB", "RFC", "StaticMom"] meta_labeling_models: - values: ["LR_two_class", "LDA", "KNN", "CART", "NB", "AB", "RFC"] + values: ["LogisticRegression_two_class", "LDA", "KNN", "CART", "NB", "AB", "RFC"] distribution: categorical own_features: values: [['single_mom', 'date_days'], [], ['level_1', 'date_days'], ['date_days', 'level_2']] diff --git a/sweep_primary_models.yaml b/sweep_primary_models.yaml index 37fa414..c34b30e 100644 --- a/sweep_primary_models.yaml +++ b/sweep_primary_models.yaml @@ -45,10 +45,10 @@ parameters: index_column: value: 'int' primary_models: - values: [['LR_two_class'], ['SVC'], ['LDA'], ['KNN'], ['CART'], ['MNB'], ['NB'], ['AB'], ['RFC'], ['XGB_two_class'], ['LGBM']] + values: [['LogisticRegression_two_class'], ['SVC'], ['LDA'], ['KNN'], ['CART'], ['MNB'], ['NB'], ['AB'], ['RFC'], ['XGB_two_class'], ['LGBM']] distribution: categorical meta_labeling_models: - value: ["LGBM", "LR_two_class"] + value: ["LGBM", "LogisticRegression_two_class"] own_features: value: ['date_days', 'level_2', 'lags_up_to_5'] other_features: diff --git a/tests/test_evaluation.py b/tests/test_evaluation.py index 419df86..028c906 100644 --- a/tests/test_evaluation.py +++ b/tests/test_evaluation.py @@ -77,14 +77,15 @@ def test_evaluation(): expanding_window=False, window_size=window_length, retrain_every=10, - transformations=[]) + transformations=[], + preloaded_transformations=None) predictions, _ = walk_forward_inference( model_name='test', model_over_time=model_over_time, transformations_over_time=transformations_over_time, X=X, expanding_window=False, - window_size=window_length + window_size=window_length, ) # verify if predictions are the same as y diff --git a/tests/test_walk_forward.py b/tests/test_walk_forward.py index e5c1e44..0d841cd 100644 --- a/tests/test_walk_forward.py +++ b/tests/test_walk_forward.py @@ -75,7 +75,9 @@ def test_walk_forward_train_test(): expanding_window=False, window_size=window_length, retrain_every=10, - transformations=[]) + transformations=[], + preloaded_transformations=None + ) predictions, _ = walk_forward_inference( model_name='test', model_over_time=model_over_time, diff --git a/training/meta_labeling.py b/training/meta_labeling.py index ffeed84..51fa478 100644 --- a/training/meta_labeling.py +++ b/training/meta_labeling.py @@ -1,9 +1,8 @@ from utils.evaluate import discretize_threeway_threshold, evaluate_predictions -from utils.helpers import random_string, equal_except_nan, drop_until_first_valid_index +from utils.helpers import equal_except_nan from training.primary_model import train_primary_model -from feature_selection.feature_selection import select_features import pandas as pd -from models.model_map import get_model_map +from models.model_map import default_feature_selector_classification, default_feature_selector_regression from models.base import Model from reporting.types import Reporting from typing import Union @@ -23,22 +22,14 @@ def train_meta_labeling_model( preloaded_models: Union[list[Reporting.Single_Model], None] = None ) -> tuple[pd.Series, pd.Series, pd.DataFrame, list[Reporting.Single_Model]]: - _, _, _, default_feature_selector_regression, default_feature_selector_classification = get_model_map(model_config) discretize = discretize_threeway_threshold(0.33) discretized_predictions = input_predictions.apply(discretize) meta_y: pd.Series = pd.concat([discretized_predictions, y], axis=1).apply(equal_except_nan, axis = 1) - print("Feature Selection started") - backup_model = default_feature_selector_regression if data_config['method'] == 'regression' else default_feature_selector_classification - meta_feature_selection_input_X, meta_feature_selection_input_y = drop_until_first_valid_index(X, meta_y) - feature_selection_output = select_features(X = meta_feature_selection_input_X, y = meta_feature_selection_input_y, model = models[0][1], n_features_to_select = training_config['n_features_to_select'], backup_model = backup_model, scaling = training_config['scaler']) - meta_selected_features_X = X[feature_selection_output.columns] - - meta_X = pd.concat([meta_selected_features_X, input_predictions, discretized_predictions], axis = 1) + meta_X = pd.concat([X, input_predictions, discretized_predictions], axis = 1) _, meta_preds, meta_probabilities, all_models_single_asset = train_primary_model( ticker_to_predict = "prediction_correct", - original_X = meta_X, X = meta_X, y = meta_y, target_returns = target_returns, diff --git a/training/primary_model.py b/training/primary_model.py index 7d9dea1..533ddcf 100644 --- a/training/primary_model.py +++ b/training/primary_model.py @@ -6,10 +6,11 @@ from models.base import Model from utils.scaler import get_scaler from utils.types import ScalerTypes from reporting.types import Reporting +from transformations.rfe import RFETransformation +from transformations.pca import PCATransformation def train_primary_model( ticker_to_predict: str, - original_X: pd.DataFrame, X: pd.DataFrame, y: pd.Series, target_returns: pd.Series, @@ -29,28 +30,35 @@ def train_primary_model( predictions = pd.DataFrame(index=y.index) probabilities = pd.DataFrame(index=y.index) all_models_single_asset: list[Reporting.Single_Model] = [] - + if preloaded_models is not None: models = preloaded_models + transformations_over_time = None + for model_name, model in models: if preloaded_models is None: model_over_time, transformations_over_time = walk_forward_train( model_name=model_name, model = model, - X = X if model.feature_selection == 'on' else original_X, + X = X, y = y, target_returns = target_returns, expanding_window = expanding_window, window_size = sliding_window_size, retrain_every = retrain_every, - transformations= [get_scaler(scaler)], + transformations= [ + get_scaler(scaler), + PCATransformation(ratio_components_to_keep=0.5, sliding_window_size=sliding_window_size), + RFETransformation(n_feature_to_select=40, model=model) + ], + preloaded_transformations=transformations_over_time, ) preds, probs = walk_forward_inference( model_name = model_name, model_over_time= model_over_time if preloaded_models is None else pd.Series(model), transformations_over_time = transformations_over_time, - X = X if model.feature_selection == 'on' else original_X, + X = X, expanding_window = expanding_window, window_size = sliding_window_size ) diff --git a/training/training_steps.py b/training/training_steps.py index be507ff..a4c46c1 100644 --- a/training/training_steps.py +++ b/training/training_steps.py @@ -11,7 +11,6 @@ from typing import Union def primary_step( X: pd.DataFrame, y:pd.Series, - original_X:pd.DataFrame, asset:list, target_returns:pd.Series, configs: dict, @@ -24,7 +23,6 @@ def primary_step( # 3. Train Primary models current_result, current_predictions, current_probabilities, all_models_for_single_asset = train_primary_model( ticker_to_predict = asset[1], - original_X = original_X, X = X, y = y, target_returns = target_returns, @@ -48,7 +46,7 @@ def primary_step( primary_model_predictions = current_predictions[model_name] primary_meta_result, primary_meta_preds, primary_meta_probabilities, meta_labeling_models = train_meta_labeling_model( target_asset=asset[1], - X = original_X, + X = X, input_predictions= primary_model_predictions, y = y, target_returns = target_returns, @@ -75,7 +73,6 @@ def primary_step( def secondary_step( X:pd.DataFrame, y:pd.Series, - original_X:pd.DataFrame, current_predictions:pd.DataFrame, asset:list, target_returns:pd.Series, @@ -89,7 +86,6 @@ def secondary_step( if model_config['ensemble_model'] is not None: ensemble_result, ensemble_predictions, _, ensemble_models_one_asset = train_primary_model( ticker_to_predict = asset[1], - original_X = current_predictions, X = current_predictions, y = y, target_returns = target_returns, @@ -116,7 +112,7 @@ def secondary_step( # 3. Train a Meta-labeling model on the averaged level-1 model predictions ensemble_meta_result, ensemble_meta_predictions, ensemble_meta_probabilities, ensemble_meta_labeling_models = train_meta_labeling_model( target_asset=asset[1], - X = original_X, + X = X, input_predictions= ensemble_predictions, y = y, target_returns = target_returns, diff --git a/training/walk_forward.py b/training/walk_forward.py index ac4dfc5..c53bd9b 100644 --- a/training/walk_forward.py +++ b/training/walk_forward.py @@ -3,10 +3,8 @@ from models.base import Model import numpy as np from utils.helpers import get_first_valid_return_index from tqdm import tqdm -from sklearn.preprocessing import MinMaxScaler, Normalizer, StandardScaler -from typing import Union -from sklearn.base import clone from transformations.base import Transformation +from typing import Optional def walk_forward_train( model_name: str, @@ -18,6 +16,7 @@ def walk_forward_train( window_size: int, retrain_every: int, transformations: list[Transformation], + preloaded_transformations: Optional[list[pd.Series]], ) -> tuple[pd.Series, list[pd.Series]]: assert len(X) == len(y) models_over_time = pd.Series(index=y.index).rename(model_name) @@ -44,9 +43,12 @@ def walk_forward_train( X_expanding_window = X[first_nonzero_return:train_window_end] y_expanding_window = y[first_nonzero_return:train_window_end] - current_transformations = [t.clone() for t in transformations] - for transformation_index, transformation in enumerate(current_transformations): - transformation.fit_transform(X_expanding_window, y_expanding_window) + if preloaded_transformations is not None and len(transformations) > 0: + current_transformations = [transformation_over_time[index] for transformation_over_time in preloaded_transformations] + else: + current_transformations = [t.clone() for t in transformations] + for transformation_index, transformation in enumerate(current_transformations): + X_expanding_window = transformation.fit_transform(X_expanding_window, y_expanding_window) X_slice = X[train_window_start:train_window_end] diff --git a/transformations/pca.py b/transformations/pca.py new file mode 100644 index 0000000..e358802 --- /dev/null +++ b/transformations/pca.py @@ -0,0 +1,39 @@ +from __future__ import annotations +from transformations.base import Transformation +from typing import Optional +from copy import deepcopy +from sklearn.decomposition import PCA +import pandas as pd + +class PCATransformation(Transformation): + + pca: PCA + + def __init__(self, ratio_components_to_keep: float, sliding_window_size: int): + self.ratio_components_to_keep = ratio_components_to_keep + self.sliding_window_size = sliding_window_size + + def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None) -> None: + self.pca = PCA(n_components = min(int(len(X.columns) * self.ratio_components_to_keep), self.sliding_window_size)) + self.pca.fit(X, y) + + def fit_transform(self, X: pd.DataFrame, y: Optional[pd.Series] = None) -> pd.DataFrame: + self.fit(X, y) + return self.transform(X) + + def transform(self, X: pd.DataFrame) -> pd.DataFrame: + X = pd.DataFrame(self.pca.transform(X), index = X.index) + X.columns = ['PCA_' + str(i) for i in range(1, len(X.columns)+1)] + return X + + def clone(self) -> PCATransformation: + return deepcopy(self) + + def get_name(self) -> str: + return "PCA" + + + + + + diff --git a/transformations/rfe.py b/transformations/rfe.py new file mode 100644 index 0000000..2c2b05a --- /dev/null +++ b/transformations/rfe.py @@ -0,0 +1,46 @@ +from __future__ import annotations +from transformations.base import Transformation +from typing import Optional +from copy import deepcopy +from sklearn.feature_selection import RFE +import pandas as pd +from models.base import Model +from models.model_map import default_feature_selector_classification + +class RFETransformation(Transformation): + + rfe: RFE + n_feature_to_select: int + + def __init__(self, n_feature_to_select: int, model: Model, step = 0.1): + self.n_feature_to_keep = n_feature_to_select + self.model = model + if hasattr(self.model, 'model') == False: return + if hasattr(self.model.model, 'feature_importances_') == False and hasattr(self.model.model, 'coef_') == False: + model = default_feature_selector_classification + self.rfe = RFE(model.model, n_features_to_select= n_feature_to_select, step=step) + + def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None) -> None: + if self.rfe is None: return + self.rfe.fit(X, y) + + def fit_transform(self, X: pd.DataFrame, y: Optional[pd.Series] = None) -> pd.DataFrame: + if self.rfe is None: return X + self.fit(X, y) + return self.transform(X) + + def transform(self, X: pd.DataFrame) -> pd.DataFrame: + if self.rfe is None: return X + return pd.DataFrame(X[X.columns[self.rfe.support_]], index= X.index) + + def clone(self) -> RFETransformation: + return deepcopy(self) + + def get_name(self) -> str: + return "RFE" + + + + + +