feat(Transformations): replaced feature selection pre-processing step with online version (with cache) (#170)

* feat(Transformations): removed feature-selection pre-processing step completely

* fix(Core): removed unnecessary `original_X`

* fix(Transformations): use the X_expanding_window to transform subsequent data

* fix(RFE): should check for model correctly

* fix(Config): only re-train the model every 40 timestamp

* fix(MetaLabeling): pass in the correct X to meta-labeling step

* fix(Transformation): PCA should at least keep as many features as sliding_window_size

* feat(Transformations): cache transformations across the same asset

* fix(Tests): missing preloaded_transformations arg

* chore(Config): got rid of unnecessary 'classification_models' and 'regression_models' dictionary keys
This commit is contained in:
Mark Aron Szulyovszky
2022-01-17 11:43:51 +01:00
committed by GitHub
parent 31dc847be1
commit 6b26643ece
27 changed files with 240 additions and 278 deletions
+7 -7
View File
@@ -30,7 +30,7 @@ def get_dev_config() -> tuple[dict, dict, dict]:
)
regression_models = ["Lasso"]
classification_models = ["LR_two_class"]
classification_models = ["LogisticRegression_two_class"]
model_config = dict(
primary_models = regression_models if data_config['method'] == 'regression' else classification_models,
@@ -52,7 +52,7 @@ def get_default_ensemble_config() -> tuple[dict, dict, dict]:
expanding_window_meta_labeling = True,
sliding_window_size_primary = 380,
sliding_window_size_meta_labeling = 240,
retrain_every = 20,
retrain_every = 40,
scaler = 'minmax', # 'normalize' 'minmax' 'standardize'
)
@@ -73,8 +73,8 @@ def get_default_ensemble_config() -> tuple[dict, dict, dict]:
)
regression_models = ["Lasso", "KNN", "RFR"]
classification_models = ["LR_two_class", "LDA", "NB", "RFC", "XGB_two_class", "LGBM", "StaticMom"]
meta_labeling_models = ['LR_two_class', 'LGBM']
classification_models = ["LogisticRegression_two_class", "LDA", "NB", "RFC", "XGB_two_class", "LGBM", "StaticMom"]
meta_labeling_models = ['LogisticRegression_two_class', 'LGBM']
ensemble_model = 'Average'
model_config = dict(
@@ -97,7 +97,7 @@ def get_lightweight_ensemble_config() -> tuple[dict, dict, dict]:
expanding_window_meta_labeling = True,
sliding_window_size_primary = 380,
sliding_window_size_meta_labeling = 240,
retrain_every = 20,
retrain_every = 40,
scaler = 'minmax', # 'normalize' 'minmax' 'standardize'
)
@@ -118,8 +118,8 @@ def get_lightweight_ensemble_config() -> tuple[dict, dict, dict]:
)
regression_models = ["Lasso", "KNN"]
classification_models = ['LR_two_class', 'SVC']
meta_labeling_models = ['LR_two_class', 'LGBM']
classification_models = ['LogisticRegression_two_class', 'SVC']
meta_labeling_models = ['LogisticRegression_two_class', 'LGBM']
ensemble_model = 'Average'
model_config = dict(
+3 -3
View File
@@ -21,10 +21,10 @@ def __preprocess_feature_extractors_config(data_dict: dict) -> dict:
return data_dict
def __preprocess_model_config(model_config:dict, method:str) -> dict:
model_map, _, _, _, _ = get_model_map(model_config)
model_config['primary_models'] = [(model_name, model_map[method + '_models'][model_name]) for model_name in model_config['primary_models']]
model_map = get_model_map(model_config)
model_config['primary_models'] = [(model_name, model_map['primary_models'][model_name]) for model_name in model_config['primary_models']]
if len(model_config['meta_labeling_models']) > 0:
model_config['meta_labeling_models'] = [(model_name, model_map[method + '_models'][model_name]) for model_name in model_config['meta_labeling_models']]
model_config['meta_labeling_models'] = [(model_name, model_map['primary_models'][model_name]) for model_name in model_config['meta_labeling_models']]
if model_config['ensemble_model'] is not None:
model_config['ensemble_model'] = (model_config['ensemble_model'], model_map['ensemble_models'][model_config['ensemble_model']])