feat(FeatureExtraction): added fractionally differentiated returns to remove lagged returns (#95)

* feat(FeatureExtraction): added fractionally differentiated returns to remove lagged returns

* fix(Sweep): config

* fix(Sweep): name

* fix(Sweep): grid

* feat(Config): separated sliding_window_size_level1 & sliding_window_size_level2

* feat(Dependencies): added ray, now using it to parallel process feature extraction

* fix(Dependencies): added pip explicitly

* fix(Dependencies): removed ray from root

* fix(Models): average model was probably not taking the right timestamp to average

* feat(Config): separated expanding_window_level1 & expanding_window_level2

* fix(Config): set n_features_to_select to the optimal 30
This commit is contained in:
Mark Aron Szulyovszky
2021-12-28 22:50:09 +01:00
committed by GitHub
parent cc70d3f907
commit f762ceed2a
13 changed files with 626 additions and 417 deletions
+2 -2
View File
@@ -4,7 +4,7 @@ import pandas as pd
from models.base import Model, SKLearnModel
from sklearn.decomposition import PCA
def select_features(X: pd.DataFrame, y: pd.Series, model: Model, min_features_to_select: int, backup_model: SKLearnModel) -> pd.DataFrame:
def select_features(X: pd.DataFrame, y: pd.Series, model: Model, n_features_to_select: int, backup_model: SKLearnModel) -> pd.DataFrame:
''' Select features using RFECV, returns a pd.DataFrame (X) with only the selected features.'''
if model.model_type != 'ml': return X
@@ -16,7 +16,7 @@ def select_features(X: pd.DataFrame, y: pd.Series, model: Model, min_features_to
feat_selector_model = backup_model.model
# selector = RFECV(feat_selector_model, cv = cv, step=5, min_features_to_select=min_features_to_select)
selector = RFE(feat_selector_model, n_features_to_select=10)
selector = RFE(feat_selector_model, n_features_to_select= n_features_to_select)
selector = selector.fit(X, y)
print("Kept %d features out of %d" % (selector.n_features_, X.shape[1]))