mirror of
https://github.com/webclinic017/drift.git
synced 2026-07-27 18:57:55 +00:00
feat(Project): use SKLearn models directly, removed custom ensembling, use 5 minute data, batch inference, numba cusum filter (#192)
* feat(Project): use 5 minute data, running training in parallel, sped up cusum filter by 10x with numba * fix(WalkForward): inference mini-batch parallelization * fix(WalkForward): don't use the parallel version of any of the functions * feat(CI): download the data required * fix(Project): 5min_crypto folder added * fix(Evaluate): make sure we have numerical stability in returns * feat(Models): use SKLearn models directly to enable composability * feat(Inference): batched inference now working, added forecasting_horizon * fix(Inference): works again * fix(Inference) * chore(Models): remove unused Ensemble model * fix(Labeller): don't just forward shift returns, also take the sum of the data happened until then * Update test.yml
This commit is contained in:
committed by
GitHub
parent
5c94af8b01
commit
9d47ee942d
@@ -4,21 +4,17 @@ from typing import Optional
|
||||
from copy import deepcopy
|
||||
from sklearn.feature_selection import RFE
|
||||
import pandas as pd
|
||||
from models.base import Model
|
||||
from models.model_map import default_feature_selector_classification
|
||||
from models.sklearn import SKLearnModel
|
||||
|
||||
class RFETransformation(Transformation):
|
||||
|
||||
rfe: RFE
|
||||
n_feature_to_select: int
|
||||
|
||||
def __init__(self, n_feature_to_select: int, model: Model, step = 0.1):
|
||||
def __init__(self, n_feature_to_select: int, model: SKLearnModel, step = 0.1):
|
||||
self.n_feature_to_keep = n_feature_to_select
|
||||
self.model = model
|
||||
if hasattr(self.model, 'model') == False: return
|
||||
if hasattr(self.model.model, 'feature_importances_') == False and hasattr(self.model.model, 'coef_') == False:
|
||||
model = default_feature_selector_classification
|
||||
self.rfe = RFE(model.model, n_features_to_select= n_feature_to_select, step=step)
|
||||
self.rfe = RFE(model, n_features_to_select= n_feature_to_select, step=step)
|
||||
|
||||
def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None) -> None:
|
||||
if self.rfe is None: return
|
||||
|
||||
Reference in New Issue
Block a user