mirror of
https://github.com/webclinic017/drift.git
synced 2026-07-28 03:08:01 +00:00
3eb3ea94e3
* refactor(Training): added InferenceResult & TrainedModel types * refactor(Pipeline): introduced TrainingOutcome, BetSizingWithMetaOutcome, etc. * fix(Pipeline): getting it to compile * refactor(WalkForward): separate preprocessing step * feat(Pipeline): separate out transformations processing step * refactor(Pipeline): use the Directional model terminology, put bet_sizing into pipeline instead of hiding it in a step * refactor(WalkForward): moved functions to separate folder * fix(WalkForward): use sparse array to store models, process transformations in parallel (lot faster) * fix(Tests): and evaluation * fix(Tests): for realz * fix(Inference): preloading everything now, renamed primary models to directional models * fix(BetSizing): was running transformations on the wrong data, oops * fix(BetSizing): concatenated on the wrong axis accidentally * fix(Reporting): able to use the new Stats type * fix(BetSizing): renamed int column names * fix(Portfolio): name the column properly * fix(Reporting): rename the correct Series, lol * fix(Inference): walk_forwad_inference() can deal with models not being aligned with the starting index * fix(WalkForward): accidentally using the wrong index * fix(WalkForward): use the correct indicies to fetch last model/transformations * fix(CI): changed the name of the results
19 lines
728 B
Python
19 lines
728 B
Python
import pandas as pd
|
|
from config.types import Config
|
|
import warnings
|
|
from utils.helpers import get_first_valid_return_index
|
|
from data_loader.types import XDataFrame
|
|
|
|
def check_data(X: XDataFrame, config: Config) -> bool:
|
|
""" Returns True if data is valid, else returns False."""
|
|
|
|
if has_enough_samples_to_train(X, config) == False:
|
|
warnings.warn("Not enough samples to train")
|
|
return False
|
|
|
|
return True
|
|
|
|
def has_enough_samples_to_train(X: XDataFrame, config: Config) -> bool:
|
|
first_valid_index = get_first_valid_return_index(X.iloc[:,0])
|
|
samples_to_train = len(X) - first_valid_index
|
|
return samples_to_train > config.sliding_window_size_base + config.sliding_window_size_meta + 100 |