mirror of
https://github.com/webclinic017/drift.git
synced 2026-08-15 11:58:07 +00:00
feature(MetaLabeling): replaced previous non-functional Ensembling method with Meta-labeling method available for both lvl1 and lvl2 models (#110)
* feature(MetaLabeling): added hacky prototype * fix(MetaLabeling): drop index until first valid X & y * fix(MetaLabeling): transform both X & y before feature selection * fix(MetaLabeling): got feature selection to work * fix(MetaLabeling): correct values for meta_y * feat(MetaLabeling): created predictions multiplied by bet sizes * feat(Pipeline): print out averaged result * fix(Evaluation): correctly deal with non-discretized data * fix(Pipeline): use the right column names * refactor(Pipeline): move out meta-labeling * refactor(Pipeline): complete refactoring * feat(CI): post results to PR * fix(Pipeline): use the correct filename * chore(Config): removed now redundant feature_selection flag * feat(Models): added SVC * fix(Pipeline): accidentally switched two return values * feat(Sweep): prepared sweep_meta.yaml, moved report_results() into a separate file * fix(Pipeline): wrong function name * fix(Sweep): yaml + run_sweep * fix(Sweep): typo in name * fix(Reporting): only save averaged results * feat(MetaLabeling): use optional meta-labeling step for every lvl1 models, before averaging * feat(Reporting): print out sharpe improvement in meta-labeling step * fix(Sweep): adjusted config, defaulted to good defaults * fix(Sweep): adjusted sweep
This commit is contained in:
+21
-3
@@ -1,12 +1,15 @@
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
import os
|
||||
import string
|
||||
import random
|
||||
from typing import Union
|
||||
|
||||
def get_files_from_dir(path: str) -> list[str]:
|
||||
return [f for f in os.listdir(path) if os.path.isfile(os.path.join(path,f)) and not f.startswith('.')]
|
||||
|
||||
def get_first_valid_return_index(series: pd.Series) -> int:
|
||||
double_nested_results = np.where(np.logical_and(series != 0, np.logical_not(np.isnan(series))))
|
||||
double_nested_results = np.where(np.logical_and(series != 0, np.logical_not(pd.isna(series))))
|
||||
if len(double_nested_results) == 0:
|
||||
return 0
|
||||
nested_result = double_nested_results[0]
|
||||
@@ -17,7 +20,7 @@ def get_first_valid_return_index(series: pd.Series) -> int:
|
||||
def flatten(list_of_lists: list) -> list:
|
||||
return [item for sublist in list_of_lists for item in sublist]
|
||||
|
||||
def weighted_average(df: pd.DataFrame, weights_source: str) -> pd.DataFrame:
|
||||
def weighted_average(df: pd.DataFrame, weights_source: str) -> pd.Series:
|
||||
if df.shape[0] == 0:
|
||||
return df
|
||||
mean_df = df.iloc[:,0]
|
||||
@@ -35,4 +38,19 @@ def drop_columns_if_exist(df: pd.DataFrame, columns: list) -> pd.DataFrame:
|
||||
for column in columns:
|
||||
if column in df.columns:
|
||||
df = df.drop(column, axis=1)
|
||||
return df
|
||||
return df
|
||||
|
||||
def random_string(n: int) -> str:
|
||||
return ''.join(random.choices(string.ascii_uppercase + string.digits, k=n))
|
||||
|
||||
def equal_except_nan(row: pd.Series):
|
||||
if np.isnan(row.iloc[0]) or np.isnan(row.iloc[1]):
|
||||
return np.nan
|
||||
if row.iloc[0] == row.iloc[1]:
|
||||
return 1.
|
||||
else:
|
||||
return 0.
|
||||
|
||||
def drop_until_first_valid_index(df: pd.DataFrame, series: pd.Series) -> tuple[pd.DataFrame, pd.Series]:
|
||||
first_valid_index = max(get_first_valid_return_index(df.iloc[:,0]), get_first_valid_return_index(series))
|
||||
return df.iloc[first_valid_index:], series.iloc[first_valid_index:]
|
||||
Reference in New Issue
Block a user