Files
drift/utils/helpers.py
T
Mark Aron Szulyovszky b1c04afb13 refactor(Naming): use new convention, added Ensemble model parameter back, support multiple Meta-Labeling models (#132)
* refactor(Naming): use `primary_models` & `meta_labeling_models`

* refactor(Naming): using primary * meta_labeling across config and in pipeline

* feat(Pipeline): added back Ensemble models

* fix(Pipeline): compiler error

* fix(Config): typo

* chore(Pipeline): removed unused averaging step

* revert the changes in discretizing

* chore(Pipeline): remove sharpe improvement logging

* fix(Pipeline): ensemble predictions should be a pd.Series instead of a DataFrame

* fix(Pipeline): discard unnecessary ensemble_probabilities

* fix(Pipeline): fixes regarding various meta-labeling ensemble bugs

* fix(Reporting): use the new naming convention

* fix(Reporting): use the right variable

* feat(Sweep): new sweep for ensemble models

* fix(Sweep): config reference

* fix(Config): simplified dev config

* fix(Models): use the faster LR model

* fix(Models): use LGBM in the meta-labeling model for speed

* fix(Selection): always use the first model for feature selection, commented out caching from select_features() as it's close to redundant in terms of speed
2022-01-09 17:21:06 +01:00

56 lines
1.9 KiB
Python

import pandas as pd
import numpy as np
import os
import string
import random
from typing import Union
def get_files_from_dir(path: str) -> list[str]:
return [f for f in os.listdir(path) if os.path.isfile(os.path.join(path,f)) and not f.startswith('.')]
def get_first_valid_return_index(series: pd.Series) -> int:
double_nested_results = np.where(np.logical_and(series != 0, np.logical_not(pd.isna(series))))
if len(double_nested_results) == 0:
return 0
nested_result = double_nested_results[0]
if len(nested_result) == 0:
return 0
return nested_result[0]
def flatten(list_of_lists: list) -> list:
return [item for sublist in list_of_lists for item in sublist]
def weighted_average(df: pd.DataFrame, weights_source: str) -> pd.Series:
if df.shape[1] == 0:
return df
mean_df = df.iloc[:,0]
weights = df.loc[weights_source]
for i, row in df.iterrows():
if i == weights_source: continue
mean_df.loc[i] = (row * weights).sum() / df.loc[weights_source].sum()
return mean_df
def deduplicate_indexes(df: pd.DataFrame) -> pd.DataFrame: return df[~df.index.duplicated(keep='last')]
def drop_columns_if_exist(df: pd.DataFrame, columns: list) -> pd.DataFrame:
for column in columns:
if column in df.columns:
df = df.drop(column, axis=1)
return df
def random_string(n: int) -> str:
return ''.join(random.choices(string.ascii_uppercase + string.digits, k=n))
def equal_except_nan(row: pd.Series):
if np.isnan(row.iloc[0]) or np.isnan(row.iloc[1]):
return np.nan
if row.iloc[0] == row.iloc[1]:
return 1.
else:
return 0.
def drop_until_first_valid_index(df: pd.DataFrame, series: pd.Series) -> tuple[pd.DataFrame, pd.Series]:
first_valid_index = max(get_first_valid_return_index(df.iloc[:,0]), get_first_valid_return_index(series))
return df.iloc[first_valid_index:], series.iloc[first_valid_index:]