2021-12-17 14:32:17 +01:00
|
|
|
import pandas as pd
|
|
|
|
|
import numpy as np
|
2021-12-31 19:04:27 +01:00
|
|
|
import os
|
2022-01-06 16:36:45 +01:00
|
|
|
import string
|
|
|
|
|
import random
|
2022-03-15 14:43:16 +01:00
|
|
|
from itertools import chain, combinations, dropwhile
|
2021-12-31 19:04:27 +01:00
|
|
|
|
2022-02-17 19:22:17 +01:00
|
|
|
|
2021-12-31 19:04:27 +01:00
|
|
|
def get_files_from_dir(path: str) -> list[str]:
|
2022-02-17 19:22:17 +01:00
|
|
|
return [
|
|
|
|
|
f
|
|
|
|
|
for f in os.listdir(path)
|
|
|
|
|
if os.path.isfile(os.path.join(path, f)) and not f.startswith(".")
|
|
|
|
|
]
|
|
|
|
|
|
2021-12-17 14:32:17 +01:00
|
|
|
|
|
|
|
|
def get_first_valid_return_index(series: pd.Series) -> int:
|
2022-02-17 19:22:17 +01:00
|
|
|
double_nested_results = np.where(
|
|
|
|
|
np.logical_and(series != 0, np.logical_not(pd.isna(series)))
|
|
|
|
|
)
|
2021-12-26 12:15:11 +01:00
|
|
|
if len(double_nested_results) == 0:
|
|
|
|
|
return 0
|
|
|
|
|
nested_result = double_nested_results[0]
|
|
|
|
|
if len(nested_result) == 0:
|
|
|
|
|
return 0
|
|
|
|
|
return nested_result[0]
|
2021-12-23 10:35:20 +01:00
|
|
|
|
2022-02-17 19:22:17 +01:00
|
|
|
|
2022-01-29 06:41:40 +01:00
|
|
|
def get_last_non_na_index(series: pd.Series, index: int) -> int:
|
2022-02-17 19:22:17 +01:00
|
|
|
return next(
|
|
|
|
|
dropwhile(lambda x: pd.isna(x[1]), enumerate(reversed(series[: index + 1])))
|
|
|
|
|
)[0]
|
2022-01-26 23:22:43 +01:00
|
|
|
|
2022-01-09 20:00:03 +01:00
|
|
|
|
2021-12-23 10:35:20 +01:00
|
|
|
def flatten(list_of_lists: list) -> list:
|
2021-12-26 12:15:11 +01:00
|
|
|
return [item for sublist in list_of_lists for item in sublist]
|
|
|
|
|
|
2022-02-17 19:22:17 +01:00
|
|
|
|
2022-01-06 16:36:45 +01:00
|
|
|
def weighted_average(df: pd.DataFrame, weights_source: str) -> pd.Series:
|
2022-01-09 17:21:06 +01:00
|
|
|
if df.shape[1] == 0:
|
2021-12-27 21:59:22 +01:00
|
|
|
return df
|
2022-02-17 19:22:17 +01:00
|
|
|
mean_df = df.iloc[:, 0]
|
2021-12-26 12:15:11 +01:00
|
|
|
weights = df.loc[weights_source]
|
|
|
|
|
|
|
|
|
|
for i, row in df.iterrows():
|
2022-02-17 19:22:17 +01:00
|
|
|
if i == weights_source:
|
|
|
|
|
continue
|
2021-12-26 12:15:11 +01:00
|
|
|
mean_df.loc[i] = (row * weights).sum() / df.loc[weights_source].sum()
|
|
|
|
|
|
|
|
|
|
return mean_df
|
|
|
|
|
|
2022-02-17 19:22:17 +01:00
|
|
|
|
2022-01-03 13:57:36 +01:00
|
|
|
def drop_columns_if_exist(df: pd.DataFrame, columns: list) -> pd.DataFrame:
|
|
|
|
|
for column in columns:
|
|
|
|
|
if column in df.columns:
|
|
|
|
|
df = df.drop(column, axis=1)
|
2022-01-06 16:36:45 +01:00
|
|
|
return df
|
|
|
|
|
|
2022-02-17 19:22:17 +01:00
|
|
|
|
2022-01-06 16:36:45 +01:00
|
|
|
def random_string(n: int) -> str:
|
2022-02-17 19:22:17 +01:00
|
|
|
return "".join(random.choices(string.ascii_uppercase + string.digits, k=n))
|
|
|
|
|
|
2022-01-06 16:36:45 +01:00
|
|
|
|
|
|
|
|
def equal_except_nan(row: pd.Series):
|
|
|
|
|
if np.isnan(row.iloc[0]) or np.isnan(row.iloc[1]):
|
|
|
|
|
return np.nan
|
|
|
|
|
if row.iloc[0] == row.iloc[1]:
|
2022-02-17 19:22:17 +01:00
|
|
|
return 1.0
|
2022-01-06 16:36:45 +01:00
|
|
|
else:
|
2022-02-17 19:22:17 +01:00
|
|
|
return 0.0
|
|
|
|
|
|
2022-01-06 16:36:45 +01:00
|
|
|
|
2022-02-17 19:22:17 +01:00
|
|
|
def drop_until_first_valid_index(
|
|
|
|
|
df: pd.DataFrame, series: pd.Series
|
|
|
|
|
) -> tuple[pd.DataFrame, pd.Series]:
|
|
|
|
|
first_valid_index = max(
|
|
|
|
|
get_first_valid_return_index(df.iloc[:, 0]),
|
|
|
|
|
get_first_valid_return_index(series),
|
|
|
|
|
)
|
|
|
|
|
return df.iloc[first_valid_index:], series.iloc[first_valid_index:]
|
2022-03-15 14:43:16 +01:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def powerset(input: list) -> list[list]:
|
|
|
|
|
p_set = list(chain.from_iterable(combinations(input, r) for r in range(len(input) + 1))) # type: ignore
|
|
|
|
|
return [list(item) for item in p_set if len(item) > 0]
|