mirror of
https://github.com/webclinic017/drift.git
synced 2026-08-02 13:47:47 +00:00
feat(Project): use SKLearn models directly, removed custom ensembling, use 5 minute data, batch inference, numba cusum filter (#192)
* feat(Project): use 5 minute data, running training in parallel, sped up cusum filter by 10x with numba * fix(WalkForward): inference mini-batch parallelization * fix(WalkForward): don't use the parallel version of any of the functions * feat(CI): download the data required * fix(Project): 5min_crypto folder added * fix(Evaluate): make sure we have numerical stability in returns * feat(Models): use SKLearn models directly to enable composability * feat(Inference): batched inference now working, added forecasting_horizon * fix(Inference): works again * fix(Inference) * chore(Models): remove unused Ensemble model * fix(Labeller): don't just forward shift returns, also take the sum of the data happened until then * Update test.yml
This commit is contained in:
committed by
GitHub
parent
5c94af8b01
commit
9d47ee942d
+1
-1
@@ -49,7 +49,7 @@ def evaluate_predictions(
|
||||
return len(series[series != 0])
|
||||
no_of_samples = count_non_zero(df.y_pred)
|
||||
scorecard['no_of_samples'] = no_of_samples
|
||||
sharpe = sharpe_ratio(df.result)
|
||||
sharpe = sharpe_ratio(df.result + 1e-20)
|
||||
scorecard['sharpe'] = sharpe
|
||||
benchmark_sharpe = sharpe_ratio(df.forward_returns)
|
||||
scorecard['benchmark_sharpe'] = benchmark_sharpe
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
from tqdm import tqdm
|
||||
import ray
|
||||
|
||||
def parallel_compute_with_bar(computations) -> list:
|
||||
|
||||
def to_iterator(obj_ids):
|
||||
while obj_ids:
|
||||
done, obj_ids = ray.wait(obj_ids)
|
||||
yield ray.get(done[0])
|
||||
|
||||
ret = []
|
||||
for x in tqdm(to_iterator(computations), total=len(computations)):
|
||||
ret.append(x)
|
||||
|
||||
return ret
|
||||
@@ -0,0 +1,9 @@
|
||||
import pandas as pd
|
||||
|
||||
def resample_ohlc(df, period):
|
||||
output = pd.DataFrame()
|
||||
output['open'] = df.open.resample(period).first()
|
||||
output['high'] = df.high.resample(period).max()
|
||||
output['low'] = df.low.resample(period).min()
|
||||
output['close'] = df.close.resample(period).last()
|
||||
return output
|
||||
Reference in New Issue
Block a user