mirror of
https://github.com/PyP-Quant/quant-trading-strategy-templates.git
synced 2026-08-21 14:48:08 +00:00
153 lines
5.6 KiB
Python
153 lines
5.6 KiB
Python
import numpy as np
|
|
import pandas as pd
|
|
from sklearn.ensemble import RandomForestClassifier
|
|
from sklearn.pipeline import Pipeline
|
|
from sklearn.preprocessing import StandardScaler
|
|
|
|
|
|
SYMBOL = "AUDUSD"
|
|
MODEL_NAME = "audusd-multi-timeframe-trend-filter-1m"
|
|
|
|
|
|
def _normalise(data):
|
|
df = data.copy()
|
|
df.columns = [str(c).lower() for c in df.columns]
|
|
if "volume" not in df.columns:
|
|
df["volume"] = 1.0
|
|
for col in ["open", "high", "low", "close", "volume"]:
|
|
df[col] = pd.to_numeric(df[col], errors="coerce")
|
|
return df.dropna(subset=["open", "high", "low", "close"]).reset_index(drop=True)
|
|
|
|
|
|
def _features(df):
|
|
c = df["close"]
|
|
h = df["high"]
|
|
l = df["low"]
|
|
v = df["volume"]
|
|
|
|
# Basic returns
|
|
out = pd.DataFrame(index=df.index)
|
|
out["ret1"] = c.pct_change()
|
|
out["ret3"] = c.pct_change(3)
|
|
out["ret6"] = c.pct_change(6)
|
|
out["ret12"] = c.pct_change(12)
|
|
|
|
# ATR for normalization
|
|
tr = np.maximum(h - l, np.maximum(abs(h - c.shift(1)), abs(l - c.shift(1))))
|
|
atr = pd.Series(tr).rolling(14).mean()
|
|
out["atr"] = atr
|
|
|
|
# ATR-normalized candle range and close location value
|
|
out["range_pct"] = (h - l) / c
|
|
out["body_pct"] = (c - df["open"]) / (h - l).replace(0, np.nan)
|
|
out["close_pos"] = (c - l) / (h - l).replace(0, np.nan)
|
|
|
|
# Multi-timeframe trend filters
|
|
# Short-term trend (5-period EMA vs 13-period EMA)
|
|
ema_5 = c.ewm(span=5, adjust=False).mean()
|
|
ema_13 = c.ewm(span=13, adjust=False).mean()
|
|
out["ema_5_13"] = (ema_5 - ema_13) / c
|
|
|
|
# Medium-term trend (13-period EMA vs 34-period EMA)
|
|
ema_34 = c.ewm(span=34, adjust=False).mean()
|
|
out["ema_13_34"] = (ema_13 - ema_34) / c
|
|
|
|
# Long-term trend (34-period EMA vs 89-period EMA)
|
|
ema_89 = c.ewm(span=89, adjust=False).mean()
|
|
out["ema_34_89"] = (ema_34 - ema_89) / c
|
|
|
|
# Price position relative to EMAs
|
|
out["price_vs_ema5"] = (c - ema_5) / c
|
|
out["price_vs_ema13"] = (c - ema_13) / c
|
|
out["price_vs_ema34"] = (c - ema_34) / c
|
|
out["price_vs_ema89"] = (c - ema_89) / c
|
|
|
|
# Rolling volatility percentile (from idea)
|
|
returns = c.pct_change()
|
|
out["volatility_fast"] = returns.rolling(16).std()
|
|
out["volatility_slow"] = returns.rolling(64).std()
|
|
out["volatility_ratio"] = out["volatility_fast"] / (out["volatility_slow"] + 1e-9)
|
|
out["volatility_percentile"] = out["volatility_fast"].rolling(200).apply(
|
|
lambda x: pd.Series(x).rank(pct=True).iloc[-1] if len(x) > 0 else 0.5, raw=False)
|
|
|
|
# Distance from EMAs (from idea)
|
|
out["ema20_dist"] = (c - c.ewm(span=20, adjust=False).mean()) / c
|
|
out["ema50_dist"] = (c - c.ewm(span=50, adjust=False).mean()) / c
|
|
out["ema200_dist"] = (c - c.ewm(span=200, adjust=False).mean()) / c
|
|
|
|
# Prior swing high and swing low distance (from idea)
|
|
swing_high = h.rolling(20, center=False).max().shift(1)
|
|
swing_low = l.rolling(20, center=False).min().shift(1)
|
|
out["dist_to_swing_high"] = (swing_high - c) / c
|
|
out["dist_to_swing_low"] = (c - swing_low) / c
|
|
|
|
# Volume features
|
|
out["volume_z"] = (v - v.rolling(48).mean()) / (v.rolling(48).std() + 1e-9)
|
|
out["volume_ratio"] = v / v.rolling(20).mean()
|
|
|
|
return out.replace([np.inf, -np.inf], np.nan).dropna()
|
|
|
|
|
|
def _labels(close, index, horizon, threshold):
|
|
fwd = close.pct_change(horizon).shift(-horizon)
|
|
y = pd.Series(1, index=close.index)
|
|
y[fwd > threshold] = 2
|
|
y[fwd < -threshold] = 0
|
|
return y.reindex(index).fillna(1).astype(int)
|
|
|
|
|
|
def train(data, config):
|
|
params = config.get("parameters", {})
|
|
horizon = int(params.get("horizon", 1))
|
|
threshold = float(params.get("threshold", 0.0004))
|
|
df = _normalise(data)
|
|
feat = _features(df)
|
|
y = _labels(df["close"], feat.index, horizon, threshold)
|
|
|
|
model = Pipeline([
|
|
("scaler", StandardScaler()),
|
|
("clf", RandomForestClassifier(n_estimators=200, max_depth=6, min_samples_leaf=5, class_weight="balanced_subsample", random_state=42, n_jobs=-1)),
|
|
])
|
|
model.fit(feat.values.astype(np.float32), y.values)
|
|
preds = model.predict(feat.values.astype(np.float32))
|
|
|
|
metrics = {
|
|
"training_bars": int(len(feat)),
|
|
"feature_count": int(feat.shape[1]),
|
|
"buy_signals": int((preds == 2).sum()),
|
|
"sell_signals": int((preds == 0).sum()),
|
|
"hold_signals": int((preds == 1).sum()),
|
|
}
|
|
return {"model": model, "features": list(feat.columns), "symbol": SYMBOL}, metrics
|
|
|
|
|
|
def predict(model, market_data, config):
|
|
params = config.get("parameters", {})
|
|
lookback = int(params.get("lookback", 100))
|
|
min_conf = float(params.get("min_confidence", 0.5))
|
|
candles = market_data.get("candles", [])
|
|
|
|
if len(candles) < lookback:
|
|
return {"signal": "HOLD", "confidence": 0.0, "metadata": {"reason": "not_enough_candles", "model": MODEL_NAME}}
|
|
|
|
df = _normalise(pd.DataFrame(candles, columns=["open", "high", "low", "close", "volume"]))
|
|
feat = _features(df).tail(1)
|
|
|
|
if feat.empty:
|
|
return {"signal": "HOLD", "confidence": 0.0, "metadata": {"reason": "no_features", "model": MODEL_NAME}}
|
|
|
|
proba = model["model"].predict_proba(feat.values.astype(np.float32))[0]
|
|
klass = int(np.argmax(proba))
|
|
conf = float(proba[klass])
|
|
signal = {0: "DOWN", 1: "HOLD", 2: "UP"}[klass]
|
|
|
|
if conf < min_conf:
|
|
signal = "HOLD"
|
|
|
|
return {"signal": signal, "confidence": round(conf, 4), "metadata": {
|
|
"p_sell": round(float(proba[0]), 4),
|
|
"p_hold": round(float(proba[1]), 4),
|
|
"p_buy": round(float(proba[2]), 4),
|
|
"model": MODEL_NAME,
|
|
"symbol": SYMBOL
|
|
}} |