feat: implement GBPUSD volatility compression expansion 1m strategy (closes #204)

This commit is contained in:
Stanley Isaac
2026-05-21 15:27:52 +00:00
parent a9a8b24cba
commit a4d0508e04
4 changed files with 231 additions and 1 deletions
@@ -41,4 +41,4 @@ GBPUSD may show repeatable behavior when volatility compression expansion condit
This idea is intentionally Markdown-only. A future template can add `strategy.py`, `quant.config.json`, and a focused README once PPE results justify turning the idea into executable code.
Closes #3
Closes #204
@@ -0,0 +1,47 @@
# GBPUSD Volatility Compression Expansion 1m
This strategy implements a volatility compression expansion approach for GBPUSD on 1-minute candles using LogisticRegression.
## Overview
- **Pair**: GBPUSD
- **Timeframe**: 1m
- **Model**: LogisticRegression with feature engineering focused on volatility compression/expansion patterns
- **Goal**: Capture breakouts following periods of low volatility (compression) that expand into high volatility moves
## Features Engineered
1. **Returns**: 1, 3, 6, 12 period returns
2. **ATR-normalized price action**: Range percentage, body percentage, close position
3. **Volatility compression/expansion**:
- ATR ratio (current vs 50-period average)
- ATR percentile (200-period ranking)
- Returns-based volatility metrics (fast/slow ratio, percentile)
- Bollinger Band width and ratio
- Keltner Channel width and ratio
4. **EMA distances**: From 20, 50, and 200 period EMAs
5. **Swing points**: Distance to recent swing high/low
6. **Volume features**: Z-score and ratio to moving average
## Configuration
See `quant.config.json` for hyperparameters:
- `lookback`: 100 candles for prediction
- `horizon`: 1 candle forward for labeling
- `threshold`: 0.0008 (8 pips) for ATR-normalized breakout
- `min_confidence`: 0.55 minimum probability for signal generation
## Usage
This template follows the PyP Quant Mode contract:
```python
def train(data, config):
return model, metrics
def predict(model, market_data, config):
return {"signal": "UP|DOWN|HOLD", "confidence": 0.0, "metadata": {}}
```
## Disclaimer
Educational template only. Not financial advice. Past performance does not guarantee future results.
@@ -0,0 +1,28 @@
{
"pair": "GBPUSD",
"timeframe": "1m",
"model_family": "sklearn LogisticRegression",
"runtime_target": "edge",
"artifact_format": "weights_bundle",
"parameters": {
"lookback": 100,
"horizon": 1,
"threshold": 0.0008,
"min_confidence": 0.55
},
"training_requirements": [
"numpy",
"pandas",
"scikit-learn",
"joblib"
],
"inference_requirements": [
"numpy",
"pandas",
"scikit-learn",
"joblib"
],
"symbol": "GBPUSD",
"description": "GBPUSD volatility compression expansion strategy using LogisticRegression",
"disclaimer": "Educational template only. Not financial advice."
}
@@ -0,0 +1,155 @@
import numpy as np
import pandas as pd
from sklearn.linear_model import LogisticRegression
from sklearn.pipeline import Pipeline
from sklearn.preprocessing import StandardScaler
SYMBOL = "GBPUSD"
MODEL_NAME = "gbpusd-volatility-compression-expansion-1m"
def _normalise(data):
df = data.copy()
df.columns = [str(c).lower() for c in df.columns]
if "volume" not in df.columns:
df["volume"] = 1.0
for col in ["open", "high", "low", "close", "volume"]:
df[col] = pd.to_numeric(df[col], errors="coerce")
return df.dropna(subset=["open", "high", "low", "close"]).reset_index(drop=True)
def _features(df):
c = df["close"]
h = df["high"]
l = df["low"]
v = df["volume"]
# Basic returns
out = pd.DataFrame(index=df.index)
out["ret1"] = c.pct_change()
out["ret3"] = c.pct_change(3)
out["ret6"] = c.pct_change(6)
out["ret12"] = c.pct_change(12)
# ATR for normalization
tr = np.maximum(h - l, np.maximum(abs(h - c.shift(1)), abs(l - c.shift(1))))
atr = pd.Series(tr).rolling(14).mean()
out["atr"] = atr
# ATR-normalized candle range and close location value
out["range_pct"] = (h - l) / c
out["body_pct"] = (c - df["open"]) / (h - l).replace(0, np.nan)
out["close_pos"] = (c - l) / (h - l).replace(0, np.nan)
# Volatility compression/expansion features
# Current ATR relative to historical ATR
out["atr_ratio"] = out["atr"] / out["atr"].rolling(50).mean()
out["atr_percentile"] = out["atr"].rolling(200).apply(
lambda x: pd.Series(x).rank(pct=True).iloc[-1] if len(x) > 0 else 0.5, raw=False)
# Volatility percentiles (from idea)
returns = c.pct_change()
out["volatility_fast"] = returns.rolling(16).std()
out["volatility_slow"] = returns.rolling(64).std()
out["volatility_ratio"] = out["volatility_fast"] / (out["volatility_slow"] + 1e-9)
out["volatility_percentile"] = out["volatility_fast"].rolling(200).apply(
lambda x: pd.Series(x).rank(pct=True).iloc[-1] if len(x) > 0 else 0.5, raw=False)
# Distance from EMAs
out["ema20_dist"] = (c - c.ewm(span=20, adjust=False).mean()) / c
out["ema50_dist"] = (c - c.ewm(span=50, adjust=False).mean()) / c
out["ema200_dist"] = (c - c.ewm(span=200, adjust=False).mean()) / c
# Prior swing high and swing low distance
swing_high = h.rolling(20, center=False).max().shift(1)
swing_low = l.rolling(20, center=False).min().shift(1)
out["dist_to_swing_high"] = (swing_high - c) / c
out["dist_to_swing_low"] = (c - swing_low) / c
# Volume features
out["volume_z"] = (v - v.rolling(48).mean()) / (v.rolling(48).std() + 1e-9)
out["volume_ratio"] = v / v.rolling(20).mean()
# Bollinger Band width (volatility indicator)
sma_20 = c.rolling(20).mean()
std_20 = c.rolling(20).std()
upper_bb = sma_20 + (std_20 * 2)
lower_bb = sma_20 - (std_20 * 2)
out["bb_width"] = (upper_bb - lower_bb) / sma_20
out["bb_width_ratio"] = out["bb_width"] / out["bb_width"].rolling(50).mean()
# Keltner Channel width (ATR-based volatility)
ema_20 = c.ewm(span=20, adjust=False).mean()
atr_20 = out["atr"].rolling(20).mean()
upper_kc = ema_20 + (atr_20 * 1.5)
lower_kc = ema_20 - (atr_20 * 1.5)
out["kc_width"] = (upper_kc - lower_kc) / ema_20
out["kc_width_ratio"] = out["kc_width"] / out["kc_width"].rolling(50).mean()
return out.replace([np.inf, -np.inf], np.nan).dropna()
def _labels(close, index, horizon, threshold):
fwd = close.pct_change(horizon).shift(-horizon)
y = pd.Series(1, index=close.index)
y[fwd > threshold] = 2
y[fwd < -threshold] = 0
return y.reindex(index).fillna(1).astype(int)
def train(data, config):
params = config.get("parameters", {})
horizon = int(params.get("horizon", 1))
threshold = float(params.get("threshold", 0.0008))
df = _normalise(data)
feat = _features(df)
y = _labels(df["close"], feat.index, horizon, threshold)
model = Pipeline([
("scaler", StandardScaler()),
("clf", LogisticRegression(max_iter=1000, class_weight="balanced", multi_class="auto", C=0.1)),
])
model.fit(feat.values.astype(np.float32), y.values)
preds = model.predict(feat.values.astype(np.float32))
metrics = {
"training_bars": int(len(feat)),
"feature_count": int(feat.shape[1]),
"buy_signals": int((preds == 2).sum()),
"sell_signals": int((preds == 0).sum()),
"hold_signals": int((preds == 1).sum()),
}
return {"model": model, "features": list(feat.columns), "symbol": SYMBOL}, metrics
def predict(model, market_data, config):
params = config.get("parameters", {})
lookback = int(params.get("lookback", 100))
min_conf = float(params.get("min_confidence", 0.55))
candles = market_data.get("candles", [])
if len(candles) < lookback:
return {"signal": "HOLD", "confidence": 0.0, "metadata": {"reason": "not_enough_candles", "model": MODEL_NAME}}
df = _normalise(pd.DataFrame(candles, columns=["open", "high", "low", "close", "volume"]))
feat = _features(df).tail(1)
if feat.empty:
return {"signal": "HOLD", "confidence": 0.0, "metadata": {"reason": "no_features", "model": MODEL_NAME}}
proba = model["model"].predict_proba(feat.values.astype(np.float32))[0]
klass = int(np.argmax(proba))
conf = float(proba[klass])
signal = {0: "DOWN", 1: "HOLD", 2: "UP"}[klass]
if conf < min_conf:
signal = "HOLD"
return {"signal": signal, "confidence": round(conf, 4), "metadata": {
"p_sell": round(float(proba[0]), 4),
"p_hold": round(float(proba[1]), 4),
"p_buy": round(float(proba[2]), 4),
"model": MODEL_NAME,
"symbol": SYMBOL
}}