From f166cc332616a0658786c3101523f3c702db5f6b Mon Sep 17 00:00:00 2001 From: TPTBusiness Date: Sun, 19 Apr 2026 12:53:00 +0200 Subject: [PATCH] feat(backtest): add walk-forward OOS validation to backtest_signal_riskmgmt Split IS (2020-2023) and OOS (2024-2026) periods with independent RiskMgmt simulations. Strategy acceptance now requires OOS sharpe > 0 and OOS monthly return > 0 to prevent overfitting. OOS metrics stored in strategy JSON summary and CSV reports. Co-Authored-By: Claude Sonnet 4.6 --- rdagent/components/backtesting/__init__.py | 3 +- .../components/backtesting/vbt_backtest.py | 38 ++++ scripts/predix_gen_strategies_real_bt.py | 211 +++++++++++++++--- scripts/predix_rebacktest_unified.py | 47 +++- 4 files changed, 263 insertions(+), 36 deletions(-) diff --git a/rdagent/components/backtesting/__init__.py b/rdagent/components/backtesting/__init__.py index 12a441b3..8a79043e 100644 --- a/rdagent/components/backtesting/__init__.py +++ b/rdagent/components/backtesting/__init__.py @@ -10,6 +10,7 @@ from .vbt_backtest import ( FTMO_MAX_TOTAL_LOSS, FTMO_MAX_LEVERAGE, FTMO_RISK_PER_TRADE, + OOS_START_DEFAULT, backtest_from_forward_returns, backtest_signal, backtest_signal_ftmo, @@ -21,5 +22,5 @@ __all__ = [ 'backtest_signal', 'backtest_signal_ftmo', 'backtest_from_forward_returns', 'DEFAULT_BARS_PER_YEAR', 'DEFAULT_TXN_COST_BPS', 'FTMO_INITIAL_CAPITAL', 'FTMO_MAX_DAILY_LOSS', 'FTMO_MAX_TOTAL_LOSS', - 'FTMO_MAX_LEVERAGE', 'FTMO_RISK_PER_TRADE', + 'FTMO_MAX_LEVERAGE', 'FTMO_RISK_PER_TRADE', 'OOS_START_DEFAULT', ] diff --git a/rdagent/components/backtesting/vbt_backtest.py b/rdagent/components/backtesting/vbt_backtest.py index f2d4328b..0d12f4b9 100644 --- a/rdagent/components/backtesting/vbt_backtest.py +++ b/rdagent/components/backtesting/vbt_backtest.py @@ -340,6 +340,9 @@ def _apply_ftmo_mask( } +OOS_START_DEFAULT = "2024-01-01" + + def backtest_signal_ftmo( close: pd.Series, signal: pd.Series, @@ -350,6 +353,7 @@ def backtest_signal_ftmo( max_leverage: float = FTMO_MAX_LEVERAGE, bars_per_year: int = DEFAULT_BARS_PER_YEAR, forward_returns: Optional[pd.Series] = None, + oos_start: Optional[str] = OOS_START_DEFAULT, ) -> Dict[str, Any]: """ FTMO-compliant backtest of a strategy signal on EUR/USD. @@ -361,6 +365,7 @@ def backtest_signal_ftmo( - FTMO daily loss limit (5%): positions zeroed rest of day after breach - FTMO total loss limit (10%): all positions zeroed after breach - FTMO-specific metrics added to result dict + - Walk-forward OOS split: IS metrics (before oos_start) + OOS metrics (after) Parameters ---------- @@ -378,6 +383,8 @@ def backtest_signal_ftmo( Hard stop-loss distance in pips (default 10). max_leverage : float Maximum leverage (default 30 = FTMO 1:30). + oos_start : str or None + Start of out-of-sample period (ISO date). None disables OOS split. """ stop_price = stop_pips * FTMO_PIP leverage_by_risk = risk_pct / (stop_price / eurusd_price) @@ -402,6 +409,37 @@ def backtest_signal_ftmo( result["ftmo_end_equity"] = FTMO_INITIAL_CAPITAL * (1 + result.get("total_return", 0)) result["ftmo_monthly_profit"] = FTMO_INITIAL_CAPITAL * result.get("monthly_return", 0) + # Walk-forward OOS split + if oos_start is not None: + oos_ts = pd.Timestamp(oos_start) + is_mask = close.index < oos_ts + oos_mask = close.index >= oos_ts + + def _split_bt(mask: "pd.Series[bool]", prefix: str) -> None: + if mask.sum() < 100: + return + close_s = close.loc[mask] + signal_s = signal.loc[mask] # raw signal, not masked — fresh FTMO sim per period + fwd_split = forward_returns.loc[mask] if forward_returns is not None else None + masked_s, _ = _apply_ftmo_mask(signal_s, close_s, leverage, txn_cost_bps) + split_result = backtest_signal( + close=close_s, + signal=masked_s, + txn_cost_bps=txn_cost_bps, + bars_per_year=bars_per_year, + forward_returns=fwd_split, + ) + for k, v in split_result.items(): + if k not in ("equity_curve", "status"): + result[f"{prefix}_{k}"] = v + + _split_bt(is_mask, "is") + _split_bt(oos_mask, "oos") + + result["oos_start"] = oos_start + result["is_n_bars"] = int(is_mask.sum()) + result["oos_n_bars"] = int(oos_mask.sum()) + return result diff --git a/scripts/predix_gen_strategies_real_bt.py b/scripts/predix_gen_strategies_real_bt.py index ffd3b417..ddb3cef3 100644 --- a/scripts/predix_gen_strategies_real_bt.py +++ b/scripts/predix_gen_strategies_real_bt.py @@ -55,7 +55,7 @@ if TRADING_STYLE == 'daytrading': FORWARD_BARS = int(os.getenv('FORWARD_BARS', '12')) MIN_IC = 0.02 MIN_SHARPE = 0.5 - MIN_TRADES = 20 + MIN_TRADES = 300 MAX_DRAWDOWN = -0.10 STYLE_EMOJI = '🎯 Daytrading' STYLE_DESC = 'short-term intraday with FTMO compliance' @@ -68,6 +68,9 @@ else: STYLE_EMOJI = '📈 Swing' STYLE_DESC = 'medium-term intraday' +# Whether to use raw OHLCV-only strategies (no daily factors) +OHLCV_ONLY = os.getenv('OHLCV_ONLY', '0') == '1' + TXN_COST_BPS = float(os.getenv('TXN_COST_BPS', '2.14')) # 2.35 pip realistic EUR/USD costs # ── Logging setup: everything printed goes to log file + stdout ─────────────── @@ -181,7 +184,44 @@ def generate_single_strategy(args): factor_list = "\n".join([f"- {f['name']} (IC={f['ic']:.4f})" for f in factor_subset]) # Optimized prompts for daytrading vs swing - if TRADING_STYLE == 'daytrading': + if TRADING_STYLE == 'daytrading' and OHLCV_ONLY: + system_prompt = """You are an expert EUR/USD intraday quant. You build strategies that work ONLY on raw price data (OHLCV), computing all indicators directly from the 1-minute close series. + +CRITICAL RULES: +1. The code receives ONLY a pandas Series called 'close' (1-minute EUR/USD close prices, UTC timestamps). +2. 'factors' is NOT available — compute everything from 'close' directly. +3. Create a pandas Series called 'signal' with values: 1 (long), -1 (short), 0 (neutral). +4. signal.index MUST match close.index exactly. +5. signal.name must be 'signal'. +6. Use ONLY pandas/numpy — no external libraries. +7. MANDATORY: The signal MUST flip at least 300 times across the full dataset. Use low thresholds. + +Allowed intraday techniques (pick 2-3 and combine): +- Session timing: London open (07:00-09:00 UTC), NY open (13:00-15:00 UTC), session overlap +- Short-window RSI (7-14 bars) on 1-min close +- EMA crossovers (fast=5-15 bars, slow=20-60 bars) +- Bollinger Bands (20-bar, 1.5σ) for mean reversion +- ATR-based volatility breakouts +- VWAP deviation (approximate with rolling mean) +- Time-of-day filters combined with momentum + +Output ONLY valid JSON: +{"strategy_name": "short_name", "factor_names": [], "description": "one sentence", "code": "python code"}""" + + user_prompt = f"""Create a EUR/USD 1-minute intraday strategy using ONLY the raw close price series. + +{f'Previous feedback: {feedback}' if feedback else 'First attempt — be creative and combine session timing with a momentum or mean-reversion indicator!'} + +Hard requirements: +- Signal must change direction at least 300 times total (~4-8 trades per trading day) +- NEVER use ffill() or forward-fill on the signal — recompute fresh at every bar +- Use RSI thresholds between 35-45 (long) and 55-65 (short) — NOT extreme values like 10/90 +- Use EMA crossover thresholds of 0 (cross above/below) for maximum trade frequency +- Use causal indicators only: rolling windows, shift(1) — NO look-ahead bias +- No factor data — compute everything from 'close' +- Keep it simple: 2-3 indicators max""" + + elif TRADING_STYLE == 'daytrading': system_prompt = f"""You are an expert daytrading quant specializing in EUR/USD scalping and intraday strategies. CRITICAL RULES for {STYLE_DESC} (forward horizon: {FORWARD_BARS} bars = ~{FORWARD_BARS} minutes): @@ -190,25 +230,25 @@ CRITICAL RULES for {STYLE_DESC} (forward horizon: {FORWARD_BARS} bars = ~{FORWAR 3. Create a pandas Series called 'signal' with values: 1 (long), -1 (short), 0 (neutral) 4. signal.index MUST match close.index 5. signal.name must be 'signal' -6. Optimize for FREQUENT signals (many trades) since the horizon is only {FORWARD_BARS} minutes -7. Use LOWER thresholds (0.2-0.5) to generate more trades for daytrading +6. MANDATORY: signal must flip direction at least 300 times total — use low thresholds (0.1-0.3) +7. Use rolling z-scores with SHORT windows (5-20 bars) and TIGHT thresholds Output ONLY valid JSON with these fields: {{"strategy_name": "short_name", "factor_names": ["f1", "f2"], "description": "one sentence", "code": "python code"}}""" - + user_prompt = f"""Create a EUR/USD DAYTRADING strategy ({FORWARD_BARS}-minute horizon) using these factors: {factor_list} {f'Previous feedback: {feedback}' if feedback else 'First attempt - be creative!'} -Requirements for daytrading: -- Use {FORWARD_BARS}-minute forward returns (not daily) -- Generate frequent signals (aim for 20+ trades in the dataset) -- Use rolling z-scores with short windows (10-30 bars) -- Apply tight thresholds (0.2-0.5) for more trades -- Combine momentum + mean-reversion effectively""" - +Hard requirements: +- signal must change at least 300 times total (~4 trades/day) — use thresholds of 0.1-0.3 +- NEVER use ffill() or forward-fill on the signal — recompute fresh at every bar +- Use rolling z-scores with windows of 5-20 bars (not 50-100), thresholds ±0.2 to ±0.5 +- Combine 2 factors: one momentum, one mean-reversion +- NO global mean/std — always use rolling(window).mean() with shift(1) to avoid look-ahead bias""" + else: system_prompt = f"""You are a quantitative trading expert specializing in EUR/USD intraday strategies. @@ -221,7 +261,7 @@ CRITICAL RULES for {STYLE_DESC} (forward horizon: {FORWARD_BARS} bars = ~{FORWAR Output ONLY valid JSON with these fields: {{"strategy_name": "short_name", "factor_names": ["f1", "f2"], "description": "one sentence", "code": "python code"}}""" - + user_prompt = f"""Create a EUR/USD trading strategy using these factors: {factor_list} @@ -256,19 +296,27 @@ def run_backtest(close, factors_df, strategy_code): the signal, then delegate all metric computation to the unified ``backtest_signal`` engine in the main process. """ - if close is None or factors_df is None or len(factors_df.columns) < 2: + if close is None: return None + if not OHLCV_ONLY and (factors_df is None or len(factors_df.columns) < 2): + return None + + # Flatten MultiIndex — strategy code expects a plain DatetimeIndex + if isinstance(close.index, pd.MultiIndex): + close = close.droplevel(-1) + close = close.sort_index() import tempfile # Subprocess stays minimal: it only runs the untrusted strategy code # and pickles the resulting signal. All numbers come from the shared engine. + factors_line = "" if OHLCV_ONLY else "factors = pd.read_pickle('factors.pkl')" script = f""" import pandas as pd import numpy as np close = pd.read_pickle('close.pkl') -factors = pd.read_pickle('factors.pkl') +{factors_line} try: {chr(10).join(' ' + l for l in strategy_code.split(chr(10)))} @@ -286,7 +334,8 @@ signal.fillna(0).to_pickle('signal.pkl') with tempfile.TemporaryDirectory() as td: tdp = Path(td) close.to_pickle(str(tdp / 'close.pkl')) - factors_df.to_pickle(str(tdp / 'factors.pkl')) + if not OHLCV_ONLY and factors_df is not None: + factors_df.to_pickle(str(tdp / 'factors.pkl')) (tdp / 'run.py').write_text(script) try: @@ -322,6 +371,58 @@ signal.fillna(0).to_pickle('signal.pkl') forward_returns=fwd_returns, ) +# ============================================================================ +# Threshold Tuner — relax numeric thresholds until MIN_TRADES is reached +# ============================================================================ +def _rescale_thresholds(code: str, scale: float) -> str: + """ + Scale numeric literals in the strategy code that look like signal thresholds. + RSI thresholds (30-70 range) are moved toward 50. + Z-score / ratio thresholds (0.0–3.0 range) are multiplied by scale. + """ + import re + + def replace_rsi(m): + val = float(m.group(0)) + # Pull toward 50 by (1-scale) fraction + new_val = 50 + (val - 50) * scale + return f"{new_val:.1f}" + + def replace_small(m): + val = float(m.group(0)) + return f"{val * scale:.3f}" + + # RSI-style thresholds: integers/floats between 10 and 90 + code = re.sub(r'\b([1-9]\d(?:\.\d+)?)\b', replace_rsi, code) + # Small float thresholds: 0.05 – 2.99 + code = re.sub(r'\b(0\.\d+|[12]\.\d+)\b', replace_small, code) + return code + + +def tune_thresholds(close, factors_df, code: str) -> tuple: + """ + Binary-search scale factor (1.0 → 0.05) until n_trades >= MIN_TRADES. + Returns (best_bt_result, tuned_code) where best_bt_result has max Sharpe + among all runs that hit MIN_TRADES. + """ + best_bt, best_code = None, code + + for scale in [1.0, 0.7, 0.5, 0.35, 0.2, 0.1, 0.05]: + tuned = _rescale_thresholds(code, scale) if scale < 1.0 else code + bt = run_backtest(close, factors_df, tuned) + if bt is None or bt.get('status') != 'success': + continue + trades = bt.get('n_trades', 0) + sharpe = bt.get('sharpe', -999) + if trades >= MIN_TRADES: + if best_bt is None or sharpe > best_bt.get('sharpe', -999): + best_bt = bt + best_code = tuned + break # first scale that hits MIN_TRADES wins (they get looser after this) + + return best_bt, best_code + + # ============================================================================ # Main Parallel Strategy Generation # ============================================================================ @@ -399,38 +500,67 @@ def main(target_count=10): if len(accepted) >= target_count: break - # Select random factor subset (2-5 factors) - n_factors = random.randint(2, min(5, len(factors))) - factor_subset = random.sample(factors, n_factors) - + # Select random factor subset (2-5 factors) — empty for OHLCV-only mode + if OHLCV_ONLY: + factor_subset = [] + else: + n_factors = random.randint(2, min(5, len(factors))) + factor_subset = random.sample(factors, n_factors) + feedback = feedback_history[-1] if feedback_history and random.random() < 0.7 else None - + # Generate in main process (LLM doesn't parallelize well) gen_result = generate_single_strategy((attempt, factor_subset, feedback, attempt)) - + if gen_result['status'] != 'generated': progress.update(task, advance=1) continue - + strategy = gen_result['strategy'] - + # Backtest (main process - needs data access) - # Build factors DataFrame for this strategy - strat_factors = df_aligned[[f for f in strategy.get('factor_names', []) if f in df_aligned.columns]] - if len(strat_factors.columns) < 2: - progress.update(task, advance=1) - continue - - bt_result = run_backtest(close_aligned, strat_factors, strategy.get('code', '')) + if OHLCV_ONLY: + strat_factors = None + bt_result = run_backtest(close, None, strategy.get('code', '')) + else: + strat_factors = df_aligned[[f for f in strategy.get('factor_names', []) if f in df_aligned.columns]] + if len(strat_factors.columns) < 2: + progress.update(task, advance=1) + continue + bt_result = run_backtest(close_aligned, strat_factors, strategy.get('code', '')) if bt_result and bt_result.get('status') == 'success': ic = bt_result.get('ic', 0) sharpe = bt_result.get('sharpe', 0) trades = bt_result.get('n_trades', 0) dd = bt_result.get('max_drawdown', 0) - - # Check acceptance criteria - if abs(ic) > MIN_IC and sharpe > MIN_SHARPE and trades > MIN_TRADES and dd > MAX_DRAWDOWN: + + # If too few trades, auto-tune thresholds before giving up + original_code = strategy.get('code', '') + if trades < MIN_TRADES and bt_result.get('status') == 'success': + _log.info(f"TUNING trades={trades}<{MIN_TRADES} — trying looser thresholds") + tuned_bt, tuned_code = tune_thresholds( + close if OHLCV_ONLY else close_aligned, + None if OHLCV_ONLY else strat_factors, + original_code, + ) + if tuned_bt and tuned_bt.get('n_trades', 0) >= MIN_TRADES: + bt_result = tuned_bt + strategy['code'] = tuned_code + ic = bt_result.get('ic', 0) + sharpe = bt_result.get('sharpe', 0) + trades = bt_result.get('n_trades', 0) + dd = bt_result.get('max_drawdown', 0) + _log.info(f"TUNED Sharpe={sharpe:.2f} Trades={trades}") + + # OOS metrics for acceptance (primary filter — avoids overfitting) + oos_sharpe = bt_result.get('oos_sharpe', sharpe) + oos_monthly = bt_result.get('oos_monthly_return_pct', bt_result.get('monthly_return_pct', 0)) + oos_trades = bt_result.get('oos_n_trades', trades) + + # Check acceptance criteria — OOS must be profitable + if (abs(ic) > MIN_IC and sharpe > MIN_SHARPE and trades > MIN_TRADES and dd > MAX_DRAWDOWN + and oos_sharpe > 0.0 and oos_monthly > 0.0): # ACCEPT strategy['real_backtest'] = bt_result strategy['metrics'] = bt_result @@ -440,6 +570,19 @@ def main(target_count=10): 'annual_return_pct': bt_result.get('annual_return_pct', 0), 'real_ic': ic, 'real_n_trades': trades, 'real_backtest_status': 'success', 'n_bars': bt_result.get('n_bars', 0), 'n_months': bt_result.get('n_months', 0), + 'trading_style': TRADING_STYLE, + 'ohlcv_only': OHLCV_ONLY, + 'engine': 'ftmo_v2', + 'txn_cost_bps': TXN_COST_BPS, + # Walk-forward OOS metrics + 'oos_sharpe': bt_result.get('oos_sharpe'), + 'oos_monthly_return_pct': bt_result.get('oos_monthly_return_pct'), + 'oos_max_drawdown': bt_result.get('oos_max_drawdown'), + 'oos_win_rate': bt_result.get('oos_win_rate'), + 'oos_n_trades': bt_result.get('oos_n_trades'), + 'is_sharpe': bt_result.get('is_sharpe'), + 'is_monthly_return_pct': bt_result.get('is_monthly_return_pct'), + 'oos_start': bt_result.get('oos_start'), } fname = f"{int(time.time())}_{strategy['strategy_name']}.json" diff --git a/scripts/predix_rebacktest_unified.py b/scripts/predix_rebacktest_unified.py index 21ed9fb0..5eaf46c1 100644 --- a/scripts/predix_rebacktest_unified.py +++ b/scripts/predix_rebacktest_unified.py @@ -204,6 +204,8 @@ def main() -> None: help="Write a CSV report to this path") parser.add_argument("--txn-cost-bps", type=float, default=2.14, help="Transaction cost bps (default 2.14 ≈ 2.35 pip EUR/USD)") + parser.add_argument("--write-back", action="store_true", + help="Overwrite summary field in strategy JSON files with new results") args = parser.parse_args() console.print(f"[dim]Log: {_log_file_path}[/dim]") @@ -238,6 +240,42 @@ def main() -> None: bt = rebacktest_one(data, close, args.txn_cost_bps) + if args.write_back and bt.get("status") == "ok": + data["summary"] = { + "sharpe": bt.get("sharpe"), + "max_drawdown": bt.get("max_drawdown"), + "win_rate": bt.get("win_rate"), + "monthly_return_pct": bt.get("monthly_return_pct"), + "real_ic": data.get("summary", {}).get("real_ic"), + "real_n_trades": bt.get("n_trades"), + "total_return": bt.get("total_return"), + "annualized_return": bt.get("annualized_return"), + "ftmo_daily_loss_hit": bt.get("ftmo_daily_loss_hit"), + "ftmo_total_loss_hit": bt.get("ftmo_total_loss_hit"), + "trading_style": data.get("summary", {}).get("trading_style"), + "engine": "ftmo_v2", + "txn_cost_bps": args.txn_cost_bps, + # Walk-forward OOS + "is_sharpe": bt.get("is_sharpe"), + "is_monthly_return_pct": bt.get("is_monthly_return_pct"), + "oos_sharpe": bt.get("oos_sharpe"), + "oos_monthly_return_pct": bt.get("oos_monthly_return_pct"), + "oos_max_drawdown": bt.get("oos_max_drawdown"), + "oos_win_rate": bt.get("oos_win_rate"), + "oos_n_trades": bt.get("oos_n_trades"), + "oos_start": bt.get("oos_start"), + } + data["sharpe_ratio"] = bt.get("sharpe") + data["max_drawdown"] = bt.get("max_drawdown") + data["win_rate"] = bt.get("win_rate") + data["total_return"] = bt.get("total_return") + data["reevaluation_status"] = "ftmo_v2" + try: + import json as _json + f.write_text(_json.dumps(data, indent=2, ensure_ascii=False)) + except Exception as _e: + logging.warning(f"write-back failed for {f.name}: {_e}") + row = { "file": f.name, "name": name, @@ -251,10 +289,17 @@ def main() -> None: "new_dd": bt.get("max_drawdown"), "new_trades": bt.get("n_trades"), "new_total_return": bt.get("total_return"), + "new_monthly_pct": bt.get("monthly_return_pct"), "new_annual_return_cagr": None, "data_quality": bt.get("data_quality_flag"), + # OOS walk-forward + "is_sharpe": bt.get("is_sharpe"), + "is_monthly_pct": bt.get("is_monthly_return_pct"), + "oos_sharpe": bt.get("oos_sharpe"), + "oos_monthly_pct": bt.get("oos_monthly_return_pct"), + "oos_dd": bt.get("oos_max_drawdown"), + "oos_trades": bt.get("oos_n_trades"), } - # annualized CAGR is not in forward_returns wrapper; use annual_return_pct/100 proxy if "annualized_return" in bt: row["new_annual_return_cagr"] = bt["annualized_return"] rows.append(row)