63a829cc46
主要内容: - Phase 8 PROMOTE: finalist #1 (trial #324) registry 条目,自动生成 - Optuna objective warmup bug 修复 (shared/optimizer/objective.py) - studies/ 目录按用途重组为 optuna/ + finalists/ + features/ 三层 - reports/ 加入 Optuna 中文 dashboard (5 主图 + 18 slice + 15 contour) - 新增 PROJECT_GUIDE.md 项目说明文档 - 新增 build_registry_entry.py / build_optuna_dashboard.py / build_feature_datasets.py - .gitignore: 允许提交 studies/*.db (Optuna DB) 和 reports/*.html (MT5 + dashboard)
234 lines
9.4 KiB
Python
234 lines
9.4 KiB
Python
"""Phase 6 — Optuna optimization for GoldScalperPro (doc 06).
|
||
|
||
Two modes via CLI flags:
|
||
|
||
--smoke 30 trials, IS = 2025 H1 only (~6 months). Validates the
|
||
objective + storage + scoring end-to-end in <2 min before
|
||
committing to the full study. Run this first, always.
|
||
|
||
(default) 500 trials, IS = 2025 full year (with a 2-week purge gap
|
||
before year-end so OOS walk-forward is leak-free). TPE
|
||
sampler, SQLite-persisted so the study resumes/inspects
|
||
mid-run. Runs in the foreground with progress bar; for a
|
||
long run launch with `start /b python scripts\\optimize.py`
|
||
(Windows) and poll the .db file separately.
|
||
|
||
Why M1 bars are mandatory here (doc 03 §7 / §8, CLAUDE.md standing rule):
|
||
GoldScalperPro uses break-even + trailing stops, so the bar-level engine
|
||
produces a −40% to −50% hidden gap vs MT5. The objective passes ``m1_bars``
|
||
to ``engine.run`` so the engine switches to tick-level exit simulation.
|
||
|
||
Window design (doc 06 §4 walk-forward):
|
||
IS = 2025-01-01 00:00 → 2025-12-15 00:00 (exclusive end, ~11.5 months)
|
||
purge = 2025-12-15 .. 2025-12-31 (2-week gap, no trades counted either side)
|
||
OOS = 2026-01-01 00:00 → 2026-07-01 00:00 (~6 months, fixed finalist params)
|
||
|
||
Constraints (user-confirmed "strict" preset):
|
||
min_trades=40, min_profit_factor=1.5, max_equity_dd_pct=0.25
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
PROJECT = Path(__file__).resolve().parent.parent
|
||
sys.path.insert(0, str(PROJECT))
|
||
|
||
import optuna
|
||
import pandas as pd
|
||
|
||
from shared.core.engine import SizingInputs
|
||
from shared.optimizer.objective import (
|
||
Constraints,
|
||
ObjectiveConfig,
|
||
build_objective,
|
||
)
|
||
from shared.optimizer.selector import select_diverse_topn
|
||
from strategies.gold_scalper_pro.instruments import XAUUSD_REAL
|
||
from strategies.gold_scalper_pro.scalper_engine import ScalperEngine
|
||
from strategies.gold_scalper_pro.search_space import (
|
||
FROZEN_BASELINE,
|
||
INT_PARAMS,
|
||
SEARCH_SPACE,
|
||
)
|
||
from strategies.gold_scalper_pro.signals import build_signals
|
||
from strategies.gold_scalper_pro.scalper_engine import engine_kwargs_from_params
|
||
|
||
|
||
# ── Windows ────────────────────────────────────────────────────────────────
|
||
IS_START = pd.Timestamp("2025-01-01 00:00:00")
|
||
IS_END = pd.Timestamp("2025-12-15 00:00:00") # exclusive end (purge after)
|
||
OOS_START = pd.Timestamp("2026-01-01 00:00:00")
|
||
OOS_END = pd.Timestamp("2026-07-01 00:00:00") # exclusive end
|
||
INITIAL_DEPOSIT = 1000.0
|
||
SEED = 42
|
||
|
||
|
||
def slice_window(df: pd.DataFrame, start: pd.Timestamp, end: pd.Timestamp) -> pd.DataFrame:
|
||
"""Slice bars to [start, end) — exclusive end matches MT5 tester semantics."""
|
||
return df[(df["timestamp"] >= start) & (df["timestamp"] < end)].reset_index(drop=True)
|
||
|
||
|
||
def make_objective_config(
|
||
bars_is: pd.DataFrame,
|
||
m1_is: pd.DataFrame,
|
||
full_m5: pd.DataFrame,
|
||
) -> ObjectiveConfig:
|
||
"""Assemble the ObjectiveConfig for the IS window.
|
||
|
||
``bars_is`` / ``m1_is`` are trimmed to the IS evaluation window — the
|
||
engine runs on these (initial_deposit reset, no open position at IS_START,
|
||
matching MT5 Strategy Tester).
|
||
|
||
``full_m5`` is the FULL M5 history (data starts 2024-06-26 → ~6 months of
|
||
pre-IS warmup, well beyond EMA(160)'s ~14h requirement). It is passed as
|
||
``signals_full_bars`` so the objective builds signals on it, then slices
|
||
the signal arrays to ``bars_is``' time range — mirroring MT5 tester's
|
||
pre-test chart-history indicator warmup. Without this, EMA/RSI/ATR would
|
||
only start warming up at IS_START and the first Python trade would land
|
||
~14h late vs MT5 (the original warmup bug — see reeval_finalist_forward.py
|
||
docstring).
|
||
"""
|
||
constraints = Constraints(
|
||
min_trades=40,
|
||
min_profit_factor=1.5,
|
||
max_equity_dd_pct=0.25,
|
||
)
|
||
return ObjectiveConfig(
|
||
engine=ScalperEngine(),
|
||
bars=bars_is,
|
||
instrument=XAUUSD_REAL,
|
||
sizing=SizingInputs(),
|
||
initial_deposit=INITIAL_DEPOSIT,
|
||
search_space=SEARCH_SPACE,
|
||
int_params=INT_PARAMS,
|
||
frozen_baseline=FROZEN_BASELINE,
|
||
constraints=constraints,
|
||
dd_weight=1.0,
|
||
build_signals=build_signals,
|
||
build_engine_kwargs=engine_kwargs_from_params,
|
||
m1_bars=m1_is, # mandatory for trailing/BE EA
|
||
signals_full_bars=full_m5, # indicator warmup (doc 03 §8)
|
||
)
|
||
|
||
|
||
def run_smoke() -> int:
|
||
"""30 trials, IS H1 2025 only. Validates the script end-to-end."""
|
||
print("=== Phase 6 SMOKE TEST ===")
|
||
m5 = load_m5()
|
||
m1 = load_m1()
|
||
# Shorter IS window for the smoke run.
|
||
bars_is = slice_window(m5, pd.Timestamp("2025-01-01"), pd.Timestamp("2025-07-01"))
|
||
m1_is = slice_window(m1, pd.Timestamp("2025-01-01"), pd.Timestamp("2025-07-01"))
|
||
print(f" IS bars : {len(bars_is):,} M1 bars: {len(m1_is):,}")
|
||
print(f" warmup : {len(m5):,} full M5 bars (signals_full_bars)")
|
||
|
||
cfg = make_objective_config(bars_is, m1_is, full_m5=m5)
|
||
objective = build_objective(cfg)
|
||
|
||
study = optuna.create_study(
|
||
direction="maximize",
|
||
sampler=optuna.samplers.TPESampler(seed=SEED),
|
||
)
|
||
print(" running 30 trials ...")
|
||
study.optimize(objective, n_trials=30, show_progress_bar=False)
|
||
print(f" done. best value = {study.best_value:.2f}")
|
||
print(f" best params: {study.best_params}")
|
||
print(f" best attrs : net={study.best_trial.user_attrs['net_profit']:.2f}, "
|
||
f"PF={study.best_trial.user_attrs['profit_factor']:.2f}, "
|
||
f"trades={study.best_trial.user_attrs['total_trades']}, "
|
||
f"DD%={study.best_trial.user_attrs['max_equity_dd_pct']:.2%}")
|
||
return 0
|
||
|
||
|
||
def run_full(study_db: Path, n_trials: int) -> int:
|
||
"""Full study, IS = 2025 (with 2-week purge). SQLite-persisted."""
|
||
print(f"=== Phase 6 FULL STUDY ({n_trials} trials) ===")
|
||
m5 = load_m5()
|
||
m1 = load_m1()
|
||
bars_is = slice_window(m5, IS_START, IS_END)
|
||
m1_is = slice_window(m1, IS_START, IS_END)
|
||
print(f" IS window: {IS_START.date()} → {IS_END.date()} (exclusive)")
|
||
print(f" IS bars : {len(bars_is):,} M1 bars: {len(m1_is):,}")
|
||
print(f" warmup : {len(m5):,} full M5 bars (signals_full_bars)")
|
||
|
||
cfg = make_objective_config(bars_is, m1_is, full_m5=m5)
|
||
objective = build_objective(cfg)
|
||
|
||
study_db.parent.mkdir(parents=True, exist_ok=True)
|
||
storage = f"sqlite:///{study_db}"
|
||
study = optuna.create_study(
|
||
direction="maximize",
|
||
sampler=optuna.samplers.TPESampler(seed=SEED),
|
||
storage=storage,
|
||
study_name="gold_scalper_pro_is2025",
|
||
load_if_exists=True,
|
||
)
|
||
n_existing = len([t for t in study.trials if t.state.name == "COMPLETE"])
|
||
if n_existing > 0:
|
||
print(f" resumed existing study: {n_existing} complete trials so far")
|
||
print(f" running {n_trials} trials (foreground; Ctrl+C to stop — study is saved) ...")
|
||
study.optimize(objective, n_trials=n_trials, show_progress_bar=True)
|
||
|
||
print(f"\n === best trial ===")
|
||
print(f" value = {study.best_value:.2f}")
|
||
print(f" params:")
|
||
for k, v in study.best_params.items():
|
||
print(f" {k:24s} = {v}")
|
||
a = study.best_trial.user_attrs
|
||
print(f" metrics: net={a['net_profit']:.2f}, PF={a['profit_factor']:.2f}, "
|
||
f"trades={a['total_trades']}, DD%={a['max_equity_dd_pct']:.2%}")
|
||
if a.get("violations"):
|
||
print(f" violations: {a['violations']}")
|
||
|
||
# Diverse top-3 finalists (doc 06 §3).
|
||
print(f"\n === diverse top-3 finalists ===")
|
||
finalists = select_diverse_topn(study, n=3, ranges=SEARCH_SPACE)
|
||
if not finalists:
|
||
print(" no constraint-passing trials found.")
|
||
return 1
|
||
for i, t in enumerate(finalists, 1):
|
||
d = t.user_attrs
|
||
print(f" finalist #{i}: trial #{t.number} value={t.value:.2f}")
|
||
print(f" net={d['net_profit']:.2f}, PF={d['profit_factor']:.2f}, "
|
||
f"trades={d['total_trades']}, DD%={d['max_equity_dd_pct']:.2%}")
|
||
print(f" params: {t.params}")
|
||
print(f"\n study DB: {study_db}")
|
||
return 0
|
||
|
||
|
||
def load_m5() -> pd.DataFrame:
|
||
from shared.data.loaders import load_bars
|
||
p = PROJECT / "data" / "XAUUSD_M5_2024-06-26_2026-06-26.parquet"
|
||
if not p.exists():
|
||
sys.exit(f"missing M5 data: {p}")
|
||
return load_bars(p)
|
||
|
||
|
||
def load_m1() -> pd.DataFrame:
|
||
from shared.data.loaders import load_bars
|
||
p = PROJECT / "data" / "XAUUSD_M1_2024-06-26_2026-06-26.parquet"
|
||
if not p.exists():
|
||
sys.exit(f"missing M1 data: {p} — run scripts/download_xauusd_m1.py first")
|
||
return load_bars(p)
|
||
|
||
|
||
def main() -> int:
|
||
ap = argparse.ArgumentParser()
|
||
ap.add_argument("--smoke", action="store_true",
|
||
help="30 trials on IS H1 2025; validate the script end-to-end")
|
||
ap.add_argument("--trials", type=int, default=500,
|
||
help="trial budget for the full study (default 500)")
|
||
ap.add_argument("--db", type=Path,
|
||
default=PROJECT / "studies" / "optuna" / "gold_scalper_pro_is2025.db",
|
||
help="SQLite path for the full study (resumable)")
|
||
args = ap.parse_args()
|
||
if args.smoke:
|
||
return run_smoke()
|
||
return run_full(args.db, args.trials)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|