feat: MT5-Quant MCP server for backtesting and optimization

MCP server exposing MetaTrader 5 strategy development tools to AI
assistants (Claude, Cursor, etc.) on macOS (CrossOver) and Linux (Wine).

Tools:
- run_backtest: full pipeline — compile EA, clean cache, backtest,
  parse HTML/XML report, analyze deals → metrics.json + analysis.json
- run_optimization: background genetic optimization with nohup/disown,
  UTF-16LE .set file handling, OptMode reset
- compile_ea: MQL5 compilation via MetaEditor with auto-detected
  include/ directory sync
- get_backtest_status / get_optimization_status: job polling
- verify_environment: Wine/MT5 path validation

Analytics:
- extract.py: MT5 HTML and SpreadsheetML XML report parser
- analyze.py: deal-level analysis (drawdown events, grid depth,
  loss sequences, monthly P&L) → analysis.json
- optimize_parser.py: optimization result parser with convergence analysis

Platform support:
- macOS CrossOver (GUI mode, no Xvfb needed)
- Linux Wine + Xvfb (headless, CI/CD compatible)
- Auto-detection of Wine executable and MT5 terminal paths
This commit is contained in:
Devid HW
2026-04-18 11:41:41 +07:00
commit 3f763827f4
25 changed files with 9824 additions and 0 deletions
View File
+1000
View File
File diff suppressed because it is too large Load Diff
+316
View File
@@ -0,0 +1,316 @@
#!/usr/bin/env python3
"""
extract.py — Single-pass MT5 report parser.
Reads MT5 backtest report (.htm or .htm.xml / SpreadsheetML) and produces:
- metrics.json (aggregate summary)
- deals.csv (all deals, 13 columns)
- deals.json (deals as JSON array)
Usage:
python3 analytics/extract.py report.htm --output-dir reports/20250101_123456/
"""
import argparse
import csv
import json
import os
import re
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
from typing import Optional
# MT5 backtest report deals table columns (actual order from HTML):
# Time, Deal, Symbol, Type, Direction, Volume, Price, Order, Commission, Swap, Profit, Balance, Comment
DEAL_COLUMNS = [
"time", "deal", "symbol", "type", "entry", "volume", "price",
"order", "commission", "swap", "profit", "balance", "comment"
]
def detect_format(path: str) -> str:
"""Return 'xml' for SpreadsheetML, 'html' for legacy HTML report."""
if path.endswith('.xml') or path.endswith('.htm.xml'):
return 'xml'
# Peek at file header
with open(path, 'rb') as f:
header = f.read(512)
if b'<?xml' in header or b'Workbook' in header:
return 'xml'
return 'html'
def read_text(path: str) -> str:
"""Read file, handling UTF-16 (MT5 default) and latin-1 fallback."""
with open(path, 'rb') as f:
raw = f.read()
for encoding in ('utf-16', 'utf-8', 'latin-1'):
try:
return raw.decode(encoding)
except (UnicodeDecodeError, LookupError):
continue
return raw.decode('latin-1', errors='replace')
def strip_tags(html: str) -> str:
return re.sub(r'<[^>]+>', '', html).strip()
# ── HTML parser ───────────────────────────────────────────────────────────────
def parse_html(path: str) -> tuple[dict, list[dict]]:
text = read_text(path)
metrics = _parse_metrics_html(text)
deals = _parse_deals_html(text)
return metrics, deals
def _parse_metrics_html(text: str) -> dict:
"""Extract aggregate metrics from the summary table."""
m = {}
# MT5 report HTML format: MetricLabel:</td>\r\n<td nowrap><b>VALUE</b></td>
# Helper patterns — values always wrapped in <b>...</b>
_b = r'[^<]*</td>\s*<td[^>]*>\s*<b>([-\d\s.,]+)</b>' # plain number
_b_pct = r'[^<]*</td>\s*<td[^>]*>\s*<b>[^(]*\(([\d.,]+)%\)' # "abs (pct%)" — capture pct
patterns = {
'net_profit': r'Net\s+Profit' + _b,
'profit_factor': r'Profit\s+Factor' + _b,
'max_dd_pct': r'Equity\s+Drawdown\s+Maximal' + _b_pct,
'sharpe_ratio': r'Sharpe\s+Ratio' + _b,
'total_trades': r'Total\s+Trades' + _b,
'recovery_factor': r'Recovery\s+Factor' + _b,
'win_rate_pct': r'Profit\s+Trades\s+\(%' + _b_pct,
'gross_profit': r'Gross\s+Profit' + _b,
'gross_loss': r'Gross\s+Loss' + _b,
}
for key, pattern in patterns.items():
match = re.search(pattern, text, re.IGNORECASE | re.DOTALL)
if match:
val = match.group(1).replace(' ', '').replace(',', '').strip()
try:
m[key] = float(val)
except ValueError:
pass
# Trades needs int
if 'total_trades' in m:
m['total_trades'] = int(m['total_trades'])
return m
def _parse_deals_html(text: str) -> list[dict]:
"""Extract deal rows from the deals table."""
# Find deals section (after "Deals" header)
deals_section = re.search(
r'<tr[^>]*>.*?Deal.*?Time.*?Type.*?Direction.*?Volume.*?</tr>(.*)',
text, re.DOTALL | re.IGNORECASE
)
if not deals_section:
return []
rows = re.findall(
r'<tr[^>]*>(.*?)</tr>',
deals_section.group(1),
re.DOTALL | re.IGNORECASE
)
deals = []
for row in rows:
cells = re.findall(r'<td[^>]*>(.*?)</td>', row, re.DOTALL | re.IGNORECASE)
cells = [strip_tags(c).replace(',', '') for c in cells]
if len(cells) < 3 or not cells[0]:
continue
# Skip balance/deposit/credit rows — 'balance' appears in the Type column (index 3)
# or sometimes in index 1; check first 5 cells
if any(c.strip().lower() in ('balance', 'credit') for c in cells[:5]):
continue
deal = {}
for i, col in enumerate(DEAL_COLUMNS):
deal[col] = cells[i] if i < len(cells) else ''
deals.append(deal)
return deals
# ── XML parser (SpreadsheetML) ─────────────────────────────────────────────────
def parse_xml(path: str) -> tuple[dict, list[dict]]:
"""Parse MT5 SpreadsheetML optimization/report XML."""
tree = ET.parse(path)
root = tree.getroot()
# Namespace handling — MT5 XML uses Excel namespace
ns = {}
ns_match = re.match(r'\{([^}]+)\}', root.tag)
if ns_match:
ns['ss'] = ns_match.group(1)
def tag(name):
return f"{{{ns['ss']}}}{name}" if ns else name
metrics = {}
deals = []
in_deals_sheet = False
for sheet in root.iter(tag('Worksheet')):
sheet_name = sheet.get(f"{{{ns['ss']}}}Name" if ns else 'Name', '')
if 'result' in sheet_name.lower() or 'report' in sheet_name.lower():
metrics = _parse_metrics_xml(sheet, tag)
elif 'deal' in sheet_name.lower() or 'trade' in sheet_name.lower():
deals = _parse_deals_xml(sheet, tag)
elif sheet_name == '':
# Unnamed sheet — check if it has deal-like structure
rows = list(sheet.iter(tag('Row')))
if len(rows) > 5:
# Try to parse as deals
candidate = _parse_deals_xml(sheet, tag)
if candidate:
deals = candidate
return metrics, deals
def _cell_value(cell, tag) -> str:
data = cell.find(tag('Data'))
return data.text.strip() if data is not None and data.text else ''
def _parse_metrics_xml(sheet, tag) -> dict:
m = {}
for row in sheet.iter(tag('Row')):
cells = [_cell_value(c, tag) for c in row.iter(tag('Cell'))]
if len(cells) < 2:
continue
key = cells[0].lower()
val = cells[1].replace(',', '').strip()
try:
fval = float(val)
if 'net profit' in key or 'net_profit' in key:
m['net_profit'] = fval
elif 'profit factor' in key:
m['profit_factor'] = fval
elif 'drawdown' in key and '%' in cells[1]:
m['max_dd_pct'] = fval
elif 'sharpe' in key:
m['sharpe_ratio'] = fval
elif 'total trades' in key:
m['total_trades'] = int(fval)
except (ValueError, AttributeError):
pass
return m
def _parse_deals_xml(sheet, tag) -> list[dict]:
deals = []
header_found = False
col_map = {}
for row in sheet.iter(tag('Row')):
cells = [_cell_value(c, tag) for c in row.iter(tag('Cell'))]
if not header_found:
# Detect header row
if any(h in str(cells).lower() for h in ('time', 'type', 'volume', 'profit')):
header_found = True
for i, h in enumerate(cells):
h_lower = h.lower().strip()
for col in DEAL_COLUMNS:
if col in h_lower or h_lower in col:
col_map[i] = col
break
continue
if not cells or not cells[0]:
continue
deal = {}
for i, val in enumerate(cells):
col = col_map.get(i)
if col:
deal[col] = val.replace(',', '')
if deal:
deals.append(deal)
return deals
# ── Writer ────────────────────────────────────────────────────────────────────
def write_outputs(metrics: dict, deals: list[dict], output_dir: str) -> dict:
os.makedirs(output_dir, exist_ok=True)
metrics_path = os.path.join(output_dir, 'metrics.json')
deals_csv_path = os.path.join(output_dir, 'deals.csv')
deals_json_path = os.path.join(output_dir, 'deals.json')
with open(metrics_path, 'w') as f:
json.dump(metrics, f, indent=2)
with open(deals_json_path, 'w') as f:
json.dump(deals, f, indent=2)
if deals:
all_keys = DEAL_COLUMNS
with open(deals_csv_path, 'w', newline='') as f:
writer = csv.DictWriter(f, fieldnames=all_keys, extrasaction='ignore')
writer.writeheader()
writer.writerows(deals)
else:
# Write empty CSV with headers
with open(deals_csv_path, 'w', newline='') as f:
writer = csv.writer(f)
writer.writerow(DEAL_COLUMNS)
return {
'metrics': metrics_path,
'deals_csv': deals_csv_path,
'deals_json': deals_json_path,
}
# ── Main ──────────────────────────────────────────────────────────────────────
def main():
parser = argparse.ArgumentParser(description='Extract MT5 backtest report')
parser.add_argument('report', help='Path to report.htm or report.htm.xml')
parser.add_argument('--output-dir', default='.', help='Output directory')
parser.add_argument('--stdout', action='store_true',
help='Print metrics JSON to stdout instead of writing files')
args = parser.parse_args()
fmt = detect_format(args.report)
if fmt == 'xml':
metrics, deals = parse_xml(args.report)
else:
metrics, deals = parse_html(args.report)
if not metrics:
print(f"WARNING: No aggregate metrics found in report", file=sys.stderr)
if not deals:
print(f"WARNING: No deals found in report (check date range and symbol)", file=sys.stderr)
if args.stdout:
json.dump({'metrics': metrics, 'deals_count': len(deals)}, sys.stdout, indent=2)
print()
return
paths = write_outputs(metrics, deals, args.output_dir)
print(f"Extracted: {len(deals)} deals, {len(metrics)} metrics")
for name, path in paths.items():
print(f" {name}: {path}")
if __name__ == '__main__':
main()
+349
View File
@@ -0,0 +1,349 @@
#!/usr/bin/env python3
"""
optimize_parser.py — Parse MT5 genetic optimization results.
Handles both HTML (.htm) and SpreadsheetML XML (.htm.xml) formats.
Usage:
python3 analytics/optimize_parser.py --job opt_20250619_143022
python3 analytics/optimize_parser.py --file reports/opt_dir/optimization.htm
python3 analytics/optimize_parser.py --file report.htm.xml --top 30 --sort profit
"""
import argparse
import json
import os
import re
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
ROOT_DIR = Path(__file__).parent.parent
def find_report(job_id: str) -> str:
"""Locate optimization report from job metadata."""
jobs_dir = ROOT_DIR / '.mt5mcp_jobs'
meta_path = jobs_dir / f'{job_id}.json'
if not meta_path.exists():
raise FileNotFoundError(f"Job not found: {job_id}. Check .mt5mcp_jobs/")
with open(meta_path) as f:
meta = json.load(f)
wine_prefix = meta.get('wine_prefix', '')
base = os.path.join(wine_prefix, 'drive_c', 'mt5mcp_opt_report')
for ext in ('.htm', '.htm.xml', '.html'):
candidate = base + ext
if os.path.exists(candidate):
return candidate
raise FileNotFoundError(
f"Optimization report not found. Expected: {base}.htm or {base}.htm.xml\n"
f"Is MT5 optimization still running? Check log: {meta.get('log_file', '')}"
)
def detect_format(path: str) -> str:
if path.endswith('.xml') or path.endswith('.htm.xml'):
return 'xml'
with open(path, 'rb') as f:
header = f.read(512)
if b'<?xml' in header or b'Workbook' in header:
return 'xml'
return 'html'
def read_text(path: str) -> str:
with open(path, 'rb') as f:
raw = f.read()
for enc in ('utf-16', 'utf-8', 'latin-1'):
try:
return raw.decode(enc)
except (UnicodeDecodeError, LookupError):
continue
return raw.decode('latin-1', errors='replace')
# ── HTML parser ───────────────────────────────────────────────────────────────
def parse_html(path: str) -> list[dict]:
text = read_text(path)
rows = re.findall(r'<tr[^>]*>(.*?)</tr>', text, re.DOTALL | re.IGNORECASE)
results = []
headers = []
for row in rows:
cells = re.findall(r'<t[dh][^>]*>(.*?)</t[dh]>', row, re.DOTALL | re.IGNORECASE)
cells = [re.sub(r'<[^>]+>', '', c).strip().replace(',', '') for c in cells]
if not cells:
continue
# Header row detection
if not headers and cells[0].lower() in ('pass', '#', 'result', 'run'):
headers = cells
continue
# Data row: first cell is pass number (digit)
if headers and cells[0].isdigit():
row_data = dict(zip(headers, cells))
results.append(row_data)
elif not headers and cells[0].isdigit() and len(cells) > 5:
# No header — use positional mapping (common MT5 layout)
results.append(_positional_row(cells))
return results
def _positional_row(cells: list[str]) -> dict:
"""Map cells by position for headerless optimization tables."""
# MT5 optimization table columns (typical order):
# Pass | Profit | Expected Payoff | Profit Factor | Recovery Factor | Sharpe | Custom | DD% | Trades | ...params
pos_names = ['pass', 'profit', 'expected_payoff', 'profit_factor',
'recovery_factor', 'sharpe_ratio', 'custom', 'max_dd_pct', 'total_trades']
row = {}
for i, name in enumerate(pos_names):
if i < len(cells):
row[name] = cells[i]
# Remaining are parameters
row['_params_raw'] = cells[len(pos_names):]
return row
# ── XML parser ────────────────────────────────────────────────────────────────
def parse_xml(path: str) -> list[dict]:
tree = ET.parse(path)
root = tree.getroot()
ns = {}
ns_match = re.match(r'\{([^}]+)\}', root.tag)
if ns_match:
ns['ss'] = ns_match.group(1)
def tag(name):
return f"{{{ns['ss']}}}{name}" if ns else name
def cell_val(cell):
data = cell.find(tag('Data'))
return data.text.strip() if data is not None and data.text else ''
results = []
headers = []
for sheet in root.iter(tag('Worksheet')):
for row in sheet.iter(tag('Row')):
cells = [cell_val(c) for c in row.iter(tag('Cell'))]
cells = [c.replace(',', '').strip() for c in cells]
if not cells:
continue
if not headers:
if any(h.lower() in ('pass', 'result', 'profit') for h in cells):
headers = cells
continue
if cells[0].isdigit():
if headers:
row_data = {}
for i, h in enumerate(headers):
row_data[h.lower().replace(' ', '_')] = cells[i] if i < len(cells) else ''
results.append(row_data)
else:
results.append(_positional_row(cells))
return results
# ── Normalizer ────────────────────────────────────────────────────────────────
def normalize(raw_results: list[dict]) -> list[dict]:
"""Convert raw parsed rows to typed dicts with consistent keys."""
normalized = []
for r in raw_results:
def fget(keys, default=0.0):
for k in keys:
for rk, rv in r.items():
if k in rk.lower().replace(' ', '_'):
try:
return float(rv)
except (ValueError, TypeError):
pass
return default
def iget(keys, default=0):
v = fget(keys, default)
return int(v)
# Extract known fields
entry = {
'pass': iget(['pass', '#']),
'net_profit': fget(['profit', 'net_profit']),
'profit_factor': fget(['profit_factor']),
'max_dd_pct': fget(['dd', 'drawdown']),
'total_trades': iget(['trades']),
'sharpe_ratio': fget(['sharpe']),
'recovery_factor': fget(['recovery']),
}
# Remaining keys are parameters
known_keys = {'pass', 'profit', 'net_profit', 'profit_factor', 'expected_payoff',
'dd', 'drawdown', 'max_dd_pct', 'trades', 'total_trades',
'sharpe', 'sharpe_ratio', 'recovery', 'recovery_factor',
'custom', '#', '_params_raw'}
params = {}
for k, v in r.items():
if not any(kw in k.lower() for kw in known_keys):
try:
params[k] = float(v)
except (ValueError, TypeError):
params[k] = v
entry['params'] = params
normalized.append(entry)
return normalized
# ── Convergence analysis ──────────────────────────────────────────────────────
def convergence_analysis(results: list[dict], top_n: int = 10) -> dict:
top = results[:top_n]
if not top:
return {}
all_param_keys = set()
for r in top:
all_param_keys.update(r.get('params', {}).keys())
strong = {} # Same value across all top-N
uncertain = [] # Varies
for key in all_param_keys:
values = set()
for r in top:
v = r.get('params', {}).get(key)
if v is not None:
values.add(v)
if len(values) == 1:
strong[key] = list(values)[0]
else:
uncertain.append(key)
return {
'top_n_agreement': strong,
'high_variance_params': uncertain,
}
# ── Display ───────────────────────────────────────────────────────────────────
def display_results(results: list[dict], top_n: int, dd_threshold: float, conv: dict):
print(f"\nTotal passes: {len(results)}")
print(f"Showing top {min(top_n, len(results))} by profit:\n")
print(f"{'Rank':<5} {'Profit':>10} {'PF':>6} {'DD%':>6} {'Sharpe':>7} {'Trades':>7} Params")
print("" * 80)
for i, r in enumerate(results[:top_n], 1):
dd = r['max_dd_pct']
risk_flag = '' if dd > dd_threshold else ''
params_str = ' '.join(f"{k}={v}" for k, v in list(r.get('params', {}).items())[:4])
print(
f"#{i:<4} ${r['net_profit']:>9,.2f} "
f"{r['profit_factor']:>5.2f} "
f"{dd:>5.2f}%"
f"{risk_flag} "
f"{r['sharpe_ratio']:>6.2f} "
f"{r['total_trades']:>7} "
f"{params_str}"
)
if conv:
print(f"\nConvergence (top-{min(top_n, len(results))} agreement):")
if conv.get('top_n_agreement'):
print(" Stable params:", ', '.join(f"{k}={v}" for k, v in conv['top_n_agreement'].items()))
if conv.get('high_variance_params'):
print(" Uncertain params:", ', '.join(conv['high_variance_params']))
# ── Main ──────────────────────────────────────────────────────────────────────
def main():
parser = argparse.ArgumentParser(description='Parse MT5 optimization results')
parser.add_argument('--job', help='Job ID from optimize.sh output')
parser.add_argument('--file', help='Direct path to optimization.htm or .htm.xml')
parser.add_argument('--top', type=int, default=20, help='Show top N results')
parser.add_argument('--sort', choices=['profit', 'profit_factor', 'sharpe'],
default='profit', help='Sort metric')
parser.add_argument('--dd-threshold', type=float, default=20.0,
help='Flag DD above this % as high-risk')
parser.add_argument('--output', help='Save results as JSON')
args = parser.parse_args()
# Locate report
if args.file:
report_path = args.file
elif args.job:
try:
report_path = find_report(args.job)
except FileNotFoundError as e:
print(f"ERROR: {e}", file=sys.stderr)
sys.exit(1)
else:
print("ERROR: Provide --job or --file", file=sys.stderr)
sys.exit(1)
if not os.path.exists(report_path):
print(f"ERROR: Report not found: {report_path}", file=sys.stderr)
sys.exit(1)
# Parse
fmt = detect_format(report_path)
if fmt == 'xml':
raw = parse_xml(report_path)
else:
raw = parse_html(report_path)
results = normalize(raw)
if not results:
print("ERROR: No optimization passes found in report.", file=sys.stderr)
sys.exit(1)
# Sort
sort_key = {
'profit': 'net_profit',
'profit_factor': 'profit_factor',
'sharpe': 'sharpe_ratio',
}[args.sort]
results.sort(key=lambda r: r.get(sort_key, 0), reverse=True)
# Convergence analysis
conv = convergence_analysis(results, top_n=10)
# Display
display_results(results, args.top, args.dd_threshold, conv)
# Optional JSON output
if args.output:
output = {
'total_passes': len(results),
'results': results[:args.top],
'convergence': conv,
}
with open(args.output, 'w') as f:
json.dump(output, f, indent=2)
print(f"\nSaved: {args.output}")
if __name__ == '__main__':
main()