fix: 6 workflow flaws detected via full system test

1. CRITICAL FIX - Reports 500 Internal Server Error:
   Root cause: {{}} in onclick was Jinja2 template expression
   Fix: Changed to single {} in reports_index.html onclick

2. Score history chart always empty:
   Root cause: score_update only emitted inside wfv.passed block
   Fix: Emit score_update after every hypothesis run with {promoted:false}
   Fix: Baseline run also pushes to chart via run_complete handler

3. IS gate too strict - optimizer produces 0 candidates forever:
   Root cause: min_calmar=0.35 but best EA has calmar=0.165
   Fix: Two-tier IS check - passes if composite_score improves vs baseline
   (absolute thresholds still apply as alternative pass condition)

4. Findings emitted twice (double UI display):
   Root cause: baseline analyzed once after run, then re-analyzed in iter 1
   Fix: Cache baseline findings, reuse in iteration 1 without re-emitting

5. Chart labels improved:
   'Baseline' for first point, 'It1·h1' for hypothesis runs
   Calmar normalized -0.5..2.0 -> 0..1 for chart display

6. Best score header only updates on actual promotion (not per-hypothesis)
This commit is contained in:
LEGSTECH Optimizer
2026-04-13 03:45:14 +00:00
parent cf8613eb49
commit b70a6760ae
4 changed files with 67 additions and 17 deletions
+21 -6
View File
@@ -146,6 +146,7 @@ class OptimizerLoop:
self.best_score = baseline_metrics.composite_score
current_params = default_params.copy()
current_metrics = baseline_metrics
current_findings = findings # reuse in iter 1, avoid double-emit
no_improve_count = 0
max_iter = cfg["optimization"]["max_iterations"]
@@ -168,11 +169,14 @@ class OptimizerLoop:
if trades_df.empty and not baseline_trades.empty:
trades_df = baseline_trades
# Analysis
self.phase = "analyze"
findings = self._run_analysis(
current_metrics.run_id, trades_df, current_metrics, analyzers, store
)
# Analysis — reuse cached findings when re-analyzing same run_id
if current_findings is not None and trades_df.empty:
findings = current_findings
else:
findings = self._run_analysis(
current_metrics.run_id, trades_df, current_metrics, analyzers, store
)
current_findings = None # only reuse once
if not findings:
self._emit("log", {"level": "warn", "msg": "No actionable findings. Stopping."})
@@ -251,6 +255,17 @@ class OptimizerLoop:
store.update_hypothesis_status(hyp.hypothesis_id, "tested", run_id)
self.session_tested_deltas.append(hyp.param_delta) # track for session dedup
# Emit score update for every run so chart populates
self._emit("score_update", {
"iteration": self.iteration,
"run_id": run_id,
"score": round(test_metrics.composite_score, 4),
"calmar": round(test_metrics.calmar_ratio, 4),
"pf": round(test_metrics.profit_factor, 4),
"ts": datetime.utcnow().isoformat(),
"promoted": False,
})
if iteration_best is None or test_metrics.composite_score > iteration_best.composite_score:
iteration_best = test_metrics
iteration_best_params = test_params
@@ -262,7 +277,7 @@ class OptimizerLoop:
# Validation gate
self.phase = "validate"
gate_result = gate.run_is_check(iteration_best)
gate_result = gate.run_is_check(iteration_best, baseline_score=current_metrics.composite_score)
if not gate_result.passed:
self._emit("log", {"level": "error", "msg": f"IS gate failed: {gate_result.reason}"})
store.update_hypothesis_status(iteration_best_hyp.hypothesis_id, "rejected")
+17 -4
View File
@@ -258,9 +258,15 @@ function addCandidate(d) {
function pushChartPoint(d) {
const labels = scoreChart.data.labels;
labels.push(`Iter ${d.iteration}`);
// Label: 'Baseline' for first point, 'Iter N · run_id' for hypothesis runs
const lbl = d.run_id
? (d.run_id.startsWith('baseline') ? 'Baseline' : `It${d.iteration}·${d.run_id.split('_').pop()}`)
: `Iter ${d.iteration}`;
labels.push(lbl);
scoreChart.data.datasets[0].data.push(d.score);
scoreChart.data.datasets[1].data.push(d.calmar > 1 ? 1 : d.calmar / 4); // normalize calmar to 0-1
// Normalize calmar: clamp -0.5..2.0 → 0..1 for display
const calmarNorm = Math.max(0, Math.min(1, (d.calmar + 0.5) / 2.5));
scoreChart.data.datasets[1].data.push(calmarNorm);
if (labels.length > 50) {
labels.shift();
scoreChart.data.datasets.forEach(ds => ds.data.shift());
@@ -309,10 +315,14 @@ socket.on('run_started', d => {
socket.on('run_complete', d => {
updateMetrics(d);
const sign = d.score > 0 ? '' : '';
const sign = d.score > 0 ? '\u2713' : '\u2022';
addLog(d.score > 0.3 ? 'success' : 'info',
`${sign} ${d.run_id}: Score=${d.score} | Calmar=${d.calmar} | PF=${d.profit_factor} | DD=${d.drawdown_pct}%`
);
// Push baseline run to chart immediately (hypothesis runs pushed via score_update)
if (d.run_id && d.run_id.startsWith('baseline')) {
pushChartPoint({ iteration: 0, run_id: d.run_id, score: d.score, calmar: d.calmar });
}
});
socket.on('run_failed', d => {
@@ -333,7 +343,10 @@ socket.on('hypothesis_testing', d => {
socket.on('score_update', d => {
pushChartPoint(d);
document.getElementById('hdr-score').textContent = d.score.toFixed(4);
// Update best score header only when a candidate is actually promoted
if (d.promoted) {
document.getElementById('hdr-score').textContent = d.score.toFixed(4);
}
});
socket.on('candidate_promoted', d => {
+1 -1
View File
@@ -38,7 +38,7 @@
<h1>📁 Optimization Reports</h1>
<div class="sub">{{ runs|length }} run(s) recorded · Click any card to open the full report</div>
</div>
<a class="back-link" href="/" onclick="if(document.referrer){{history.back();return false;}}"> Back to Dashboard</a>
<a class="back-link" href="/" onclick="if(document.referrer){history.back();return false;}">&#8592; Back to Dashboard</a>
</header>
{% if not runs %}
+28 -6
View File
@@ -38,16 +38,38 @@ class ValidationGate:
# ── Phase 1: IS Check ─────────────────────────────────────────────────────
def run_is_check(self, metrics: RunMetrics) -> GateResult:
def run_is_check(
self,
metrics: RunMetrics,
baseline_score: float = 0.0,
) -> GateResult:
"""
Hard minimum thresholds. All must pass.
Two-tier IS check:
- MUST: enough trades for statistical confidence
- MUST: composite_score is better than baseline (or meets abs thresholds)
Absolute calmar/PF thresholds are logged as warnings but are NOT blockers
when the hypothesis shows clear improvement over baseline.
"""
min_trades_ok = metrics.total_trades >= self.thresh["min_trades"]
abs_pf_ok = metrics.profit_factor >= self.thresh["min_profit_factor"]
abs_calmar_ok = metrics.calmar_ratio >= self.thresh["min_calmar"]
# Score-based relative pass: hypothesis is better than baseline
score_improve_ok = metrics.composite_score > baseline_score * 1.0 + 0.01
# Gate passes if:
# a) Enough trades AND (abs thresholds met OR clearly better than baseline)
passed = min_trades_ok and (
(abs_pf_ok and abs_calmar_ok) # standard absolute pass
or score_improve_ok # OR better than baseline
)
checks = {
"min_trades": metrics.total_trades >= self.thresh["min_trades"],
"min_pf": metrics.profit_factor >= self.thresh["min_profit_factor"],
"min_calmar": metrics.calmar_ratio >= self.thresh["min_calmar"],
"min_trades": min_trades_ok,
"min_pf": abs_pf_ok,
"min_calmar": abs_calmar_ok,
"score_improve": score_improve_ok,
}
passed = all(checks.values())
reason = None
if not passed:
failed = [k for k, v in checks.items() if not v]