release: v0.19.0

This commit is contained in:
github-actions[bot]
2026-08-23 13:31:37 +00:00
parent a5f51e2fde
commit 44f8ed1a91
43 changed files with 2666 additions and 205 deletions
+190
View File
@@ -0,0 +1,190 @@
"""Look-ahead — the leak the detector cannot see.
The classic beginner's leak, and the one that survives review: computing a
statistic over the *whole* dataset in the notebook, then using it as a strategy
parameter. Here it is the mean price of the entire asset, used from bar 0 —
where 749 of the 750 days it summarises are still in the future.
**There IS a leak in this strategy. It is deliberate.** The question the
example answers is not whether the leak exists, but which audit methods can
see it — and two of the three cannot. Read every "sees nothing" below as a
statement about the *method*, never as a clean bill of health for the
strategy.
What this example shows, in order:
1. the leak triples the return and doubles the Sharpe;
2. `detect_lookahead` sees nothing;
3. perturbing every future bar sees nothing either;
4. the one method that catches it: re-derive the parameter on truncated data.
**Why the first two are blind.** Both re-run the *same strategy* on shorter or
altered data. The mean is not recomputed by the engine: it is a number baked
into the strategy at research time. Re-running cannot see it, because the leak
already happened, in the notebook, before the backtest existed.
That is not a bug to fix in the detector, it is the shape of the problem. No
re-run method can audit a constant. The only defence is to treat every
parameter derived from data as part of the pipeline, and re-derive it on
whatever window you are testing.
Self-contained: generates its own synthetic data in a temp store.
Demonstrates:
- a parameter computed over the whole dataset, used from bar 0
- two audit methods that do not see it, and why
- the one that does: re-deriving the parameter per window
Data: synthetic (seed 11) — generated by this file, reproducible.
The fixture DETERMINES the outcome: see examples/README.md.
Usage:
python examples/25_lookahead_trap.py
"""
import os
import tempfile
import numpy as np
import pandas as pd
import manifoldbt as mbt
from manifoldbt.indicators import close
from manifoldbt.helpers import Interval, Slippage
# -- A mean-reverting series: exactly where knowing the mean is gold ----------
N_DAYS = 750
rng = np.random.default_rng(11)
level = np.cumsum(rng.normal(0.0, 0.018, N_DAYS))
px = 100.0 * np.exp(level - np.linspace(0, level[-1], N_DAYS))
o = px
c = np.roll(px, -1)
c[-1] = px[-1]
amp = np.abs(rng.normal(0.0, 0.004, N_DAYS))
frame = pd.DataFrame({
"timestamp": pd.date_range("2022-01-01", periods=N_DAYS, freq="1D", tz="UTC"),
"open": o,
"high": np.maximum(o, c) * (1 + amp),
"low": np.minimum(o, c) * (1 - amp),
"close": c,
"volume": rng.uniform(1_000, 5_000, N_DAYS),
})
def store_of(df, tag):
root = tempfile.mkdtemp(prefix=f"mbt_ex25_{tag}_")
return mbt.import_dataframe(
df, symbol="SYNTH", symbol_id=1, interval="1d",
data_root=os.path.join(root, "data"),
metadata_db=os.path.join(root, "meta.sqlite"),
)
def config_of(df):
ts = pd.DatetimeIndex(df["timestamp"])
return mbt.BacktestConfig(
universe=[1],
time_range_start=int(ts[0].value),
time_range_end=int(ts[-1].value) + 86_400_000_000_000,
bar_interval=Interval.days(1),
initial_capital=10_000,
execution=mbt.ExecutionConfig(
signal_delay=1, max_position_pct=1.0,
allow_short=True, position_sizing_mode="FractionOfEquity",
),
slippage=Slippage.fixed_bps(0),
warmup_bars=0,
)
def leaky(mean_price):
"""THE LEAK: `mean_price` comes from `df.close.mean()` over everything."""
return (
mbt.Strategy.create("global_mean_leak")
.signal("edge", close)
.size(mbt.when(close < mean_price, 1.0, -1.0))
.describe("Long below the global mean, short above")
)
def causal(window):
"""The honest twin: a rolling mean knows only the past."""
return (
mbt.Strategy.create("rolling_mean")
.signal("edge", close)
.size(mbt.when(close < close.rolling_mean(window), 1.0, -1.0))
.describe("Long below the rolling mean")
)
def equity(result):
return np.array([float(x) for x in result.equity_curve])
# The two words this example turns on. "SEES NOTHING" describes the METHOD,
# never the strategy: the leak below is there whatever any audit reports.
BLIND, CAUGHT = "SEES NOTHING", "CATCHES THE LEAK"
if __name__ == "__main__":
global_mean = float(frame["close"].mean())
store = store_of(frame, "full")
# -- 1. The seduction -----------------------------------------------------
leaked = mbt.run(leaky(global_mean), config_of(frame), store)
honest = mbt.run(causal(60), config_of(frame), store)
print(f"Global mean over all {N_DAYS} days: {global_mean:.4f}")
print("(at bar 0, 749 of those days have not happened yet)")
print("\nThis strategy IS leaking. Below: which audits notice.\n")
for label, r in (("with the leak ", leaked), ("rolling mean ", honest)):
m = r.metrics
print(f" {label} return {100 * m['total_return']:+8.2f}% "
f"sharpe {m['sharpe']:5.2f} trades {r.trade_count}")
# -- 2. The detector -------------------------------------------------------
from manifoldbt.diagnostics import detect_lookahead
report = detect_lookahead(leaky(global_mean), config_of(frame), store, mode="all")
compared = sum(r.total_trades_overlap for r in report.reports)
verdict = BLIND if report.passed else CAUGHT
print(f"\n [1] detect_lookahead .......... {verdict}")
print(f" ({compared} trades compared, so the verdict is not empty)")
# -- 3. Perturbing the future ---------------------------------------------
split = 500
reference = equity(mbt.run(leaky(global_mean), config_of(frame), store))
corrupted = frame.copy()
tail = slice(split + 1, None)
factor = 1.0 + np.random.default_rng(99).uniform(-0.05, 0.05, N_DAYS - split - 1)
for col in ("open", "high", "low", "close"):
corrupted.loc[corrupted.index[tail], col] = corrupted[col].to_numpy()[tail] * factor
perturbed = equity(mbt.run(leaky(global_mean), config_of(frame),
store_of(corrupted, "pert")))
n = min(len(reference), len(perturbed), split + 1)
drift = float(np.abs(reference[:n] - perturbed[:n]).max())
verdict = CAUGHT if drift > 0 else BLIND
print(f" [2] future perturbation ....... {verdict}")
print(f" (past moved by {drift:.3e} while the future moved by "
f"{abs(reference[-1] - perturbed[-1]):.0f})")
# -- 4. The method that works ---------------------------------------------
# Treat the mean as what it is: a pipeline step, not a constant. A
# researcher standing at bar 500 only has the first 500 bars.
truncated = frame.iloc[:split + 1]
honest_mean = float(truncated["close"].mean())
with_future = equity(mbt.run(leaky(global_mean), config_of(truncated),
store_of(truncated, "t1")))
with_past = equity(mbt.run(leaky(honest_mean), config_of(truncated),
store_of(truncated, "t2")))
m = min(len(with_future), len(with_past))
gap = float(np.abs(with_future[:m] - with_past[:m]).max())
verdict = CAUGHT if gap > 0 else BLIND
print(f" [3] re-deriving the parameter . {verdict}")
print(f" mean from the whole set : {global_mean:.4f} -> "
f"final equity {with_future[-1]:,.0f}")
print(f" mean from the past only : {honest_mean:.4f} -> "
f"final equity {with_past[-1]:,.0f}")
print(f" same window, same bars, equity differs by {gap:.0f}")
print("\n Verdict: the leak is real, and only [3] found it. A")
print(" 'SEES NOTHING' is a statement about the method, not about")
print(" the strategy.")