mirror of
https://github.com/manifoldbt/manifoldbt.git
synced 2026-08-24 14:38:04 +00:00
75 lines
2.4 KiB
Python
75 lines
2.4 KiB
Python
"""Deterministic synthetic OHLCV bars, identical for every engine.
|
|||
|
|
|
||
|
|
One generator, one seed, one fingerprint. Both engines receive the *same*
|
||
|
|
DataFrame object; nothing about the data can differ between them.
|
||
|
|
|
||
|
|
Two properties are deliberate, not incidental:
|
||
|
|
|
||
|
|
* ``open == previous close`` (no overnight gap). A bar that gaps through a stop
|
||
|
|
is the one place where two engines can legitimately disagree on the fill price
|
||
|
|
while both being correct. Removing gaps removes that whole class of false
|
||
|
|
parity failures, so a real semantic drift is the only thing left that can trip
|
||
|
|
the gate.
|
||
|
|
* the intrabar range is wide enough that percentage stops and targets actually
|
||
|
|
trigger, otherwise the bracket workload would measure an empty branch.
|
||
|
|
"""
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import hashlib
|
||
|
|
|
||
|
|
import numpy as np
|
||
|
|
import pandas as pd
|
||
|
|
|
||
|
|
DEFAULT_SEED = 20260816
|
||
|
|
|
||
|
|
|
||
|
|
def make_ohlcv(
|
||
|
|
rows: int,
|
||
|
|
*,
|
||
|
|
seed: int = DEFAULT_SEED,
|
||
|
|
freq: str = "1min",
|
||
|
|
start: str = "2020-01-01",
|
||
|
|
vol: float = 3e-4,
|
||
|
|
drift: float = 2e-7,
|
||
|
|
) -> pd.DataFrame:
|
||
|
|
"""A gap-free random walk of ``rows`` bars, reproducible from ``seed``."""
|
||
|
|
rng = np.random.default_rng(seed)
|
||
|
|
|
||
|
|
log_ret = rng.normal(drift, vol, size=rows)
|
||
|
|
close = 100.0 * np.exp(np.cumsum(log_ret))
|
||
|
|
|
||
|
|
open_ = np.empty(rows, dtype=np.float64)
|
||
|
|
open_[0] = 100.0
|
||
|
|
open_[1:] = close[:-1]
|
||
|
|
|
||
|
|
# Intrabar excursion beyond the open/close body, as a fraction of price.
|
||
|
|
wick = rng.uniform(0.2, 1.8, size=rows) * vol * close
|
||
|
|
body_hi = np.maximum(open_, close)
|
||
|
|
body_lo = np.minimum(open_, close)
|
||
|
|
high = body_hi + wick
|
||
|
|
low = np.maximum(body_lo - wick, 1e-8)
|
||
|
|
|
||
|
|
return pd.DataFrame(
|
||
|
|
{
|
||
|
|
"timestamp": pd.date_range(start, periods=rows, freq=freq, tz="UTC"),
|
||
|
|
"open": open_,
|
||
|
|
"high": high,
|
||
|
|
"low": low,
|
||
|
|
"close": close,
|
||
|
|
"volume": rng.uniform(100.0, 10_000.0, size=rows),
|
||
|
|
}
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
def digest(df: pd.DataFrame) -> str:
|
||
|
|
"""Short content fingerprint of the bars, recorded in the result envelope.
|
||
|
|
|
||
|
|
Anyone re-running the harness can compare this before comparing timings: a
|
||
|
|
different digest means a different dataset, which makes the numbers
|
||
|
|
incomparable no matter how clean the machine was.
|
||
|
|
"""
|
||
|
|
h = hashlib.sha256()
|
||
|
|
for column in ("open", "high", "low", "close", "volume"):
|
||
|
|
h.update(np.ascontiguousarray(df[column].to_numpy(dtype=np.float64)).tobytes())
|
||
|
|
return h.hexdigest()[:16]
|