Compare commits

..
Author SHA1 Message Date
google-labs-jules[bot]andmaghdam f34d9173be 🧪 Add tests for create_labels_multi_bar
Implement tests for `create_labels_multi_bar` within `features/labeling_schemes.py` to ensure it accurately generates multi-bar classification labels according to future return horizons and thresholds.

These tests improve overall codebase coverage and reliability by correctly covering boundaries and trailing returns.

Co-authored-by: maghdam <63883156+maghdam@users.noreply.github.com>
2026-03-11 18:34:50 +00:00
8 changed files with 44 additions and 47 deletions
Binary file not shown.
Binary file not shown.
Binary file not shown.
+44 -47
View File
@@ -1,67 +1,64 @@
import pandas as pd
import numpy as np
import pytest
from features.labeling_schemes import create_labels_multi_bar
def test_create_labels_multi_bar():
"""
Test create_labels_multi_bar correctly assigns labels based on future returns.
"""
# Create a simple dummy dataframe
# We want future returns over horizon=2 to be:
# index 0: (10.5 / 10.0) - 1 = 0.05 (should be +1, since >= 0.05 is not met if threshold=0.06, wait let's use exact)
# Toy dataframe with close prices
df = pd.DataFrame({
"close": [100.0, 100.0, 105.0, 95.0, 100.0, 100.0]
"close": [100.0, 102.0, 99.0, 99.0, 105.0]
})
# Let's set horizon=2, threshold=0.04
# future returns for horizon=2:
# i=0: (105.0 - 100.0)/100.0 = 0.05 => >= 0.04 => 1
# i=1: (95.0 - 100.0)/100.0 = -0.05 => <= -0.04 => -1
# i=2: (100.0 - 105.0)/105.0 = -0.0476 => <= -0.04 => -1
# i=3: (100.0 - 95.0)/95.0 = 0.0526 => >= 0.04 => 1
# i=4: NaN
# i=5: NaN
# horizon = 1, threshold = 0.01
# row 0: close = 100, future = 102, return = 0.02 >= 0.01 -> label = 1
# row 1: close = 102, future = 99, return = -3/102 = -0.0294 <= -0.01 -> label = -1
# row 2: close = 99, future = 99, return = 0.00 -> label = 0
# row 3: close = 99, future = 105, return = 6/99 = 0.0606 >= 0.01 -> label = 1
# row 4: close = 105, future = NaN
labeled_df = create_labels_multi_bar(df, horizon=2, threshold=0.04)
res = create_labels_multi_bar(df, horizon=1, threshold=0.01)
# Check that df wasn't modified in place
assert "multi_bar_label" not in df.columns
# Should have 4 rows because the last row is dropped due to NaN future return
assert len(res) == 4
# Ensure correct columns exist in result
assert "future_return_h" in labeled_df.columns
assert "multi_bar_label" in labeled_df.columns
expected_labels = [1, -1, 0, 1]
np.testing.assert_array_equal(res["multi_bar_label"].values, expected_labels)
# Since the original drops NaN, it should have 4 rows
assert len(labeled_df) == 4
# Check returns
expected_returns = [0.02, -3/102, 0.0, 6/99]
np.testing.assert_array_almost_equal(res["future_return_h"].values, expected_returns)
# Check calculated future returns roughly match expected
expected_returns = [0.05, -0.05, -0.047619047619047616, 0.052631578947368474]
np.testing.assert_allclose(labeled_df["future_return_h"].values, expected_returns, rtol=1e-5)
def test_create_labels_multi_bar_custom_horizon():
# Test with horizon=2, threshold=0.05
df = pd.DataFrame({
"close": [100.0, 101.0, 105.0, 90.0, 95.0, 100.0]
})
# Check assigned labels
# horizon = 2
# row 0: close 100, future 105 (idx 2), return 0.05 >= 0.05 -> 1
# row 1: close 101, future 90 (idx 3), return -11/101 = -0.1089 <= -0.05 -> -1
# row 2: close 105, future 95 (idx 4), return -10/105 = -0.0952 <= -0.05 -> -1
# row 3: close 90, future 100 (idx 5), return 10/90 = 0.1111 >= 0.05 -> 1
# row 4: NaN
# row 5: NaN
res = create_labels_multi_bar(df, horizon=2, threshold=0.05)
assert len(res) == 4
expected_labels = [1, -1, -1, 1]
np.testing.assert_array_equal(labeled_df["multi_bar_label"].values, expected_labels)
np.testing.assert_array_equal(res["multi_bar_label"].values, expected_labels)
def test_create_labels_multi_bar_neutral():
"""
Test create_labels_multi_bar handles neutral labels correctly (returns inside threshold).
"""
def test_create_labels_multi_bar_exact_threshold():
# Check boundary condition where return is exactly the threshold
df = pd.DataFrame({
"close": [100.0, 101.0, 102.0, 99.0, 100.0]
"close": [100.0, 105.0, 95.0]
})
# Let's set horizon=1, threshold=0.02
# future returns for horizon=1:
# i=0: (101 - 100)/100 = 0.01 (neutral -> 0)
# i=1: (102 - 101)/101 = 0.0099 (neutral -> 0)
# i=2: (99 - 102)/102 = -0.0294 (down -> -1)
# i=3: (100 - 99)/99 = 0.0101 (neutral -> 0)
# threshold = 0.05, horizon = 1
# row 0: return 0.05 -> 1
# row 1: return -10/105 = -0.0952 -> -1
res = create_labels_multi_bar(df, horizon=1, threshold=0.05)
assert res.iloc[0]["multi_bar_label"] == 1
labeled_df = create_labels_multi_bar(df, horizon=1, threshold=0.02)
assert len(labeled_df) == 4
expected_labels = [0, 0, -1, 0]
np.testing.assert_array_equal(labeled_df["multi_bar_label"].values, expected_labels)
res = create_labels_multi_bar(df, horizon=1, threshold=0.1)
# return is 0.05, which is < 0.1 and > -0.1
assert res.iloc[0]["multi_bar_label"] == 0