451 KiB
451 KiB
In [41]:
%pip install pandas -q
%pip install numpy -q
%pip install seaborn -q
%pip install matplotlib -q
%pip install yfinance -q
%pip install scikit-learn -q
%pip install yellowbrick[1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m A new release of pip is available: [0m[31;49m23.0.1[0m[39;49m -> [0m[32;49m23.2.1[0m [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m To update, run: [0m[32;49mpip install --upgrade pip[0m Note: you may need to restart the kernel to use updated packages. [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m A new release of pip is available: [0m[31;49m23.0.1[0m[39;49m -> [0m[32;49m23.2.1[0m [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m To update, run: [0m[32;49mpip install --upgrade pip[0m Note: you may need to restart the kernel to use updated packages. [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m A new release of pip is available: [0m[31;49m23.0.1[0m[39;49m -> [0m[32;49m23.2.1[0m [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m To update, run: [0m[32;49mpip install --upgrade pip[0m Note: you may need to restart the kernel to use updated packages. [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m A new release of pip is available: [0m[31;49m23.0.1[0m[39;49m -> [0m[32;49m23.2.1[0m [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m To update, run: [0m[32;49mpip install --upgrade pip[0m Note: you may need to restart the kernel to use updated packages. [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m A new release of pip is available: [0m[31;49m23.0.1[0m[39;49m -> [0m[32;49m23.2.1[0m [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m To update, run: [0m[32;49mpip install --upgrade pip[0m Note: you may need to restart the kernel to use updated packages. [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m A new release of pip is available: [0m[31;49m23.0.1[0m[39;49m -> [0m[32;49m23.2.1[0m [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m To update, run: [0m[32;49mpip install --upgrade pip[0m Note: you may need to restart the kernel to use updated packages. Collecting yellowbrick Using cached yellowbrick-1.5-py3-none-any.whl (282 kB) Requirement already satisfied: scipy>=1.0.0 in ./venv/lib/python3.10/site-packages (from yellowbrick) (1.11.1) Requirement already satisfied: cycler>=0.10.0 in ./venv/lib/python3.10/site-packages (from yellowbrick) (0.11.0) Requirement already satisfied: numpy>=1.16.0 in ./venv/lib/python3.10/site-packages (from yellowbrick) (1.25.2) Requirement already satisfied: matplotlib!=3.0.0,>=2.0.2 in ./venv/lib/python3.10/site-packages (from yellowbrick) (3.7.2) Requirement already satisfied: scikit-learn>=1.0.0 in ./venv/lib/python3.10/site-packages (from yellowbrick) (1.3.0) Requirement already satisfied: kiwisolver>=1.0.1 in ./venv/lib/python3.10/site-packages (from matplotlib!=3.0.0,>=2.0.2->yellowbrick) (1.4.4) Requirement already satisfied: fonttools>=4.22.0 in ./venv/lib/python3.10/site-packages (from matplotlib!=3.0.0,>=2.0.2->yellowbrick) (4.42.0) Requirement already satisfied: contourpy>=1.0.1 in ./venv/lib/python3.10/site-packages (from matplotlib!=3.0.0,>=2.0.2->yellowbrick) (1.1.0) Requirement already satisfied: pyparsing<3.1,>=2.3.1 in ./venv/lib/python3.10/site-packages (from matplotlib!=3.0.0,>=2.0.2->yellowbrick) (3.0.9) Requirement already satisfied: pillow>=6.2.0 in ./venv/lib/python3.10/site-packages (from matplotlib!=3.0.0,>=2.0.2->yellowbrick) (10.0.0) Requirement already satisfied: python-dateutil>=2.7 in ./venv/lib/python3.10/site-packages (from matplotlib!=3.0.0,>=2.0.2->yellowbrick) (2.8.2) Requirement already satisfied: packaging>=20.0 in ./venv/lib/python3.10/site-packages (from matplotlib!=3.0.0,>=2.0.2->yellowbrick) (23.1) Requirement already satisfied: threadpoolctl>=2.0.0 in ./venv/lib/python3.10/site-packages (from scikit-learn>=1.0.0->yellowbrick) (3.2.0) Requirement already satisfied: joblib>=1.1.1 in ./venv/lib/python3.10/site-packages (from scikit-learn>=1.0.0->yellowbrick) (1.3.2) Requirement already satisfied: six>=1.5 in ./venv/lib/python3.10/site-packages (from python-dateutil>=2.7->matplotlib!=3.0.0,>=2.0.2->yellowbrick) (1.16.0) Installing collected packages: yellowbrick Successfully installed yellowbrick-1.5 [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m A new release of pip is available: [0m[31;49m23.0.1[0m[39;49m -> [0m[32;49m23.2.1[0m [1m[[0m[34;49mnotice[0m[1;39;49m][0m[39;49m To update, run: [0m[32;49mpip install --upgrade pip[0m Note: you may need to restart the kernel to use updated packages.
In [59]:
import pandas as pd
import numpy as np
import seaborn as sns
import matplotlib.pyplot as plt
import yfinance as yf
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from sklearn.linear_model import LogisticRegression
from sklearn.svm import SVC
from sklearn.metrics import classification_report, accuracy_score
from yellowbrick.classifier import ConfusionMatrixIn [3]:
ticker = "ITUB4.SA"
data = yf.download(ticker, start="2008-01-01", progress=False)In [4]:
data.head()Out [4]:
| Open | High | Low | Close | Adj Close | Volume | |
|---|---|---|---|---|---|---|
| Date | ||||||
| 2008-01-02 | 15.100806 | 15.223335 | 14.306027 | 14.411997 | 8.390958 | 6442543 |
| 2008-01-03 | 14.372258 | 14.736532 | 14.117267 | 14.140448 | 8.232860 | 7212266 |
| 2008-01-04 | 14.315962 | 14.431867 | 13.799355 | 14.239795 | 8.290697 | 7374122 |
| 2008-01-07 | 14.292780 | 14.570953 | 14.007985 | 14.239795 | 8.290697 | 7597580 |
| 2008-01-08 | 14.534526 | 14.703416 | 14.256353 | 14.372258 | 8.367819 | 5372057 |
In [5]:
data.shapeOut [5]:
(3876, 6)
In [6]:
data.describe()Out [6]:
| Open | High | Low | Close | Adj Close | Volume | |
|---|---|---|---|---|---|---|
| count | 3876.000000 | 3876.000000 | 3876.000000 | 3876.000000 | 3876.000000 | 3.876000e+03 |
| mean | 21.517539 | 21.799286 | 21.222586 | 21.512633 | 16.731696 | 2.267386e+07 |
| std | 7.123009 | 7.186481 | 7.053429 | 7.115916 | 7.935137 | 1.393839e+07 |
| min | 7.244082 | 7.653890 | 6.999853 | 7.244082 | 4.217649 | 0.000000e+00 |
| 25% | 15.766475 | 15.967374 | 15.577260 | 15.775082 | 9.843326 | 1.333177e+07 |
| 50% | 19.382919 | 19.682645 | 19.144298 | 19.423195 | 13.558575 | 1.949876e+07 |
| 75% | 26.915833 | 27.295000 | 26.580000 | 26.912500 | 24.145989 | 2.864823e+07 |
| max | 38.669998 | 39.790001 | 38.400002 | 39.689999 | 33.508636 | 1.606699e+08 |
In [7]:
data.info()<class 'pandas.core.frame.DataFrame'> DatetimeIndex: 3876 entries, 2008-01-02 to 2023-08-11 Data columns (total 6 columns): # Column Non-Null Count Dtype --- ------ -------------- ----- 0 Open 3876 non-null float64 1 High 3876 non-null float64 2 Low 3876 non-null float64 3 Close 3876 non-null float64 4 Adj Close 3876 non-null float64 5 Volume 3876 non-null int64 dtypes: float64(5), int64(1) memory usage: 212.0 KB
In [8]:
plt.figure(figsize=(15, 5))
plt.plot(data["Adj Close"])
plt.title("ITUB4 close price", fontsize=24)
plt.ylabel("Price (in R$)")
plt.show()In [9]:
dataOut [9]:
| Open | High | Low | Close | Adj Close | Volume | |
|---|---|---|---|---|---|---|
| Date | ||||||
| 2008-01-02 | 15.100806 | 15.223335 | 14.306027 | 14.411997 | 8.390958 | 6442543 |
| 2008-01-03 | 14.372258 | 14.736532 | 14.117267 | 14.140448 | 8.232860 | 7212266 |
| 2008-01-04 | 14.315962 | 14.431867 | 13.799355 | 14.239795 | 8.290697 | 7374122 |
| 2008-01-07 | 14.292780 | 14.570953 | 14.007985 | 14.239795 | 8.290697 | 7597580 |
| 2008-01-08 | 14.534526 | 14.703416 | 14.256353 | 14.372258 | 8.367819 | 5372057 |
| ... | ... | ... | ... | ... | ... | ... |
| 2023-08-07 | 27.930000 | 28.280001 | 27.620001 | 27.660000 | 27.660000 | 33001900 |
| 2023-08-08 | 27.360001 | 28.010000 | 27.110001 | 27.600000 | 27.600000 | 44517500 |
| 2023-08-09 | 27.500000 | 27.600000 | 26.920000 | 27.500000 | 27.500000 | 35060900 |
| 2023-08-10 | 27.650000 | 27.940001 | 27.520000 | 27.590000 | 27.590000 | 30620600 |
| 2023-08-11 | 27.629999 | 27.799999 | 27.459999 | 27.639999 | 27.639999 | 19424700 |
3876 rows × 6 columns
In [10]:
data[data["Close"] == data["Adj Close"]].shapeOut [10]:
(9, 6)
In [11]:
data.isnull().sum()Out [11]:
Open 0 High 0 Low 0 Close 0 Adj Close 0 Volume 0 dtype: int64
In [12]:
features = ["Open", "High", "Low", "Close", "Adj Close", "Volume"]
plt.subplots(figsize=(20, 10))
for i, column in enumerate(features):
plt.subplot(2, 3, i+1)
# https://stackoverflow.com/questions/67638590/emulating-deprecated-seaborn-distplots
_, FD_bins = np.histogram(data[column], bins="fd")
bin_nr = min(len(FD_bins)-1, 50)
sns.histplot(data=data, x=column, bins=bin_nr,
stat="density", alpha=0.4, kde=True, kde_kws={"cut": 3})
plt.show()/tmp/ipykernel_22901/980192968.py:6: MatplotlibDeprecationWarning: Auto-removal of overlapping axes is deprecated since 3.6 and will be removed two minor releases later; explicitly call ax.remove() as needed. plt.subplot(2, 3, i+1)
In [13]:
plt.subplots(figsize=(20,10))
for i, column in enumerate(features):
plt.subplot(2,3,i+1)
sns.boxplot(data[column]).set_title(column)
plt.show()/tmp/ipykernel_22901/2157267840.py:4: MatplotlibDeprecationWarning: Auto-removal of overlapping axes is deprecated since 3.6 and will be removed two minor releases later; explicitly call ax.remove() as needed. plt.subplot(2,3,i+1)
In [18]:
data['day'] = data.index.day.astype('int')
data['month'] = data.index.month.astype('int')
data['year'] = data.index.year.astype('int')
data.head()Out [18]:
| Open | High | Low | Close | Adj Close | Volume | day | month | year | |
|---|---|---|---|---|---|---|---|---|---|
| Date | |||||||||
| 2008-01-02 | 15.100806 | 15.223335 | 14.306027 | 14.411997 | 8.390958 | 6442543 | 2 | 1 | 2008 |
| 2008-01-03 | 14.372258 | 14.736532 | 14.117267 | 14.140448 | 8.232860 | 7212266 | 3 | 1 | 2008 |
| 2008-01-04 | 14.315962 | 14.431867 | 13.799355 | 14.239795 | 8.290697 | 7374122 | 4 | 1 | 2008 |
| 2008-01-07 | 14.292780 | 14.570953 | 14.007985 | 14.239795 | 8.290697 | 7597580 | 7 | 1 | 2008 |
| 2008-01-08 | 14.534526 | 14.703416 | 14.256353 | 14.372258 | 8.367819 | 5372057 | 8 | 1 | 2008 |
In [25]:
data['is_quarter_end'] = np.where(data['month'] % 3 == 0, 1, 0)
data.head()Out [25]:
| Open | High | Low | Close | Adj Close | Volume | day | month | year | is_quarter_end | |
|---|---|---|---|---|---|---|---|---|---|---|
| Date | ||||||||||
| 2008-01-02 | 15.100806 | 15.223335 | 14.306027 | 14.411997 | 8.390958 | 6442543 | 2 | 1 | 2008 | 0 |
| 2008-01-03 | 14.372258 | 14.736532 | 14.117267 | 14.140448 | 8.232860 | 7212266 | 3 | 1 | 2008 | 0 |
| 2008-01-04 | 14.315962 | 14.431867 | 13.799355 | 14.239795 | 8.290697 | 7374122 | 4 | 1 | 2008 | 0 |
| 2008-01-07 | 14.292780 | 14.570953 | 14.007985 | 14.239795 | 8.290697 | 7597580 | 7 | 1 | 2008 | 0 |
| 2008-01-08 | 14.534526 | 14.703416 | 14.256353 | 14.372258 | 8.367819 | 5372057 | 8 | 1 | 2008 | 0 |
In [29]:
data_grouped = data.groupby('year').mean()
plt.subplots(figsize=(20,15))
for i, column in enumerate(['Open', 'High', 'Low', 'Close']):
plt.subplot(2,2,i+1)
data_grouped[col].plot.bar().set_title(column)
plt.show()/tmp/ipykernel_22901/1494818884.py:5: MatplotlibDeprecationWarning: Auto-removal of overlapping axes is deprecated since 3.6 and will be removed two minor releases later; explicitly call ax.remove() as needed. plt.subplot(2,2,i+1)
In [31]:
data.groupby('is_quarter_end').mean()Out [31]:
| Open | High | Low | Close | Adj Close | Volume | day | month | year | |
|---|---|---|---|---|---|---|---|---|---|
| is_quarter_end | |||||||||
| 0 | 21.562839 | 21.844573 | 21.272231 | 21.561341 | 16.760581 | 2.262913e+07 | 15.810131 | 5.978732 | 2015.344934 |
| 1 | 21.426729 | 21.708502 | 21.123064 | 21.414992 | 16.673791 | 2.276353e+07 | 15.598450 | 7.327907 | 2015.277519 |
In [32]:
data['open-close'] = data['Open'] - data['Adj Close']
data['low-high'] = data['Low'] - data['High']
data['target'] = np.where(data['Adj Close'].shift(-1) > data['Adj Close'], 1, 0) # "1" if the "Adj Close" of the previous day is higher than the current "Adj Close"In [33]:
plt.pie(
data['target'].value_counts().values,
labels=[0, 1],
autopct='%1.1f%%'
)
plt.show()In [35]:
plt.figure(figsize=(10, 10))
# As our concern is with the highly
# correlated features only so, we will visualize
# our heatmap as per that criteria only.
sns.heatmap(data.corr() > 0.9, annot=True, cbar=False)
plt.show()In [94]:
features = data[['open-close', 'low-high', 'is_quarter_end']]
target = data['target']
scaler = StandardScaler()
features = scaler.fit_transform(features)
X_train, X_test, y_train, y_test = train_test_split(
features, target, test_size=0.1, random_state=2022
)
X_train.shape, X_test.shapeOut [94]:
((3488, 3), (388, 3))
In [116]:
model = SVC(kernel='sigmoid', C=1, probability=True)
model.fit(X_train, y_train)
predicts = model.predict(X_test)In [117]:
cm = ConfusionMatrix(model)
cm.fit(X_train, y_train)
cm.score(X_test, y_test)Out [117]:
0.5180412371134021
In [118]:
print(classification_report(y_test, predicts)) precision recall f1-score support
0 0.48 0.52 0.50 181
1 0.55 0.51 0.53 207
accuracy 0.52 388
macro avg 0.52 0.52 0.52 388
weighted avg 0.52 0.52 0.52 388
In [119]:
accuracy_score(Y_valid, predicts)Out [119]:
0.5180412371134021
In [ ]: