# APPLY REGRESSION TECHNIQUES FOR TIME SERIES DATA
import matplotlib.pyplot as plt
import numpy as np
import pandas as pd
from sklearn.linear_model import LinearRegression
from sklearn.metrics import r2_score, mean_absolute_error
# ==========================================================
# LOAD DATASET
# ==========================================================
dataset = pd.read_csv("/content/dow-jones-index.data")
print(dataset.head())
# ==========================================================
# PREPROCESS THE DATA
# ==========================================================
observations = {}
for el in dataset[["stock", "close"]].iterrows():
    stock = el[1].stock
    close = float(
        el[1].close.replace("$", "")
    )
    try:
        observations[stock].append(close)
    except KeyError:
        observations[stock] = [close]
print("\nObservations:")
print(observations)
# ==========================================================
# CREATE ARRAY
# ==========================================================
X = []
stock_names = sorted(
    observations.keys()
)
for stock in stock_names:
    X.append(
        observations[stock]
    )
X = np.array(X)
print("\nX Shape:")
print(X.shape)
# ==========================================================
# TRAINING DATA
# ==========================================================
# First 12 weeks are used as input
X_train = X[:, :12]
# 13th week is used as output
y_train = X[:, 12]
# ==========================================================
# LINEAR REGRESSION
# ==========================================================
reg_l = LinearRegression()
# Train the model
reg_l.fit(
    X_train,
    y_train
)
# ==========================================================
# TEST MODEL FOR DIFFERENT WEEKS
# ==========================================================
plot_vals = []
for offset in range(
    0,
    X.shape[1] - X_train.shape[1]
):
    # Input values
    X_test = X[
        :,
        offset:12 + offset
    ]
    # Actual output
    y_test = X[
        :,
        12 + offset
    ]
    # Prediction
    y_pred = reg_l.predict(
        X_test
    )
    # R2 Score
    r2 = r2_score(
        y_test,
        y_pred
    )
    # Mean Absolute Error
    mae = mean_absolute_error(
        y_test,
        y_pred
    )
    print(
        "Offset =",
        offset,
        "R2-Score =",
        r2
    )
    print(
        "Offset =",
        offset,
        "MAE =",
        mae
    )
    plot_vals.append(
        (
            offset,
            r2,
            mae
        )
    )
# ==========================================================
# DISPLAY MEAN AND VARIANCE
# ==========================================================
print()
print(
    "R2-Score : Mean =",
    np.mean(
        [x[1] for x in plot_vals]
    ),
    "Variance =",
    np.var(
        [x[1] for x in plot_vals]
    )
)
print(
    "MAE-Score : Mean =",
    np.mean(
        [x[2] for x in plot_vals]
    ),
    "Variance =",
    np.var(
        [x[2] for x in plot_vals]
    )
)
# ==========================================================
# PLOT GRAPH
# ==========================================================
fig, ax1 = plt.subplots()
# Plot R2 Score
ax1.plot(
    [x[0] for x in plot_vals],
    [x[1] for x in plot_vals],
    "b-"
)
ax1.set_xlabel(
    "Test week"
)
ax1.set_ylabel(
    "R2-Score",
    color="b"
)
for t1 in ax1.get_yticklabels():
    t1.set_color("b")
ax1.set_ylim(
    [0, 1]
)
# ==========================================================
# SECOND Y AXIS FOR MAE
# ==========================================================
ax2 = ax1.twinx()
ax2.plot(
    [x[0] for x in plot_vals],
    [x[2] for x in plot_vals],
    "r-"
)
ax2.set_ylabel(
    "MAE-Score",
    color="r"
)
for t2 in ax2.get_yticklabels():
    t2.set_color("r")
ax2.set_ylim(
    [0, 3.3]
)
plt.xlim(
    [-0.1, 12.1]
)
plt.show()
