import pandas as pd

from matplotlib import pyplot as plt

from scipy import stats

from scipy.stats import skew, kurtosis

df = pd.read_csv("Retail_Data.csv")

df.columns = df.columns.str.strip()

for column in ["Perindex", "Growth", "Retailer_Age"]:
    if column in df.columns:
        df[column] = pd.to_numeric(df[column], errors="coerce")

df.head()

df.info()

df.describe(include = 'all')

df['Perindex'].mean().round(2)

df['Perindex'].median()

df['Growth'].mean().round(2)

df['Growth'].median()

trimmed_mean_PI = stats.trim_mean(df['Perindex'].dropna(), 0.1).round(2)

trimmed_mean_PI

freq = df['Zone'].value_counts()

freq

round(df['Perindex'].std(), 2)

round(df['Perindex'].var(), 2)

df['Perindex'].std() / df['Perindex'].mean() * 100 if df['Perindex'].mean() != 0 else float('nan')

df['Growth'].skew().round(2)

skew(df['Growth'].dropna(), bias = False)

df['Growth'].kurtosis().round(2)

kurtosis(df['Growth'].dropna(), bias = False)

plt.figure(figsize = (5, 4))

plt.boxplot(df['Perindex'].dropna())

plt.title('Box Plot(Performance Index)')

plt.ylabel('Perindex')

plt.xlabel('Retail observations')

plt.show()

plt.figure(figsize=(7, 5))

plt.hist(df['Growth'].dropna(), bins=10)

plt.xlabel('Growth')

plt.ylabel('Frequency')

plt.title('Distribution of Growth')

plt.show()

plt.figure(figsize=(7, 5))

plt.scatter(df['Retailer_Age'], df['Perindex'])

plt.xlabel('Retailer Age')

plt.ylabel('Perindex')

plt.title('Retailer Age vs Perindex')

plt.show()
