import pandas as pd
from sklearn.preprocessing import StandardScaler
import sklearn.preprocessing
import numpy as np
import matplotlib.pyplot as plt
from sklearn.cluster import KMeans
from sklearn.impute import SimpleImputer
import warnings
warnings.filterwarnings("ignore", category = FutureWarning)
custsales = pd.read_csv("K MEANS DATA.csv")
custsales.head()
custsales.shape
custsales = custsales.dropna()
custsales_cl = custsales.drop(columns = ['Custid', 'region'])
custsales_cl.head()
df_scaled = sklearn.preprocessing.scale(custsales_cl)
df_scaled
#elbow method to determine the optimal value of k
distortions = []
K = range(2,10)
for k in K:
    kmeanModel = KMeans(n_clusters = k, n_init = 10)
    kmeanModel.fit(df_scaled)
    distortions.append(kmeanModel.inertia_)
plt.figure(figsize = (8,6))
plt.plot(K, distortions, 'bx-')
plt.xlabel('k')
plt.ylabel('Distortion')
plt.title('The Elbow Method showing the optimal k')
plt.show()
np.random.seed(123)
CL = KMeans(n_clusters = 3)
CL.fit(df_scaled)
labels = CL.labels_
cluster_sizes = np.bincount(labels)
print(cluster_sizes)
segment = pd.DataFrame({'segment': labels})
custsales = pd.concat([custsales_cl.reset_index(drop=True), segment], axis = 1)
custsales.head()
custsales['segment'] = custsales['segment'] + 1
custsales.head()
seg_analysis = custsales.groupby('segment')[['nsv',
'n_brands',
'n_bills',
'growth']].mean().reset_index().round(2)
seg_analysis
col_list = ['nsv', 'n_brands', 'n_bills', 'growth']
#now create a figure
fig, axes = plt.subplots(nrows = 2, ncols = 2, figsize = (15, 8))
axe = axes.ravel()
#Now plot each zone on a particular axis
for i, cols in enumerate(col_list):
    seg_analysis.plot.bar('segment', cols, color = 'blue', ax = axe[i], sharey = True)
plt.tight_layout()
plt.show()
