{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Clustering Algorithm","metadata":{}},{"cell_type":"markdown","source":"**Import Libraries**","metadata":{}},{"cell_type":"code","source":"# for preprocessing\nimport pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing  import MinMaxScaler\nfrom sklearn.decomposition import PCA\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n\n# for visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom termcolor import colored\n\n\n# for clustering\nfrom sklearn.cluster import KMeans\n\nfrom sklearn.cluster import DBSCAN\nfrom scipy.stats import shapiro\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:16:56.839793Z","iopub.execute_input":"2022-07-13T12:16:56.840403Z","iopub.status.idle":"2022-07-13T12:17:00.493289Z","shell.execute_reply.started":"2022-07-13T12:16:56.840259Z","shell.execute_reply":"2022-07-13T12:17:00.491566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the Datset\ndf = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv')\ndf.head()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:17:00.495876Z","iopub.execute_input":"2022-07-13T12:17:00.496443Z","iopub.status.idle":"2022-07-13T12:17:02.034255Z","shell.execute_reply.started":"2022-07-13T12:17:00.496373Z","shell.execute_reply":"2022-07-13T12:17:02.032795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df = pd.DataFrame()\nfinal_df['id'] = df['id']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop('id', axis  =1, inplace = True)    # drop the id column as it is no use  for cluster formation\ndf","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the shape of dataset\nprint('Dataset have {} rows and {} columns'.format(df.shape[0], df.shape[1]))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# null value percentage \npercentage_of_null_values = ((df.isnull().sum())*100/len(df)).sort_values(ascending = False)\npercentage_of_null_values","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Inference:** There is no null value present in dataset","metadata":{}},{"cell_type":"code","source":"# basic statistical properties of numeric variable\ndf.describe()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test the dataset for normal distribution","metadata":{}},{"cell_type":"code","source":"for col in df.columns:\n    stat, p_value = shapiro(df[col])\n    alpha = 0.05\n    if p_value > alpha:\n        result = colored('Accepted', 'green')\n    else:\n        result = colored('Rejected', 'red')\n    print('Feature: {}\\t Hypothesis: {}'.format(col, result))\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Inference:** It is found from the shapiro Wilk test that 12 variable are not normally distributed","metadata":{}},{"cell_type":"markdown","source":"### Correlation between the variable","metadata":{}},{"cell_type":"code","source":"sns.set_style('darkgrid')\nplt.figure(figsize = (12,7))\nsns.heatmap(df.corr().abs(), cmap = 'Greens')\nplt.title('Correlation')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Inference:** Most of the variable are independent only variable from f_07 to f_13 shows the week correlation between them","metadata":{}},{"cell_type":"markdown","source":"## We will use different Cluster Algorithm and see which one is able to cluster  data better other.","metadata":{}},{"cell_type":"markdown","source":"### Kmeans Clustering  \nThe K-means clustering algorithm computes centroids and repeats until the optimal centroid is found. It is presumptively known how many clusters there are. It is also known as the flat clustering algorithm. The number of clusters found from data by the method is denoted by the letter ‘K’ in K-means.","metadata":{}},{"cell_type":"code","source":"kmeans = KMeans(n_clusters = 3)   # randomly selected how many cluster needed\nkmeans.fit_predict(df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kmeans.labels_    # adding cluster in dataframe\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we do some optimize the k-means by using some properties \nscale the data: It is used to \nelbow method: This method is used to find the optimize number of cluster needed to classify them.\n","metadata":{}},{"cell_type":"code","source":"# Scaling the dataset\n\ncolumns = df.columns\n\nscaled = MinMaxScaler()\nscaled.fit(df[columns])\ndf[columns] = scaled.transform(df[columns])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculating the inertia for finding optimal k value\ninertia = []\nfor i in range(1,11):\n    kmeans = KMeans(\n        n_clusters=i, init=\"k-means++\",\n        n_init=10,\n        tol=1e-04, random_state=42\n    )\n    kmeans.fit(df)\n    inertia.append(kmeans.inertia_)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(data=go.Scatter(x=np.arange(1,11),y=inertia))\nfig.update_layout(title=\"Elbow Method\",xaxis=dict(range=[0,11],title=\"Cluster Number\"),\n                  yaxis={'title':'Inertia'})\n       ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Infernce:** Inertia is a sum of square distance between the data points and there center, from the graph optimal number of  \n    of cluster will be **5** as by then increasing the cluster there is not much decrease in inertia.","metadata":{}},{"cell_type":"code","source":"kmeans = KMeans(n_clusters = 5, random_state = 42)   \nkmeans.fit_predict(df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kmeans = pd.DataFrame(kmeans.labels_)\nfinal_df['kmeans'] = kmeans\nfinal_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### DSCAN Clustering \nDBSCAN clustering algorithm considered as one of the best algorithm as in this we didn't need to pre-define the number of  \ncluster required and it is robust to outliers.","metadata":{}},{"cell_type":"code","source":"model_2 = DBSCAN()\nmodel_2.fit_predict(df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DBSCAN = pd.DataFrame(model_2.labels_)\nfinal_df['DBSCAN'] = DBSCAN","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Gaussian Mixture Model","metadata":{}},{"cell_type":"code","source":"%%time\n\nmodel_3 = GaussianMixture(n_components = 5, random_state = 42)\nlabel = model_3.fit_predict(df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GMM = pd.DataFrame(label)\nfinal_df['GMM'] = GMM","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualize the Clustering","metadata":{}},{"cell_type":"code","source":"reduced_data = PCA(n_components = 2).fit_transform(df)\ndata = pd.DataFrame(reduced_data, columns = ['PCA1', 'PCA2'])\nfinal_df = final_df.join(data)\nfinal_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (12,7))\nsns.scatterplot(data = final_df, x = 'PCA1', y = 'PCA2',  hue = 'kmeans', palette = 'Set2')\nplt.title('Kmeans');","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (12,7))\nsns.scatterplot(data = final_df, x = 'PCA1', y = 'PCA2',  hue = 'DBSCAN', palette = 'vlag')\nplt.title('DBSCAN');","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (12,7))\nsns.scatterplot(data = final_df, x = 'PCA1', y = 'PCA2',  hue = 'GMM', palette = 'Spectral')\nplt.title('GMM');","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['GMM'] = final_df['GMM']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"polar=df.groupby(\"GMM\").mean().reset_index()\npolar=pd.melt(polar,id_vars=[\"GMM\"])\nfig4 = px.line_polar(polar, r=\"value\", theta=\"variable\", color=\"GMM\", line_close=True,height=800,width=1400)\nfig4.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"final_df = final_df[['id', 'GMM']]\nfinal_df.columns = ['id', 'Predicted']\nfinal_df.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Refernce  \nhttps://www.kaggle.com/code/samuelcortinhas/tps-july-22-unsupervised-clustering/notebook\n\nhttps://towardsdatascience.com/clustering-with-more-than-two-features-try-this-to-explain-your-findings-b053007d680a","metadata":{}}]}